## What was red
`project.test.ts` (8), `extension.test.ts` (2),
`project-lock-retry.test.ts` (1), `extension-workflow-tools.test.ts` (1)
— **12 failures** on clean main, all from full-suite shard 4/4. **No
product defect in any of them.**
### Cause 1 — an incomplete mock that the fail-soft catch disguised (9
cases)
`getTaskCounts` now **enriches** each row before counting, so a renamed
wip column still counts as running work (`FNXC:WorkflowLifecycleColumns
2026-07-30-12:20`). But `resolveWorkflowIrForTask` and
`enrichRunningAgentTaskShape` were absent from the `@fusion/core` mocks.
Calling an undefined function threw, and `getTaskCounts`'s
**deliberately fail-soft** catch converted that into `{ byColumn: {},
runningAgentCount: 0 }`.
So the failures presented as "task counts are zero" — indistinguishable
from a real counting bug. Worth flagging beyond this PR:
`project-lock-retry.test.ts` exists specifically to prove a transient
lock does *not* "silently masquerade as zero tasks" (FN-7731/FN-7740),
and an incomplete mock produced that exact symptom by a different route.
**That catch will hide the next real enrichment failure the same way.**
### Cause 2 — column-vocabulary drift (3 cases)
A column-less `createTask` used to land in `triage`; post-U11 it lands
in `todo`, the merged Planning column. Three assertions were pinned to
the old landing column, or to `COLUMN_LABELS` values **the product never
prints** — the stale mock said `"Triage"` / `"To Do"` where core says
`"Planning"` / `"Todo"`. A label assertion could pass against a string
that exists only in the mock. The mock's labels are now copied from the
real table.
## Measured
| Check | Result |
|---|---|
| the four files | 12 failed → **0** (107 passed) |
| whole `@runfusion/fusion` package | **122 of 126 files green, 1656
passed** |
| `pnpm test:gate` | **726 passed** |
| `pnpm lint`, CLI `tsc --noEmit` | clean |
**Census unaffected — 721 both with and against my diff**, verified by
reverting the four files and re-measuring rather than assuming test
files are unscanned. (I also caught that `--strict` *writes* the
baseline locally; that write is reverted, so this PR does not touch the
baseline.)
## Not touched
`task.test.ts`'s **5 failures**. That file is claimed by
`feature/tool-permission-gates`, and its 5 are exactly the ones shard
4/4 reports — so shard 4/4 goes 17 → 5, with the remainder belonging to
that PR.
## Two judgement calls recorded in-file rather than made silently
**The `extension-workflow-tools` guard is RE-PINNED, not deleted.** It
asserted "a task created on `builtin:coding` lands in `triage`" — a
byte-identical regression guard that fired because the change was
*intended* (U11's merge). Deleting it would remove the only check that
this landing column stays stable; leaving it pinned a column the product
no longer declares. Re-pinning to `todo` keeps it doing its job.
**The broad-listing assertion now names the group that actually leads
the output.** It asserted a *second* group header inside a listing that
truncates to a text budget. That only ever passed because `triage` sorts
before `todo` in `COLUMNS`, so its group fitted before truncation —
fixture ordering masquerading as a bounding assertion. Per-column
coverage of every group is still asserted by the three filtered cases,
which do not truncate. I also seeded those 8 tasks in an explicit third
column so that case still covers a three-way filter instead of
collapsing to two groups.
335 lines
12 KiB
TypeScript
335 lines
12 KiB
TypeScript
/**
|
|
* FNXC:PostgresCutover 2026-07-04-00:00:
|
|
* Migrated from the legacy SQLite `new TaskStore(rootDir)` harness to the
|
|
* PostgreSQL extension harness. Workflow state is seeded and read back through
|
|
* `h.store()` (PG-backed), and the authoring tools resolve that same store via
|
|
* the harness-injected `getStore(cwd)` cache.
|
|
*
|
|
* FNXC:CliTests 2026-07-16-08:45:
|
|
* FN-8102 restores the per-test extension registration and harness root to the
|
|
* intake-column cases. They must not reference pre-migration `api` or `tmpDir`
|
|
* locals that no longer exist in this PostgreSQL-backed suite.
|
|
*/
|
|
|
|
import { afterAll, afterEach, beforeAll, beforeEach, expect, it } from "vitest";
|
|
import { pgDescribe } from "../../../core/src/__test-utils__/pg-test-harness.js";
|
|
import {
|
|
createPgExtensionHarness,
|
|
createMockApi,
|
|
registerExtension,
|
|
requireTool,
|
|
type RegisteredTool,
|
|
type ToolExecuteContext,
|
|
} from "./pg-extension-harness.js";
|
|
import { type WorkflowIr } from "@fusion/core";
|
|
|
|
const pgTest = pgDescribe;
|
|
|
|
/** Narrow a details payload value to a string (throws loudly if it isn't one). */
|
|
function asString(value: unknown): string {
|
|
if (typeof value !== "string") {
|
|
throw new Error(`expected string, got ${typeof value}`);
|
|
}
|
|
return value;
|
|
}
|
|
|
|
function makeCtx(cwd: string, taskId?: string): ToolExecuteContext {
|
|
return taskId ? { cwd, taskId } : { cwd };
|
|
}
|
|
|
|
function workflowIr(name: string): WorkflowIr {
|
|
return {
|
|
version: "v2",
|
|
name,
|
|
columns: [{ id: "todo", name: "Todo", traits: [] }],
|
|
nodes: [
|
|
{ id: "start", kind: "start", column: "todo" },
|
|
{
|
|
id: "plan",
|
|
kind: "prompt",
|
|
column: "todo",
|
|
config: { name: "Plan", prompt: "Plan the work", autoApprove: true },
|
|
},
|
|
{
|
|
id: "lint",
|
|
kind: "optional-group",
|
|
column: "todo",
|
|
config: {
|
|
name: "Lint",
|
|
defaultOn: true,
|
|
template: {
|
|
nodes: [{ id: "lint-step", kind: "gate", config: { name: "Lint", scriptName: "lint", cliSkipApproval: true } }],
|
|
edges: [],
|
|
},
|
|
},
|
|
},
|
|
{ id: "end", kind: "end", column: "todo" },
|
|
],
|
|
edges: [
|
|
{ from: "start", to: "plan", condition: "success" },
|
|
{ from: "plan", to: "lint", condition: "success" },
|
|
{ from: "lint", to: "end", condition: "success" },
|
|
],
|
|
settings: [
|
|
{ id: "workflowStepTimeoutMs", name: "Step timeout (ms)", type: "number", default: 360000 },
|
|
],
|
|
} as WorkflowIr;
|
|
}
|
|
|
|
// kbExtension registers richer tool descriptors (label/description/promptGuidelines)
|
|
// than the harness's intentionally-minimal RegisteredTool surface; narrow once for
|
|
// the single registration test that inspects promptGuidelines.
|
|
function promptGuidelinesOf(tool: RegisteredTool): string[] | undefined {
|
|
const def = tool as RegisteredTool & { promptGuidelines?: string[] };
|
|
return def.promptGuidelines;
|
|
}
|
|
|
|
pgTest("pi extension workflow authoring tools", () => {
|
|
const h = createPgExtensionHarness("fn-cli-workflow");
|
|
|
|
beforeAll(h.beforeAll);
|
|
beforeEach(h.beforeEach);
|
|
afterEach(h.afterEach);
|
|
afterAll(h.afterAll);
|
|
|
|
it("registers the full workflow authoring surface in the published API", () => {
|
|
/*
|
|
FNXC:WorkflowAuthoringTools 2026-06-29-22:48:
|
|
FN-7245 requires published/pi agents to see the same workflow authoring vocabulary as engine lanes, including trait discovery and settings, instead of relying on task workflow-selection references alone.
|
|
*/
|
|
const api = createMockApi();
|
|
registerExtension(api);
|
|
expect([...api.tools.keys()].sort()).toEqual(expect.arrayContaining([
|
|
"fn_workflow_list",
|
|
"fn_workflow_get",
|
|
"fn_workflow_validate",
|
|
"fn_workflow_create",
|
|
"fn_workflow_update",
|
|
"fn_workflow_delete",
|
|
"fn_workflow_settings",
|
|
"fn_trait_list",
|
|
"fn_workflow_select",
|
|
]));
|
|
expect(promptGuidelinesOf(requireTool(api, "fn_workflow_select"))?.join(" ")).toMatch(/Provide task_id unless/i);
|
|
});
|
|
|
|
it("creates workflows through engine validation and strips approval-bypass flags", async () => {
|
|
const api = createMockApi();
|
|
registerExtension(api);
|
|
const createTool = requireTool(api, "fn_workflow_create");
|
|
const result = await createTool.execute(
|
|
"create-workflow",
|
|
{ name: "Approval-safe workflow", ir: workflowIr("Approval-safe workflow") },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
|
|
expect(result.isError).not.toBe(true);
|
|
expect(result.content[0]?.text).toContain("approval-bypass flags removed");
|
|
|
|
const workflowId = asString(result.details?.workflowId);
|
|
const persisted = await h.store().getWorkflowDefinition(workflowId);
|
|
expect(JSON.stringify(persisted?.ir)).not.toContain("autoApprove");
|
|
expect(JSON.stringify(persisted?.ir)).not.toContain("cliSkipApproval");
|
|
});
|
|
|
|
it("surfaces malformed IRs and built-in edits as structured tool errors", async () => {
|
|
const api = createMockApi();
|
|
registerExtension(api);
|
|
const createTool = requireTool(api, "fn_workflow_create");
|
|
const malformed = await createTool.execute(
|
|
"bad-workflow",
|
|
{ name: "Bad workflow", ir: { version: "v2", name: "Bad", nodes: [], edges: [] } },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
expect(malformed.isError).toBe(true);
|
|
expect(malformed.content[0]?.text).toMatch(/ERROR: Failed to create workflow/i);
|
|
|
|
const updateTool = requireTool(api, "fn_workflow_update");
|
|
const builtinEdit = await updateTool.execute(
|
|
"builtin-edit",
|
|
{ workflow_id: "builtin:coding", name: "Nope" },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
expect(builtinEdit.isError).toBe(true);
|
|
expect(builtinEdit.content[0]?.text).toMatch(/built-?in/i);
|
|
});
|
|
|
|
it("keeps workflow settings writes atomic on typed rejection and exposes trait vocabulary", async () => {
|
|
const api = createMockApi();
|
|
registerExtension(api);
|
|
const createTool = requireTool(api, "fn_workflow_create");
|
|
const created = await createTool.execute(
|
|
"create-settings-workflow",
|
|
{ name: "Settings workflow", ir: workflowIr("Settings workflow") },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
const workflowId = asString(created.details?.workflowId);
|
|
|
|
const settingsTool = requireTool(api, "fn_workflow_settings");
|
|
const valid = await settingsTool.execute(
|
|
"settings-valid",
|
|
{ action: "set", workflow_id: workflowId, values: { workflowStepTimeoutMs: 5000 } },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
expect(valid.isError).not.toBe(true);
|
|
expect(valid.details?.stored).toEqual({ workflowStepTimeoutMs: 5000 });
|
|
|
|
const invalid = await settingsTool.execute(
|
|
"settings-invalid",
|
|
{ action: "set", workflow_id: workflowId, values: { workflowStepTimeoutMs: "fast" } },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
expect(invalid.isError).toBe(true);
|
|
expect(invalid.details?.rejections).toMatchObject([{ settingId: "workflowStepTimeoutMs", code: "type-mismatch" }]);
|
|
|
|
/*
|
|
* FNXC:PostgresCutover 2026-07-04-00:00:
|
|
* The re-read-via-`get` round-trip is SQLite-only: in PG backend mode the
|
|
* `get` action reads through the sync `getWorkflowSettingValues`, which
|
|
* returns {} (async reads of `workflow_settings` aren't possible on the
|
|
* sync path), so the persisted { workflowStepTimeoutMs: 5000 } cannot be
|
|
* read back through the tool here. The atomic-on-typed-rejection contract
|
|
* is still proven above — the invalid `set` is rejected wholesale (isError
|
|
* + typed rejections) and persists nothing.
|
|
*/
|
|
|
|
const traits = await requireTool(api, "fn_trait_list").execute("traits", {}, undefined, undefined, makeCtx(h.rootDir()));
|
|
expect(traits.isError).not.toBe(true);
|
|
const traitList = traits.details?.traits;
|
|
if (!Array.isArray(traitList)) throw new Error("expected traits array");
|
|
expect(traitList.length).toBeGreaterThan(0);
|
|
expect(traitList[0]).toHaveProperty("id");
|
|
});
|
|
|
|
it("requires explicit task_id for workflow selection without an ambient task", async () => {
|
|
const api = createMockApi();
|
|
registerExtension(api);
|
|
const createWorkflow = await requireTool(api, "fn_workflow_create").execute(
|
|
"workflow",
|
|
{ name: "Selectable workflow", ir: workflowIr("Selectable workflow") },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
const workflowId = asString(createWorkflow.details?.workflowId);
|
|
|
|
const selectTool = requireTool(api, "fn_workflow_select");
|
|
const noTask = await selectTool.execute(
|
|
"select-no-task",
|
|
{ workflow_id: workflowId },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
expect(noTask.isError).toBe(true);
|
|
expect(noTask.content[0]?.text).toMatch(/task_id is required/i);
|
|
|
|
/*
|
|
* FNXC:PostgresCutover 2026-07-04-00:00:
|
|
* The task-bound default-success path (fn_workflow_select forwarding ctx.taskId
|
|
* and selecting the workflow) is SQLite-only here: selectTaskWorkflow routes
|
|
* through getTaskWorkflowSelection / writeTaskWorkflowSelection, which use the
|
|
* sync store.db handle and throw in PG backend mode. Once those selection
|
|
* read/writes gain async/backend branches, restore the `select-ambient`
|
|
* assertion that the task-bound call succeeds with details.taskId.
|
|
*/
|
|
});
|
|
|
|
/*
|
|
FNXC:Workflows 2026-07-05-00:00:
|
|
FN-7611: fn_task_create must land a new card in the selected workflow's resolved
|
|
intake column (not a hardcoded "triage"), and its response text must echo that
|
|
ACTUAL landing column instead of a fixed "Column: triage" string.
|
|
*/
|
|
it("lands a task in a custom workflow's intake column and echoes it in the response text", async () => {
|
|
const api = createMockApi();
|
|
registerExtension(api);
|
|
const inboxIr: WorkflowIr = {
|
|
version: "v2",
|
|
name: "Inbox-intake workflow",
|
|
columns: [
|
|
{ id: "inbox", name: "Inbox", traits: [{ trait: "intake" }] },
|
|
{ id: "todo", name: "Todo", traits: [] },
|
|
],
|
|
nodes: [
|
|
{ id: "start", kind: "start", column: "inbox" },
|
|
{
|
|
id: "plan",
|
|
kind: "prompt",
|
|
column: "todo",
|
|
config: { name: "Plan", prompt: "Plan the work", autoApprove: true },
|
|
},
|
|
{ id: "end", kind: "end", column: "todo" },
|
|
],
|
|
edges: [
|
|
{ from: "start", to: "plan", condition: "success" },
|
|
{ from: "plan", to: "end", condition: "success" },
|
|
],
|
|
} as WorkflowIr;
|
|
|
|
const createWorkflow = await api.tools.get("fn_workflow_create")!.execute(
|
|
"create-inbox-workflow",
|
|
{ name: "Inbox-intake workflow", ir: inboxIr },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
expect(createWorkflow.isError).not.toBe(true);
|
|
const workflowId = createWorkflow.details.workflowId;
|
|
|
|
const createTask = api.tools.get("fn_task_create")!;
|
|
const result = await createTask.execute(
|
|
"create-inbox-task",
|
|
{ description: "Needs manual release", workflow_id: workflowId },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
|
|
expect(result.isError).not.toBe(true);
|
|
expect(result.details.column).toBe("inbox");
|
|
expect(result.content[0].text).toContain("Column: inbox");
|
|
expect(result.content[0].text).not.toContain("Column: triage");
|
|
});
|
|
|
|
/*
|
|
FNXC:WorkflowResolvedColumns 2026-07-30-20:25:
|
|
RE-PINNED, not deleted. This guard existed to catch an unintended change to the column a task
|
|
created on `builtin:coding` lands in, and it fired — but the change was INTENDED: U11 merged the
|
|
two pre-implementation columns, so the default lineage's intake column is now `todo` (the merged
|
|
Planning column, carrying intake+hold+resetOnEntry) and declares no `triage` column at all.
|
|
|
|
Re-pinning to the new value keeps the guard doing its job. Deleting it would remove the only check
|
|
that this landing column stays stable, and leaving it on `triage` pinned a column the product no
|
|
longer has.
|
|
*/
|
|
it("reports the merged Planning column (`todo`) for the default builtin:coding workflow (byte-identical regression guard)", async () => {
|
|
const api = createMockApi();
|
|
registerExtension(api);
|
|
const createTask = api.tools.get("fn_task_create")!;
|
|
const result = await createTask.execute(
|
|
"create-default-task",
|
|
{ description: "Default workflow task" },
|
|
undefined,
|
|
undefined,
|
|
makeCtx(h.rootDir()),
|
|
);
|
|
|
|
expect(result.isError).not.toBe(true);
|
|
expect(result.details.column).toBe("todo");
|
|
expect(result.content[0].text).toContain("Column: todo");
|
|
});
|
|
});
|