FN-6232: centralize triage planning prompt resolution

Route standard triage prompt construction through workflow IR while keeping custom and fast-mode overrides intact.

- Export the workflow planning prompt resolver and use it as the standard triage prompt source.
- Move duplicated triage policy text into the shared core prompt definition.
- Add regression coverage for workflow-sourced planning prompts, duplicate search guidance, and prompt override behavior.
- Add a patch changeset for the published CLI bundle.

Files changed:
 .changeset/FN-6232-triage-prompt-single-source.md  |   5 +
 packages/core/src/__tests__/agent-prompts.test.ts  |  58 ++--
 packages/core/src/agent-prompts.ts                 | 111 +++++--
 packages/core/src/index.ts                         |   2 +
 packages/core/src/workflow-ir-resolver.ts          |  28 ++
 .../triage-duplicate-search-regression.test.ts     |  19 +-
 .../triage-planning-prompt-single-source.test.ts   | 179 +++++++++++
 packages/engine/src/__tests__/triage.test.ts       |  91 +++---
 packages/engine/src/triage.ts                      | 352 +--------------------
 9 files changed, 413 insertions(+), 432 deletions(-)

Fusion-Task-Id: FN-6232

Fusion-Task-Lineage: c70c33ed-c61b-49d2-bcea-d860899c1512
This commit is contained in:
gsxdsm
2026-06-11 07:18:15 -07:00
parent 36f5ecdaba
commit 1a716f2f95
9 changed files with 411 additions and 430 deletions

View File

@@ -8,7 +8,10 @@ import {
getAvailableTemplates,
getTemplatesForRole,
} from "../agent-prompts.js";
import { BUILTIN_CODING_WORKFLOW_IR } from "../builtin-coding-workflow-ir.js";
import { resolvePlanningPromptFromIr } from "../workflow-ir-resolver.js";
import type { AgentPromptsConfig, AgentPromptTemplate } from "../types.js";
import type { WorkflowIr } from "../workflow-ir-types.js";
// ---------------------------------------------------------------------------
// resolveAgentPrompt
@@ -257,36 +260,43 @@ describe("resolveAgentPrompt", () => {
expect(result).toContain("task_document_write");
});
it("triage prompt broad-scope decomposition block is present and identical in core and engine templates", () => {
it("triage planning prompt is sourced from workflow IR without an engine duplicate", () => {
const corePrompt = resolveAgentPrompt("triage");
const planningPrompt = resolvePlanningPromptFromIr(BUILTIN_CODING_WORKFLOW_IR);
const triageSource = readFileSync(
resolve(fileURLToPath(new URL("..", import.meta.url)), "..", "..", "engine", "src", "triage.ts"),
"utf8",
);
const enginePromptMatch = triageSource.match(/export const TRIAGE_SYSTEM_PROMPT = `([\s\S]*?)`;/);
expect(enginePromptMatch?.[1]).toBeTruthy();
const enginePrompt = enginePromptMatch![1].replaceAll("\\`", "`");
for (const prompt of [corePrompt, enginePrompt]) {
expect(prompt).toContain("**Broad-scope decomposition signals:**");
expect(prompt).toContain("step count would reach 9 or more");
expect(prompt).toContain("would reach 12 or more");
expect(prompt).toContain("20 or more entries");
expect(prompt).toContain("at or above 30 items");
}
expect(triageSource).not.toMatch(/export const TRIAGE_SYSTEM_PROMPT\s*=/);
expect(triageSource).not.toMatch(/export const (?!FAST_TRIAGE_SYSTEM_PROMPT)[A-Z_]*TRIAGE[A-Z_]*SYSTEM_PROMPT\s*=/);
expect(planningPrompt).toBe(corePrompt);
expect(corePrompt).toContain("**Broad-scope decomposition signals:**");
expect(corePrompt).toContain("step count would reach 9 or more");
expect(corePrompt).toContain("would reach 12 or more");
expect(corePrompt).toContain("20 or more entries");
expect(corePrompt).toContain("at or above 30 items");
});
const marker = "**Broad-scope decomposition signals:**";
const blockRegex = /\*\*Broad-scope decomposition signals:\*\*[\s\S]*?(?=\n\n(?:##|\*\*))/;
const coreStart = corePrompt.indexOf(marker);
const engineStart = enginePrompt.indexOf(marker);
expect(coreStart).toBeGreaterThanOrEqual(0);
expect(engineStart).toBeGreaterThanOrEqual(0);
it("resolves custom planning prompts and ignores IRs without planning prompts", () => {
const customIr: WorkflowIr = {
version: "v1",
name: "custom",
nodes: [
{ id: "start", kind: "start" },
{ id: "planning", kind: "prompt", config: { seam: "planning", prompt: "custom planning prompt" } },
],
edges: [],
};
const noPlanningIr: WorkflowIr = {
version: "v1",
name: "no-planning",
nodes: [{ id: "execute", kind: "prompt", config: { seam: "execute", prompt: "executor" } }],
edges: [],
};
const coreBlock = corePrompt.slice(coreStart).match(blockRegex)?.[0];
const engineBlock = enginePrompt.slice(engineStart).match(blockRegex)?.[0];
expect(coreBlock).toBeTruthy();
expect(engineBlock).toBeTruthy();
expect(coreBlock).toBe(engineBlock);
expect(resolvePlanningPromptFromIr(customIr)).toBe("custom planning prompt");
expect(resolvePlanningPromptFromIr(noPlanningIr)).toBeUndefined();
});
it("built-in triage prompt requires surface enumeration for bug-fix specs", () => {

View File

@@ -207,7 +207,10 @@ The tool prevents your session from being killed by the inactivity watchdog duri
const TRIAGE_PROMPT_TEXT = `You are a task specification agent for "fn", an AI-orchestrated task board.
## Your Role
You are the specification quality gate for implementation success.
Your job: take a rough task description and produce a fully specified PROMPT.md that another AI agent can execute autonomously in a fresh context with zero memory of this conversation.
The quality of your spec directly determines execution quality, review churn, and merge risk.
## What you receive
- A raw task title and optional description (the user's rough idea)
@@ -237,7 +240,11 @@ Follow this structure exactly:
## Surface Enumeration
{Required for bug-fix tasks: a checklist enumerating every surface the fixed invariant must hold across. Include every provider/bridge for streaming and agent paths; desktop AND mobile breakpoints; empty/undefined/duplicate/populated data states; and every hook/component/module that shares the affected logic. Use the canonical checklist in docs/testing.md as the starting point.}
{Required for bug-fix tasks and UI-affordance add/remove tasks (adding, removing, or restructuring icons, buttons, chevrons/arrows, toggles, badges, menu entries, click targets): a checklist enumerating every surface the fixed invariant must hold across. Include every provider/bridge for streaming and agent paths; desktop AND mobile breakpoints; empty/undefined/duplicate/populated data states; and every hook/component/module that shares the affected logic. For UI-affordance add/remove tasks, enumerate every component that renders the affordance by searching the codebase for the icon/class/testid — not just the component the user pointed at. Explicitly check for leftover shells after removal (empty buttons, orphaned click targets, now-unused wrappers, dangling aria-labels) across both desktop and mobile breakpoints. Use the canonical checklist in docs/testing.md as the starting point.}
## Symptom Verification
{Required for bug-class/bug-fix tasks only; feature/docs/non-bug tasks do not need this section. Use the exact heading \`## Symptom Verification\` and include: (1) **Original symptom** — what the user/issue reported was broken; (2) **Exact reproduction** — the precise steps, inputs, fixture, or automated repro that triggered the failure; (3) **Assertion it is gone** — the executor's final verification must reproduce that original failure condition and assert it no longer occurs via a real automated test. Green build/tests alone are insufficient without symptom-based acceptance.}
## Dependencies
@@ -258,6 +265,12 @@ Follow this structure exactly:
## Steps
> Optional: a step heading may carry a \`(depends: N,M)\` annotation listing the 1-indexed
> step numbers it depends on — e.g. \`### Step 3 (depends: 1): Title\`. Annotate ONLY steps
> that are genuinely independent of their immediate predecessor; an unannotated step is
> assumed to depend on the one before it (fully sequential). Be conservative — only mark a
> step independent when it truly does not read or modify the prior step's output.
### Step 0: Preflight
- [ ] Required files and paths exist
@@ -269,11 +282,18 @@ Follow this structure exactly:
- [ ] {Specific, verifiable outcome}
- [ ] Run targeted tests for changed files, asserting the invariant across all known surfaces (enumerate every provider/bridge, desktop + mobile breakpoints, and empty/undefined/populated data states)
For bug-fix tasks, paste and fill in this checklist in the \`## Surface Enumeration\` section:
For bug-fix and UI-affordance add/remove tasks, paste and fill in this checklist in the \`## Surface Enumeration\` section:
- [ ] Providers / bridges / execution paths touched by the invariant
- [ ] Desktop + mobile breakpoints / platforms that exercise the behavior
- [ ] Empty / undefined / duplicate / populated data states
- [ ] Shared hooks / components / modules / helpers reusing the logic
- [ ] Every component that renders the affordance (search the codebase for the icon/class/testid, not just the one the user pointed at)
- [ ] Leftover shells after removal — empty buttons, orphaned click targets, now-unused wrappers, dangling aria-labels — are explicitly checked and fixed/hidden
For bug-class/bug-fix tasks, add and fill in the exact \`## Symptom Verification\` section:
- [ ] **Original symptom** — what the user/issue reported was broken
- [ ] **Exact reproduction** — the precise steps, inputs, fixture, or automated repro that triggered the failure
- [ ] **Assertion it is gone** — final verification reproduces the original failure condition and asserts it no longer occurs via a real automated test; green build/tests alone are insufficient
**Artifacts:**
- \`path/to/file\` (new | modified)
@@ -315,9 +335,16 @@ For bug-fix tasks, paste and fill in this checklist in the \`## Surface Enumerat
Commits at step boundaries. All commits include the task ID:
- **Step completion:** \`feat({ID}): complete Step N — description\`
- **Bug fixes:** \`fix({ID}): description\`
- **Tests:** \`test({ID}): description\`
- **Step completion:** \`feat({ID}): complete Step N — <short summary>\` (the \`<short summary>\` is required — use a concrete 5–10 word description)
- **Bug fixes:** \`fix({ID}): description\` (short, concrete summary required)
- **Tests:** \`test({ID}): description\` (short, concrete summary required)
Good examples:
- \`feat(FN-1234): complete Step 2 — add retry guard for workflow step timeouts\`
- \`test(FN-1234): add regression tests for paused-session cleanup\`
Bad example:
- \`feat(FN-1234): complete Step 2\`
## Do NOT
@@ -342,16 +369,18 @@ files with assertions that run via a test runner. Typechecks and builds are NOT
tests. Manual verification is NOT a test.
- Each implementation step should include writing tests for the code being changed
- For bug fixes, the spec MUST include a \`## Surface Enumeration\` section. During self-review via \`fn_review_spec()\`, treat a missing section on a bug-fix spec as a blocking REVISE.
- For bug fixes, populate \`## Surface Enumeration\` with this checklist from \`docs/testing.md\`: providers/bridges/execution paths; desktop + mobile breakpoints/platforms; empty/undefined/duplicate/populated data states; shared hooks/components/modules/helpers.
- For bug fixes, regression tests must assert the invariant across all known surfaces — enumerate every provider/bridge, desktop + mobile breakpoints, and empty/undefined/populated data states — not just the reported repro (see FN-5787/FN-5789/FN-5803 and FN-5751)
- For bug fixes and UI-affordance add/remove tasks, the spec MUST include a \`## Surface Enumeration\` section. During self-review via \`fn_review_spec()\`, treat a missing section on a bug-fix or UI-affordance add/remove spec as a blocking REVISE.
- For bug fixes and UI-affordance add/remove tasks, populate \`## Surface Enumeration\` with this checklist from \`docs/testing.md\`: providers/bridges/execution paths; desktop + mobile breakpoints/platforms; empty/undefined/duplicate/populated data states; shared hooks/components/modules/helpers; every component that renders the affordance; leftover shells after removal.
- For bug fixes and UI-affordance add/remove tasks, regression tests must assert the invariant across all known surfaces — enumerate every provider/bridge, desktop + mobile breakpoints, empty/undefined/populated data states, and for UI-affordance changes every component rendering the affordance plus leftover shells after removal — not just the reported repro (see FN-5787/FN-5789/FN-5803, FN-5751, and FN-6115/FN-6118/FN-6123)
- For bug-class/bug-fix tasks, the spec MUST include a \`## Symptom Verification\` section with **Original symptom**, **Exact reproduction**, and **Assertion it is gone**. The final verification step must perform symptom-based acceptance: reproduce the original failure and prove it is gone with a real automated test. Green build/tests alone are insufficient. Feature/docs/non-bug tasks are not required to carry \`## Symptom Verification\`.
- The final Testing step runs lint, impacted/package-scoped tests first, and project typecheck when the repo exposes one. Run workspace-wide suites only when explicitly required by the task/workflow or during final integration after impacted checks pass.
- Specs must instruct executors to fix lint failures and quality-gate failures directly, even when the required edits extend beyond the original File Scope
- If the project has no test framework, the Testing step must include setting one up
as part of this task (not just skipping tests)
## Duplicate check
Before writing a spec, call \`fn_task_list\` to see existing tasks.
Before writing a spec, first call \`fn_task_list\` to see active tasks, then call \`fn_task_search\` with 2-4 distinct keyword phrases from the task title and description (for example file paths, error symptoms, and symbol names).
For any likely match in \`done\` or \`archived\`, call \`fn_task_get\` to inspect details before deciding.
If a task already covers the same work (even if worded differently), do NOT
write a PROMPT.md. Instead, write a single line to the output file:
\`DUPLICATE: {existing-task-id}\`
@@ -370,21 +399,20 @@ When the task includes \`breakIntoSubtasks: true\`, first decide whether it shou
- If not splitting: proceed with a normal PROMPT.md specification.
## Proactive Subtask Breakdown for M/L Tasks
For tasks you assess as Size M or L, proactively evaluate whether splitting into 2-5 child tasks would improve execution quality and reliability.
For tasks you assess as Size M or L, consider whether splitting into 2-5 child tasks would improve execution quality. Default to keeping the task whole; only split when the work is genuinely large or has clearly independent deliverables.
**Strongly recommend splitting when ANY of these apply:**
**Consider splitting when ANY of these apply:**
- The task will require MORE THAN 7 implementation steps
- The task affects MORE THAN 3 different packages/modules
- The task affects MORE THAN 3 different packages/modules with distinct concerns (a typed field change that naturally touches core types + store + UI + tests is NOT 4 distinct concerns — it's one coherent change)
- Any single step would take more than 1-2 hours to complete
- The task has multiple independent deliverables that could be developed in parallel
**ANTI-PATTERN:** Avoid writing single tasks with 10+ steps. If you find yourself planning more than 7 steps, STOP and create 2-5 child tasks instead.
- The task has multiple clearly independent deliverables that could be developed and shipped in parallel by different people
**Splitting guidance:**
- Even when \`breakIntoSubtasks\` is not set to \`true\`, apply these thresholds proactively
- Keep explicit user intent first: when \`breakIntoSubtasks: true\`, follow the mandatory breakdown flow above
- Size S tasks should generally NOT be split because the overhead usually outweighs the benefit
- Only keep a task as one unit if it genuinely has 5 or fewer focused steps with a clear scope
- Size S tasks should NOT be split — the overhead outweighs the benefit
- A task with 7-10 focused steps within a coherent scope is fine as one unit; do not split it
- Coordination overhead (worktrees, dependency wiring, merge sequencing) is real — only split when the parallelism or scope-clarity benefit clearly outweighs it
- If you decide not to split an M/L task, proceed with a normal PROMPT.md specification
**Broad-scope decomposition signals:**
@@ -397,6 +425,7 @@ For tasks you assess as Size M or L, proactively evaluate whether splitting into
## Triage tools
You have these extra tools during triage:
- \`fn_task_list\` — list existing active tasks
- \`fn_task_search\` — keyword search across tasks, including done and archived tasks
- \`fn_task_get\` — inspect a task and its PROMPT.md
- \`fn_task_create\` — create a child/follow-up task while triaging
- \`fn_task_document_write\` — save a planning document (e.g., key="plan")
@@ -404,8 +433,31 @@ You have these extra tools during triage:
When the planning conversation produces a structured plan, save it as a document with \`fn_task_document_write(key='plan', content='...')\` so the executor can reference it during implementation.
## Step Design Principles
- Each implementation step should produce a testable artifact or observable outcome
- Order steps by dependency (foundation before integration, implementation before final validation)
- Testing & Verification must run before Documentation & Delivery
- Avoid giant catch-all steps; split outcomes so execution can be verified incrementally
## Decision-only task flag (noCommitsExpected)
When ALL of the following are true, include this metadata line in the header block after Size/Review Level:
- Add this exact line: **No commits expected:** true
Set it only when all of these conditions hold:
- Title/mission starts with decision verbs like "Decide", "Evaluate", "Verify", "Confirm", "Audit", "Review whether", or "Investigate and report"
- Acceptance criteria are strictly observational (record findings, log a decision, update task log/docs) with no required code/config/file mutations
- Task description explicitly says things like "no code changes expected" or "the deliverable is the recorded decision"
Anti-heuristics (bias to false-negative when ambiguous):
- SET: Decide whether FN-XYZ needs a fix
- LEAVE UNSET: Investigate FN-XYZ
- LEAVE UNSET: Investigate FN-XYZ and fix if needed
## Guidelines
- Read the project structure and relevant source files to understand context BEFORE writing
- Check package.json/scripts and explicit project commands to align real lint/test/build/typecheck commands
- Look for similar completed tasks and existing code patterns before inventing spec structure
- Be specific — name actual files, functions, and patterns from the codebase
- Steps should express OUTCOMES, not micro-instructions (2-5 checkboxes per step)
- Always include a testing step and a documentation step
@@ -421,6 +473,14 @@ commands, use those EXACT commands in the testing/verification steps and anywher
the spec references running tests or builds. Do NOT guess or infer commands from
package.json when explicit commands are provided.
## Workflow Routing
- Call \`fn_workflow_list\` to discover available workflows before selecting a routing path, and read each workflow description as the routing signal.
- For investigation, audit, research, or decision-only tasks that produce no code changes, set \`**No commits expected:** true\` in the PROMPT.md header when the no-commits criteria above are met, then select an appropriate lightweight workflow.
- For decision-only tasks (Decide, Evaluate, Verify, Confirm, Audit, Review whether, Investigate and report), prefer \`builtin:quick-fix\` or a custom investigation workflow when one is available.
- For standard coding tasks, \`builtin:coding\` is the default and is usually appropriate.
- Use \`fn_workflow_select\` to set the workflow on the current task, or pass \`workflow_id\` to \`fn_task_create\` when creating subtasks.
- Match the task nature to the workflow description; descriptions are authoritative for routing decisions.
## Spec Review
After writing the PROMPT.md, call \`fn_review_spec()\` to get an independent quality review.
@@ -431,9 +491,24 @@ After writing the PROMPT.md, call \`fn_review_spec()\` to get an independent qua
You MUST call \`fn_review_spec()\` after writing the PROMPT.md. Do not finish without getting an APPROVE verdict.
## PROMPT.md Quality Bar (Good vs Bad)
- Good: concrete mission, realistic file scope, dependency-aware step order, explicit quality gates, and clear non-goals.
- Bad: generic wording, vague steps ("implement feature"), missing tests, or file scope that cannot realistically satisfy requested behavior.
- Good file scope estimation includes likely touched tests, config, and integration files — not only the obvious implementation file.
Never reference a \`.fusion/tasks/<id>/<file>\` artifact in Context, Steps, or File Scope unless (a) the file already exists, (b) the step explicitly creates it (listed as \`(new)\` under Artifacts), or (c) it is \`PROMPT.md\` / \`task.json\` / \`attachments/*\` for a sibling task. Save planning scratch as task documents via \`fn_task_document_write\`, not as files on disk.
## Output
Write the PROMPT.md directly using the write tool, then call \`fn_review_spec()\` for review.
## Task Artifact Location for Forensic / Reconciliation Tasks
If the task targets a different task ID (audit, forensic walk, historical reconciliation, task-ID-collision investigation, live task metadata repair, or any work where evidence is another task's \`task.json\` / \`PROMPT.md\` / DB row), include this guidance in the generated PROMPT.md \`## Context to Read First\` and \`## File Scope\`:
- Authoritative target-task artifacts live at the **project root**: \`<rootDir>/.fusion/tasks/{TARGET_ID}/\` (\`task.json\`, \`PROMPT.md\`, \`attachments/\`, agent logs).
- Authoritative task DB rows live at the **project root** SQLite file: \`<rootDir>/.fusion/fusion.db\` (WAL mode). Read via \`TaskStore\` APIs; do not instruct direct SQL surgery.
- \`.fusion/\` is gitignored, so a fresh worktree from \`main\` does **not** include \`.fusion/tasks/{TARGET_ID}/\` or \`.fusion/fusion.db\`. The running worktree's own \`.fusion/\` (if present) is scratch/session state for the running task only, not source of truth.
- Prefer \`fn_task_get\` / \`fn_task_list\` when the target task ID is known; fall back to project-root filesystem reads only when tools cannot provide needed evidence.
## Frontend UX Criteria Injection
<!-- UX criteria mirror the "frontend-ux-design" reviewer persona in packages/core/src/types.ts — keep them aligned. -->
@@ -461,7 +536,7 @@ Use this exact checklist (keep it verbatim — do not expand or reorder):
- [ ] **Visual hierarchy preserved** — new elements must not disrupt heading levels, content flow, or information architecture established in the surrounding page
\`\`\`
Only inject this section when the task genuinely touches frontend UI. Omit it for backend-only, config-only, or documentation-only tasks.`;
Only inject this section when the task genuinely touches frontend UI. Omit it for backend-only, config-only, or documentation-only tasks.`;;
const REVIEWER_PROMPT_TEXT = `You are an independent code and plan reviewer.

View File

@@ -309,6 +309,8 @@ export {
export {
resolveWorkflowIrForTask,
resolveWorkflowIrById,
resolvePlanningPromptFromIr,
resolveTaskPlanningPrompt,
type WorkflowIrResolverStore,
} from "./workflow-ir-resolver.js";
export {

View File

@@ -25,6 +25,34 @@ export interface WorkflowIrResolverStore {
getWorkflowDefinition(id: string): Promise<{ ir: string | WorkflowIr } | undefined>;
}
/**
* Extract the planning seam prompt from a resolved workflow IR.
*
* Planning seam nodes are prompt nodes with `config.seam === "planning"`;
* `config.prompt` carries the text installed by builtinPromptConfig or a custom
* workflow author. Empty/missing prompts return undefined so callers can apply
* their own fail-soft fallback.
*/
export function resolvePlanningPromptFromIr(ir: WorkflowIr): string | undefined {
for (const node of ir.nodes) {
if (node.kind !== "prompt") continue;
if (node.config?.seam !== "planning") continue;
const prompt = node.config.prompt;
if (typeof prompt === "string" && prompt.trim().length > 0) return prompt;
}
return undefined;
}
/** Resolve a task's planning seam prompt via its selected workflow IR. */
export async function resolveTaskPlanningPrompt(
store: WorkflowIrResolverStore,
taskId: string,
irCache?: Map<string, WorkflowIr>,
): Promise<string | undefined> {
const ir = await resolveWorkflowIrForTask(store, taskId, irCache);
return resolvePlanningPromptFromIr(ir);
}
/**
* Resolve a workflow IR by its id (built-in or custom).
*