feat(chat): ground agent chat answers in live codebase investigation (#2416)

## Summary

- Add `CHAT_CODEBASE_ACCURACY_GUIDANCE` so agent chat (direct +
multi-agent rooms) investigates the live checkout with tools before
answering code/architecture questions.
- Soften chat’s pure brevity default for repo questions: still lead
short, but keep real path/symbol evidence (parity with the investigation
pressure that makes Planning Mode more accurate).
- Cover the constant and assembly via unit tests; include a patch
changeset for `@runfusion/fusion`.

## Why

Users reported Planning Mode was more accurate about the codebase than
agent chat. Plan mode inherits the triage seam’s “read/grep first, name
real files” contract; chat only had a short helpful-assistant persona
plus a brevity default, so models often answered from priors.

## Test plan

- [x] `pnpm --filter @fusion/dashboard exec vitest run
src/__tests__/chat-system-prompt.test.ts
src/__tests__/chat-manager.test.ts`
- [ ] Manually ask agent chat a project-specific architecture question
and confirm it greps/reads before answering with real paths
- [ ] Confirm non-code chat still stays short/crisp
- [ ] Confirm multi-agent room responders also receive the new guidance

<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

* **Improvements**
* Improved repository/code answers by investigating the live codebase
first and prioritizing verified paths and symbols over speculation.
* Refined response-length behavior so code questions stay
evidence-focused, while non-code questions remain concise.
* Applied consistent accuracy guidance across both direct and room-based
conversations.
* **Bug Fixes**
* Prevented chat instructions from depending on unavailable mailbox
functionality, improving reliability for agentless/room flows.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->
This commit is contained in:
gsxdsm
2026-07-23 00:00:22 -07:00
committed by GitHub
parent fc4f5aa0e4
commit 016221c6a8
4 changed files with 129 additions and 3 deletions

View File

@@ -0,0 +1,7 @@
---
"@runfusion/fusion": minor
---
summary: Agent chat now investigates the live codebase with tools before answering architecture and code questions.
category: feature
dev: Adds CHAT_CODEBASE_ACCURACY_GUIDANCE and appends it in direct and room chat system-prompt assembly; response-length policy yields to path/symbol evidence on repo questions. Mailbox long-form path is conditional when fn_send_message is registered; find is bounded to the project checkout.

View File

@@ -21,6 +21,7 @@ import {
__getChatDiagnostics, __getChatDiagnostics,
__setChatDiagnostics, __setChatDiagnostics,
CHAT_ASK_QUESTION_GUIDANCE, CHAT_ASK_QUESTION_GUIDANCE,
CHAT_CODEBASE_ACCURACY_GUIDANCE,
} from "../chat.js"; } from "../chat.js";
// ── Mock Setup ────────────────────────────────────────────────────────────── // ── Mock Setup ──────────────────────────────────────────────────────────────
@@ -1873,6 +1874,7 @@ describe("ChatManager.sendMessage", () => {
expect(customTools.map((tool: { name: string }) => tool.name)).toContain("fn_ask_question"); expect(customTools.map((tool: { name: string }) => tool.name)).toContain("fn_ask_question");
expect(customTools.map((tool: { name: string }) => tool.name)).not.toContain("fn_send_message"); expect(customTools.map((tool: { name: string }) => tool.name)).not.toContain("fn_send_message");
expect(customTools.map((tool: { name: string }) => tool.name)).not.toContain("fn_read_messages"); expect(customTools.map((tool: { name: string }) => tool.name)).not.toContain("fn_read_messages");
expect(createOptions?.systemPrompt).toContain(CHAT_CODEBASE_ACCURACY_GUIDANCE);
expect(createOptions?.systemPrompt).toContain(CHAT_ASK_QUESTION_GUIDANCE); expect(createOptions?.systemPrompt).toContain(CHAT_ASK_QUESTION_GUIDANCE);
}); });
@@ -2709,6 +2711,7 @@ describe("ChatManager.sendMessage", () => {
expect(createOptions.systemPrompt).toContain("Be calm and precise."); expect(createOptions.systemPrompt).toContain("Be calm and precise.");
expect(createOptions.systemPrompt).toContain("Your chat reply is the primary response to the user."); expect(createOptions.systemPrompt).toContain("Your chat reply is the primary response to the user.");
expect(createOptions.systemPrompt).toContain("Use `fn_send_message` only when either (a) the user explicitly asks"); expect(createOptions.systemPrompt).toContain("Use `fn_send_message` only when either (a) the user explicitly asks");
expect(createOptions.systemPrompt).toContain(CHAT_CODEBASE_ACCURACY_GUIDANCE);
}); });
it("includes guidance to avoid double-sending mailbox copies by default", async () => { it("includes guidance to avoid double-sending mailbox copies by default", async () => {
@@ -3792,7 +3795,7 @@ describe("ChatManager generation isolation", () => {
]); ]);
mockAgentStore.getAgent.mockResolvedValue({ id: "agent-001", name: "Avery", role: "executor", state: "idle" }); mockAgentStore.getAgent.mockResolvedValue({ id: "agent-001", name: "Avery", role: "executor", state: "idle" });
__setCreateResolvedAgentSession(async () => ({ const createResolvedSession = vi.fn(async () => ({
session: { session: {
prompt: vi.fn().mockResolvedValue(undefined), prompt: vi.fn().mockResolvedValue(undefined),
dispose: vi.fn(), dispose: vi.fn(),
@@ -3802,6 +3805,7 @@ describe("ChatManager generation isolation", () => {
model: "test", model: "test",
fallbackInfo: undefined, fallbackInfo: undefined,
} as any)); } as any));
__setCreateResolvedAgentSession(createResolvedSession);
const chatManager = createChatManager(); const chatManager = createChatManager();
await chatManager.sendRoomMessage("room-1", "hello @Avery"); await chatManager.sendRoomMessage("room-1", "hello @Avery");
@@ -3815,6 +3819,11 @@ describe("ChatManager generation isolation", () => {
senderAgentId: "agent-001", senderAgentId: "agent-001",
content: "Room answer", content: "Room answer",
}); });
/*
FNXC:ChatCodebaseAccuracy 2026-07-23-05:20:
Room responder assembly must keep the investigate-first contract; capture the system prompt so the append site cannot regress unnoticed (PR #2416 review).
*/
expect(createResolvedSession.mock.calls[0]?.[0]?.systemPrompt).toContain(CHAT_CODEBASE_ACCURACY_GUIDANCE);
}); });
it("sendRoomMessage still resolves room member responders when listAgents is unavailable", async () => { it("sendRoomMessage still resolves room member responders when listAgents is unavailable", async () => {

View File

@@ -1,6 +1,11 @@
import { describe, expect, it } from "vitest"; import { describe, expect, it } from "vitest";
import { FUSION_RUNTIME_SELF_AWARENESS } from "@fusion/core"; import { FUSION_RUNTIME_SELF_AWARENESS } from "@fusion/core";
import { CHAT_AGENT_MESSAGE_ROUTING_GUIDANCE, CHAT_SYSTEM_PROMPT } from "../chat.js"; import {
CHAT_AGENT_MESSAGE_ROUTING_GUIDANCE,
CHAT_ASK_QUESTION_GUIDANCE,
CHAT_CODEBASE_ACCURACY_GUIDANCE,
CHAT_SYSTEM_PROMPT,
} from "../chat.js";
describe("chat system prompt guidance", () => { describe("chat system prompt guidance", () => {
it.each(["short", "crisp", "few sentences"])("includes brevity direction: %s", (token) => { it.each(["short", "crisp", "few sentences"])("includes brevity direction: %s", (token) => {
@@ -13,6 +18,13 @@ describe("chat system prompt guidance", () => {
expect(CHAT_SYSTEM_PROMPT).toContain('to_id: "dashboard"'); expect(CHAT_SYSTEM_PROMPT).toContain('to_id: "dashboard"');
}); });
it("yields brevity to correctness for repository code questions", () => {
const lower = CHAT_SYSTEM_PROMPT.toLowerCase();
expect(lower).toContain("prioritize correctness");
expect(lower).toContain("cite real paths/symbols");
expect(lower).toContain("key citations");
});
it("authorizes the full coding workspace toolset for user-directed changes", () => { it("authorizes the full coding workspace toolset for user-directed changes", () => {
const lower = CHAT_SYSTEM_PROMPT.toLowerCase(); const lower = CHAT_SYSTEM_PROMPT.toLowerCase();
@@ -44,6 +56,59 @@ describe("chat system prompt guidance", () => {
}); });
}); });
describe("chat codebase accuracy guidance", () => {
it("requires tool-grounded investigation for codebase questions", () => {
const lower = CHAT_CODEBASE_ACCURACY_GUIDANCE.toLowerCase();
expect(lower).toContain("live checkout");
expect(lower).toContain("before");
expect(lower).toContain("grep");
expect(lower).toContain("read");
expect(lower).toContain("do not invent");
expect(lower).toContain("trust the tools");
});
it("bounds find to the project checkout and forbids OS temp-root walks", () => {
const lower = CHAT_CODEBASE_ACCURACY_GUIDANCE.toLowerCase();
expect(lower).toContain("restrict `find`");
expect(lower).toContain("project checkout");
expect(lower).toContain("never recurse through the os temp root");
expect(CHAT_CODEBASE_ACCURACY_GUIDANCE).toContain("$TMPDIR");
});
it("makes long-form mailbox guidance conditional on fn_send_message availability", () => {
const lower = CHAT_CODEBASE_ACCURACY_GUIDANCE.toLowerCase();
expect(lower).toContain("if `fn_send_message` is available in this session");
expect(lower).toContain("if it is not available");
expect(lower).toContain("instead of calling a missing tool");
expect(CHAT_SYSTEM_PROMPT.toLowerCase()).toContain("when `fn_send_message` is available in this session");
expect(CHAT_SYSTEM_PROMPT.toLowerCase()).toContain("when that tool is not available");
});
it("keeps conversational brevity for non-code questions", () => {
expect(CHAT_CODEBASE_ACCURACY_GUIDANCE.toLowerCase()).toContain("conversational");
expect(CHAT_CODEBASE_ACCURACY_GUIDANCE.toLowerCase()).toContain("short/crisp");
});
it("does not turn chat into a planning-mode interview", () => {
const lower = CHAT_CODEBASE_ACCURACY_GUIDANCE.toLowerCase();
expect(lower).toContain("do **not** start a planning mode interview".toLowerCase());
expect(lower).toContain("prompt.md");
});
it("combined chat guidance keeps mailbox routing, accuracy, and ask-question", () => {
const combined = [
CHAT_SYSTEM_PROMPT,
CHAT_AGENT_MESSAGE_ROUTING_GUIDANCE,
CHAT_CODEBASE_ACCURACY_GUIDANCE,
CHAT_ASK_QUESTION_GUIDANCE,
].join("\n\n");
expect(combined).toContain("fn_send_message");
expect(combined).toContain("## Codebase accuracy");
expect(combined).toContain("fn_ask_question");
expect(combined.toLowerCase()).toContain("trust the tools");
});
});
// --------------------------------------------------------------------------- // ---------------------------------------------------------------------------
// Runtime self-awareness preamble (FN-7675) // Runtime self-awareness preamble (FN-7675)
// --------------------------------------------------------------------------- // ---------------------------------------------------------------------------

View File

@@ -243,13 +243,48 @@ async function ensureEngineReady(): Promise<void> {
/** /**
* FNXC:DashboardChat 2026-07-15-00:00: * FNXC:DashboardChat 2026-07-15-00:00:
* FN-7984 keeps the live checkout's branch sticky because chat commands run in the user's project directory. Agents may inspect Git state, but must not switch branches unless the user explicitly requests it. * FN-7984 keeps the live checkout's branch sticky because chat commands run in the user's project directory. Agents may inspect Git state, but must not switch branches unless the user explicitly requests it.
*
* FNXC:ChatCodebaseAccuracy 2026-07-22-12:00:
* Response-length policy yields to correctness on repo questions: users reported Planning Mode more accurate than agent chat because chat only had a brevity default. Code questions still lead short, but must keep path/symbol evidence; long traces still go via mailbox when the tool is registered.
*
* FNXC:ChatCodebaseAccuracy 2026-07-23-05:20:
* PR #2416 review: room responders and agentless direct chat do not register `fn_send_message`. Long-form guidance must be conditional on tool availability so those sessions put detail in the chat reply instead of calling a missing mailbox tool.
*/ */
export const CHAT_SYSTEM_PROMPT = `${FUSION_RUNTIME_SELF_AWARENESS} export const CHAT_SYSTEM_PROMPT = `${FUSION_RUNTIME_SELF_AWARENESS}
You are a helpful AI assistant integrated into the fn task board system. You help users with questions about their project, code, architecture, and tasks. You have coding workspace tools on the project checkout: \`read\`, \`write\`, \`edit\`, \`bash\`, \`grep\`, \`find\`, and \`ls\`. Use \`write\`, \`edit\`, and \`bash\` for user-requested code changes, file edits, or shell investigation; prefer minimal, user-directed mutations and respect any pending-approval or blocked tool result. Do not claim that you only have read access. Do not change the branch the working directory is checked out on: do not run \`git checkout <branch>\` or \`git switch <branch>\` to a different branch unless the user explicitly asks. Read-only Git and branch inspection, such as \`git status\`, \`git branch\`, and \`git log\`, is allowed. Response length policy: default to a short, crisp reply (a few sentences or a short bulleted list) that directly answers the user; avoid preamble, restating the question, and filler. If a thorough answer genuinely needs long-form content (for example multi-step plans, design proposals, deep analyses, or long file excerpts), keep the chat reply brief with a one- or two-sentence summary and then send the full write-up via \`fn_send_message\` using \`type: "agent-to-user"\` and \`to_id: "dashboard"\`. That mailbox follow-up must add new substantive detail and must not duplicate the chat reply.`; You are a helpful AI assistant integrated into the fn task board system. You help users with questions about their project, code, architecture, and tasks. You have coding workspace tools on the project checkout: \`read\`, \`write\`, \`edit\`, \`bash\`, \`grep\`, \`find\`, and \`ls\`. Use \`write\`, \`edit\`, and \`bash\` for user-requested code changes, file edits, or shell investigation; prefer minimal, user-directed mutations and respect any pending-approval or blocked tool result. Do not claim that you only have read access. Do not change the branch the working directory is checked out on: do not run \`git checkout <branch>\` or \`git switch <branch>\` to a different branch unless the user explicitly asks. Read-only Git and branch inspection, such as \`git status\`, \`git branch\`, and \`git log\`, is allowed. Response length policy: default to a short, crisp reply (a few sentences or a short bulleted list) that directly answers the user; avoid preamble, restating the question, and filler. For questions about this repository's code or architecture, prioritize correctness and cite real paths/symbols over maximum brevity — still lead short, but do not omit the evidence needed to be accurate. If a thorough answer genuinely needs long-form content (for example multi-step plans, design proposals, deep multi-file traces, deep analyses, or long file excerpts), keep the chat reply brief with a one- or two-sentence summary plus key citations. When \`fn_send_message\` is available in this session, send the full write-up via \`fn_send_message\` using \`type: "agent-to-user"\` and \`to_id: "dashboard"\` (additive detail that must not duplicate the chat reply). When that tool is not available, put the necessary detail in the chat reply itself rather than inventing a mailbox path.`;
export const CHAT_AGENT_MESSAGE_ROUTING_GUIDANCE = `## Messaging Semantics\n\nYour chat reply is the primary response to the user. Do not also call \`fn_send_message\` with the same content just to mirror your chat response into mailbox.\n\nUse \`fn_send_message\` only when either (a) the user explicitly asks for mailbox/inbox/notification delivery (for example: "send me this in mail", "ntfy me when…", or "leave me a note in my inbox"), or (b) you are sending a genuinely longer follow-up that did not fit in a short chat reply. In either case, send with \`type: "agent-to-user"\` and target the dashboard user alias (\`to_id: "dashboard"\` is preferred), and ensure the mailbox message is additive rather than a duplicate of the chat reply. Never route that as a user/CLI → agent message.`; export const CHAT_AGENT_MESSAGE_ROUTING_GUIDANCE = `## Messaging Semantics\n\nYour chat reply is the primary response to the user. Do not also call \`fn_send_message\` with the same content just to mirror your chat response into mailbox.\n\nUse \`fn_send_message\` only when either (a) the user explicitly asks for mailbox/inbox/notification delivery (for example: "send me this in mail", "ntfy me when…", or "leave me a note in my inbox"), or (b) you are sending a genuinely longer follow-up that did not fit in a short chat reply. In either case, send with \`type: "agent-to-user"\` and target the dashboard user alias (\`to_id: "dashboard"\` is preferred), and ensure the mailbox message is additive rather than a duplicate of the chat reply. Never route that as a user/CLI → agent message.`;
/*
FNXC:ChatCodebaseAccuracy 2026-07-22-12:00:
Users reported Planning Mode is more accurate about the codebase than agent chat. Plan mode inherits the triage seam's "read/grep first, name real files" contract; chat only had a short helpful-assistant persona plus a brevity default, so models answered from priors. This section is a lean port of those investigation rules — not the full PROMPT.md interview — so chat stays conversational while still grounding code/architecture claims in the live checkout. Append during direct and room chat prompt assembly.
FNXC:ChatCodebaseAccuracy 2026-07-23-05:20:
PR #2416 review: (1) room/agentless sessions lack `fn_send_message` — long-form mailbox guidance is conditional on the tool being registered; (2) `find` must stay bounded to the project checkout / known merge path and never walk the OS temp root, matching AGENTS.md.
*/
export const CHAT_CODEBASE_ACCURACY_GUIDANCE = `## Codebase accuracy
When the user asks about **this project's** code, architecture, behavior, APIs, files, tests, settings, or "how does X work here", ground the answer in the live checkout **before** you reply. Do not answer from training-data memory or generic framework knowledge as if it were this repo.
### Investigate first
1. Use readonly tools first: \`grep\`, \`find\`, \`ls\`, \`read\` (and readonly git such as \`git status\` / \`git log\` / \`git branch\` when relevant). Prefer \`grep\`/\`find\` to locate symbols, then \`read\` the concrete files. Restrict \`find\` (and directory walks) to the project checkout or a known merge/worktree path with bounded, prefix-filtered scans; never recurse through the OS temp root (\`$TMPDIR\`, \`/tmp\`, macOS \`/var/folders/...\`) or an unbounded path.
2. Open the real definitions and call sites you will cite. For behavioral questions, also check tests and docs under \`docs/\` / \`CONCEPTS.md\` when they exist.
3. If board/task context matters, use \`fn_task_list\` / \`fn_task_show\` (and search tools if available) rather than inventing task state.
4. Only after you have evidence, write the chat reply.
### How to answer
- Name **real** paths, symbols, and modules from the checkout (e.g. \`packages/foo/src/bar.ts\`, \`functionName\`). Prefer evidence over abstraction.
- Prefer a short answer that is **correct and cited** over a fluent guess. A crisp 3–6 bullet list with file paths is better than a long ungrounded narrative.
- If tools conflict with your prior assumptions, **trust the tools**.
- If you cannot find something after a reasonable search, say what you searched and that it may not exist — do not invent modules, routes, tables, or APIs.
- Distinguish **verified in this checkout** from **general recommendation**. Label speculation explicitly.
### When brevity still applies
- Purely conversational, product-how-to, or non-code questions: keep the existing short/crisp default.
- Code/architecture questions: still lead with a short answer, but include the grounding (paths / symbols). For long excerpts, multi-file traces, or deep designs: lead with a brief chat summary; if \`fn_send_message\` is available in this session, send the full write-up via that tool (\`type: "agent-to-user"\`, \`to_id: "dashboard"\`); if it is not available, keep the necessary detail in the chat reply instead of calling a missing tool.
- Do **not** start a Planning Mode interview, write PROMPT.md, or invent steps/file-scope specs unless the user asks to plan or create a task.`;
/** /**
* FNXC:ChatAskQuestion 2026-06-17-13:17: * FNXC:ChatAskQuestion 2026-06-17-13:17:
* Only the dashboard chat lane registers `fn_ask_question`, so append this guidance during sendMessage prompt assembly instead of baking it into room-responder prompts that do not receive the tool. * Only the dashboard chat lane registers `fn_ask_question`, so append this guidance during sendMessage prompt assembly instead of baking it into room-responder prompts that do not receive the tool.
@@ -1953,6 +1988,11 @@ export class ChatManager {
systemPrompt = `${systemPrompt}\n\n${mentionContext}`; systemPrompt = `${systemPrompt}\n\n${mentionContext}`;
} }
systemPrompt = `${systemPrompt}\n\n${CHAT_AGENT_MESSAGE_ROUTING_GUIDANCE}`; systemPrompt = `${systemPrompt}\n\n${CHAT_AGENT_MESSAGE_ROUTING_GUIDANCE}`;
/*
FNXC:ChatCodebaseAccuracy 2026-07-22-12:00:
Room responders share the same investigate-first contract as direct chat so multi-agent rooms do not answer codebase questions from priors.
*/
systemPrompt = `${systemPrompt}\n\n${CHAT_CODEBASE_ACCURACY_GUIDANCE}`;
const roomCompactionSettings = await this.getRoomCompactionSettings(); const roomCompactionSettings = await this.getRoomCompactionSettings();
const roomMessages = await this.chatStore.getRoomMessages(input.roomId, { limit: roomCompactionSettings.fetchLimit }); const roomMessages = await this.chatStore.getRoomMessages(input.roomId, { limit: roomCompactionSettings.fetchLimit });
@@ -2360,6 +2400,11 @@ export class ChatManager {
diagnostics.warn(`Failed to build enriched system prompt for ${agent.id}: ${message}`); diagnostics.warn(`Failed to build enriched system prompt for ${agent.id}: ${message}`);
} }
} }
/*
FNXC:ChatCodebaseAccuracy 2026-07-22-12:00:
Append for every direct chat turn (with or without a durable agent) so generic project chat and agent chat both ground code answers in the live checkout.
*/
systemPrompt = `${systemPrompt}\n\n${CHAT_CODEBASE_ACCURACY_GUIDANCE}`;
systemPrompt = `${systemPrompt}\n\n${CHAT_ASK_QUESTION_GUIDANCE}`; systemPrompt = `${systemPrompt}\n\n${CHAT_ASK_QUESTION_GUIDANCE}`;
const taskPlannerChatTaskId = typeof session.agentId === "string" && session.agentId.startsWith(TASK_PLANNER_CHAT_AGENT_ID_PREFIX) const taskPlannerChatTaskId = typeof session.agentId === "string" && session.agentId.startsWith(TASK_PLANNER_CHAT_AGENT_ID_PREFIX)