feat(FN-1617): add verification output summarization for concise test failure reports

- Add summarizeVerificationOutput helper function to condense test output
- Update runVerificationCommand to use summarization for cleaner merge reports
- Improve readability of test failure summaries in the dashboard
This commit is contained in:
gsxdsm
2026-04-12 13:31:27 -07:00
parent 2fc4fccd00
commit c42ea664f8
2 changed files with 220 additions and 2 deletions

View File

@@ -79,6 +79,7 @@ import {
extractFileScope,
validateDiffScope,
shouldSyncDependenciesForMerge,
summarizeVerificationOutput,
type ConflictCategory,
} from "./merger.js";
import { createKbAgent } from "./pi.js";
@@ -2263,6 +2264,14 @@ describe("aiMergeTask — deterministic merge verification", () => {
expect.stringContaining("Deterministic test verification failed"),
"VerificationError",
);
// Verify log entry contains summary (not raw output) with engine logs reference
const logCalls = (store.logEntry as ReturnType<typeof vi.fn>).mock.calls;
const verificationFailCall = logCalls.find((call: any[]) =>
typeof call[1] === "string" && call[1].includes("[verification] test command failed"),
);
expect(verificationFailCall).toBeTruthy();
expect(verificationFailCall![1]).toContain("full output available in engine logs");
});
it("does not fail verification when verbose test output exceeds buffer after exit 0", async () => {
@@ -3565,3 +3574,63 @@ describe("aiMergeTask — context limit recovery with truncation", () => {
expect(vi.mocked(compactSessionContext)).not.toHaveBeenCalled();
});
});
describe("summarizeVerificationOutput", () => {
it("extracts vitest-style test summary with failure names", () => {
const output = [
"some setup output...",
"FAIL src/utils.test.ts",
" ✗ should validate input",
" ✗ should handle edge case",
"Tests: 2 failed, 48 passed, 50 total",
].join("\n");
const result = summarizeVerificationOutput(output, "test");
expect(result).toContain("Tests: 2 failed, 48 passed, 50 total");
expect(result).toContain("should validate input");
expect(result).toContain("full output available in engine logs");
});
it("limits failure names to 5 with overflow indicator", () => {
// Build output with 7 FAIL lines
const output = [
"FAIL test1",
"FAIL test2",
"FAIL test3",
"FAIL test4",
"FAIL test5",
"FAIL test6",
"FAIL test7",
"Tests: 7 failed, 0 passed, 7 total",
].join("\n");
const result = summarizeVerificationOutput(output, "test");
expect(result).toContain("test5");
expect(result).toContain("... and 2 more failures");
expect(result).not.toContain("test6");
});
it("falls back to first 500 chars for unstructured output", () => {
const output = "A".repeat(1000);
const result = summarizeVerificationOutput(output, "build");
expect(result.length).toBeLessThan(600);
expect(result).toContain("full output available in engine logs");
});
it("returns generic message for empty output", () => {
const result = summarizeVerificationOutput("", "test");
expect(result).toContain("no output");
expect(result).toContain("full output available in engine logs");
});
it("deduplicates identical failure names", () => {
const output = [
"FAIL src/a.test.ts",
"FAIL src/a.test.ts", // duplicate
"FAIL src/b.test.ts",
"Tests: 3 failed, 0 passed",
].join("\n");
const result = summarizeVerificationOutput(output, "test");
// Should contain only unique names (src/a.test.ts once, src/b.test.ts)
const bulletMatches = result.match(/• /g);
expect(bulletMatches?.length).toBe(2);
});
});

View File

@@ -90,6 +90,154 @@ function truncateVerificationOutput(output: string): string {
return `... output truncated to last ${VERIFICATION_LOG_MAX_CHARS} characters ...\n${output.slice(-VERIFICATION_LOG_MAX_CHARS)}`;
}
/**
* Summarize test/build verification failure output into a concise message.
* Extracts test counts and failure names from common test runner formats,
* falls back to truncated output for unstructured output.
*/
export function summarizeVerificationOutput(output: string, type: "test" | "build"): string {
const lines = output.split("\n");
let summaryLine: string | null = null;
const failureNames = new Set<string>();
// 1. Extract summary line
for (const line of lines) {
// vitest/jest: "Tests: 2 failed, 48 passed, 50 total"
const testsMatch = line.match(/^Tests:\s*(\d+)\s+failed,\s*(\d+)\s+passed(?:,\s*(\d+)\s+total)?/i);
if (testsMatch) {
const failed = testsMatch[1];
const passed = testsMatch[2];
const total = testsMatch[3] ? `, ${testsMatch[3]} total` : "";
summaryLine = `Tests: ${failed} failed, ${passed} passed${total}`;
break;
}
// Generic: "X tests failed, Y passed, Z total"
const genericMatch = line.match(/^(\d+)\s+tests?\s+failed,\s*(\d+)\s+passed,\s*(\d+)\s+total/i);
if (genericMatch) {
summaryLine = `${genericMatch[1]} tests failed, ${genericMatch[2]} passed, ${genericMatch[3]} total`;
break;
}
// Various runners: "X failing" / "X failures" / "X failed"
const failCountMatch = line.match(/^(\d+)\s+(failings?|failures?|failed)/i);
if (failCountMatch) {
summaryLine = `${failCountMatch[1]} ${failCountMatch[2]}`;
break;
}
}
// 2. Extract failure names (up to 5 unique names)
// Priority: markers (✗, ●, -) provide descriptive names, FAIL lines provide file context
// Process markers first (they give actual test names), then FAIL lines (file context)
const markerLines: string[] = [];
const failLines: string[] = [];
for (const line of lines) {
// FAIL <file> — vitest file-level failure header (at start of line)
const failMatch = line.match(/^(FAIL)\s+(.+)/);
if (failMatch) {
failLines.push(failMatch[2].trim());
continue;
}
// Trim leading whitespace for marker detection (vitest indents failure details)
const trimmedLine = line.trimStart();
// Unicode cross markers: ✗ or ✕ or × (possibly indented)
const crossMatch = trimmedLine.match(/^[✗✕×]\s*(.+)/);
if (crossMatch) {
markerLines.push(crossMatch[1].trim());
continue;
}
// Jest failure bullet: ● (possibly indented)
const bulletMatch = trimmedLine.match(/^●\s*(.+)/);
if (bulletMatch) {
markerLines.push(bulletMatch[1].trim());
continue;
}
// Jest/Mocha indented test name: - MyTest should do something (indented)
const dashMatch = trimmedLine.match(/^-\s+(\S[\s\S]*?)$/);
if (dashMatch) {
const potential = dashMatch[1].trim();
// Only include lines that look like test names (contain common test patterns)
if (/[\s>]|(should|cannot|does|doesn|to|not|throws)/i.test(potential)) {
markerLines.push(potential);
}
continue;
}
// AssertionError — generic assertion failures (possibly indented)
const assertionMatch = trimmedLine.match(/^(AssertionError|AssertionError:.*)$/i);
if (assertionMatch) {
markerLines.push(assertionMatch[1]);
}
}
// Add marker names first (higher priority - they give actual test names)
for (const name of markerLines) {
const truncated = name.length > 120 ? name.slice(0, 120) : name;
failureNames.add(truncated);
}
// Fill remaining slots with FAIL file names (lower priority - just file context)
for (const name of failLines) {
const truncated = name.length > 120 ? name.slice(0, 120) : name;
failureNames.add(truncated);
}
// 3. Build the summary string
const footer = "(full output available in engine logs)";
const parts: string[] = [];
if (summaryLine) {
parts.push(summaryLine);
}
if (failureNames.size > 0) {
const names = Array.from(failureNames);
if (names.length <= 5) {
for (const name of names) {
parts.push(`${name}`);
}
} else {
// Show first 5 and note overflow
for (let i = 0; i < 5; i++) {
parts.push(`${names[i]}`);
}
parts.push(` • ... and ${names.length - 5} more failures`);
}
}
if (parts.length > 0) {
parts.push(footer);
return parts.join("\n");
}
// 4. Fallback — no structured data found
const trimmed = output.trim();
if (!trimmed) {
return `Verification command failed with no output\n${footer}`;
}
if (trimmed.length <= 500) {
return `${trimmed}\n${footer}`;
}
// Truncate at last space or newline boundary
let cutoff = 500;
for (let i = 500; i < trimmed.length; i++) {
if (trimmed[i] === " " || trimmed[i] === "\n") {
cutoff = i;
break;
}
}
return `${trimmed.slice(0, cutoff)}...\n${footer}`;
}
function truncateWorkflowScriptOutput(output: string): string {
if (output.length <= WORKFLOW_SCRIPT_OUTPUT_MAX_CHARS) return output;
return `... output truncated to last ${WORKFLOW_SCRIPT_OUTPUT_MAX_CHARS} characters ...\n${output.slice(-WORKFLOW_SCRIPT_OUTPUT_MAX_CHARS)}`;
@@ -369,11 +517,12 @@ async function runVerificationCommand(
// Keep command output out of process logs. The bounded excerpt is stored on
// the task for diagnostics without dumping test output to the engine stdout.
const summary = truncateVerificationOutput(result.stderr || result.stdout || error.message || "Unknown error");
const output = result.stderr || result.stdout || error.message || "Unknown error";
const summary = summarizeVerificationOutput(output, type);
mergerLog.error(`${taskId}: ${type} command failed (exit ${result.exitCode}); output captured in task log`);
await store.logEntry(
taskId,
`[verification] ${type} command failed (exit ${result.exitCode}): ${summary.trim()}`,
`[verification] ${type} command failed (exit ${result.exitCode}):\n${summary}`,
);
}