feat(engine): default-to-fail posture + non-mutating verification run (U2,U3)
Behavioral/bug assertions now default to fail unless a bounded verification run confirms them; static assertions keep the existing read-only judging path. Adds mission-verification.ts (injected capability): disposable git checkout at a trusted revision, explicit isolating sandbox backend with fail-closed + scrubbed env, fixed command template + validated test path, merge-base pre-fix baseline (rejects pass-on-both), git-clean post-condition, inconclusive as a first-class verdict. Judge session stays tools:readonly.
This commit is contained in:
@@ -1070,6 +1070,9 @@ export {
|
||||
FEATURE_LOOP_STATES,
|
||||
VALIDATOR_RUN_STATUSES,
|
||||
MISSION_ASSERTION_STATUSES,
|
||||
MISSION_ASSERTION_TYPES,
|
||||
DEFAULT_MISSION_ASSERTION_TYPE,
|
||||
normalizeMissionAssertionType,
|
||||
MILESTONE_VALIDATION_STATES,
|
||||
} from "./mission-types.js";
|
||||
export type {
|
||||
@@ -1116,6 +1119,7 @@ export type {
|
||||
MissionFeatureLoopSnapshot,
|
||||
// Contract assertion types
|
||||
MissionAssertionStatus,
|
||||
MissionAssertionType,
|
||||
MilestoneValidationState,
|
||||
MissionContractAssertion,
|
||||
FeatureAssertionLink,
|
||||
|
||||
350
packages/engine/src/__tests__/mission-verification.test.ts
Normal file
350
packages/engine/src/__tests__/mission-verification.test.ts
Normal file
@@ -0,0 +1,350 @@
|
||||
/**
|
||||
* Tests for the behavioral-verification capability (U3).
|
||||
*
|
||||
* Covers the isolation/safety contract before any real execution:
|
||||
* - command-template rejects shell metacharacters (R19)
|
||||
* - fail-closed when no isolating sandbox backend is available (R18)
|
||||
* - env scrubbed to a minimal allowlist (R18)
|
||||
* - pass-on-both regression test rejected (R5/AE5)
|
||||
* - source tree git-clean post-condition asserted (R17)
|
||||
* - no integration SHA → inconclusive (R11 fail-closed)
|
||||
*/
|
||||
|
||||
import { describe, it, expect, vi, afterEach } from "vitest";
|
||||
import {
|
||||
validateTestPath,
|
||||
buildVerificationCommand,
|
||||
selectIsolatingBackend,
|
||||
scrubEnv,
|
||||
VERIFICATION_ENV_ALLOWLIST,
|
||||
TestExecutionVerificationCapability,
|
||||
type CheckoutMaterializer,
|
||||
type IsolatingBackendProbe,
|
||||
type VerificationRequest,
|
||||
} from "../mission-verification.js";
|
||||
import {
|
||||
__resetSandboxBackendForTests,
|
||||
type SandboxBackend,
|
||||
type SandboxStreamingResult,
|
||||
} from "../sandbox/index.js";
|
||||
import type { TaskStore } from "@fusion/core";
|
||||
|
||||
afterEach(() => {
|
||||
__resetSandboxBackendForTests();
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
// ── validateTestPath (R19) ────────────────────────────────────────────────────
|
||||
|
||||
describe("validateTestPath", () => {
|
||||
it("accepts a plain relative test path", () => {
|
||||
expect(validateTestPath("packages/engine/src/__tests__/foo.test.ts")).toBe(
|
||||
"packages/engine/src/__tests__/foo.test.ts",
|
||||
);
|
||||
});
|
||||
|
||||
it.each([
|
||||
"foo.test.ts; rm -rf /",
|
||||
"foo.test.ts && curl evil",
|
||||
"foo.test.ts | cat",
|
||||
"$(whoami).test.ts",
|
||||
"`id`.test.ts",
|
||||
"foo.test.ts\nrm x",
|
||||
"a${b}.test.ts",
|
||||
"foo>out.test.ts",
|
||||
])("rejects shell metacharacters: %s", (p) => {
|
||||
expect(validateTestPath(p)).toBeNull();
|
||||
});
|
||||
|
||||
it("rejects absolute paths, parent escapes, flag-like, and non-strings", () => {
|
||||
expect(validateTestPath("/etc/passwd")).toBeNull();
|
||||
expect(validateTestPath("../../etc/passwd")).toBeNull();
|
||||
expect(validateTestPath("a/../../b")).toBeNull();
|
||||
expect(validateTestPath("--config=evil")).toBeNull();
|
||||
expect(validateTestPath("")).toBeNull();
|
||||
expect(validateTestPath(42 as unknown)).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildVerificationCommand", () => {
|
||||
it("substitutes a validated test path into the template", () => {
|
||||
expect(buildVerificationCommand("pnpm vitest run {testPath}", "src/a.test.ts")).toBe(
|
||||
"pnpm vitest run src/a.test.ts",
|
||||
);
|
||||
});
|
||||
|
||||
it("produces a whole-suite command when no path is supplied", () => {
|
||||
expect(buildVerificationCommand("pnpm vitest run {testPath}")).toBe("pnpm vitest run");
|
||||
});
|
||||
|
||||
it("throws if asked to substitute an unsafe path (defense in depth)", () => {
|
||||
expect(() => buildVerificationCommand("pnpm vitest run {testPath}", "a; rm -rf /")).toThrow();
|
||||
});
|
||||
});
|
||||
|
||||
// ── selectIsolatingBackend (R18 fail-closed) ───────────────────────────────────
|
||||
|
||||
describe("selectIsolatingBackend", () => {
|
||||
it("selects bubblewrap on linux when available", () => {
|
||||
expect(
|
||||
selectIsolatingBackend({ platform: "linux", bubblewrapAvailable: true, sandboxExecAvailable: false }).backendId,
|
||||
).toBe("bubblewrap");
|
||||
});
|
||||
|
||||
it("selects sandbox-exec on darwin when available", () => {
|
||||
expect(
|
||||
selectIsolatingBackend({ platform: "darwin", bubblewrapAvailable: false, sandboxExecAvailable: true }).backendId,
|
||||
).toBe("sandbox-exec");
|
||||
});
|
||||
|
||||
it("fails closed (null) when no isolating backend is available", () => {
|
||||
const sel = selectIsolatingBackend({ platform: "linux", bubblewrapAvailable: false, sandboxExecAvailable: false });
|
||||
expect(sel.backendId).toBeNull();
|
||||
expect(sel.reason).toMatch(/no isolating sandbox backend/);
|
||||
});
|
||||
});
|
||||
|
||||
// ── scrubEnv (R18) ─────────────────────────────────────────────────────────────
|
||||
|
||||
describe("scrubEnv", () => {
|
||||
it("keeps only allowlisted keys and forces CI=1", () => {
|
||||
const result = scrubEnv({
|
||||
PATH: "/usr/bin",
|
||||
HOME: "/home/u",
|
||||
ANTHROPIC_API_KEY: "secret",
|
||||
DATABASE_URL: "postgres://secret",
|
||||
AWS_SECRET_ACCESS_KEY: "secret",
|
||||
});
|
||||
expect(result.PATH).toBe("/usr/bin");
|
||||
expect(result.HOME).toBe("/home/u");
|
||||
expect(result.CI).toBe("1");
|
||||
expect(result.ANTHROPIC_API_KEY).toBeUndefined();
|
||||
expect(result.DATABASE_URL).toBeUndefined();
|
||||
expect(result.AWS_SECRET_ACCESS_KEY).toBeUndefined();
|
||||
});
|
||||
|
||||
it("only ever emits allowlisted keys plus CI", () => {
|
||||
const result = scrubEnv({ FOO: "x", BAR: "y", PATH: "/bin" });
|
||||
const allowed = new Set<string>([...VERIFICATION_ENV_ALLOWLIST, "CI"]);
|
||||
for (const k of Object.keys(result)) {
|
||||
expect(allowed.has(k)).toBe(true);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// ── TestExecutionVerificationCapability ────────────────────────────────────────
|
||||
|
||||
function makeMockStore(): TaskStore {
|
||||
return {
|
||||
logEntry: vi.fn().mockResolvedValue(undefined),
|
||||
appendAgentLog: vi.fn().mockResolvedValue(undefined),
|
||||
} as unknown as TaskStore;
|
||||
}
|
||||
|
||||
/** A fake materializer that records dispose/clean calls without touching git. */
|
||||
function makeFakeMaterializer(): CheckoutMaterializer & {
|
||||
disposed: number;
|
||||
cleanCalls: number;
|
||||
setDirty(dirty: boolean): void;
|
||||
} {
|
||||
let dirty = false;
|
||||
const state = {
|
||||
disposed: 0,
|
||||
cleanCalls: 0,
|
||||
setDirty(d: boolean) {
|
||||
dirty = d;
|
||||
},
|
||||
async materialize(_rootDir: string, _revision: string) {
|
||||
return {
|
||||
dir: "/tmp/fake-checkout",
|
||||
dispose: async () => {
|
||||
state.disposed += 1;
|
||||
},
|
||||
};
|
||||
},
|
||||
async assertSourceClean(_rootDir: string) {
|
||||
state.cleanCalls += 1;
|
||||
if (dirty) throw new Error("Source tree is not git-clean after verification run");
|
||||
},
|
||||
};
|
||||
return state;
|
||||
}
|
||||
|
||||
/** A sandbox backend whose runStreaming returns a scripted outcome. */
|
||||
function makeScriptedBackend(outcomes: SandboxStreamingResult[]): SandboxBackend {
|
||||
let i = 0;
|
||||
return {
|
||||
capabilities: () => ({
|
||||
id: "bubblewrap",
|
||||
supportsNetworkPolicy: true,
|
||||
supportsFilesystemPolicy: true,
|
||||
supportsStreaming: true,
|
||||
platform: "any",
|
||||
}),
|
||||
prepare: vi.fn().mockResolvedValue(undefined),
|
||||
run: vi.fn(),
|
||||
runStreaming: vi.fn(async () => {
|
||||
const out = outcomes[Math.min(i, outcomes.length - 1)];
|
||||
i += 1;
|
||||
return out;
|
||||
}),
|
||||
dispose: vi.fn().mockResolvedValue(undefined),
|
||||
} as unknown as SandboxBackend;
|
||||
}
|
||||
|
||||
const isolatingProbe = (): Promise<IsolatingBackendProbe> =>
|
||||
Promise.resolve({ platform: "linux", bubblewrapAvailable: true, sandboxExecAvailable: false });
|
||||
|
||||
function baseRequest(overrides: Partial<VerificationRequest> = {}): VerificationRequest {
|
||||
return {
|
||||
assertionId: "CA-1",
|
||||
assertion: "the bug no longer reproduces",
|
||||
taskId: "FN-1",
|
||||
integrationSha: "abc123",
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe("TestExecutionVerificationCapability", () => {
|
||||
it("fails closed to inconclusive when no isolating backend is available (R18)", async () => {
|
||||
const materializer = makeFakeMaterializer();
|
||||
const cap = new TestExecutionVerificationCapability({
|
||||
store: makeMockStore(),
|
||||
rootDir: "/repo",
|
||||
commandTemplate: "pnpm vitest run {testPath}",
|
||||
materializer,
|
||||
probeBackends: async () => ({ platform: "linux", bubblewrapAvailable: false, sandboxExecAvailable: false }),
|
||||
});
|
||||
|
||||
const outcome = await cap.verifyBehavioralAssertion(baseRequest());
|
||||
expect(outcome.verdict).toBe("inconclusive");
|
||||
expect(outcome.reason).toMatch(/no isolating sandbox backend/);
|
||||
// Never materialized / executed.
|
||||
expect(materializer.disposed).toBe(0);
|
||||
});
|
||||
|
||||
it("returns inconclusive when no integration SHA is available (R11)", async () => {
|
||||
const cap = new TestExecutionVerificationCapability({
|
||||
store: makeMockStore(),
|
||||
rootDir: "/repo",
|
||||
commandTemplate: "pnpm vitest run {testPath}",
|
||||
materializer: makeFakeMaterializer(),
|
||||
probeBackends: isolatingProbe,
|
||||
});
|
||||
const outcome = await cap.verifyBehavioralAssertion(baseRequest({ integrationSha: undefined }));
|
||||
expect(outcome.verdict).toBe("inconclusive");
|
||||
expect(outcome.reason).toMatch(/integration SHA/);
|
||||
});
|
||||
|
||||
it("rejects an agent test path with shell metacharacters before execution (R19)", async () => {
|
||||
const materializer = makeFakeMaterializer();
|
||||
const cap = new TestExecutionVerificationCapability({
|
||||
store: makeMockStore(),
|
||||
rootDir: "/repo",
|
||||
commandTemplate: "pnpm vitest run {testPath}",
|
||||
materializer,
|
||||
probeBackends: isolatingProbe,
|
||||
});
|
||||
const outcome = await cap.verifyBehavioralAssertion(
|
||||
baseRequest({ proof: { testFilePath: "a.test.ts; rm -rf /" } }),
|
||||
);
|
||||
expect(outcome.verdict).toBe("inconclusive");
|
||||
expect(outcome.reason).toMatch(/shell metacharacters|rejected/);
|
||||
expect(materializer.disposed).toBe(0);
|
||||
});
|
||||
|
||||
it("passes when the whole-suite run succeeds, and asserts source git-clean (R17)", async () => {
|
||||
const materializer = makeFakeMaterializer();
|
||||
const cap = new TestExecutionVerificationCapability({
|
||||
store: makeMockStore(),
|
||||
rootDir: "/repo",
|
||||
commandTemplate: "pnpm vitest run {testPath}",
|
||||
materializer,
|
||||
probeBackends: isolatingProbe,
|
||||
backendFactory: () =>
|
||||
makeScriptedBackend([{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }]),
|
||||
});
|
||||
const outcome = await cap.verifyBehavioralAssertion(baseRequest());
|
||||
expect(outcome.verdict).toBe("pass");
|
||||
expect(materializer.disposed).toBe(1);
|
||||
expect(materializer.cleanCalls).toBe(1);
|
||||
});
|
||||
|
||||
it("rejects a regression test that passes on BOTH baseline and implementation (R5/AE5)", async () => {
|
||||
const materializer = makeFakeMaterializer();
|
||||
const cap = new TestExecutionVerificationCapability({
|
||||
store: makeMockStore(),
|
||||
rootDir: "/repo",
|
||||
commandTemplate: "pnpm vitest run {testPath}",
|
||||
materializer,
|
||||
probeBackends: isolatingProbe,
|
||||
// impl run (1st) success, baseline run (2nd) success → pass-on-both.
|
||||
backendFactory: () =>
|
||||
makeScriptedBackend([
|
||||
{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false },
|
||||
{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false },
|
||||
]),
|
||||
});
|
||||
const outcome = await cap.verifyBehavioralAssertion(
|
||||
baseRequest({ proof: { testFilePath: "src/a.test.ts" }, mergeBaseSha: "base000" }),
|
||||
);
|
||||
expect(outcome.verdict).toBe("fail");
|
||||
expect(outcome.reason).toMatch(/both/);
|
||||
});
|
||||
|
||||
it("passes when a regression test fails on baseline and passes on implementation (R5)", async () => {
|
||||
const materializer = makeFakeMaterializer();
|
||||
const cap = new TestExecutionVerificationCapability({
|
||||
store: makeMockStore(),
|
||||
rootDir: "/repo",
|
||||
commandTemplate: "pnpm vitest run {testPath}",
|
||||
materializer,
|
||||
probeBackends: isolatingProbe,
|
||||
// impl run (1st) success, baseline run (2nd) non-zero-exit → genuine proof.
|
||||
backendFactory: () =>
|
||||
makeScriptedBackend([
|
||||
{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false },
|
||||
{ outcome: "non-zero-exit", stdout: "", stderr: "boom", exitCode: 1, signal: null },
|
||||
]),
|
||||
});
|
||||
const outcome = await cap.verifyBehavioralAssertion(
|
||||
baseRequest({ proof: { testFilePath: "src/a.test.ts" }, mergeBaseSha: "base000" }),
|
||||
);
|
||||
expect(outcome.verdict).toBe("pass");
|
||||
});
|
||||
|
||||
it("inconclusive when the run times out (R9), still asserts source clean", async () => {
|
||||
const materializer = makeFakeMaterializer();
|
||||
const cap = new TestExecutionVerificationCapability({
|
||||
store: makeMockStore(),
|
||||
rootDir: "/repo",
|
||||
commandTemplate: "pnpm vitest run {testPath}",
|
||||
materializer,
|
||||
probeBackends: isolatingProbe,
|
||||
backendFactory: () =>
|
||||
makeScriptedBackend([{ outcome: "timeout", stdout: "", stderr: "", timeoutMs: 1000 }]),
|
||||
});
|
||||
// A timeout surfaces as a thrown ETIMEDOUT inside runVerificationCommand,
|
||||
// which the capability catches and maps to a non-pass. For the whole-suite
|
||||
// channel a non-success result is a behavioral fail; but a timeout throw is
|
||||
// caught by the capability's try/catch → inconclusive.
|
||||
const outcome = await cap.verifyBehavioralAssertion(baseRequest());
|
||||
expect(["inconclusive", "fail"]).toContain(outcome.verdict);
|
||||
expect(materializer.cleanCalls).toBe(1);
|
||||
});
|
||||
|
||||
it("throws if the source tree is dirty after a run (R17 post-condition)", async () => {
|
||||
const materializer = makeFakeMaterializer();
|
||||
materializer.setDirty(true);
|
||||
const cap = new TestExecutionVerificationCapability({
|
||||
store: makeMockStore(),
|
||||
rootDir: "/repo",
|
||||
commandTemplate: "pnpm vitest run {testPath}",
|
||||
materializer,
|
||||
probeBackends: isolatingProbe,
|
||||
backendFactory: () =>
|
||||
makeScriptedBackend([{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }]),
|
||||
});
|
||||
await expect(cap.verifyBehavioralAssertion(baseRequest())).rejects.toThrow(/git-clean/);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,402 @@
|
||||
/**
|
||||
* Behavioral-verification posture in the Validator Run (U2 + U3).
|
||||
*
|
||||
* Verifies the default-to-fail posture for behavioral assertions and the
|
||||
* non-mutating verification step that confirms them, while static assertions
|
||||
* keep the exact legacy judge path.
|
||||
*
|
||||
* Covers:
|
||||
* - AE2: behavioral assertion with no verification evidence → fail, even when
|
||||
* the judge text claims pass (capability absent).
|
||||
* - AE3: static assertion → unchanged static verdict, no verification invoked.
|
||||
* - Mixed set: static and behavioral each take their correct path.
|
||||
* - Behavioral pass via an injected verification capability.
|
||||
* - Behavioral inconclusive → blocked verdict, NO fix feature.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
|
||||
import type {
|
||||
Mission,
|
||||
Milestone,
|
||||
Slice,
|
||||
MissionFeature,
|
||||
MissionValidatorRun,
|
||||
} from "@fusion/core";
|
||||
import type { VerificationCapability, VerificationOutcome } from "../../mission-verification.js";
|
||||
|
||||
// ── Mock AI dependencies (mirror mission-execution-loop.test.ts) ───────────────
|
||||
const mockSessionHolder: {
|
||||
session: { state: { messages: Array<{ role: string; content: string }> }; dispose: ReturnType<typeof vi.fn> };
|
||||
} = { session: { state: { messages: [] }, dispose: vi.fn() } };
|
||||
|
||||
vi.mock("../../pi.js", () => ({
|
||||
createFnAgent: vi.fn(() => Promise.resolve({ session: mockSessionHolder.session })),
|
||||
promptWithFallback: vi.fn().mockResolvedValue(undefined),
|
||||
}));
|
||||
|
||||
vi.mock("../../logger.js", () => ({
|
||||
createLogger: vi.fn(() => ({ log: vi.fn(), warn: vi.fn(), error: vi.fn() })),
|
||||
}));
|
||||
|
||||
vi.mock("../../agent-session-helpers.js", async (importOriginal) => {
|
||||
const actual = await importOriginal<typeof import("../../agent-session-helpers.js")>();
|
||||
return {
|
||||
...actual,
|
||||
createResolvedAgentSession: vi.fn(async () => ({
|
||||
session: mockSessionHolder.session as any,
|
||||
sessionFile: undefined,
|
||||
runtimeId: "test-runtime",
|
||||
wasConfigured: true,
|
||||
})),
|
||||
};
|
||||
});
|
||||
|
||||
import { createResolvedAgentSession } from "../../agent-session-helpers.js";
|
||||
import { MissionExecutionLoop } from "../../mission-execution-loop.js";
|
||||
|
||||
type AssertionRow = {
|
||||
id: string;
|
||||
milestoneId: string;
|
||||
title: string;
|
||||
assertion: string;
|
||||
status: "pending" | "passed" | "failed" | "blocked";
|
||||
type?: "static" | "behavioral";
|
||||
orderIndex: number;
|
||||
createdAt: string;
|
||||
updatedAt: string;
|
||||
sourceFeatureId?: string;
|
||||
};
|
||||
|
||||
function now() {
|
||||
return new Date().toISOString();
|
||||
}
|
||||
|
||||
function createMockMission(): Mission {
|
||||
return {
|
||||
id: "M-TEST1",
|
||||
title: "Test Mission",
|
||||
status: "active",
|
||||
interviewState: "not_started",
|
||||
autopilotEnabled: true,
|
||||
autopilotState: "inactive",
|
||||
createdAt: now(),
|
||||
updatedAt: now(),
|
||||
};
|
||||
}
|
||||
|
||||
function createMockMilestone(overrides: Partial<Milestone> = {}): Milestone {
|
||||
return {
|
||||
id: "MS-001",
|
||||
missionId: "M-TEST1",
|
||||
title: "Test Milestone",
|
||||
status: "active",
|
||||
orderIndex: 0,
|
||||
interviewState: "not_started",
|
||||
dependencies: [],
|
||||
createdAt: now(),
|
||||
updatedAt: now(),
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function createMockSlice(overrides: Partial<Slice> = {}): Slice {
|
||||
return {
|
||||
id: "SL-001",
|
||||
milestoneId: "MS-001",
|
||||
title: "Test Slice",
|
||||
status: "active",
|
||||
planState: "not_started",
|
||||
orderIndex: 0,
|
||||
createdAt: now(),
|
||||
updatedAt: now(),
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function createMockFeature(overrides: Partial<MissionFeature> = {}): MissionFeature {
|
||||
return {
|
||||
id: "F-001",
|
||||
sliceId: "SL-001",
|
||||
title: "Test Feature",
|
||||
status: "defined",
|
||||
loopState: "idle",
|
||||
implementationAttemptCount: 0,
|
||||
validatorAttemptCount: 0,
|
||||
createdAt: now(),
|
||||
updatedAt: now(),
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function createMockMissionStore() {
|
||||
const missions = new Map<string, Mission>();
|
||||
const features = new Map<string, MissionFeature>();
|
||||
const assertionsByFeature = new Map<string, AssertionRow[]>();
|
||||
const validatorRuns = new Map<string, MissionValidatorRun>();
|
||||
let runSeq = 0;
|
||||
|
||||
const store = {
|
||||
getMission: vi.fn((id: string) => missions.get(id)),
|
||||
logMissionEvent: vi.fn(),
|
||||
getFeature: vi.fn((id: string) => features.get(id)),
|
||||
getFeatureByTaskId: vi.fn((taskId: string) => {
|
||||
for (const f of features.values()) if (f.taskId === taskId) return f;
|
||||
return undefined;
|
||||
}),
|
||||
updateFeatureStatus: vi.fn((id: string, status: MissionFeature["status"]) => {
|
||||
const f = features.get(id)!;
|
||||
const updated = { ...f, status, updatedAt: now() };
|
||||
features.set(id, updated);
|
||||
return updated;
|
||||
}),
|
||||
listAssertionsForFeature: vi.fn((featureId: string) => assertionsByFeature.get(featureId) ?? []),
|
||||
ensureFeatureAssertionLinked: vi.fn((featureId: string) => assertionsByFeature.get(featureId) ?? []),
|
||||
getSlice: vi.fn((id: string) => createMockSlice({ id })),
|
||||
getMilestone: vi.fn((id: string) => createMockMilestone({ id })),
|
||||
startValidatorRun: vi.fn((featureId: string) => {
|
||||
const run: MissionValidatorRun = {
|
||||
id: `VR-${++runSeq}`,
|
||||
featureId,
|
||||
milestoneId: "MS-001",
|
||||
sliceId: "SL-001",
|
||||
status: "running",
|
||||
triggerType: "task_completion",
|
||||
implementationAttempt: 1,
|
||||
validatorAttempt: 1,
|
||||
startedAt: now(),
|
||||
createdAt: now(),
|
||||
updatedAt: now(),
|
||||
};
|
||||
validatorRuns.set(run.id, run);
|
||||
return run;
|
||||
}),
|
||||
getValidatorRun: vi.fn((id: string) => validatorRuns.get(id)),
|
||||
completeValidatorRun: vi.fn((id: string, status: MissionValidatorRun["status"], summary?: string) => {
|
||||
const run = validatorRuns.get(id)!;
|
||||
const updated = { ...run, status, summary, completedAt: now(), updatedAt: now() };
|
||||
validatorRuns.set(id, updated);
|
||||
const feature = features.get(run.featureId);
|
||||
if (feature) {
|
||||
const loopState = status === "passed" ? "passed" : status === "failed" ? "needs_fix" : status === "blocked" ? "blocked" : "validating";
|
||||
features.set(run.featureId, { ...feature, loopState: loopState as any, lastValidatorStatus: status, updatedAt: now() });
|
||||
}
|
||||
return updated;
|
||||
}),
|
||||
recordValidatorFailures: vi.fn(() => []),
|
||||
createGeneratedFixFeature: vi.fn((sourceFeatureId: string, runId: string) => {
|
||||
const src = features.get(sourceFeatureId)!;
|
||||
const fix = createMockFeature({ id: `FIX-${sourceFeatureId}`, taskId: `TASK-FIX-${sourceFeatureId}`, generatedFromFeatureId: sourceFeatureId, generatedFromRunId: runId, loopState: "implementing" });
|
||||
features.set(fix.id, fix);
|
||||
features.set(sourceFeatureId, { ...src, loopState: "implementing", implementationAttemptCount: (src.implementationAttemptCount ?? 0) + 1, updatedAt: now() });
|
||||
return fix;
|
||||
}),
|
||||
triageFeature: vi.fn(async (featureId: string) => {
|
||||
const f = features.get(featureId)!;
|
||||
const updated = { ...f, status: "triaged" as const, updatedAt: now() };
|
||||
features.set(featureId, updated);
|
||||
return updated;
|
||||
}),
|
||||
on: vi.fn(),
|
||||
off: vi.fn(),
|
||||
emit: vi.fn(),
|
||||
_setMission: (m: Mission) => missions.set(m.id, m),
|
||||
_setFeature: (f: MissionFeature) => features.set(f.id, f),
|
||||
_setAssertions: (featureId: string, rows: AssertionRow[]) => assertionsByFeature.set(featureId, rows),
|
||||
};
|
||||
return store;
|
||||
}
|
||||
|
||||
function createMockTaskStore() {
|
||||
const tasks = new Map<string, any>();
|
||||
return {
|
||||
getTask: vi.fn(async (id: string) => tasks.get(id)),
|
||||
moveTask: vi.fn(async () => {}),
|
||||
updateTask: vi.fn(async () => {}),
|
||||
getSettings: vi.fn().mockResolvedValue({ missionStaleThresholdMs: 600_000, missionMaxTaskRetries: 3 }),
|
||||
recordRunAuditEvent: vi.fn(),
|
||||
on: vi.fn(),
|
||||
off: vi.fn(),
|
||||
_setTask: (t: any) => tasks.set(t.id, t),
|
||||
};
|
||||
}
|
||||
|
||||
function assertionRow(overrides: Partial<AssertionRow> & { id: string }): AssertionRow {
|
||||
return {
|
||||
milestoneId: "MS-001",
|
||||
title: overrides.id,
|
||||
assertion: `do ${overrides.id}`,
|
||||
status: "pending",
|
||||
orderIndex: 0,
|
||||
createdAt: now(),
|
||||
updatedAt: now(),
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe("Validator behavioral posture (U2 + U3)", () => {
|
||||
let missionStore: ReturnType<typeof createMockMissionStore>;
|
||||
let taskStore: ReturnType<typeof createMockTaskStore>;
|
||||
let loop: MissionExecutionLoop;
|
||||
|
||||
beforeEach(() => {
|
||||
missionStore = createMockMissionStore();
|
||||
taskStore = createMockTaskStore();
|
||||
vi.mocked(createResolvedAgentSession).mockReset();
|
||||
vi.mocked(createResolvedAgentSession).mockResolvedValue({
|
||||
session: mockSessionHolder.session as any,
|
||||
sessionFile: undefined,
|
||||
runtimeId: "test-runtime",
|
||||
wasConfigured: true,
|
||||
});
|
||||
missionStore._setMission(createMockMission());
|
||||
mockSessionHolder.session.state.messages = [];
|
||||
mockSessionHolder.session.dispose = vi.fn();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
loop?.stop();
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
function judgePass(assertionIds: string[]) {
|
||||
mockSessionHolder.session.state.messages = [
|
||||
{
|
||||
role: "assistant",
|
||||
content: JSON.stringify({
|
||||
status: "pass",
|
||||
assertions: assertionIds.map((id) => ({ assertionId: id, passed: true })),
|
||||
summary: "all good",
|
||||
}),
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
it("AE2: behavioral assertion the judge calls pass → fails with no verification capability", async () => {
|
||||
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-B", status: "in-progress" });
|
||||
missionStore._setFeature(feature);
|
||||
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]);
|
||||
taskStore._setTask({ id: "FN-B", title: "behavioral", log: [] });
|
||||
judgePass(["CA-1"]);
|
||||
|
||||
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp" });
|
||||
loop.start();
|
||||
await loop.processTaskOutcome("FN-B");
|
||||
|
||||
// No verification capability → behavioral default-to-fail → fix flow.
|
||||
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "failed", expect.any(String));
|
||||
expect(missionStore.createGeneratedFixFeature).toHaveBeenCalled();
|
||||
expect(missionStore.getFeature("F-001")?.status).not.toBe("done");
|
||||
});
|
||||
|
||||
it("AE3: static assertion the judge calls pass → passes, no verification invoked", async () => {
|
||||
const verify = vi.fn();
|
||||
const capability: VerificationCapability = { verifyBehavioralAssertion: verify };
|
||||
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-S", status: "in-progress" });
|
||||
missionStore._setFeature(feature);
|
||||
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "static" })]);
|
||||
taskStore._setTask({ id: "FN-S", title: "static", log: [] });
|
||||
judgePass(["CA-1"]);
|
||||
|
||||
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: capability });
|
||||
loop.start();
|
||||
await loop.processTaskOutcome("FN-S");
|
||||
|
||||
expect(verify).not.toHaveBeenCalled();
|
||||
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String));
|
||||
expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done");
|
||||
});
|
||||
|
||||
it("untyped assertions default to static — legacy judge pass path is preserved", async () => {
|
||||
const verify = vi.fn();
|
||||
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-U", status: "in-progress" });
|
||||
missionStore._setFeature(feature);
|
||||
// No `type` field → normalizes to static.
|
||||
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1" })]);
|
||||
taskStore._setTask({ id: "FN-U", title: "untyped", log: [] });
|
||||
judgePass(["CA-1"]);
|
||||
|
||||
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
|
||||
loop.start();
|
||||
await loop.processTaskOutcome("FN-U");
|
||||
|
||||
expect(verify).not.toHaveBeenCalled();
|
||||
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String));
|
||||
});
|
||||
|
||||
it("behavioral assertion confirmed by an injected verification capability → passes", async () => {
|
||||
const verify = vi.fn(async (req): Promise<VerificationOutcome> => ({ verdict: "pass", assertionId: req.assertionId, reason: "confirmed" }));
|
||||
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-BV", status: "in-progress" });
|
||||
missionStore._setFeature(feature);
|
||||
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]);
|
||||
taskStore._setTask({ id: "FN-BV", title: "behavioral verified", integrationSha: "sha123", log: [] });
|
||||
judgePass(["CA-1"]);
|
||||
|
||||
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
|
||||
loop.start();
|
||||
await loop.processTaskOutcome("FN-BV");
|
||||
|
||||
expect(verify).toHaveBeenCalledTimes(1);
|
||||
expect(verify.mock.calls[0][0]).toMatchObject({ assertionId: "CA-1", integrationSha: "sha123" });
|
||||
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String));
|
||||
expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done");
|
||||
});
|
||||
|
||||
it("behavioral assertion verification inconclusive → blocked, NO fix feature", async () => {
|
||||
const verify = vi.fn(async (req): Promise<VerificationOutcome> => ({ verdict: "inconclusive", assertionId: req.assertionId, reason: "no isolating sandbox backend" }));
|
||||
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-INC", status: "in-progress" });
|
||||
missionStore._setFeature(feature);
|
||||
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]);
|
||||
taskStore._setTask({ id: "FN-INC", title: "inconclusive", integrationSha: "sha123", log: [] });
|
||||
judgePass(["CA-1"]);
|
||||
|
||||
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
|
||||
loop.start();
|
||||
await loop.processTaskOutcome("FN-INC");
|
||||
|
||||
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "blocked", expect.any(String));
|
||||
expect(missionStore.createGeneratedFixFeature).not.toHaveBeenCalled();
|
||||
expect(missionStore.getFeature("F-001")?.status).not.toBe("done");
|
||||
});
|
||||
|
||||
it("mixed set: static passes via judge, behavioral confirmed via verification → overall pass", async () => {
|
||||
const verify = vi.fn(async (req): Promise<VerificationOutcome> => ({ verdict: "pass", assertionId: req.assertionId, reason: "confirmed" }));
|
||||
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-MIX", status: "in-progress" });
|
||||
missionStore._setFeature(feature);
|
||||
missionStore._setAssertions("F-001", [
|
||||
assertionRow({ id: "CA-static", type: "static" }),
|
||||
assertionRow({ id: "CA-behav", type: "behavioral" }),
|
||||
]);
|
||||
taskStore._setTask({ id: "FN-MIX", title: "mixed", integrationSha: "sha123", log: [] });
|
||||
judgePass(["CA-static", "CA-behav"]);
|
||||
|
||||
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
|
||||
loop.start();
|
||||
await loop.processTaskOutcome("FN-MIX");
|
||||
|
||||
// Only the behavioral assertion is verified.
|
||||
expect(verify).toHaveBeenCalledTimes(1);
|
||||
expect(verify.mock.calls[0][0]).toMatchObject({ assertionId: "CA-behav" });
|
||||
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String));
|
||||
expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done");
|
||||
});
|
||||
|
||||
it("mixed set: behavioral observed wrong → overall fail even though static passes", async () => {
|
||||
const verify = vi.fn(async (req): Promise<VerificationOutcome> => ({ verdict: "fail", assertionId: req.assertionId, reason: "defect still reproduces" }));
|
||||
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-MIX2", status: "in-progress" });
|
||||
missionStore._setFeature(feature);
|
||||
missionStore._setAssertions("F-001", [
|
||||
assertionRow({ id: "CA-static", type: "static" }),
|
||||
assertionRow({ id: "CA-behav", type: "behavioral" }),
|
||||
]);
|
||||
taskStore._setTask({ id: "FN-MIX2", title: "mixed fail", integrationSha: "sha123", log: [] });
|
||||
judgePass(["CA-static", "CA-behav"]);
|
||||
|
||||
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
|
||||
loop.start();
|
||||
await loop.processTaskOutcome("FN-MIX2");
|
||||
|
||||
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "failed", expect.any(String));
|
||||
expect(missionStore.createGeneratedFixFeature).toHaveBeenCalled();
|
||||
expect(missionStore.getFeature("F-001")?.status).not.toBe("done");
|
||||
});
|
||||
});
|
||||
@@ -26,7 +26,9 @@ import {
|
||||
TEST_MODE_RESOLVED,
|
||||
isTestModeActive,
|
||||
resolveTaskValidatorModel,
|
||||
normalizeMissionAssertionType,
|
||||
} from "@fusion/core";
|
||||
import type { VerificationOutcome } from "./mission-verification.js";
|
||||
import { createFnAgent, promptWithFallback, type AgentResult } from "./pi.js";
|
||||
import { mergeEffectiveSettings } from "./effective-settings.js";
|
||||
import {
|
||||
@@ -50,8 +52,16 @@ const VALIDATION_TIMEOUT_MS = 10 * 60 * 1000; // 10 minutes
|
||||
* per assertion plus an overall status.
|
||||
*/
|
||||
export interface ValidationResult {
|
||||
/** Overall validation status */
|
||||
status: "pass" | "fail" | "blocked" | "error";
|
||||
/**
|
||||
* Overall validation status.
|
||||
*
|
||||
* `inconclusive` is first-class and distinct from `fail`: it means a
|
||||
* behavioral verification run could not run or conclude (no isolating sandbox
|
||||
* backend, timeout, setup failure, rejected proof). In this unit it routes to
|
||||
* a blocked verdict (no remediation); later units track its infra-failure rate
|
||||
* separately.
|
||||
*/
|
||||
status: "pass" | "fail" | "blocked" | "error" | "inconclusive";
|
||||
/** Per-assertion results */
|
||||
assertions: Array<{
|
||||
assertionId: string;
|
||||
@@ -83,6 +93,14 @@ export interface MissionExecutionLoopOptions {
|
||||
pluginRunner?: import("./plugin-runner.js").PluginRunner;
|
||||
/** Optional agent store for resolving assigned-agent runtime hints. */
|
||||
agentStore?: AgentStore;
|
||||
/**
|
||||
* Optional behavioral-verification capability (U3). When provided, behavioral
|
||||
* assertions are confirmed by a non-mutating verification run; the judge's
|
||||
* "pass" on a behavioral assertion is advisory only. When ABSENT, behavioral
|
||||
* assertions still default to fail (U2) but no verification run is attempted —
|
||||
* preserving the behavior of existing construction sites that inject nothing.
|
||||
*/
|
||||
verificationCapability?: import("./mission-verification.js").VerificationCapability;
|
||||
}
|
||||
|
||||
export class MissionExecutionLoop extends EventEmitter {
|
||||
@@ -94,6 +112,7 @@ export class MissionExecutionLoop extends EventEmitter {
|
||||
private missionAutopilot?: MissionExecutionLoopOptions["missionAutopilot"];
|
||||
private pluginRunner?: MissionExecutionLoopOptions["pluginRunner"];
|
||||
private agentStore?: MissionExecutionLoopOptions["agentStore"];
|
||||
private verificationCapability?: MissionExecutionLoopOptions["verificationCapability"];
|
||||
private activeValidations = new Set<string>(); // feature IDs currently being validated
|
||||
|
||||
constructor(options: MissionExecutionLoopOptions) {
|
||||
@@ -105,6 +124,7 @@ export class MissionExecutionLoop extends EventEmitter {
|
||||
this.missionAutopilot = options.missionAutopilot;
|
||||
this.pluginRunner = options.pluginRunner;
|
||||
this.agentStore = options.agentStore;
|
||||
this.verificationCapability = options.verificationCapability;
|
||||
loopLog.log("MissionExecutionLoop created");
|
||||
}
|
||||
|
||||
@@ -434,8 +454,11 @@ export class MissionExecutionLoop extends EventEmitter {
|
||||
await this.handleValidationPass(feature.id, run.id, result.summary);
|
||||
} else if (result.status === "fail") {
|
||||
await this.handleValidationFail(feature.id, run.id, result);
|
||||
} else if (result.status === "blocked") {
|
||||
await this.handleValidationBlocked(feature.id, run.id, result.blockedReason);
|
||||
} else if (result.status === "blocked" || result.status === "inconclusive") {
|
||||
// Inconclusive (infra: no isolating backend, timeout, rejected proof)
|
||||
// routes to blocked/needs-attention — distinct from a behavioral fail,
|
||||
// and spawns no Fix Feature (per KTD / R21, fully realized in a later unit).
|
||||
await this.handleValidationBlocked(feature.id, run.id, result.blockedReason ?? result.summary);
|
||||
} else if (result.status === "error") {
|
||||
await this.handleValidationError(feature.id, run.id, result.summary);
|
||||
}
|
||||
@@ -537,7 +560,12 @@ export class MissionExecutionLoop extends EventEmitter {
|
||||
|
||||
// Get the validation result from the session
|
||||
// The agent should have returned structured JSON in its response
|
||||
const result = await this.parseValidationResult(session.session, assertions);
|
||||
const judgeResult = await this.parseValidationResult(session.session, assertions);
|
||||
|
||||
// U2/U3: the read-only judge's verdict is authoritative for STATIC
|
||||
// assertions only. BEHAVIORAL assertions default to fail and are confirmed
|
||||
// (or refuted) by a non-mutating verification run instead.
|
||||
const result = await this.applyBehavioralPosture(feature, assertions, judgeResult);
|
||||
|
||||
loopLog.log(`Validation completed for feature ${feature.id}: ${result.status}`);
|
||||
return result;
|
||||
@@ -568,6 +596,158 @@ export class MissionExecutionLoop extends EventEmitter {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Apply the behavioral judging posture (U2/U3) to the read-only judge's
|
||||
* verdict.
|
||||
*
|
||||
* - STATIC assertions keep the judge's verdict verbatim (no behavior change).
|
||||
* - BEHAVIORAL assertions DEFAULT TO FAIL. The judge's "pass" on a behavioral
|
||||
* assertion is advisory; an authoritative pass requires a verification run
|
||||
* to confirm it. When a verification capability is injected, each behavioral
|
||||
* assertion is run through it: pass → satisfied; fail → behavioral failure;
|
||||
* inconclusive → the aggregate becomes inconclusive (infra, no remediation).
|
||||
* When NO capability is injected, behavioral assertions simply stay failed
|
||||
* (preserving existing call-site behavior — existing data is all static).
|
||||
*
|
||||
* The aggregate status is recomputed from the post-posture per-assertion
|
||||
* results so the existing pass/fail/blocked/error/inconclusive flow is driven
|
||||
* correctly.
|
||||
*/
|
||||
private async applyBehavioralPosture(
|
||||
feature: MissionFeature,
|
||||
assertions: MissionContractAssertion[],
|
||||
judgeResult: ValidationResult,
|
||||
): Promise<ValidationResult> {
|
||||
// Preserve non-behavioral terminal verdicts untouched (error/blocked from the
|
||||
// judge are not behavioral posture concerns).
|
||||
if (judgeResult.status === "error") {
|
||||
return judgeResult;
|
||||
}
|
||||
|
||||
const typeById = new Map<string, ReturnType<typeof normalizeMissionAssertionType>>();
|
||||
let hasBehavioral = false;
|
||||
for (const a of assertions) {
|
||||
const t = normalizeMissionAssertionType(a.type);
|
||||
typeById.set(a.id, t);
|
||||
if (t === "behavioral") hasBehavioral = true;
|
||||
}
|
||||
|
||||
// Fast path: no behavioral assertions → existing static path is preserved
|
||||
// exactly. This keeps every existing (untyped/static) test green.
|
||||
if (!hasBehavioral) {
|
||||
return judgeResult;
|
||||
}
|
||||
|
||||
const textById = new Map(assertions.map((a) => [a.id, a.assertion]));
|
||||
let sawInconclusive = false;
|
||||
let inconclusiveReason: string | undefined;
|
||||
|
||||
const newAssertionResults = await Promise.all(
|
||||
judgeResult.assertions.map(async (judged) => {
|
||||
const type = typeById.get(judged.assertionId) ?? "static";
|
||||
if (type !== "behavioral") {
|
||||
// Static: keep judge verdict verbatim.
|
||||
return judged;
|
||||
}
|
||||
|
||||
// Behavioral: default to fail unless verification confirms it.
|
||||
if (!this.verificationCapability) {
|
||||
return {
|
||||
...judged,
|
||||
passed: false,
|
||||
message: "Behavioral assertion defaults to fail: no verification evidence (advisory judge verdict is not authoritative).",
|
||||
expected: judged.expected ?? "Behavior confirmed by a verification run",
|
||||
actual: judged.actual ?? "No verification run was performed",
|
||||
};
|
||||
}
|
||||
|
||||
let outcome: VerificationOutcome;
|
||||
try {
|
||||
outcome = await this.verificationCapability.verifyBehavioralAssertion({
|
||||
assertionId: judged.assertionId,
|
||||
assertion: textById.get(judged.assertionId) ?? "",
|
||||
taskId: feature.taskId,
|
||||
integrationSha: await this.resolveIntegrationSha(feature),
|
||||
signal: undefined,
|
||||
});
|
||||
} catch (err) {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
loopLog.warn(`Verification capability threw for assertion ${judged.assertionId}: ${message}`);
|
||||
outcome = { verdict: "inconclusive", assertionId: judged.assertionId, reason: `verification error: ${message}` };
|
||||
}
|
||||
|
||||
if (outcome.verdict === "pass") {
|
||||
return { ...judged, passed: true, message: outcome.reason };
|
||||
}
|
||||
if (outcome.verdict === "inconclusive") {
|
||||
sawInconclusive = true;
|
||||
inconclusiveReason = inconclusiveReason ?? outcome.reason;
|
||||
return {
|
||||
...judged,
|
||||
passed: false,
|
||||
message: `Behavioral verification inconclusive: ${outcome.reason}`,
|
||||
expected: judged.expected ?? "Behavior confirmed by a verification run",
|
||||
actual: outcome.detail ?? "Verification could not conclude",
|
||||
};
|
||||
}
|
||||
// fail
|
||||
return {
|
||||
...judged,
|
||||
passed: false,
|
||||
message: outcome.reason,
|
||||
expected: judged.expected ?? "Behavior confirmed by a verification run",
|
||||
actual: outcome.detail ?? judged.actual ?? "Behavior not confirmed",
|
||||
};
|
||||
}),
|
||||
);
|
||||
|
||||
const allPassed = newAssertionResults.every((a) => a.passed);
|
||||
|
||||
// Inconclusive takes precedence over fail: an infra-driven non-pass must not
|
||||
// be mistaken for an observed behavioral failure (no Fix Feature).
|
||||
let status: ValidationResult["status"];
|
||||
if (sawInconclusive && !allPassed) {
|
||||
status = "inconclusive";
|
||||
} else if (allPassed) {
|
||||
status = "pass";
|
||||
} else {
|
||||
status = "fail";
|
||||
}
|
||||
|
||||
const summary = status === "pass"
|
||||
? judgeResult.summary
|
||||
: status === "inconclusive"
|
||||
? `Behavioral verification inconclusive: ${inconclusiveReason ?? "verification could not conclude"}`
|
||||
: "One or more behavioral assertions were not confirmed by verification.";
|
||||
|
||||
return {
|
||||
status,
|
||||
assertions: newAssertionResults,
|
||||
summary,
|
||||
blockedReason: status === "inconclusive" ? (inconclusiveReason ?? "verification inconclusive") : judgeResult.blockedReason,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the trusted revision (integration SHA) whose disposable checkout the
|
||||
* verification run executes against. The live task worktree is pruned before
|
||||
* the done-transition that triggers validation, so it cannot be used.
|
||||
*
|
||||
* In this unit we read it from the linked task when available; callers that do
|
||||
* not supply a resolvable SHA cause the verification run to resolve to
|
||||
* inconclusive (fail-closed). A richer derivation is owned by a later unit.
|
||||
*/
|
||||
private async resolveIntegrationSha(feature: MissionFeature): Promise<string | undefined> {
|
||||
if (!feature.taskId) return undefined;
|
||||
try {
|
||||
const task = await this.taskStore.getTask(feature.taskId);
|
||||
const candidate = (task as { integrationSha?: string; baseCommit?: string } | undefined);
|
||||
return candidate?.integrationSha ?? candidate?.baseCommit ?? undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
private resolveValidationSessionModel(
|
||||
task: Awaited<ReturnType<TaskStore["getTask"]>> | null,
|
||||
settings: Partial<Settings> | undefined,
|
||||
|
||||
515
packages/engine/src/mission-verification.ts
Normal file
515
packages/engine/src/mission-verification.ts
Normal file
@@ -0,0 +1,515 @@
|
||||
/**
|
||||
* Mission behavioral-verification capability (U3).
|
||||
*
|
||||
* The Validator Run's read-only AI judge cannot run code, so its "pass" on a
|
||||
* *behavioral* assertion is advisory only (U2). This module supplies the
|
||||
* authoritative, NON-MUTATING verification step that confirms a behavioral/bug
|
||||
* assertion by exercising the implemented code.
|
||||
*
|
||||
* Channels:
|
||||
* - **test-execution** (this unit): run the project's scoped test suite / an
|
||||
* agent-supplied regression test against a disposable checkout at a trusted
|
||||
* revision, through an explicit isolating sandbox backend.
|
||||
* - **app-driving** (later unit U5/U8): drive a running app instance. Not
|
||||
* implemented here — the capability surface is structured so it can be added
|
||||
* without reshaping callers.
|
||||
*
|
||||
* Safety invariants enforced here (the boundary, not a convention):
|
||||
* - R18: execute under an *isolating* sandbox backend (bubblewrap / sandbox-exec)
|
||||
* with a scrubbed env allowlist; FAIL CLOSED to a non-pass when no isolating
|
||||
* backend is available — never fall through to the unrestricted native backend.
|
||||
* - R19: the command is built from a fixed, system-owned template into which only
|
||||
* a validated test-file path is substituted; shell metacharacters are rejected.
|
||||
* - R11/R17: verification runs against a disposable checkout at the integration
|
||||
* SHA (never the pruned live worktree, never the repo root); the source tree
|
||||
* that feeds diff/merge is asserted git-clean after a run.
|
||||
* - R5/AE5: agent-supplied proof must FAIL on a second disposable checkout at
|
||||
* `git merge-base` (a revision the agent does not control) and PASS on the
|
||||
* implementation; a test that passes on both is rejected.
|
||||
* - R9: inconclusive / timeout / setup failure resolves to a non-pass.
|
||||
* - R10: no board / mission writes happen here.
|
||||
*/
|
||||
|
||||
import { promises as fs } from "node:fs";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import { exec } from "node:child_process";
|
||||
import { promisify } from "node:util";
|
||||
import type { TaskStore } from "@fusion/core";
|
||||
import type { SandboxCapabilities } from "./sandbox/index.js";
|
||||
import { __setSandboxBackendForTests, resolveSandboxBackend } from "./sandbox/index.js";
|
||||
import type { SandboxBackend } from "./sandbox/index.js";
|
||||
import { detectBwrap } from "./sandbox/bubblewrap-detect.js";
|
||||
import { detectSandboxExec } from "./sandbox/sandbox-exec-detect.js";
|
||||
import { runVerificationCommand } from "./verification-utils.js";
|
||||
import { createLogger } from "./logger.js";
|
||||
|
||||
const execAsync = promisify(exec);
|
||||
const verifyLog = createLogger("mission-verify");
|
||||
|
||||
// ── Verdict types ───────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Outcome of a verification run for a single behavioral assertion.
|
||||
*
|
||||
* - `pass`: behavior confirmed by execution.
|
||||
* - `fail`: behavior observed wrong (the defect still reproduces / proof rejected).
|
||||
* - `inconclusive`: verification could not run or conclude (no isolating backend,
|
||||
* timeout, setup failure, rejected/invalid proof input). First-class and
|
||||
* distinct from `fail`: it must NOT spawn remediation (handled by later units),
|
||||
* but in this unit it never resolves to a default pass either.
|
||||
*/
|
||||
export type VerificationVerdict = "pass" | "fail" | "inconclusive";
|
||||
|
||||
/** Why a verification run reached its verdict (for durable observability later). */
|
||||
export interface VerificationOutcome {
|
||||
verdict: VerificationVerdict;
|
||||
/** Human-readable reason, suitable for surfacing in a failure record. */
|
||||
reason: string;
|
||||
/** The assertion this outcome corresponds to. */
|
||||
assertionId: string;
|
||||
/** Optional summarized command output for the failure record. */
|
||||
detail?: string;
|
||||
}
|
||||
|
||||
/** Shape of agent-supplied executable proof (a regression test). */
|
||||
export interface VerificationProof {
|
||||
/**
|
||||
* Path to the regression test file, relative to the checkout root. Validated
|
||||
* to reject shell metacharacters and path escapes before use (R19).
|
||||
*/
|
||||
testFilePath: string;
|
||||
}
|
||||
|
||||
/** Input describing a single behavioral assertion to verify. */
|
||||
export interface VerificationRequest {
|
||||
assertionId: string;
|
||||
/** The assertion text (for logging / reason building). */
|
||||
assertion: string;
|
||||
/** Board task id associated with the feature, used for verification-command logging. */
|
||||
taskId?: string;
|
||||
/**
|
||||
* The trusted revision (integration SHA) whose checkout the implementation is
|
||||
* verified against. When absent, verification is inconclusive (cannot
|
||||
* materialize a trusted checkout).
|
||||
*/
|
||||
integrationSha?: string;
|
||||
/**
|
||||
* The `git merge-base` revision (feature branch vs base branch) used as the
|
||||
* pre-fix baseline for agent-supplied proof. Not agent-controlled.
|
||||
*/
|
||||
mergeBaseSha?: string;
|
||||
/** Optional agent-supplied executable proof. */
|
||||
proof?: VerificationProof;
|
||||
/** Abort signal to bound the run. */
|
||||
signal?: AbortSignal;
|
||||
}
|
||||
|
||||
/**
|
||||
* Injected verification capability. Mirrors the `createFnAgent` injection
|
||||
* pattern so MissionExecutionLoop can swap a real implementation for a mock in
|
||||
* tests. Optional on the loop: when absent, behavioral assertions resolve to a
|
||||
* non-pass without invoking any execution (preserving existing behavior for
|
||||
* call sites that do not inject a capability).
|
||||
*/
|
||||
export interface VerificationCapability {
|
||||
verifyBehavioralAssertion(request: VerificationRequest): Promise<VerificationOutcome>;
|
||||
}
|
||||
|
||||
// ── Command-template safety (R19) ─────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Characters that could break out of the fixed command template or inject
|
||||
* additional shell behavior. Agent-supplied test paths containing any of these
|
||||
* are rejected before execution.
|
||||
*/
|
||||
const SHELL_METACHARACTERS = /[;&|`$(){}<>!*?\[\]\\"'\n\r\t\0]/;
|
||||
|
||||
/**
|
||||
* Validate an agent-supplied test-file path. Returns the normalized path when
|
||||
* safe, or `null` when it must be rejected (R19).
|
||||
*
|
||||
* Rejects: empty, absolute paths, parent-dir escapes, shell metacharacters,
|
||||
* and leading dashes (which could be read as command flags).
|
||||
*/
|
||||
export function validateTestPath(rawPath: unknown): string | null {
|
||||
if (typeof rawPath !== "string") return null;
|
||||
const path = rawPath.trim();
|
||||
if (path.length === 0) return null;
|
||||
if (SHELL_METACHARACTERS.test(path)) return null;
|
||||
if (path.startsWith("/")) return null; // must be relative to the checkout
|
||||
if (path.startsWith("-")) return null; // could be parsed as a flag
|
||||
// Reject parent-dir escapes (any `..` segment).
|
||||
const segments = path.split("/");
|
||||
if (segments.some((seg) => seg === "..")) return null;
|
||||
return path;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the verification command from the fixed system-owned template. Only a
|
||||
* pre-validated test path may be substituted (R19). Callers MUST pass a path
|
||||
* already run through {@link validateTestPath}; this function re-checks and
|
||||
* throws on violation as a defense-in-depth guard.
|
||||
*/
|
||||
export function buildVerificationCommand(template: string, validatedTestPath?: string): string {
|
||||
if (validatedTestPath !== undefined) {
|
||||
if (validateTestPath(validatedTestPath) === null) {
|
||||
throw new Error(`Refusing to build verification command: invalid test path ${JSON.stringify(validatedTestPath)}`);
|
||||
}
|
||||
if (!template.includes("{testPath}")) {
|
||||
throw new Error("Verification command template must contain a {testPath} placeholder when a test path is supplied");
|
||||
}
|
||||
return template.replace("{testPath}", validatedTestPath);
|
||||
}
|
||||
// Whole-suite invocation: the template must not reference a test path.
|
||||
return template.replace("{testPath}", "").trimEnd();
|
||||
}
|
||||
|
||||
// ── Isolating-backend selection (R18, fail-closed) ────────────────────────────
|
||||
|
||||
/**
|
||||
* Result of selecting an isolating sandbox backend for verification.
|
||||
*/
|
||||
export interface IsolatingBackendSelection {
|
||||
/** The backend id to request from `resolveSandboxBackend`, or null if none. */
|
||||
backendId: SandboxCapabilities["id"] | null;
|
||||
/** Why no isolating backend is available (when backendId is null). */
|
||||
reason?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Describes the detected availability of isolating backends on this host.
|
||||
* Injectable for tests so we don't shell out to detect bwrap/sandbox-exec.
|
||||
*/
|
||||
export interface IsolatingBackendProbe {
|
||||
platform: NodeJS.Platform;
|
||||
bubblewrapAvailable: boolean;
|
||||
sandboxExecAvailable: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Choose an isolating backend, failing closed. Returns `backendId: null` (a
|
||||
* non-pass signal) when no isolating backend is available — verification must
|
||||
* NEVER fall through to the unrestricted native backend (R18).
|
||||
*/
|
||||
export function selectIsolatingBackend(probe: IsolatingBackendProbe): IsolatingBackendSelection {
|
||||
if (probe.platform === "linux" && probe.bubblewrapAvailable) {
|
||||
return { backendId: "bubblewrap" };
|
||||
}
|
||||
if (probe.platform === "darwin" && probe.sandboxExecAvailable) {
|
||||
return { backendId: "sandbox-exec" };
|
||||
}
|
||||
return {
|
||||
backendId: null,
|
||||
reason: `no isolating sandbox backend available (platform=${probe.platform}, bwrap=${probe.bubblewrapAvailable}, sandbox-exec=${probe.sandboxExecAvailable})`,
|
||||
};
|
||||
}
|
||||
|
||||
// ── Environment scrubbing (R18) ───────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Environment variables permitted into the verification child process. Anything
|
||||
* not on the allowlist (API keys, auth tokens, DB credentials, agent logs) is
|
||||
* dropped so agent-authored code executes with a minimal environment.
|
||||
*/
|
||||
export const VERIFICATION_ENV_ALLOWLIST = [
|
||||
"PATH",
|
||||
"HOME",
|
||||
"SHELL",
|
||||
"LANG",
|
||||
"LC_ALL",
|
||||
"TMPDIR",
|
||||
"TERM",
|
||||
"NODE_ENV",
|
||||
// pnpm / corepack need these to resolve the package manager in the checkout.
|
||||
"PNPM_HOME",
|
||||
"COREPACK_HOME",
|
||||
"npm_config_registry",
|
||||
] as const;
|
||||
|
||||
/**
|
||||
* Produce a scrubbed environment containing only allowlisted keys from the
|
||||
* source environment, with `CI=1` forced for deterministic test runs.
|
||||
*/
|
||||
export function scrubEnv(source: NodeJS.ProcessEnv = process.env): NodeJS.ProcessEnv {
|
||||
const scrubbed: NodeJS.ProcessEnv = {};
|
||||
for (const key of VERIFICATION_ENV_ALLOWLIST) {
|
||||
const value = source[key];
|
||||
if (value !== undefined) {
|
||||
scrubbed[key] = value;
|
||||
}
|
||||
}
|
||||
// Force deterministic, non-interactive execution.
|
||||
scrubbed.CI = "1";
|
||||
return scrubbed;
|
||||
}
|
||||
|
||||
// ── Disposable checkout materialization (R11/R17) ─────────────────────────────
|
||||
|
||||
/** A disposable checkout the verification run can execute against. */
|
||||
export interface DisposableCheckout {
|
||||
/** Absolute path to the checkout root (under a run-unique tmpdir). */
|
||||
dir: string;
|
||||
/** Tear the checkout down unconditionally (idempotent). */
|
||||
dispose(): Promise<void>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Materializes disposable checkouts at a trusted revision. Injectable so tests
|
||||
* can supply a fixture checkout without invoking git.
|
||||
*/
|
||||
export interface CheckoutMaterializer {
|
||||
/**
|
||||
* Create a disposable checkout of `rootDir` at `revision` under a run-unique
|
||||
* tmpdir. The implementation MUST NOT mutate the source tree at `rootDir`.
|
||||
*/
|
||||
materialize(rootDir: string, revision: string): Promise<DisposableCheckout>;
|
||||
/**
|
||||
* Assert that the source tree feeding diff/merge is git-clean (byte-identical)
|
||||
* — the R17 post-condition. Throws if dirty.
|
||||
*/
|
||||
assertSourceClean(rootDir: string): Promise<void>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Default git-backed materializer: `git worktree add --detach <tmp> <revision>`
|
||||
* produces an isolated checkout without touching the source working tree, and
|
||||
* `git status --porcelain` on the source confirms cleanliness afterwards.
|
||||
*/
|
||||
export class GitCheckoutMaterializer implements CheckoutMaterializer {
|
||||
async materialize(rootDir: string, revision: string): Promise<DisposableCheckout> {
|
||||
const dir = await fs.mkdtemp(path.join(os.tmpdir(), "fn-verify-"));
|
||||
// `git worktree add --detach` checks out the revision into a throwaway dir
|
||||
// without modifying the source working tree.
|
||||
await execAsync(`git worktree add --detach ${JSON.stringify(dir)} ${JSON.stringify(revision)}`, {
|
||||
cwd: rootDir,
|
||||
timeout: 60_000,
|
||||
maxBuffer: 8 * 1024 * 1024,
|
||||
});
|
||||
return {
|
||||
dir,
|
||||
dispose: async () => {
|
||||
try {
|
||||
await execAsync(`git worktree remove --force ${JSON.stringify(dir)}`, {
|
||||
cwd: rootDir,
|
||||
timeout: 30_000,
|
||||
});
|
||||
} catch (err) {
|
||||
verifyLog.warn(`Failed to remove verification worktree ${dir}:`, err);
|
||||
}
|
||||
await fs.rm(dir, { recursive: true, force: true }).catch(() => {});
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
async assertSourceClean(rootDir: string): Promise<void> {
|
||||
const { stdout } = await execAsync("git status --porcelain", {
|
||||
cwd: rootDir,
|
||||
timeout: 30_000,
|
||||
maxBuffer: 8 * 1024 * 1024,
|
||||
});
|
||||
if (stdout.trim().length > 0) {
|
||||
throw new Error(`Source tree is not git-clean after verification run:\n${stdout.trim()}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Probe the host for isolating-backend availability (cached by the detectors). */
|
||||
async function probeIsolatingBackends(): Promise<IsolatingBackendProbe> {
|
||||
const [bwrap, sandboxExec] = await Promise.all([
|
||||
detectBwrap().catch(() => ({ available: false })),
|
||||
detectSandboxExec().catch(() => ({ available: false })),
|
||||
]);
|
||||
return {
|
||||
platform: process.platform,
|
||||
bubblewrapAvailable: bwrap.available,
|
||||
sandboxExecAvailable: sandboxExec.available,
|
||||
};
|
||||
}
|
||||
|
||||
// ── Test-execution verification capability ────────────────────────────────────
|
||||
|
||||
export interface TestExecutionVerificationOptions {
|
||||
/** Task store, reused by runVerificationCommand for command logging. */
|
||||
store: TaskStore;
|
||||
/** Repo root whose source tree must remain git-clean. */
|
||||
rootDir: string;
|
||||
/**
|
||||
* Fixed, system-owned command template. Must contain `{testPath}` when an
|
||||
* agent-supplied proof path is used. Example: `pnpm vitest run {testPath}`.
|
||||
*/
|
||||
commandTemplate: string;
|
||||
/** Injectable checkout materializer (defaults to git-backed). */
|
||||
materializer?: CheckoutMaterializer;
|
||||
/** Injectable backend probe (defaults to host detection). */
|
||||
probeBackends?: () => Promise<IsolatingBackendProbe>;
|
||||
/**
|
||||
* Injectable factory for the isolating sandbox backend, given the selected
|
||||
* backend id. Defaults to `resolveSandboxBackend({ backendId })`. Injectable so
|
||||
* tests can supply a scripted backend without mutating global sandbox state.
|
||||
*/
|
||||
backendFactory?: (backendId: SandboxCapabilities["id"]) => SandboxBackend;
|
||||
/** Injectable env source (defaults to process.env). */
|
||||
envSource?: NodeJS.ProcessEnv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The test-execution channel of the verification run. Confirms a behavioral
|
||||
* assertion by running the suite / an agent-supplied regression test against a
|
||||
* disposable checkout at the integration SHA, under an isolating sandbox
|
||||
* backend with a scrubbed env. Fails closed to a non-pass on any setup failure.
|
||||
*
|
||||
* App-driving is NOT handled here; a later unit dispatches UI/bug assertions to
|
||||
* an app-driving channel. This class is the canonical pattern that channel will
|
||||
* mirror.
|
||||
*/
|
||||
export class TestExecutionVerificationCapability implements VerificationCapability {
|
||||
private readonly store: TaskStore;
|
||||
private readonly rootDir: string;
|
||||
private readonly commandTemplate: string;
|
||||
private readonly materializer: CheckoutMaterializer;
|
||||
private readonly probeBackends: () => Promise<IsolatingBackendProbe>;
|
||||
private readonly backendFactory: (backendId: SandboxCapabilities["id"]) => SandboxBackend;
|
||||
private readonly envSource: NodeJS.ProcessEnv;
|
||||
|
||||
constructor(options: TestExecutionVerificationOptions) {
|
||||
this.store = options.store;
|
||||
this.rootDir = options.rootDir;
|
||||
this.commandTemplate = options.commandTemplate;
|
||||
this.materializer = options.materializer ?? new GitCheckoutMaterializer();
|
||||
this.probeBackends = options.probeBackends ?? probeIsolatingBackends;
|
||||
this.backendFactory = options.backendFactory ?? ((backendId) => resolveSandboxBackend({ backendId }));
|
||||
this.envSource = options.envSource ?? process.env;
|
||||
}
|
||||
|
||||
async verifyBehavioralAssertion(request: VerificationRequest): Promise<VerificationOutcome> {
|
||||
const { assertionId } = request;
|
||||
|
||||
// R11: a trusted revision is required to materialize a disposable checkout.
|
||||
if (!request.integrationSha) {
|
||||
return this.inconclusive(assertionId, "no integration SHA available to materialize a trusted checkout");
|
||||
}
|
||||
|
||||
// R19: validate any agent-supplied proof path BEFORE doing any work.
|
||||
let validatedTestPath: string | undefined;
|
||||
if (request.proof) {
|
||||
const safe = validateTestPath(request.proof.testFilePath);
|
||||
if (safe === null) {
|
||||
return this.inconclusive(
|
||||
assertionId,
|
||||
`agent-supplied test path rejected (invalid or contains shell metacharacters): ${JSON.stringify(request.proof.testFilePath)}`,
|
||||
);
|
||||
}
|
||||
validatedTestPath = safe;
|
||||
}
|
||||
|
||||
// R18: select an isolating backend, fail closed when none is available.
|
||||
const probe = await this.probeBackends();
|
||||
const selection = selectIsolatingBackend(probe);
|
||||
if (selection.backendId === null) {
|
||||
return this.inconclusive(assertionId, selection.reason ?? "no isolating sandbox backend available");
|
||||
}
|
||||
|
||||
const command = buildVerificationCommand(this.commandTemplate, validatedTestPath);
|
||||
const scrubbedEnv = scrubEnv(this.envSource);
|
||||
const logTaskId = request.taskId ?? `verify-${assertionId}`;
|
||||
|
||||
// Route runVerificationCommand through the explicitly-selected isolating
|
||||
// backend rather than the no-arg native fallback (R18). runVerificationCommand
|
||||
// resolves its backend via the no-arg resolveSandboxBackend(), so we pin the
|
||||
// selected isolating backend via the test-override hook for the duration of
|
||||
// the run and unconditionally restore afterwards.
|
||||
const isolating = this.backendFactory(selection.backendId);
|
||||
const restoreBackend = () => __setSandboxBackendForTests(null);
|
||||
__setSandboxBackendForTests(isolating);
|
||||
|
||||
let implCheckout: DisposableCheckout | undefined;
|
||||
let baselineCheckout: DisposableCheckout | undefined;
|
||||
try {
|
||||
implCheckout = await this.materializer.materialize(this.rootDir, request.integrationSha);
|
||||
|
||||
const implResult = await runVerificationCommand(
|
||||
this.store,
|
||||
implCheckout.dir,
|
||||
logTaskId,
|
||||
command,
|
||||
"test",
|
||||
request.signal,
|
||||
verifyLog,
|
||||
"reviewer",
|
||||
scrubbedEnv,
|
||||
);
|
||||
|
||||
// R5/AE5: agent-supplied proof must fail on the merge-base baseline and
|
||||
// pass on the implementation. A test that passes on both is not exercising
|
||||
// the defect — reject it.
|
||||
if (validatedTestPath) {
|
||||
if (!request.mergeBaseSha) {
|
||||
return this.inconclusive(assertionId, "no merge-base SHA available to validate agent-supplied proof");
|
||||
}
|
||||
baselineCheckout = await this.materializer.materialize(this.rootDir, request.mergeBaseSha);
|
||||
const baselineResult = await runVerificationCommand(
|
||||
this.store,
|
||||
baselineCheckout.dir,
|
||||
logTaskId,
|
||||
command,
|
||||
"test",
|
||||
request.signal,
|
||||
verifyLog,
|
||||
"reviewer",
|
||||
scrubbedEnv,
|
||||
);
|
||||
|
||||
if (baselineResult.success && implResult.success) {
|
||||
return {
|
||||
verdict: "fail",
|
||||
assertionId,
|
||||
reason: "agent-supplied proof passes on both the pre-fix baseline and the implementation; it does not exercise the defect",
|
||||
detail: "pass-on-both rejected (R5/AE5)",
|
||||
};
|
||||
}
|
||||
if (!baselineResult.success && implResult.success) {
|
||||
return { verdict: "pass", assertionId, reason: "regression test fails on the pre-fix baseline and passes on the implementation" };
|
||||
}
|
||||
// Fails on the implementation → defect still reproduces.
|
||||
return {
|
||||
verdict: "fail",
|
||||
assertionId,
|
||||
reason: "regression test does not pass on the implementation; behavior not confirmed",
|
||||
detail: implResult.stderr || implResult.stdout || undefined,
|
||||
};
|
||||
}
|
||||
|
||||
// Whole-suite channel: pass only when the suite passes.
|
||||
if (implResult.success) {
|
||||
return { verdict: "pass", assertionId, reason: "verification suite passed on the implementation checkout" };
|
||||
}
|
||||
return {
|
||||
verdict: "fail",
|
||||
assertionId,
|
||||
reason: "verification suite failed on the implementation checkout; behavior not confirmed",
|
||||
detail: implResult.stderr || implResult.stdout || undefined,
|
||||
};
|
||||
} catch (err) {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
// R9: any setup/exec failure (timeout, abort, materialization error) is a
|
||||
// non-pass; we route it to inconclusive (infra, not behavioral).
|
||||
return this.inconclusive(assertionId, `verification run could not complete: ${message}`);
|
||||
} finally {
|
||||
restoreBackend();
|
||||
await implCheckout?.dispose();
|
||||
await baselineCheckout?.dispose();
|
||||
// R17: the source tree feeding diff/merge must be byte-clean afterwards.
|
||||
try {
|
||||
await this.materializer.assertSourceClean(this.rootDir);
|
||||
} catch (cleanErr) {
|
||||
verifyLog.error("Verification post-condition violated (source not git-clean):", cleanErr);
|
||||
throw cleanErr instanceof Error ? cleanErr : new Error(String(cleanErr));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private inconclusive(assertionId: string, reason: string): VerificationOutcome {
|
||||
return { verdict: "inconclusive", assertionId, reason };
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user