feat(engine): default-to-fail posture + non-mutating verification run (U2,U3)

Behavioral/bug assertions now default to fail unless a bounded verification run
confirms them; static assertions keep the existing read-only judging path. Adds
mission-verification.ts (injected capability): disposable git checkout at a
trusted revision, explicit isolating sandbox backend with fail-closed + scrubbed
env, fixed command template + validated test path, merge-base pre-fix baseline
(rejects pass-on-both), git-clean post-condition, inconclusive as a first-class
verdict. Judge session stays tools:readonly.
This commit is contained in:
gsxdsm
2026-06-12 01:16:34 -07:00
parent 0522cbccc4
commit 483c71b014
5 changed files with 1456 additions and 5 deletions

View File

@@ -1070,6 +1070,9 @@ export {
FEATURE_LOOP_STATES,
VALIDATOR_RUN_STATUSES,
MISSION_ASSERTION_STATUSES,
MISSION_ASSERTION_TYPES,
DEFAULT_MISSION_ASSERTION_TYPE,
normalizeMissionAssertionType,
MILESTONE_VALIDATION_STATES,
} from "./mission-types.js";
export type {
@@ -1116,6 +1119,7 @@ export type {
MissionFeatureLoopSnapshot,
// Contract assertion types
MissionAssertionStatus,
MissionAssertionType,
MilestoneValidationState,
MissionContractAssertion,
FeatureAssertionLink,

View File

@@ -0,0 +1,350 @@
/**
* Tests for the behavioral-verification capability (U3).
*
* Covers the isolation/safety contract before any real execution:
* - command-template rejects shell metacharacters (R19)
* - fail-closed when no isolating sandbox backend is available (R18)
* - env scrubbed to a minimal allowlist (R18)
* - pass-on-both regression test rejected (R5/AE5)
* - source tree git-clean post-condition asserted (R17)
* - no integration SHA → inconclusive (R11 fail-closed)
*/
import { describe, it, expect, vi, afterEach } from "vitest";
import {
validateTestPath,
buildVerificationCommand,
selectIsolatingBackend,
scrubEnv,
VERIFICATION_ENV_ALLOWLIST,
TestExecutionVerificationCapability,
type CheckoutMaterializer,
type IsolatingBackendProbe,
type VerificationRequest,
} from "../mission-verification.js";
import {
__resetSandboxBackendForTests,
type SandboxBackend,
type SandboxStreamingResult,
} from "../sandbox/index.js";
import type { TaskStore } from "@fusion/core";
afterEach(() => {
__resetSandboxBackendForTests();
vi.restoreAllMocks();
});
// ── validateTestPath (R19) ────────────────────────────────────────────────────
describe("validateTestPath", () => {
it("accepts a plain relative test path", () => {
expect(validateTestPath("packages/engine/src/__tests__/foo.test.ts")).toBe(
"packages/engine/src/__tests__/foo.test.ts",
);
});
it.each([
"foo.test.ts; rm -rf /",
"foo.test.ts && curl evil",
"foo.test.ts | cat",
"$(whoami).test.ts",
"`id`.test.ts",
"foo.test.ts\nrm x",
"a${b}.test.ts",
"foo>out.test.ts",
])("rejects shell metacharacters: %s", (p) => {
expect(validateTestPath(p)).toBeNull();
});
it("rejects absolute paths, parent escapes, flag-like, and non-strings", () => {
expect(validateTestPath("/etc/passwd")).toBeNull();
expect(validateTestPath("../../etc/passwd")).toBeNull();
expect(validateTestPath("a/../../b")).toBeNull();
expect(validateTestPath("--config=evil")).toBeNull();
expect(validateTestPath("")).toBeNull();
expect(validateTestPath(42 as unknown)).toBeNull();
});
});
describe("buildVerificationCommand", () => {
it("substitutes a validated test path into the template", () => {
expect(buildVerificationCommand("pnpm vitest run {testPath}", "src/a.test.ts")).toBe(
"pnpm vitest run src/a.test.ts",
);
});
it("produces a whole-suite command when no path is supplied", () => {
expect(buildVerificationCommand("pnpm vitest run {testPath}")).toBe("pnpm vitest run");
});
it("throws if asked to substitute an unsafe path (defense in depth)", () => {
expect(() => buildVerificationCommand("pnpm vitest run {testPath}", "a; rm -rf /")).toThrow();
});
});
// ── selectIsolatingBackend (R18 fail-closed) ───────────────────────────────────
describe("selectIsolatingBackend", () => {
it("selects bubblewrap on linux when available", () => {
expect(
selectIsolatingBackend({ platform: "linux", bubblewrapAvailable: true, sandboxExecAvailable: false }).backendId,
).toBe("bubblewrap");
});
it("selects sandbox-exec on darwin when available", () => {
expect(
selectIsolatingBackend({ platform: "darwin", bubblewrapAvailable: false, sandboxExecAvailable: true }).backendId,
).toBe("sandbox-exec");
});
it("fails closed (null) when no isolating backend is available", () => {
const sel = selectIsolatingBackend({ platform: "linux", bubblewrapAvailable: false, sandboxExecAvailable: false });
expect(sel.backendId).toBeNull();
expect(sel.reason).toMatch(/no isolating sandbox backend/);
});
});
// ── scrubEnv (R18) ─────────────────────────────────────────────────────────────
describe("scrubEnv", () => {
it("keeps only allowlisted keys and forces CI=1", () => {
const result = scrubEnv({
PATH: "/usr/bin",
HOME: "/home/u",
ANTHROPIC_API_KEY: "secret",
DATABASE_URL: "postgres://secret",
AWS_SECRET_ACCESS_KEY: "secret",
});
expect(result.PATH).toBe("/usr/bin");
expect(result.HOME).toBe("/home/u");
expect(result.CI).toBe("1");
expect(result.ANTHROPIC_API_KEY).toBeUndefined();
expect(result.DATABASE_URL).toBeUndefined();
expect(result.AWS_SECRET_ACCESS_KEY).toBeUndefined();
});
it("only ever emits allowlisted keys plus CI", () => {
const result = scrubEnv({ FOO: "x", BAR: "y", PATH: "/bin" });
const allowed = new Set<string>([...VERIFICATION_ENV_ALLOWLIST, "CI"]);
for (const k of Object.keys(result)) {
expect(allowed.has(k)).toBe(true);
}
});
});
// ── TestExecutionVerificationCapability ────────────────────────────────────────
function makeMockStore(): TaskStore {
return {
logEntry: vi.fn().mockResolvedValue(undefined),
appendAgentLog: vi.fn().mockResolvedValue(undefined),
} as unknown as TaskStore;
}
/** A fake materializer that records dispose/clean calls without touching git. */
function makeFakeMaterializer(): CheckoutMaterializer & {
disposed: number;
cleanCalls: number;
setDirty(dirty: boolean): void;
} {
let dirty = false;
const state = {
disposed: 0,
cleanCalls: 0,
setDirty(d: boolean) {
dirty = d;
},
async materialize(_rootDir: string, _revision: string) {
return {
dir: "/tmp/fake-checkout",
dispose: async () => {
state.disposed += 1;
},
};
},
async assertSourceClean(_rootDir: string) {
state.cleanCalls += 1;
if (dirty) throw new Error("Source tree is not git-clean after verification run");
},
};
return state;
}
/** A sandbox backend whose runStreaming returns a scripted outcome. */
function makeScriptedBackend(outcomes: SandboxStreamingResult[]): SandboxBackend {
let i = 0;
return {
capabilities: () => ({
id: "bubblewrap",
supportsNetworkPolicy: true,
supportsFilesystemPolicy: true,
supportsStreaming: true,
platform: "any",
}),
prepare: vi.fn().mockResolvedValue(undefined),
run: vi.fn(),
runStreaming: vi.fn(async () => {
const out = outcomes[Math.min(i, outcomes.length - 1)];
i += 1;
return out;
}),
dispose: vi.fn().mockResolvedValue(undefined),
} as unknown as SandboxBackend;
}
const isolatingProbe = (): Promise<IsolatingBackendProbe> =>
Promise.resolve({ platform: "linux", bubblewrapAvailable: true, sandboxExecAvailable: false });
function baseRequest(overrides: Partial<VerificationRequest> = {}): VerificationRequest {
return {
assertionId: "CA-1",
assertion: "the bug no longer reproduces",
taskId: "FN-1",
integrationSha: "abc123",
...overrides,
};
}
describe("TestExecutionVerificationCapability", () => {
it("fails closed to inconclusive when no isolating backend is available (R18)", async () => {
const materializer = makeFakeMaterializer();
const cap = new TestExecutionVerificationCapability({
store: makeMockStore(),
rootDir: "/repo",
commandTemplate: "pnpm vitest run {testPath}",
materializer,
probeBackends: async () => ({ platform: "linux", bubblewrapAvailable: false, sandboxExecAvailable: false }),
});
const outcome = await cap.verifyBehavioralAssertion(baseRequest());
expect(outcome.verdict).toBe("inconclusive");
expect(outcome.reason).toMatch(/no isolating sandbox backend/);
// Never materialized / executed.
expect(materializer.disposed).toBe(0);
});
it("returns inconclusive when no integration SHA is available (R11)", async () => {
const cap = new TestExecutionVerificationCapability({
store: makeMockStore(),
rootDir: "/repo",
commandTemplate: "pnpm vitest run {testPath}",
materializer: makeFakeMaterializer(),
probeBackends: isolatingProbe,
});
const outcome = await cap.verifyBehavioralAssertion(baseRequest({ integrationSha: undefined }));
expect(outcome.verdict).toBe("inconclusive");
expect(outcome.reason).toMatch(/integration SHA/);
});
it("rejects an agent test path with shell metacharacters before execution (R19)", async () => {
const materializer = makeFakeMaterializer();
const cap = new TestExecutionVerificationCapability({
store: makeMockStore(),
rootDir: "/repo",
commandTemplate: "pnpm vitest run {testPath}",
materializer,
probeBackends: isolatingProbe,
});
const outcome = await cap.verifyBehavioralAssertion(
baseRequest({ proof: { testFilePath: "a.test.ts; rm -rf /" } }),
);
expect(outcome.verdict).toBe("inconclusive");
expect(outcome.reason).toMatch(/shell metacharacters|rejected/);
expect(materializer.disposed).toBe(0);
});
it("passes when the whole-suite run succeeds, and asserts source git-clean (R17)", async () => {
const materializer = makeFakeMaterializer();
const cap = new TestExecutionVerificationCapability({
store: makeMockStore(),
rootDir: "/repo",
commandTemplate: "pnpm vitest run {testPath}",
materializer,
probeBackends: isolatingProbe,
backendFactory: () =>
makeScriptedBackend([{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }]),
});
const outcome = await cap.verifyBehavioralAssertion(baseRequest());
expect(outcome.verdict).toBe("pass");
expect(materializer.disposed).toBe(1);
expect(materializer.cleanCalls).toBe(1);
});
it("rejects a regression test that passes on BOTH baseline and implementation (R5/AE5)", async () => {
const materializer = makeFakeMaterializer();
const cap = new TestExecutionVerificationCapability({
store: makeMockStore(),
rootDir: "/repo",
commandTemplate: "pnpm vitest run {testPath}",
materializer,
probeBackends: isolatingProbe,
// impl run (1st) success, baseline run (2nd) success → pass-on-both.
backendFactory: () =>
makeScriptedBackend([
{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false },
{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false },
]),
});
const outcome = await cap.verifyBehavioralAssertion(
baseRequest({ proof: { testFilePath: "src/a.test.ts" }, mergeBaseSha: "base000" }),
);
expect(outcome.verdict).toBe("fail");
expect(outcome.reason).toMatch(/both/);
});
it("passes when a regression test fails on baseline and passes on implementation (R5)", async () => {
const materializer = makeFakeMaterializer();
const cap = new TestExecutionVerificationCapability({
store: makeMockStore(),
rootDir: "/repo",
commandTemplate: "pnpm vitest run {testPath}",
materializer,
probeBackends: isolatingProbe,
// impl run (1st) success, baseline run (2nd) non-zero-exit → genuine proof.
backendFactory: () =>
makeScriptedBackend([
{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false },
{ outcome: "non-zero-exit", stdout: "", stderr: "boom", exitCode: 1, signal: null },
]),
});
const outcome = await cap.verifyBehavioralAssertion(
baseRequest({ proof: { testFilePath: "src/a.test.ts" }, mergeBaseSha: "base000" }),
);
expect(outcome.verdict).toBe("pass");
});
it("inconclusive when the run times out (R9), still asserts source clean", async () => {
const materializer = makeFakeMaterializer();
const cap = new TestExecutionVerificationCapability({
store: makeMockStore(),
rootDir: "/repo",
commandTemplate: "pnpm vitest run {testPath}",
materializer,
probeBackends: isolatingProbe,
backendFactory: () =>
makeScriptedBackend([{ outcome: "timeout", stdout: "", stderr: "", timeoutMs: 1000 }]),
});
// A timeout surfaces as a thrown ETIMEDOUT inside runVerificationCommand,
// which the capability catches and maps to a non-pass. For the whole-suite
// channel a non-success result is a behavioral fail; but a timeout throw is
// caught by the capability's try/catch → inconclusive.
const outcome = await cap.verifyBehavioralAssertion(baseRequest());
expect(["inconclusive", "fail"]).toContain(outcome.verdict);
expect(materializer.cleanCalls).toBe(1);
});
it("throws if the source tree is dirty after a run (R17 post-condition)", async () => {
const materializer = makeFakeMaterializer();
materializer.setDirty(true);
const cap = new TestExecutionVerificationCapability({
store: makeMockStore(),
rootDir: "/repo",
commandTemplate: "pnpm vitest run {testPath}",
materializer,
probeBackends: isolatingProbe,
backendFactory: () =>
makeScriptedBackend([{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }]),
});
await expect(cap.verifyBehavioralAssertion(baseRequest())).rejects.toThrow(/git-clean/);
});
});

View File

@@ -0,0 +1,402 @@
/**
* Behavioral-verification posture in the Validator Run (U2 + U3).
*
* Verifies the default-to-fail posture for behavioral assertions and the
* non-mutating verification step that confirms them, while static assertions
* keep the exact legacy judge path.
*
* Covers:
* - AE2: behavioral assertion with no verification evidence → fail, even when
* the judge text claims pass (capability absent).
* - AE3: static assertion → unchanged static verdict, no verification invoked.
* - Mixed set: static and behavioral each take their correct path.
* - Behavioral pass via an injected verification capability.
* - Behavioral inconclusive → blocked verdict, NO fix feature.
*/
import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
import type {
Mission,
Milestone,
Slice,
MissionFeature,
MissionValidatorRun,
} from "@fusion/core";
import type { VerificationCapability, VerificationOutcome } from "../../mission-verification.js";
// ── Mock AI dependencies (mirror mission-execution-loop.test.ts) ───────────────
const mockSessionHolder: {
session: { state: { messages: Array<{ role: string; content: string }> }; dispose: ReturnType<typeof vi.fn> };
} = { session: { state: { messages: [] }, dispose: vi.fn() } };
vi.mock("../../pi.js", () => ({
createFnAgent: vi.fn(() => Promise.resolve({ session: mockSessionHolder.session })),
promptWithFallback: vi.fn().mockResolvedValue(undefined),
}));
vi.mock("../../logger.js", () => ({
createLogger: vi.fn(() => ({ log: vi.fn(), warn: vi.fn(), error: vi.fn() })),
}));
vi.mock("../../agent-session-helpers.js", async (importOriginal) => {
const actual = await importOriginal<typeof import("../../agent-session-helpers.js")>();
return {
...actual,
createResolvedAgentSession: vi.fn(async () => ({
session: mockSessionHolder.session as any,
sessionFile: undefined,
runtimeId: "test-runtime",
wasConfigured: true,
})),
};
});
import { createResolvedAgentSession } from "../../agent-session-helpers.js";
import { MissionExecutionLoop } from "../../mission-execution-loop.js";
type AssertionRow = {
id: string;
milestoneId: string;
title: string;
assertion: string;
status: "pending" | "passed" | "failed" | "blocked";
type?: "static" | "behavioral";
orderIndex: number;
createdAt: string;
updatedAt: string;
sourceFeatureId?: string;
};
function now() {
return new Date().toISOString();
}
function createMockMission(): Mission {
return {
id: "M-TEST1",
title: "Test Mission",
status: "active",
interviewState: "not_started",
autopilotEnabled: true,
autopilotState: "inactive",
createdAt: now(),
updatedAt: now(),
};
}
function createMockMilestone(overrides: Partial<Milestone> = {}): Milestone {
return {
id: "MS-001",
missionId: "M-TEST1",
title: "Test Milestone",
status: "active",
orderIndex: 0,
interviewState: "not_started",
dependencies: [],
createdAt: now(),
updatedAt: now(),
...overrides,
};
}
function createMockSlice(overrides: Partial<Slice> = {}): Slice {
return {
id: "SL-001",
milestoneId: "MS-001",
title: "Test Slice",
status: "active",
planState: "not_started",
orderIndex: 0,
createdAt: now(),
updatedAt: now(),
...overrides,
};
}
function createMockFeature(overrides: Partial<MissionFeature> = {}): MissionFeature {
return {
id: "F-001",
sliceId: "SL-001",
title: "Test Feature",
status: "defined",
loopState: "idle",
implementationAttemptCount: 0,
validatorAttemptCount: 0,
createdAt: now(),
updatedAt: now(),
...overrides,
};
}
function createMockMissionStore() {
const missions = new Map<string, Mission>();
const features = new Map<string, MissionFeature>();
const assertionsByFeature = new Map<string, AssertionRow[]>();
const validatorRuns = new Map<string, MissionValidatorRun>();
let runSeq = 0;
const store = {
getMission: vi.fn((id: string) => missions.get(id)),
logMissionEvent: vi.fn(),
getFeature: vi.fn((id: string) => features.get(id)),
getFeatureByTaskId: vi.fn((taskId: string) => {
for (const f of features.values()) if (f.taskId === taskId) return f;
return undefined;
}),
updateFeatureStatus: vi.fn((id: string, status: MissionFeature["status"]) => {
const f = features.get(id)!;
const updated = { ...f, status, updatedAt: now() };
features.set(id, updated);
return updated;
}),
listAssertionsForFeature: vi.fn((featureId: string) => assertionsByFeature.get(featureId) ?? []),
ensureFeatureAssertionLinked: vi.fn((featureId: string) => assertionsByFeature.get(featureId) ?? []),
getSlice: vi.fn((id: string) => createMockSlice({ id })),
getMilestone: vi.fn((id: string) => createMockMilestone({ id })),
startValidatorRun: vi.fn((featureId: string) => {
const run: MissionValidatorRun = {
id: `VR-${++runSeq}`,
featureId,
milestoneId: "MS-001",
sliceId: "SL-001",
status: "running",
triggerType: "task_completion",
implementationAttempt: 1,
validatorAttempt: 1,
startedAt: now(),
createdAt: now(),
updatedAt: now(),
};
validatorRuns.set(run.id, run);
return run;
}),
getValidatorRun: vi.fn((id: string) => validatorRuns.get(id)),
completeValidatorRun: vi.fn((id: string, status: MissionValidatorRun["status"], summary?: string) => {
const run = validatorRuns.get(id)!;
const updated = { ...run, status, summary, completedAt: now(), updatedAt: now() };
validatorRuns.set(id, updated);
const feature = features.get(run.featureId);
if (feature) {
const loopState = status === "passed" ? "passed" : status === "failed" ? "needs_fix" : status === "blocked" ? "blocked" : "validating";
features.set(run.featureId, { ...feature, loopState: loopState as any, lastValidatorStatus: status, updatedAt: now() });
}
return updated;
}),
recordValidatorFailures: vi.fn(() => []),
createGeneratedFixFeature: vi.fn((sourceFeatureId: string, runId: string) => {
const src = features.get(sourceFeatureId)!;
const fix = createMockFeature({ id: `FIX-${sourceFeatureId}`, taskId: `TASK-FIX-${sourceFeatureId}`, generatedFromFeatureId: sourceFeatureId, generatedFromRunId: runId, loopState: "implementing" });
features.set(fix.id, fix);
features.set(sourceFeatureId, { ...src, loopState: "implementing", implementationAttemptCount: (src.implementationAttemptCount ?? 0) + 1, updatedAt: now() });
return fix;
}),
triageFeature: vi.fn(async (featureId: string) => {
const f = features.get(featureId)!;
const updated = { ...f, status: "triaged" as const, updatedAt: now() };
features.set(featureId, updated);
return updated;
}),
on: vi.fn(),
off: vi.fn(),
emit: vi.fn(),
_setMission: (m: Mission) => missions.set(m.id, m),
_setFeature: (f: MissionFeature) => features.set(f.id, f),
_setAssertions: (featureId: string, rows: AssertionRow[]) => assertionsByFeature.set(featureId, rows),
};
return store;
}
function createMockTaskStore() {
const tasks = new Map<string, any>();
return {
getTask: vi.fn(async (id: string) => tasks.get(id)),
moveTask: vi.fn(async () => {}),
updateTask: vi.fn(async () => {}),
getSettings: vi.fn().mockResolvedValue({ missionStaleThresholdMs: 600_000, missionMaxTaskRetries: 3 }),
recordRunAuditEvent: vi.fn(),
on: vi.fn(),
off: vi.fn(),
_setTask: (t: any) => tasks.set(t.id, t),
};
}
function assertionRow(overrides: Partial<AssertionRow> & { id: string }): AssertionRow {
return {
milestoneId: "MS-001",
title: overrides.id,
assertion: `do ${overrides.id}`,
status: "pending",
orderIndex: 0,
createdAt: now(),
updatedAt: now(),
...overrides,
};
}
describe("Validator behavioral posture (U2 + U3)", () => {
let missionStore: ReturnType<typeof createMockMissionStore>;
let taskStore: ReturnType<typeof createMockTaskStore>;
let loop: MissionExecutionLoop;
beforeEach(() => {
missionStore = createMockMissionStore();
taskStore = createMockTaskStore();
vi.mocked(createResolvedAgentSession).mockReset();
vi.mocked(createResolvedAgentSession).mockResolvedValue({
session: mockSessionHolder.session as any,
sessionFile: undefined,
runtimeId: "test-runtime",
wasConfigured: true,
});
missionStore._setMission(createMockMission());
mockSessionHolder.session.state.messages = [];
mockSessionHolder.session.dispose = vi.fn();
});
afterEach(() => {
loop?.stop();
vi.restoreAllMocks();
});
function judgePass(assertionIds: string[]) {
mockSessionHolder.session.state.messages = [
{
role: "assistant",
content: JSON.stringify({
status: "pass",
assertions: assertionIds.map((id) => ({ assertionId: id, passed: true })),
summary: "all good",
}),
},
];
}
it("AE2: behavioral assertion the judge calls pass → fails with no verification capability", async () => {
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-B", status: "in-progress" });
missionStore._setFeature(feature);
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]);
taskStore._setTask({ id: "FN-B", title: "behavioral", log: [] });
judgePass(["CA-1"]);
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp" });
loop.start();
await loop.processTaskOutcome("FN-B");
// No verification capability → behavioral default-to-fail → fix flow.
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "failed", expect.any(String));
expect(missionStore.createGeneratedFixFeature).toHaveBeenCalled();
expect(missionStore.getFeature("F-001")?.status).not.toBe("done");
});
it("AE3: static assertion the judge calls pass → passes, no verification invoked", async () => {
const verify = vi.fn();
const capability: VerificationCapability = { verifyBehavioralAssertion: verify };
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-S", status: "in-progress" });
missionStore._setFeature(feature);
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "static" })]);
taskStore._setTask({ id: "FN-S", title: "static", log: [] });
judgePass(["CA-1"]);
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: capability });
loop.start();
await loop.processTaskOutcome("FN-S");
expect(verify).not.toHaveBeenCalled();
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String));
expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done");
});
it("untyped assertions default to static — legacy judge pass path is preserved", async () => {
const verify = vi.fn();
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-U", status: "in-progress" });
missionStore._setFeature(feature);
// No `type` field → normalizes to static.
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1" })]);
taskStore._setTask({ id: "FN-U", title: "untyped", log: [] });
judgePass(["CA-1"]);
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
loop.start();
await loop.processTaskOutcome("FN-U");
expect(verify).not.toHaveBeenCalled();
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String));
});
it("behavioral assertion confirmed by an injected verification capability → passes", async () => {
const verify = vi.fn(async (req): Promise<VerificationOutcome> => ({ verdict: "pass", assertionId: req.assertionId, reason: "confirmed" }));
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-BV", status: "in-progress" });
missionStore._setFeature(feature);
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]);
taskStore._setTask({ id: "FN-BV", title: "behavioral verified", integrationSha: "sha123", log: [] });
judgePass(["CA-1"]);
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
loop.start();
await loop.processTaskOutcome("FN-BV");
expect(verify).toHaveBeenCalledTimes(1);
expect(verify.mock.calls[0][0]).toMatchObject({ assertionId: "CA-1", integrationSha: "sha123" });
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String));
expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done");
});
it("behavioral assertion verification inconclusive → blocked, NO fix feature", async () => {
const verify = vi.fn(async (req): Promise<VerificationOutcome> => ({ verdict: "inconclusive", assertionId: req.assertionId, reason: "no isolating sandbox backend" }));
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-INC", status: "in-progress" });
missionStore._setFeature(feature);
missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]);
taskStore._setTask({ id: "FN-INC", title: "inconclusive", integrationSha: "sha123", log: [] });
judgePass(["CA-1"]);
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
loop.start();
await loop.processTaskOutcome("FN-INC");
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "blocked", expect.any(String));
expect(missionStore.createGeneratedFixFeature).not.toHaveBeenCalled();
expect(missionStore.getFeature("F-001")?.status).not.toBe("done");
});
it("mixed set: static passes via judge, behavioral confirmed via verification → overall pass", async () => {
const verify = vi.fn(async (req): Promise<VerificationOutcome> => ({ verdict: "pass", assertionId: req.assertionId, reason: "confirmed" }));
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-MIX", status: "in-progress" });
missionStore._setFeature(feature);
missionStore._setAssertions("F-001", [
assertionRow({ id: "CA-static", type: "static" }),
assertionRow({ id: "CA-behav", type: "behavioral" }),
]);
taskStore._setTask({ id: "FN-MIX", title: "mixed", integrationSha: "sha123", log: [] });
judgePass(["CA-static", "CA-behav"]);
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
loop.start();
await loop.processTaskOutcome("FN-MIX");
// Only the behavioral assertion is verified.
expect(verify).toHaveBeenCalledTimes(1);
expect(verify.mock.calls[0][0]).toMatchObject({ assertionId: "CA-behav" });
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String));
expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done");
});
it("mixed set: behavioral observed wrong → overall fail even though static passes", async () => {
const verify = vi.fn(async (req): Promise<VerificationOutcome> => ({ verdict: "fail", assertionId: req.assertionId, reason: "defect still reproduces" }));
const feature = createMockFeature({ loopState: "implementing", taskId: "FN-MIX2", status: "in-progress" });
missionStore._setFeature(feature);
missionStore._setAssertions("F-001", [
assertionRow({ id: "CA-static", type: "static" }),
assertionRow({ id: "CA-behav", type: "behavioral" }),
]);
taskStore._setTask({ id: "FN-MIX2", title: "mixed fail", integrationSha: "sha123", log: [] });
judgePass(["CA-static", "CA-behav"]);
loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } });
loop.start();
await loop.processTaskOutcome("FN-MIX2");
expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "failed", expect.any(String));
expect(missionStore.createGeneratedFixFeature).toHaveBeenCalled();
expect(missionStore.getFeature("F-001")?.status).not.toBe("done");
});
});

View File

@@ -26,7 +26,9 @@ import {
TEST_MODE_RESOLVED,
isTestModeActive,
resolveTaskValidatorModel,
normalizeMissionAssertionType,
} from "@fusion/core";
import type { VerificationOutcome } from "./mission-verification.js";
import { createFnAgent, promptWithFallback, type AgentResult } from "./pi.js";
import { mergeEffectiveSettings } from "./effective-settings.js";
import {
@@ -50,8 +52,16 @@ const VALIDATION_TIMEOUT_MS = 10 * 60 * 1000; // 10 minutes
* per assertion plus an overall status.
*/
export interface ValidationResult {
/** Overall validation status */
status: "pass" | "fail" | "blocked" | "error";
/**
* Overall validation status.
*
* `inconclusive` is first-class and distinct from `fail`: it means a
* behavioral verification run could not run or conclude (no isolating sandbox
* backend, timeout, setup failure, rejected proof). In this unit it routes to
* a blocked verdict (no remediation); later units track its infra-failure rate
* separately.
*/
status: "pass" | "fail" | "blocked" | "error" | "inconclusive";
/** Per-assertion results */
assertions: Array<{
assertionId: string;
@@ -83,6 +93,14 @@ export interface MissionExecutionLoopOptions {
pluginRunner?: import("./plugin-runner.js").PluginRunner;
/** Optional agent store for resolving assigned-agent runtime hints. */
agentStore?: AgentStore;
/**
* Optional behavioral-verification capability (U3). When provided, behavioral
* assertions are confirmed by a non-mutating verification run; the judge's
* "pass" on a behavioral assertion is advisory only. When ABSENT, behavioral
* assertions still default to fail (U2) but no verification run is attempted —
* preserving the behavior of existing construction sites that inject nothing.
*/
verificationCapability?: import("./mission-verification.js").VerificationCapability;
}
export class MissionExecutionLoop extends EventEmitter {
@@ -94,6 +112,7 @@ export class MissionExecutionLoop extends EventEmitter {
private missionAutopilot?: MissionExecutionLoopOptions["missionAutopilot"];
private pluginRunner?: MissionExecutionLoopOptions["pluginRunner"];
private agentStore?: MissionExecutionLoopOptions["agentStore"];
private verificationCapability?: MissionExecutionLoopOptions["verificationCapability"];
private activeValidations = new Set<string>(); // feature IDs currently being validated
constructor(options: MissionExecutionLoopOptions) {
@@ -105,6 +124,7 @@ export class MissionExecutionLoop extends EventEmitter {
this.missionAutopilot = options.missionAutopilot;
this.pluginRunner = options.pluginRunner;
this.agentStore = options.agentStore;
this.verificationCapability = options.verificationCapability;
loopLog.log("MissionExecutionLoop created");
}
@@ -434,8 +454,11 @@ export class MissionExecutionLoop extends EventEmitter {
await this.handleValidationPass(feature.id, run.id, result.summary);
} else if (result.status === "fail") {
await this.handleValidationFail(feature.id, run.id, result);
} else if (result.status === "blocked") {
await this.handleValidationBlocked(feature.id, run.id, result.blockedReason);
} else if (result.status === "blocked" || result.status === "inconclusive") {
// Inconclusive (infra: no isolating backend, timeout, rejected proof)
// routes to blocked/needs-attention — distinct from a behavioral fail,
// and spawns no Fix Feature (per KTD / R21, fully realized in a later unit).
await this.handleValidationBlocked(feature.id, run.id, result.blockedReason ?? result.summary);
} else if (result.status === "error") {
await this.handleValidationError(feature.id, run.id, result.summary);
}
@@ -537,7 +560,12 @@ export class MissionExecutionLoop extends EventEmitter {
// Get the validation result from the session
// The agent should have returned structured JSON in its response
const result = await this.parseValidationResult(session.session, assertions);
const judgeResult = await this.parseValidationResult(session.session, assertions);
// U2/U3: the read-only judge's verdict is authoritative for STATIC
// assertions only. BEHAVIORAL assertions default to fail and are confirmed
// (or refuted) by a non-mutating verification run instead.
const result = await this.applyBehavioralPosture(feature, assertions, judgeResult);
loopLog.log(`Validation completed for feature ${feature.id}: ${result.status}`);
return result;
@@ -568,6 +596,158 @@ export class MissionExecutionLoop extends EventEmitter {
}
}
/**
* Apply the behavioral judging posture (U2/U3) to the read-only judge's
* verdict.
*
* - STATIC assertions keep the judge's verdict verbatim (no behavior change).
* - BEHAVIORAL assertions DEFAULT TO FAIL. The judge's "pass" on a behavioral
* assertion is advisory; an authoritative pass requires a verification run
* to confirm it. When a verification capability is injected, each behavioral
* assertion is run through it: pass → satisfied; fail → behavioral failure;
* inconclusive → the aggregate becomes inconclusive (infra, no remediation).
* When NO capability is injected, behavioral assertions simply stay failed
* (preserving existing call-site behavior — existing data is all static).
*
* The aggregate status is recomputed from the post-posture per-assertion
* results so the existing pass/fail/blocked/error/inconclusive flow is driven
* correctly.
*/
private async applyBehavioralPosture(
feature: MissionFeature,
assertions: MissionContractAssertion[],
judgeResult: ValidationResult,
): Promise<ValidationResult> {
// Preserve non-behavioral terminal verdicts untouched (error/blocked from the
// judge are not behavioral posture concerns).
if (judgeResult.status === "error") {
return judgeResult;
}
const typeById = new Map<string, ReturnType<typeof normalizeMissionAssertionType>>();
let hasBehavioral = false;
for (const a of assertions) {
const t = normalizeMissionAssertionType(a.type);
typeById.set(a.id, t);
if (t === "behavioral") hasBehavioral = true;
}
// Fast path: no behavioral assertions → existing static path is preserved
// exactly. This keeps every existing (untyped/static) test green.
if (!hasBehavioral) {
return judgeResult;
}
const textById = new Map(assertions.map((a) => [a.id, a.assertion]));
let sawInconclusive = false;
let inconclusiveReason: string | undefined;
const newAssertionResults = await Promise.all(
judgeResult.assertions.map(async (judged) => {
const type = typeById.get(judged.assertionId) ?? "static";
if (type !== "behavioral") {
// Static: keep judge verdict verbatim.
return judged;
}
// Behavioral: default to fail unless verification confirms it.
if (!this.verificationCapability) {
return {
...judged,
passed: false,
message: "Behavioral assertion defaults to fail: no verification evidence (advisory judge verdict is not authoritative).",
expected: judged.expected ?? "Behavior confirmed by a verification run",
actual: judged.actual ?? "No verification run was performed",
};
}
let outcome: VerificationOutcome;
try {
outcome = await this.verificationCapability.verifyBehavioralAssertion({
assertionId: judged.assertionId,
assertion: textById.get(judged.assertionId) ?? "",
taskId: feature.taskId,
integrationSha: await this.resolveIntegrationSha(feature),
signal: undefined,
});
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
loopLog.warn(`Verification capability threw for assertion ${judged.assertionId}: ${message}`);
outcome = { verdict: "inconclusive", assertionId: judged.assertionId, reason: `verification error: ${message}` };
}
if (outcome.verdict === "pass") {
return { ...judged, passed: true, message: outcome.reason };
}
if (outcome.verdict === "inconclusive") {
sawInconclusive = true;
inconclusiveReason = inconclusiveReason ?? outcome.reason;
return {
...judged,
passed: false,
message: `Behavioral verification inconclusive: ${outcome.reason}`,
expected: judged.expected ?? "Behavior confirmed by a verification run",
actual: outcome.detail ?? "Verification could not conclude",
};
}
// fail
return {
...judged,
passed: false,
message: outcome.reason,
expected: judged.expected ?? "Behavior confirmed by a verification run",
actual: outcome.detail ?? judged.actual ?? "Behavior not confirmed",
};
}),
);
const allPassed = newAssertionResults.every((a) => a.passed);
// Inconclusive takes precedence over fail: an infra-driven non-pass must not
// be mistaken for an observed behavioral failure (no Fix Feature).
let status: ValidationResult["status"];
if (sawInconclusive && !allPassed) {
status = "inconclusive";
} else if (allPassed) {
status = "pass";
} else {
status = "fail";
}
const summary = status === "pass"
? judgeResult.summary
: status === "inconclusive"
? `Behavioral verification inconclusive: ${inconclusiveReason ?? "verification could not conclude"}`
: "One or more behavioral assertions were not confirmed by verification.";
return {
status,
assertions: newAssertionResults,
summary,
blockedReason: status === "inconclusive" ? (inconclusiveReason ?? "verification inconclusive") : judgeResult.blockedReason,
};
}
/**
* Resolve the trusted revision (integration SHA) whose disposable checkout the
* verification run executes against. The live task worktree is pruned before
* the done-transition that triggers validation, so it cannot be used.
*
* In this unit we read it from the linked task when available; callers that do
* not supply a resolvable SHA cause the verification run to resolve to
* inconclusive (fail-closed). A richer derivation is owned by a later unit.
*/
private async resolveIntegrationSha(feature: MissionFeature): Promise<string | undefined> {
if (!feature.taskId) return undefined;
try {
const task = await this.taskStore.getTask(feature.taskId);
const candidate = (task as { integrationSha?: string; baseCommit?: string } | undefined);
return candidate?.integrationSha ?? candidate?.baseCommit ?? undefined;
} catch {
return undefined;
}
}
private resolveValidationSessionModel(
task: Awaited<ReturnType<TaskStore["getTask"]>> | null,
settings: Partial<Settings> | undefined,

View File

@@ -0,0 +1,515 @@
/**
* Mission behavioral-verification capability (U3).
*
* The Validator Run's read-only AI judge cannot run code, so its "pass" on a
* *behavioral* assertion is advisory only (U2). This module supplies the
* authoritative, NON-MUTATING verification step that confirms a behavioral/bug
* assertion by exercising the implemented code.
*
* Channels:
* - **test-execution** (this unit): run the project's scoped test suite / an
* agent-supplied regression test against a disposable checkout at a trusted
* revision, through an explicit isolating sandbox backend.
* - **app-driving** (later unit U5/U8): drive a running app instance. Not
* implemented here — the capability surface is structured so it can be added
* without reshaping callers.
*
* Safety invariants enforced here (the boundary, not a convention):
* - R18: execute under an *isolating* sandbox backend (bubblewrap / sandbox-exec)
* with a scrubbed env allowlist; FAIL CLOSED to a non-pass when no isolating
* backend is available — never fall through to the unrestricted native backend.
* - R19: the command is built from a fixed, system-owned template into which only
* a validated test-file path is substituted; shell metacharacters are rejected.
* - R11/R17: verification runs against a disposable checkout at the integration
* SHA (never the pruned live worktree, never the repo root); the source tree
* that feeds diff/merge is asserted git-clean after a run.
* - R5/AE5: agent-supplied proof must FAIL on a second disposable checkout at
* `git merge-base` (a revision the agent does not control) and PASS on the
* implementation; a test that passes on both is rejected.
* - R9: inconclusive / timeout / setup failure resolves to a non-pass.
* - R10: no board / mission writes happen here.
*/
import { promises as fs } from "node:fs";
import os from "node:os";
import path from "node:path";
import { exec } from "node:child_process";
import { promisify } from "node:util";
import type { TaskStore } from "@fusion/core";
import type { SandboxCapabilities } from "./sandbox/index.js";
import { __setSandboxBackendForTests, resolveSandboxBackend } from "./sandbox/index.js";
import type { SandboxBackend } from "./sandbox/index.js";
import { detectBwrap } from "./sandbox/bubblewrap-detect.js";
import { detectSandboxExec } from "./sandbox/sandbox-exec-detect.js";
import { runVerificationCommand } from "./verification-utils.js";
import { createLogger } from "./logger.js";
const execAsync = promisify(exec);
const verifyLog = createLogger("mission-verify");
// ── Verdict types ───────────────────────────────────────────────────────────
/**
* Outcome of a verification run for a single behavioral assertion.
*
* - `pass`: behavior confirmed by execution.
* - `fail`: behavior observed wrong (the defect still reproduces / proof rejected).
* - `inconclusive`: verification could not run or conclude (no isolating backend,
* timeout, setup failure, rejected/invalid proof input). First-class and
* distinct from `fail`: it must NOT spawn remediation (handled by later units),
* but in this unit it never resolves to a default pass either.
*/
export type VerificationVerdict = "pass" | "fail" | "inconclusive";
/** Why a verification run reached its verdict (for durable observability later). */
export interface VerificationOutcome {
verdict: VerificationVerdict;
/** Human-readable reason, suitable for surfacing in a failure record. */
reason: string;
/** The assertion this outcome corresponds to. */
assertionId: string;
/** Optional summarized command output for the failure record. */
detail?: string;
}
/** Shape of agent-supplied executable proof (a regression test). */
export interface VerificationProof {
/**
* Path to the regression test file, relative to the checkout root. Validated
* to reject shell metacharacters and path escapes before use (R19).
*/
testFilePath: string;
}
/** Input describing a single behavioral assertion to verify. */
export interface VerificationRequest {
assertionId: string;
/** The assertion text (for logging / reason building). */
assertion: string;
/** Board task id associated with the feature, used for verification-command logging. */
taskId?: string;
/**
* The trusted revision (integration SHA) whose checkout the implementation is
* verified against. When absent, verification is inconclusive (cannot
* materialize a trusted checkout).
*/
integrationSha?: string;
/**
* The `git merge-base` revision (feature branch vs base branch) used as the
* pre-fix baseline for agent-supplied proof. Not agent-controlled.
*/
mergeBaseSha?: string;
/** Optional agent-supplied executable proof. */
proof?: VerificationProof;
/** Abort signal to bound the run. */
signal?: AbortSignal;
}
/**
* Injected verification capability. Mirrors the `createFnAgent` injection
* pattern so MissionExecutionLoop can swap a real implementation for a mock in
* tests. Optional on the loop: when absent, behavioral assertions resolve to a
* non-pass without invoking any execution (preserving existing behavior for
* call sites that do not inject a capability).
*/
export interface VerificationCapability {
verifyBehavioralAssertion(request: VerificationRequest): Promise<VerificationOutcome>;
}
// ── Command-template safety (R19) ─────────────────────────────────────────────
/**
* Characters that could break out of the fixed command template or inject
* additional shell behavior. Agent-supplied test paths containing any of these
* are rejected before execution.
*/
const SHELL_METACHARACTERS = /[;&|`$(){}<>!*?\[\]\\"'\n\r\t\0]/;
/**
* Validate an agent-supplied test-file path. Returns the normalized path when
* safe, or `null` when it must be rejected (R19).
*
* Rejects: empty, absolute paths, parent-dir escapes, shell metacharacters,
* and leading dashes (which could be read as command flags).
*/
export function validateTestPath(rawPath: unknown): string | null {
if (typeof rawPath !== "string") return null;
const path = rawPath.trim();
if (path.length === 0) return null;
if (SHELL_METACHARACTERS.test(path)) return null;
if (path.startsWith("/")) return null; // must be relative to the checkout
if (path.startsWith("-")) return null; // could be parsed as a flag
// Reject parent-dir escapes (any `..` segment).
const segments = path.split("/");
if (segments.some((seg) => seg === "..")) return null;
return path;
}
/**
* Build the verification command from the fixed system-owned template. Only a
* pre-validated test path may be substituted (R19). Callers MUST pass a path
* already run through {@link validateTestPath}; this function re-checks and
* throws on violation as a defense-in-depth guard.
*/
export function buildVerificationCommand(template: string, validatedTestPath?: string): string {
if (validatedTestPath !== undefined) {
if (validateTestPath(validatedTestPath) === null) {
throw new Error(`Refusing to build verification command: invalid test path ${JSON.stringify(validatedTestPath)}`);
}
if (!template.includes("{testPath}")) {
throw new Error("Verification command template must contain a {testPath} placeholder when a test path is supplied");
}
return template.replace("{testPath}", validatedTestPath);
}
// Whole-suite invocation: the template must not reference a test path.
return template.replace("{testPath}", "").trimEnd();
}
// ── Isolating-backend selection (R18, fail-closed) ────────────────────────────
/**
* Result of selecting an isolating sandbox backend for verification.
*/
export interface IsolatingBackendSelection {
/** The backend id to request from `resolveSandboxBackend`, or null if none. */
backendId: SandboxCapabilities["id"] | null;
/** Why no isolating backend is available (when backendId is null). */
reason?: string;
}
/**
* Describes the detected availability of isolating backends on this host.
* Injectable for tests so we don't shell out to detect bwrap/sandbox-exec.
*/
export interface IsolatingBackendProbe {
platform: NodeJS.Platform;
bubblewrapAvailable: boolean;
sandboxExecAvailable: boolean;
}
/**
* Choose an isolating backend, failing closed. Returns `backendId: null` (a
* non-pass signal) when no isolating backend is available — verification must
* NEVER fall through to the unrestricted native backend (R18).
*/
export function selectIsolatingBackend(probe: IsolatingBackendProbe): IsolatingBackendSelection {
if (probe.platform === "linux" && probe.bubblewrapAvailable) {
return { backendId: "bubblewrap" };
}
if (probe.platform === "darwin" && probe.sandboxExecAvailable) {
return { backendId: "sandbox-exec" };
}
return {
backendId: null,
reason: `no isolating sandbox backend available (platform=${probe.platform}, bwrap=${probe.bubblewrapAvailable}, sandbox-exec=${probe.sandboxExecAvailable})`,
};
}
// ── Environment scrubbing (R18) ───────────────────────────────────────────────
/**
* Environment variables permitted into the verification child process. Anything
* not on the allowlist (API keys, auth tokens, DB credentials, agent logs) is
* dropped so agent-authored code executes with a minimal environment.
*/
export const VERIFICATION_ENV_ALLOWLIST = [
"PATH",
"HOME",
"SHELL",
"LANG",
"LC_ALL",
"TMPDIR",
"TERM",
"NODE_ENV",
// pnpm / corepack need these to resolve the package manager in the checkout.
"PNPM_HOME",
"COREPACK_HOME",
"npm_config_registry",
] as const;
/**
* Produce a scrubbed environment containing only allowlisted keys from the
* source environment, with `CI=1` forced for deterministic test runs.
*/
export function scrubEnv(source: NodeJS.ProcessEnv = process.env): NodeJS.ProcessEnv {
const scrubbed: NodeJS.ProcessEnv = {};
for (const key of VERIFICATION_ENV_ALLOWLIST) {
const value = source[key];
if (value !== undefined) {
scrubbed[key] = value;
}
}
// Force deterministic, non-interactive execution.
scrubbed.CI = "1";
return scrubbed;
}
// ── Disposable checkout materialization (R11/R17) ─────────────────────────────
/** A disposable checkout the verification run can execute against. */
export interface DisposableCheckout {
/** Absolute path to the checkout root (under a run-unique tmpdir). */
dir: string;
/** Tear the checkout down unconditionally (idempotent). */
dispose(): Promise<void>;
}
/**
* Materializes disposable checkouts at a trusted revision. Injectable so tests
* can supply a fixture checkout without invoking git.
*/
export interface CheckoutMaterializer {
/**
* Create a disposable checkout of `rootDir` at `revision` under a run-unique
* tmpdir. The implementation MUST NOT mutate the source tree at `rootDir`.
*/
materialize(rootDir: string, revision: string): Promise<DisposableCheckout>;
/**
* Assert that the source tree feeding diff/merge is git-clean (byte-identical)
* — the R17 post-condition. Throws if dirty.
*/
assertSourceClean(rootDir: string): Promise<void>;
}
/**
* Default git-backed materializer: `git worktree add --detach <tmp> <revision>`
* produces an isolated checkout without touching the source working tree, and
* `git status --porcelain` on the source confirms cleanliness afterwards.
*/
export class GitCheckoutMaterializer implements CheckoutMaterializer {
async materialize(rootDir: string, revision: string): Promise<DisposableCheckout> {
const dir = await fs.mkdtemp(path.join(os.tmpdir(), "fn-verify-"));
// `git worktree add --detach` checks out the revision into a throwaway dir
// without modifying the source working tree.
await execAsync(`git worktree add --detach ${JSON.stringify(dir)} ${JSON.stringify(revision)}`, {
cwd: rootDir,
timeout: 60_000,
maxBuffer: 8 * 1024 * 1024,
});
return {
dir,
dispose: async () => {
try {
await execAsync(`git worktree remove --force ${JSON.stringify(dir)}`, {
cwd: rootDir,
timeout: 30_000,
});
} catch (err) {
verifyLog.warn(`Failed to remove verification worktree ${dir}:`, err);
}
await fs.rm(dir, { recursive: true, force: true }).catch(() => {});
},
};
}
async assertSourceClean(rootDir: string): Promise<void> {
const { stdout } = await execAsync("git status --porcelain", {
cwd: rootDir,
timeout: 30_000,
maxBuffer: 8 * 1024 * 1024,
});
if (stdout.trim().length > 0) {
throw new Error(`Source tree is not git-clean after verification run:\n${stdout.trim()}`);
}
}
}
/** Probe the host for isolating-backend availability (cached by the detectors). */
async function probeIsolatingBackends(): Promise<IsolatingBackendProbe> {
const [bwrap, sandboxExec] = await Promise.all([
detectBwrap().catch(() => ({ available: false })),
detectSandboxExec().catch(() => ({ available: false })),
]);
return {
platform: process.platform,
bubblewrapAvailable: bwrap.available,
sandboxExecAvailable: sandboxExec.available,
};
}
// ── Test-execution verification capability ────────────────────────────────────
export interface TestExecutionVerificationOptions {
/** Task store, reused by runVerificationCommand for command logging. */
store: TaskStore;
/** Repo root whose source tree must remain git-clean. */
rootDir: string;
/**
* Fixed, system-owned command template. Must contain `{testPath}` when an
* agent-supplied proof path is used. Example: `pnpm vitest run {testPath}`.
*/
commandTemplate: string;
/** Injectable checkout materializer (defaults to git-backed). */
materializer?: CheckoutMaterializer;
/** Injectable backend probe (defaults to host detection). */
probeBackends?: () => Promise<IsolatingBackendProbe>;
/**
* Injectable factory for the isolating sandbox backend, given the selected
* backend id. Defaults to `resolveSandboxBackend({ backendId })`. Injectable so
* tests can supply a scripted backend without mutating global sandbox state.
*/
backendFactory?: (backendId: SandboxCapabilities["id"]) => SandboxBackend;
/** Injectable env source (defaults to process.env). */
envSource?: NodeJS.ProcessEnv;
}
/**
* The test-execution channel of the verification run. Confirms a behavioral
* assertion by running the suite / an agent-supplied regression test against a
* disposable checkout at the integration SHA, under an isolating sandbox
* backend with a scrubbed env. Fails closed to a non-pass on any setup failure.
*
* App-driving is NOT handled here; a later unit dispatches UI/bug assertions to
* an app-driving channel. This class is the canonical pattern that channel will
* mirror.
*/
export class TestExecutionVerificationCapability implements VerificationCapability {
private readonly store: TaskStore;
private readonly rootDir: string;
private readonly commandTemplate: string;
private readonly materializer: CheckoutMaterializer;
private readonly probeBackends: () => Promise<IsolatingBackendProbe>;
private readonly backendFactory: (backendId: SandboxCapabilities["id"]) => SandboxBackend;
private readonly envSource: NodeJS.ProcessEnv;
constructor(options: TestExecutionVerificationOptions) {
this.store = options.store;
this.rootDir = options.rootDir;
this.commandTemplate = options.commandTemplate;
this.materializer = options.materializer ?? new GitCheckoutMaterializer();
this.probeBackends = options.probeBackends ?? probeIsolatingBackends;
this.backendFactory = options.backendFactory ?? ((backendId) => resolveSandboxBackend({ backendId }));
this.envSource = options.envSource ?? process.env;
}
async verifyBehavioralAssertion(request: VerificationRequest): Promise<VerificationOutcome> {
const { assertionId } = request;
// R11: a trusted revision is required to materialize a disposable checkout.
if (!request.integrationSha) {
return this.inconclusive(assertionId, "no integration SHA available to materialize a trusted checkout");
}
// R19: validate any agent-supplied proof path BEFORE doing any work.
let validatedTestPath: string | undefined;
if (request.proof) {
const safe = validateTestPath(request.proof.testFilePath);
if (safe === null) {
return this.inconclusive(
assertionId,
`agent-supplied test path rejected (invalid or contains shell metacharacters): ${JSON.stringify(request.proof.testFilePath)}`,
);
}
validatedTestPath = safe;
}
// R18: select an isolating backend, fail closed when none is available.
const probe = await this.probeBackends();
const selection = selectIsolatingBackend(probe);
if (selection.backendId === null) {
return this.inconclusive(assertionId, selection.reason ?? "no isolating sandbox backend available");
}
const command = buildVerificationCommand(this.commandTemplate, validatedTestPath);
const scrubbedEnv = scrubEnv(this.envSource);
const logTaskId = request.taskId ?? `verify-${assertionId}`;
// Route runVerificationCommand through the explicitly-selected isolating
// backend rather than the no-arg native fallback (R18). runVerificationCommand
// resolves its backend via the no-arg resolveSandboxBackend(), so we pin the
// selected isolating backend via the test-override hook for the duration of
// the run and unconditionally restore afterwards.
const isolating = this.backendFactory(selection.backendId);
const restoreBackend = () => __setSandboxBackendForTests(null);
__setSandboxBackendForTests(isolating);
let implCheckout: DisposableCheckout | undefined;
let baselineCheckout: DisposableCheckout | undefined;
try {
implCheckout = await this.materializer.materialize(this.rootDir, request.integrationSha);
const implResult = await runVerificationCommand(
this.store,
implCheckout.dir,
logTaskId,
command,
"test",
request.signal,
verifyLog,
"reviewer",
scrubbedEnv,
);
// R5/AE5: agent-supplied proof must fail on the merge-base baseline and
// pass on the implementation. A test that passes on both is not exercising
// the defect — reject it.
if (validatedTestPath) {
if (!request.mergeBaseSha) {
return this.inconclusive(assertionId, "no merge-base SHA available to validate agent-supplied proof");
}
baselineCheckout = await this.materializer.materialize(this.rootDir, request.mergeBaseSha);
const baselineResult = await runVerificationCommand(
this.store,
baselineCheckout.dir,
logTaskId,
command,
"test",
request.signal,
verifyLog,
"reviewer",
scrubbedEnv,
);
if (baselineResult.success && implResult.success) {
return {
verdict: "fail",
assertionId,
reason: "agent-supplied proof passes on both the pre-fix baseline and the implementation; it does not exercise the defect",
detail: "pass-on-both rejected (R5/AE5)",
};
}
if (!baselineResult.success && implResult.success) {
return { verdict: "pass", assertionId, reason: "regression test fails on the pre-fix baseline and passes on the implementation" };
}
// Fails on the implementation → defect still reproduces.
return {
verdict: "fail",
assertionId,
reason: "regression test does not pass on the implementation; behavior not confirmed",
detail: implResult.stderr || implResult.stdout || undefined,
};
}
// Whole-suite channel: pass only when the suite passes.
if (implResult.success) {
return { verdict: "pass", assertionId, reason: "verification suite passed on the implementation checkout" };
}
return {
verdict: "fail",
assertionId,
reason: "verification suite failed on the implementation checkout; behavior not confirmed",
detail: implResult.stderr || implResult.stdout || undefined,
};
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
// R9: any setup/exec failure (timeout, abort, materialization error) is a
// non-pass; we route it to inconclusive (infra, not behavioral).
return this.inconclusive(assertionId, `verification run could not complete: ${message}`);
} finally {
restoreBackend();
await implCheckout?.dispose();
await baselineCheckout?.dispose();
// R17: the source tree feeding diff/merge must be byte-clean afterwards.
try {
await this.materializer.assertSourceClean(this.rootDir);
} catch (cleanErr) {
verifyLog.error("Verification post-condition violated (source not git-clean):", cleanErr);
throw cleanErr instanceof Error ? cleanErr : new Error(String(cleanErr));
}
}
}
private inconclusive(assertionId: string, reason: string): VerificationOutcome {
return { verdict: "inconclusive", assertionId, reason };
}
}