From 483c71b014d59bb98422a67e0a9c725be0c575ea Mon Sep 17 00:00:00 2001 From: gsxdsm Date: Fri, 12 Jun 2026 01:16:34 -0700 Subject: [PATCH] feat(engine): default-to-fail posture + non-mutating verification run (U2,U3) Behavioral/bug assertions now default to fail unless a bounded verification run confirms them; static assertions keep the existing read-only judging path. Adds mission-verification.ts (injected capability): disposable git checkout at a trusted revision, explicit isolating sandbox backend with fail-closed + scrubbed env, fixed command template + validated test path, merge-base pre-fix baseline (rejects pass-on-both), git-clean post-condition, inconclusive as a first-class verdict. Judge session stays tools:readonly. --- packages/core/src/index.ts | 4 + .../__tests__/mission-verification.test.ts | 350 ++++++++++++ ...ssion-validator-behavioral-posture.test.ts | 402 ++++++++++++++ packages/engine/src/mission-execution-loop.ts | 190 ++++++- packages/engine/src/mission-verification.ts | 515 ++++++++++++++++++ 5 files changed, 1456 insertions(+), 5 deletions(-) create mode 100644 packages/engine/src/__tests__/mission-verification.test.ts create mode 100644 packages/engine/src/__tests__/reliability-interactions/mission-validator-behavioral-posture.test.ts create mode 100644 packages/engine/src/mission-verification.ts diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 4cc5e8b585..06a8b8c618 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -1070,6 +1070,9 @@ export { FEATURE_LOOP_STATES, VALIDATOR_RUN_STATUSES, MISSION_ASSERTION_STATUSES, + MISSION_ASSERTION_TYPES, + DEFAULT_MISSION_ASSERTION_TYPE, + normalizeMissionAssertionType, MILESTONE_VALIDATION_STATES, } from "./mission-types.js"; export type { @@ -1116,6 +1119,7 @@ export type { MissionFeatureLoopSnapshot, // Contract assertion types MissionAssertionStatus, + MissionAssertionType, MilestoneValidationState, MissionContractAssertion, FeatureAssertionLink, diff --git a/packages/engine/src/__tests__/mission-verification.test.ts b/packages/engine/src/__tests__/mission-verification.test.ts new file mode 100644 index 0000000000..7b32226daa --- /dev/null +++ b/packages/engine/src/__tests__/mission-verification.test.ts @@ -0,0 +1,350 @@ +/** + * Tests for the behavioral-verification capability (U3). + * + * Covers the isolation/safety contract before any real execution: + * - command-template rejects shell metacharacters (R19) + * - fail-closed when no isolating sandbox backend is available (R18) + * - env scrubbed to a minimal allowlist (R18) + * - pass-on-both regression test rejected (R5/AE5) + * - source tree git-clean post-condition asserted (R17) + * - no integration SHA → inconclusive (R11 fail-closed) + */ + +import { describe, it, expect, vi, afterEach } from "vitest"; +import { + validateTestPath, + buildVerificationCommand, + selectIsolatingBackend, + scrubEnv, + VERIFICATION_ENV_ALLOWLIST, + TestExecutionVerificationCapability, + type CheckoutMaterializer, + type IsolatingBackendProbe, + type VerificationRequest, +} from "../mission-verification.js"; +import { + __resetSandboxBackendForTests, + type SandboxBackend, + type SandboxStreamingResult, +} from "../sandbox/index.js"; +import type { TaskStore } from "@fusion/core"; + +afterEach(() => { + __resetSandboxBackendForTests(); + vi.restoreAllMocks(); +}); + +// ── validateTestPath (R19) ──────────────────────────────────────────────────── + +describe("validateTestPath", () => { + it("accepts a plain relative test path", () => { + expect(validateTestPath("packages/engine/src/__tests__/foo.test.ts")).toBe( + "packages/engine/src/__tests__/foo.test.ts", + ); + }); + + it.each([ + "foo.test.ts; rm -rf /", + "foo.test.ts && curl evil", + "foo.test.ts | cat", + "$(whoami).test.ts", + "`id`.test.ts", + "foo.test.ts\nrm x", + "a${b}.test.ts", + "foo>out.test.ts", + ])("rejects shell metacharacters: %s", (p) => { + expect(validateTestPath(p)).toBeNull(); + }); + + it("rejects absolute paths, parent escapes, flag-like, and non-strings", () => { + expect(validateTestPath("/etc/passwd")).toBeNull(); + expect(validateTestPath("../../etc/passwd")).toBeNull(); + expect(validateTestPath("a/../../b")).toBeNull(); + expect(validateTestPath("--config=evil")).toBeNull(); + expect(validateTestPath("")).toBeNull(); + expect(validateTestPath(42 as unknown)).toBeNull(); + }); +}); + +describe("buildVerificationCommand", () => { + it("substitutes a validated test path into the template", () => { + expect(buildVerificationCommand("pnpm vitest run {testPath}", "src/a.test.ts")).toBe( + "pnpm vitest run src/a.test.ts", + ); + }); + + it("produces a whole-suite command when no path is supplied", () => { + expect(buildVerificationCommand("pnpm vitest run {testPath}")).toBe("pnpm vitest run"); + }); + + it("throws if asked to substitute an unsafe path (defense in depth)", () => { + expect(() => buildVerificationCommand("pnpm vitest run {testPath}", "a; rm -rf /")).toThrow(); + }); +}); + +// ── selectIsolatingBackend (R18 fail-closed) ─────────────────────────────────── + +describe("selectIsolatingBackend", () => { + it("selects bubblewrap on linux when available", () => { + expect( + selectIsolatingBackend({ platform: "linux", bubblewrapAvailable: true, sandboxExecAvailable: false }).backendId, + ).toBe("bubblewrap"); + }); + + it("selects sandbox-exec on darwin when available", () => { + expect( + selectIsolatingBackend({ platform: "darwin", bubblewrapAvailable: false, sandboxExecAvailable: true }).backendId, + ).toBe("sandbox-exec"); + }); + + it("fails closed (null) when no isolating backend is available", () => { + const sel = selectIsolatingBackend({ platform: "linux", bubblewrapAvailable: false, sandboxExecAvailable: false }); + expect(sel.backendId).toBeNull(); + expect(sel.reason).toMatch(/no isolating sandbox backend/); + }); +}); + +// ── scrubEnv (R18) ───────────────────────────────────────────────────────────── + +describe("scrubEnv", () => { + it("keeps only allowlisted keys and forces CI=1", () => { + const result = scrubEnv({ + PATH: "/usr/bin", + HOME: "/home/u", + ANTHROPIC_API_KEY: "secret", + DATABASE_URL: "postgres://secret", + AWS_SECRET_ACCESS_KEY: "secret", + }); + expect(result.PATH).toBe("/usr/bin"); + expect(result.HOME).toBe("/home/u"); + expect(result.CI).toBe("1"); + expect(result.ANTHROPIC_API_KEY).toBeUndefined(); + expect(result.DATABASE_URL).toBeUndefined(); + expect(result.AWS_SECRET_ACCESS_KEY).toBeUndefined(); + }); + + it("only ever emits allowlisted keys plus CI", () => { + const result = scrubEnv({ FOO: "x", BAR: "y", PATH: "/bin" }); + const allowed = new Set([...VERIFICATION_ENV_ALLOWLIST, "CI"]); + for (const k of Object.keys(result)) { + expect(allowed.has(k)).toBe(true); + } + }); +}); + +// ── TestExecutionVerificationCapability ──────────────────────────────────────── + +function makeMockStore(): TaskStore { + return { + logEntry: vi.fn().mockResolvedValue(undefined), + appendAgentLog: vi.fn().mockResolvedValue(undefined), + } as unknown as TaskStore; +} + +/** A fake materializer that records dispose/clean calls without touching git. */ +function makeFakeMaterializer(): CheckoutMaterializer & { + disposed: number; + cleanCalls: number; + setDirty(dirty: boolean): void; +} { + let dirty = false; + const state = { + disposed: 0, + cleanCalls: 0, + setDirty(d: boolean) { + dirty = d; + }, + async materialize(_rootDir: string, _revision: string) { + return { + dir: "/tmp/fake-checkout", + dispose: async () => { + state.disposed += 1; + }, + }; + }, + async assertSourceClean(_rootDir: string) { + state.cleanCalls += 1; + if (dirty) throw new Error("Source tree is not git-clean after verification run"); + }, + }; + return state; +} + +/** A sandbox backend whose runStreaming returns a scripted outcome. */ +function makeScriptedBackend(outcomes: SandboxStreamingResult[]): SandboxBackend { + let i = 0; + return { + capabilities: () => ({ + id: "bubblewrap", + supportsNetworkPolicy: true, + supportsFilesystemPolicy: true, + supportsStreaming: true, + platform: "any", + }), + prepare: vi.fn().mockResolvedValue(undefined), + run: vi.fn(), + runStreaming: vi.fn(async () => { + const out = outcomes[Math.min(i, outcomes.length - 1)]; + i += 1; + return out; + }), + dispose: vi.fn().mockResolvedValue(undefined), + } as unknown as SandboxBackend; +} + +const isolatingProbe = (): Promise => + Promise.resolve({ platform: "linux", bubblewrapAvailable: true, sandboxExecAvailable: false }); + +function baseRequest(overrides: Partial = {}): VerificationRequest { + return { + assertionId: "CA-1", + assertion: "the bug no longer reproduces", + taskId: "FN-1", + integrationSha: "abc123", + ...overrides, + }; +} + +describe("TestExecutionVerificationCapability", () => { + it("fails closed to inconclusive when no isolating backend is available (R18)", async () => { + const materializer = makeFakeMaterializer(); + const cap = new TestExecutionVerificationCapability({ + store: makeMockStore(), + rootDir: "/repo", + commandTemplate: "pnpm vitest run {testPath}", + materializer, + probeBackends: async () => ({ platform: "linux", bubblewrapAvailable: false, sandboxExecAvailable: false }), + }); + + const outcome = await cap.verifyBehavioralAssertion(baseRequest()); + expect(outcome.verdict).toBe("inconclusive"); + expect(outcome.reason).toMatch(/no isolating sandbox backend/); + // Never materialized / executed. + expect(materializer.disposed).toBe(0); + }); + + it("returns inconclusive when no integration SHA is available (R11)", async () => { + const cap = new TestExecutionVerificationCapability({ + store: makeMockStore(), + rootDir: "/repo", + commandTemplate: "pnpm vitest run {testPath}", + materializer: makeFakeMaterializer(), + probeBackends: isolatingProbe, + }); + const outcome = await cap.verifyBehavioralAssertion(baseRequest({ integrationSha: undefined })); + expect(outcome.verdict).toBe("inconclusive"); + expect(outcome.reason).toMatch(/integration SHA/); + }); + + it("rejects an agent test path with shell metacharacters before execution (R19)", async () => { + const materializer = makeFakeMaterializer(); + const cap = new TestExecutionVerificationCapability({ + store: makeMockStore(), + rootDir: "/repo", + commandTemplate: "pnpm vitest run {testPath}", + materializer, + probeBackends: isolatingProbe, + }); + const outcome = await cap.verifyBehavioralAssertion( + baseRequest({ proof: { testFilePath: "a.test.ts; rm -rf /" } }), + ); + expect(outcome.verdict).toBe("inconclusive"); + expect(outcome.reason).toMatch(/shell metacharacters|rejected/); + expect(materializer.disposed).toBe(0); + }); + + it("passes when the whole-suite run succeeds, and asserts source git-clean (R17)", async () => { + const materializer = makeFakeMaterializer(); + const cap = new TestExecutionVerificationCapability({ + store: makeMockStore(), + rootDir: "/repo", + commandTemplate: "pnpm vitest run {testPath}", + materializer, + probeBackends: isolatingProbe, + backendFactory: () => + makeScriptedBackend([{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }]), + }); + const outcome = await cap.verifyBehavioralAssertion(baseRequest()); + expect(outcome.verdict).toBe("pass"); + expect(materializer.disposed).toBe(1); + expect(materializer.cleanCalls).toBe(1); + }); + + it("rejects a regression test that passes on BOTH baseline and implementation (R5/AE5)", async () => { + const materializer = makeFakeMaterializer(); + const cap = new TestExecutionVerificationCapability({ + store: makeMockStore(), + rootDir: "/repo", + commandTemplate: "pnpm vitest run {testPath}", + materializer, + probeBackends: isolatingProbe, + // impl run (1st) success, baseline run (2nd) success → pass-on-both. + backendFactory: () => + makeScriptedBackend([ + { outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }, + { outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }, + ]), + }); + const outcome = await cap.verifyBehavioralAssertion( + baseRequest({ proof: { testFilePath: "src/a.test.ts" }, mergeBaseSha: "base000" }), + ); + expect(outcome.verdict).toBe("fail"); + expect(outcome.reason).toMatch(/both/); + }); + + it("passes when a regression test fails on baseline and passes on implementation (R5)", async () => { + const materializer = makeFakeMaterializer(); + const cap = new TestExecutionVerificationCapability({ + store: makeMockStore(), + rootDir: "/repo", + commandTemplate: "pnpm vitest run {testPath}", + materializer, + probeBackends: isolatingProbe, + // impl run (1st) success, baseline run (2nd) non-zero-exit → genuine proof. + backendFactory: () => + makeScriptedBackend([ + { outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }, + { outcome: "non-zero-exit", stdout: "", stderr: "boom", exitCode: 1, signal: null }, + ]), + }); + const outcome = await cap.verifyBehavioralAssertion( + baseRequest({ proof: { testFilePath: "src/a.test.ts" }, mergeBaseSha: "base000" }), + ); + expect(outcome.verdict).toBe("pass"); + }); + + it("inconclusive when the run times out (R9), still asserts source clean", async () => { + const materializer = makeFakeMaterializer(); + const cap = new TestExecutionVerificationCapability({ + store: makeMockStore(), + rootDir: "/repo", + commandTemplate: "pnpm vitest run {testPath}", + materializer, + probeBackends: isolatingProbe, + backendFactory: () => + makeScriptedBackend([{ outcome: "timeout", stdout: "", stderr: "", timeoutMs: 1000 }]), + }); + // A timeout surfaces as a thrown ETIMEDOUT inside runVerificationCommand, + // which the capability catches and maps to a non-pass. For the whole-suite + // channel a non-success result is a behavioral fail; but a timeout throw is + // caught by the capability's try/catch → inconclusive. + const outcome = await cap.verifyBehavioralAssertion(baseRequest()); + expect(["inconclusive", "fail"]).toContain(outcome.verdict); + expect(materializer.cleanCalls).toBe(1); + }); + + it("throws if the source tree is dirty after a run (R17 post-condition)", async () => { + const materializer = makeFakeMaterializer(); + materializer.setDirty(true); + const cap = new TestExecutionVerificationCapability({ + store: makeMockStore(), + rootDir: "/repo", + commandTemplate: "pnpm vitest run {testPath}", + materializer, + probeBackends: isolatingProbe, + backendFactory: () => + makeScriptedBackend([{ outcome: "success", stdout: "ok", stderr: "", bufferOverflow: false }]), + }); + await expect(cap.verifyBehavioralAssertion(baseRequest())).rejects.toThrow(/git-clean/); + }); +}); diff --git a/packages/engine/src/__tests__/reliability-interactions/mission-validator-behavioral-posture.test.ts b/packages/engine/src/__tests__/reliability-interactions/mission-validator-behavioral-posture.test.ts new file mode 100644 index 0000000000..028599a981 --- /dev/null +++ b/packages/engine/src/__tests__/reliability-interactions/mission-validator-behavioral-posture.test.ts @@ -0,0 +1,402 @@ +/** + * Behavioral-verification posture in the Validator Run (U2 + U3). + * + * Verifies the default-to-fail posture for behavioral assertions and the + * non-mutating verification step that confirms them, while static assertions + * keep the exact legacy judge path. + * + * Covers: + * - AE2: behavioral assertion with no verification evidence → fail, even when + * the judge text claims pass (capability absent). + * - AE3: static assertion → unchanged static verdict, no verification invoked. + * - Mixed set: static and behavioral each take their correct path. + * - Behavioral pass via an injected verification capability. + * - Behavioral inconclusive → blocked verdict, NO fix feature. + */ + +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; +import type { + Mission, + Milestone, + Slice, + MissionFeature, + MissionValidatorRun, +} from "@fusion/core"; +import type { VerificationCapability, VerificationOutcome } from "../../mission-verification.js"; + +// ── Mock AI dependencies (mirror mission-execution-loop.test.ts) ─────────────── +const mockSessionHolder: { + session: { state: { messages: Array<{ role: string; content: string }> }; dispose: ReturnType }; +} = { session: { state: { messages: [] }, dispose: vi.fn() } }; + +vi.mock("../../pi.js", () => ({ + createFnAgent: vi.fn(() => Promise.resolve({ session: mockSessionHolder.session })), + promptWithFallback: vi.fn().mockResolvedValue(undefined), +})); + +vi.mock("../../logger.js", () => ({ + createLogger: vi.fn(() => ({ log: vi.fn(), warn: vi.fn(), error: vi.fn() })), +})); + +vi.mock("../../agent-session-helpers.js", async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + createResolvedAgentSession: vi.fn(async () => ({ + session: mockSessionHolder.session as any, + sessionFile: undefined, + runtimeId: "test-runtime", + wasConfigured: true, + })), + }; +}); + +import { createResolvedAgentSession } from "../../agent-session-helpers.js"; +import { MissionExecutionLoop } from "../../mission-execution-loop.js"; + +type AssertionRow = { + id: string; + milestoneId: string; + title: string; + assertion: string; + status: "pending" | "passed" | "failed" | "blocked"; + type?: "static" | "behavioral"; + orderIndex: number; + createdAt: string; + updatedAt: string; + sourceFeatureId?: string; +}; + +function now() { + return new Date().toISOString(); +} + +function createMockMission(): Mission { + return { + id: "M-TEST1", + title: "Test Mission", + status: "active", + interviewState: "not_started", + autopilotEnabled: true, + autopilotState: "inactive", + createdAt: now(), + updatedAt: now(), + }; +} + +function createMockMilestone(overrides: Partial = {}): Milestone { + return { + id: "MS-001", + missionId: "M-TEST1", + title: "Test Milestone", + status: "active", + orderIndex: 0, + interviewState: "not_started", + dependencies: [], + createdAt: now(), + updatedAt: now(), + ...overrides, + }; +} + +function createMockSlice(overrides: Partial = {}): Slice { + return { + id: "SL-001", + milestoneId: "MS-001", + title: "Test Slice", + status: "active", + planState: "not_started", + orderIndex: 0, + createdAt: now(), + updatedAt: now(), + ...overrides, + }; +} + +function createMockFeature(overrides: Partial = {}): MissionFeature { + return { + id: "F-001", + sliceId: "SL-001", + title: "Test Feature", + status: "defined", + loopState: "idle", + implementationAttemptCount: 0, + validatorAttemptCount: 0, + createdAt: now(), + updatedAt: now(), + ...overrides, + }; +} + +function createMockMissionStore() { + const missions = new Map(); + const features = new Map(); + const assertionsByFeature = new Map(); + const validatorRuns = new Map(); + let runSeq = 0; + + const store = { + getMission: vi.fn((id: string) => missions.get(id)), + logMissionEvent: vi.fn(), + getFeature: vi.fn((id: string) => features.get(id)), + getFeatureByTaskId: vi.fn((taskId: string) => { + for (const f of features.values()) if (f.taskId === taskId) return f; + return undefined; + }), + updateFeatureStatus: vi.fn((id: string, status: MissionFeature["status"]) => { + const f = features.get(id)!; + const updated = { ...f, status, updatedAt: now() }; + features.set(id, updated); + return updated; + }), + listAssertionsForFeature: vi.fn((featureId: string) => assertionsByFeature.get(featureId) ?? []), + ensureFeatureAssertionLinked: vi.fn((featureId: string) => assertionsByFeature.get(featureId) ?? []), + getSlice: vi.fn((id: string) => createMockSlice({ id })), + getMilestone: vi.fn((id: string) => createMockMilestone({ id })), + startValidatorRun: vi.fn((featureId: string) => { + const run: MissionValidatorRun = { + id: `VR-${++runSeq}`, + featureId, + milestoneId: "MS-001", + sliceId: "SL-001", + status: "running", + triggerType: "task_completion", + implementationAttempt: 1, + validatorAttempt: 1, + startedAt: now(), + createdAt: now(), + updatedAt: now(), + }; + validatorRuns.set(run.id, run); + return run; + }), + getValidatorRun: vi.fn((id: string) => validatorRuns.get(id)), + completeValidatorRun: vi.fn((id: string, status: MissionValidatorRun["status"], summary?: string) => { + const run = validatorRuns.get(id)!; + const updated = { ...run, status, summary, completedAt: now(), updatedAt: now() }; + validatorRuns.set(id, updated); + const feature = features.get(run.featureId); + if (feature) { + const loopState = status === "passed" ? "passed" : status === "failed" ? "needs_fix" : status === "blocked" ? "blocked" : "validating"; + features.set(run.featureId, { ...feature, loopState: loopState as any, lastValidatorStatus: status, updatedAt: now() }); + } + return updated; + }), + recordValidatorFailures: vi.fn(() => []), + createGeneratedFixFeature: vi.fn((sourceFeatureId: string, runId: string) => { + const src = features.get(sourceFeatureId)!; + const fix = createMockFeature({ id: `FIX-${sourceFeatureId}`, taskId: `TASK-FIX-${sourceFeatureId}`, generatedFromFeatureId: sourceFeatureId, generatedFromRunId: runId, loopState: "implementing" }); + features.set(fix.id, fix); + features.set(sourceFeatureId, { ...src, loopState: "implementing", implementationAttemptCount: (src.implementationAttemptCount ?? 0) + 1, updatedAt: now() }); + return fix; + }), + triageFeature: vi.fn(async (featureId: string) => { + const f = features.get(featureId)!; + const updated = { ...f, status: "triaged" as const, updatedAt: now() }; + features.set(featureId, updated); + return updated; + }), + on: vi.fn(), + off: vi.fn(), + emit: vi.fn(), + _setMission: (m: Mission) => missions.set(m.id, m), + _setFeature: (f: MissionFeature) => features.set(f.id, f), + _setAssertions: (featureId: string, rows: AssertionRow[]) => assertionsByFeature.set(featureId, rows), + }; + return store; +} + +function createMockTaskStore() { + const tasks = new Map(); + return { + getTask: vi.fn(async (id: string) => tasks.get(id)), + moveTask: vi.fn(async () => {}), + updateTask: vi.fn(async () => {}), + getSettings: vi.fn().mockResolvedValue({ missionStaleThresholdMs: 600_000, missionMaxTaskRetries: 3 }), + recordRunAuditEvent: vi.fn(), + on: vi.fn(), + off: vi.fn(), + _setTask: (t: any) => tasks.set(t.id, t), + }; +} + +function assertionRow(overrides: Partial & { id: string }): AssertionRow { + return { + milestoneId: "MS-001", + title: overrides.id, + assertion: `do ${overrides.id}`, + status: "pending", + orderIndex: 0, + createdAt: now(), + updatedAt: now(), + ...overrides, + }; +} + +describe("Validator behavioral posture (U2 + U3)", () => { + let missionStore: ReturnType; + let taskStore: ReturnType; + let loop: MissionExecutionLoop; + + beforeEach(() => { + missionStore = createMockMissionStore(); + taskStore = createMockTaskStore(); + vi.mocked(createResolvedAgentSession).mockReset(); + vi.mocked(createResolvedAgentSession).mockResolvedValue({ + session: mockSessionHolder.session as any, + sessionFile: undefined, + runtimeId: "test-runtime", + wasConfigured: true, + }); + missionStore._setMission(createMockMission()); + mockSessionHolder.session.state.messages = []; + mockSessionHolder.session.dispose = vi.fn(); + }); + + afterEach(() => { + loop?.stop(); + vi.restoreAllMocks(); + }); + + function judgePass(assertionIds: string[]) { + mockSessionHolder.session.state.messages = [ + { + role: "assistant", + content: JSON.stringify({ + status: "pass", + assertions: assertionIds.map((id) => ({ assertionId: id, passed: true })), + summary: "all good", + }), + }, + ]; + } + + it("AE2: behavioral assertion the judge calls pass → fails with no verification capability", async () => { + const feature = createMockFeature({ loopState: "implementing", taskId: "FN-B", status: "in-progress" }); + missionStore._setFeature(feature); + missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]); + taskStore._setTask({ id: "FN-B", title: "behavioral", log: [] }); + judgePass(["CA-1"]); + + loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp" }); + loop.start(); + await loop.processTaskOutcome("FN-B"); + + // No verification capability → behavioral default-to-fail → fix flow. + expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "failed", expect.any(String)); + expect(missionStore.createGeneratedFixFeature).toHaveBeenCalled(); + expect(missionStore.getFeature("F-001")?.status).not.toBe("done"); + }); + + it("AE3: static assertion the judge calls pass → passes, no verification invoked", async () => { + const verify = vi.fn(); + const capability: VerificationCapability = { verifyBehavioralAssertion: verify }; + const feature = createMockFeature({ loopState: "implementing", taskId: "FN-S", status: "in-progress" }); + missionStore._setFeature(feature); + missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "static" })]); + taskStore._setTask({ id: "FN-S", title: "static", log: [] }); + judgePass(["CA-1"]); + + loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: capability }); + loop.start(); + await loop.processTaskOutcome("FN-S"); + + expect(verify).not.toHaveBeenCalled(); + expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String)); + expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done"); + }); + + it("untyped assertions default to static — legacy judge pass path is preserved", async () => { + const verify = vi.fn(); + const feature = createMockFeature({ loopState: "implementing", taskId: "FN-U", status: "in-progress" }); + missionStore._setFeature(feature); + // No `type` field → normalizes to static. + missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1" })]); + taskStore._setTask({ id: "FN-U", title: "untyped", log: [] }); + judgePass(["CA-1"]); + + loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } }); + loop.start(); + await loop.processTaskOutcome("FN-U"); + + expect(verify).not.toHaveBeenCalled(); + expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String)); + }); + + it("behavioral assertion confirmed by an injected verification capability → passes", async () => { + const verify = vi.fn(async (req): Promise => ({ verdict: "pass", assertionId: req.assertionId, reason: "confirmed" })); + const feature = createMockFeature({ loopState: "implementing", taskId: "FN-BV", status: "in-progress" }); + missionStore._setFeature(feature); + missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]); + taskStore._setTask({ id: "FN-BV", title: "behavioral verified", integrationSha: "sha123", log: [] }); + judgePass(["CA-1"]); + + loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } }); + loop.start(); + await loop.processTaskOutcome("FN-BV"); + + expect(verify).toHaveBeenCalledTimes(1); + expect(verify.mock.calls[0][0]).toMatchObject({ assertionId: "CA-1", integrationSha: "sha123" }); + expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String)); + expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done"); + }); + + it("behavioral assertion verification inconclusive → blocked, NO fix feature", async () => { + const verify = vi.fn(async (req): Promise => ({ verdict: "inconclusive", assertionId: req.assertionId, reason: "no isolating sandbox backend" })); + const feature = createMockFeature({ loopState: "implementing", taskId: "FN-INC", status: "in-progress" }); + missionStore._setFeature(feature); + missionStore._setAssertions("F-001", [assertionRow({ id: "CA-1", type: "behavioral" })]); + taskStore._setTask({ id: "FN-INC", title: "inconclusive", integrationSha: "sha123", log: [] }); + judgePass(["CA-1"]); + + loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } }); + loop.start(); + await loop.processTaskOutcome("FN-INC"); + + expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "blocked", expect.any(String)); + expect(missionStore.createGeneratedFixFeature).not.toHaveBeenCalled(); + expect(missionStore.getFeature("F-001")?.status).not.toBe("done"); + }); + + it("mixed set: static passes via judge, behavioral confirmed via verification → overall pass", async () => { + const verify = vi.fn(async (req): Promise => ({ verdict: "pass", assertionId: req.assertionId, reason: "confirmed" })); + const feature = createMockFeature({ loopState: "implementing", taskId: "FN-MIX", status: "in-progress" }); + missionStore._setFeature(feature); + missionStore._setAssertions("F-001", [ + assertionRow({ id: "CA-static", type: "static" }), + assertionRow({ id: "CA-behav", type: "behavioral" }), + ]); + taskStore._setTask({ id: "FN-MIX", title: "mixed", integrationSha: "sha123", log: [] }); + judgePass(["CA-static", "CA-behav"]); + + loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } }); + loop.start(); + await loop.processTaskOutcome("FN-MIX"); + + // Only the behavioral assertion is verified. + expect(verify).toHaveBeenCalledTimes(1); + expect(verify.mock.calls[0][0]).toMatchObject({ assertionId: "CA-behav" }); + expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "passed", expect.any(String)); + expect(missionStore.updateFeatureStatus).toHaveBeenCalledWith("F-001", "done"); + }); + + it("mixed set: behavioral observed wrong → overall fail even though static passes", async () => { + const verify = vi.fn(async (req): Promise => ({ verdict: "fail", assertionId: req.assertionId, reason: "defect still reproduces" })); + const feature = createMockFeature({ loopState: "implementing", taskId: "FN-MIX2", status: "in-progress" }); + missionStore._setFeature(feature); + missionStore._setAssertions("F-001", [ + assertionRow({ id: "CA-static", type: "static" }), + assertionRow({ id: "CA-behav", type: "behavioral" }), + ]); + taskStore._setTask({ id: "FN-MIX2", title: "mixed fail", integrationSha: "sha123", log: [] }); + judgePass(["CA-static", "CA-behav"]); + + loop = new MissionExecutionLoop({ taskStore: taskStore as any, missionStore: missionStore as any, rootDir: "/tmp", verificationCapability: { verifyBehavioralAssertion: verify } }); + loop.start(); + await loop.processTaskOutcome("FN-MIX2"); + + expect(missionStore.completeValidatorRun).toHaveBeenCalledWith(expect.any(String), "failed", expect.any(String)); + expect(missionStore.createGeneratedFixFeature).toHaveBeenCalled(); + expect(missionStore.getFeature("F-001")?.status).not.toBe("done"); + }); +}); diff --git a/packages/engine/src/mission-execution-loop.ts b/packages/engine/src/mission-execution-loop.ts index cfd77f3e7e..2d4ae91d73 100644 --- a/packages/engine/src/mission-execution-loop.ts +++ b/packages/engine/src/mission-execution-loop.ts @@ -26,7 +26,9 @@ import { TEST_MODE_RESOLVED, isTestModeActive, resolveTaskValidatorModel, + normalizeMissionAssertionType, } from "@fusion/core"; +import type { VerificationOutcome } from "./mission-verification.js"; import { createFnAgent, promptWithFallback, type AgentResult } from "./pi.js"; import { mergeEffectiveSettings } from "./effective-settings.js"; import { @@ -50,8 +52,16 @@ const VALIDATION_TIMEOUT_MS = 10 * 60 * 1000; // 10 minutes * per assertion plus an overall status. */ export interface ValidationResult { - /** Overall validation status */ - status: "pass" | "fail" | "blocked" | "error"; + /** + * Overall validation status. + * + * `inconclusive` is first-class and distinct from `fail`: it means a + * behavioral verification run could not run or conclude (no isolating sandbox + * backend, timeout, setup failure, rejected proof). In this unit it routes to + * a blocked verdict (no remediation); later units track its infra-failure rate + * separately. + */ + status: "pass" | "fail" | "blocked" | "error" | "inconclusive"; /** Per-assertion results */ assertions: Array<{ assertionId: string; @@ -83,6 +93,14 @@ export interface MissionExecutionLoopOptions { pluginRunner?: import("./plugin-runner.js").PluginRunner; /** Optional agent store for resolving assigned-agent runtime hints. */ agentStore?: AgentStore; + /** + * Optional behavioral-verification capability (U3). When provided, behavioral + * assertions are confirmed by a non-mutating verification run; the judge's + * "pass" on a behavioral assertion is advisory only. When ABSENT, behavioral + * assertions still default to fail (U2) but no verification run is attempted — + * preserving the behavior of existing construction sites that inject nothing. + */ + verificationCapability?: import("./mission-verification.js").VerificationCapability; } export class MissionExecutionLoop extends EventEmitter { @@ -94,6 +112,7 @@ export class MissionExecutionLoop extends EventEmitter { private missionAutopilot?: MissionExecutionLoopOptions["missionAutopilot"]; private pluginRunner?: MissionExecutionLoopOptions["pluginRunner"]; private agentStore?: MissionExecutionLoopOptions["agentStore"]; + private verificationCapability?: MissionExecutionLoopOptions["verificationCapability"]; private activeValidations = new Set(); // feature IDs currently being validated constructor(options: MissionExecutionLoopOptions) { @@ -105,6 +124,7 @@ export class MissionExecutionLoop extends EventEmitter { this.missionAutopilot = options.missionAutopilot; this.pluginRunner = options.pluginRunner; this.agentStore = options.agentStore; + this.verificationCapability = options.verificationCapability; loopLog.log("MissionExecutionLoop created"); } @@ -434,8 +454,11 @@ export class MissionExecutionLoop extends EventEmitter { await this.handleValidationPass(feature.id, run.id, result.summary); } else if (result.status === "fail") { await this.handleValidationFail(feature.id, run.id, result); - } else if (result.status === "blocked") { - await this.handleValidationBlocked(feature.id, run.id, result.blockedReason); + } else if (result.status === "blocked" || result.status === "inconclusive") { + // Inconclusive (infra: no isolating backend, timeout, rejected proof) + // routes to blocked/needs-attention — distinct from a behavioral fail, + // and spawns no Fix Feature (per KTD / R21, fully realized in a later unit). + await this.handleValidationBlocked(feature.id, run.id, result.blockedReason ?? result.summary); } else if (result.status === "error") { await this.handleValidationError(feature.id, run.id, result.summary); } @@ -537,7 +560,12 @@ export class MissionExecutionLoop extends EventEmitter { // Get the validation result from the session // The agent should have returned structured JSON in its response - const result = await this.parseValidationResult(session.session, assertions); + const judgeResult = await this.parseValidationResult(session.session, assertions); + + // U2/U3: the read-only judge's verdict is authoritative for STATIC + // assertions only. BEHAVIORAL assertions default to fail and are confirmed + // (or refuted) by a non-mutating verification run instead. + const result = await this.applyBehavioralPosture(feature, assertions, judgeResult); loopLog.log(`Validation completed for feature ${feature.id}: ${result.status}`); return result; @@ -568,6 +596,158 @@ export class MissionExecutionLoop extends EventEmitter { } } + /** + * Apply the behavioral judging posture (U2/U3) to the read-only judge's + * verdict. + * + * - STATIC assertions keep the judge's verdict verbatim (no behavior change). + * - BEHAVIORAL assertions DEFAULT TO FAIL. The judge's "pass" on a behavioral + * assertion is advisory; an authoritative pass requires a verification run + * to confirm it. When a verification capability is injected, each behavioral + * assertion is run through it: pass → satisfied; fail → behavioral failure; + * inconclusive → the aggregate becomes inconclusive (infra, no remediation). + * When NO capability is injected, behavioral assertions simply stay failed + * (preserving existing call-site behavior — existing data is all static). + * + * The aggregate status is recomputed from the post-posture per-assertion + * results so the existing pass/fail/blocked/error/inconclusive flow is driven + * correctly. + */ + private async applyBehavioralPosture( + feature: MissionFeature, + assertions: MissionContractAssertion[], + judgeResult: ValidationResult, + ): Promise { + // Preserve non-behavioral terminal verdicts untouched (error/blocked from the + // judge are not behavioral posture concerns). + if (judgeResult.status === "error") { + return judgeResult; + } + + const typeById = new Map>(); + let hasBehavioral = false; + for (const a of assertions) { + const t = normalizeMissionAssertionType(a.type); + typeById.set(a.id, t); + if (t === "behavioral") hasBehavioral = true; + } + + // Fast path: no behavioral assertions → existing static path is preserved + // exactly. This keeps every existing (untyped/static) test green. + if (!hasBehavioral) { + return judgeResult; + } + + const textById = new Map(assertions.map((a) => [a.id, a.assertion])); + let sawInconclusive = false; + let inconclusiveReason: string | undefined; + + const newAssertionResults = await Promise.all( + judgeResult.assertions.map(async (judged) => { + const type = typeById.get(judged.assertionId) ?? "static"; + if (type !== "behavioral") { + // Static: keep judge verdict verbatim. + return judged; + } + + // Behavioral: default to fail unless verification confirms it. + if (!this.verificationCapability) { + return { + ...judged, + passed: false, + message: "Behavioral assertion defaults to fail: no verification evidence (advisory judge verdict is not authoritative).", + expected: judged.expected ?? "Behavior confirmed by a verification run", + actual: judged.actual ?? "No verification run was performed", + }; + } + + let outcome: VerificationOutcome; + try { + outcome = await this.verificationCapability.verifyBehavioralAssertion({ + assertionId: judged.assertionId, + assertion: textById.get(judged.assertionId) ?? "", + taskId: feature.taskId, + integrationSha: await this.resolveIntegrationSha(feature), + signal: undefined, + }); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + loopLog.warn(`Verification capability threw for assertion ${judged.assertionId}: ${message}`); + outcome = { verdict: "inconclusive", assertionId: judged.assertionId, reason: `verification error: ${message}` }; + } + + if (outcome.verdict === "pass") { + return { ...judged, passed: true, message: outcome.reason }; + } + if (outcome.verdict === "inconclusive") { + sawInconclusive = true; + inconclusiveReason = inconclusiveReason ?? outcome.reason; + return { + ...judged, + passed: false, + message: `Behavioral verification inconclusive: ${outcome.reason}`, + expected: judged.expected ?? "Behavior confirmed by a verification run", + actual: outcome.detail ?? "Verification could not conclude", + }; + } + // fail + return { + ...judged, + passed: false, + message: outcome.reason, + expected: judged.expected ?? "Behavior confirmed by a verification run", + actual: outcome.detail ?? judged.actual ?? "Behavior not confirmed", + }; + }), + ); + + const allPassed = newAssertionResults.every((a) => a.passed); + + // Inconclusive takes precedence over fail: an infra-driven non-pass must not + // be mistaken for an observed behavioral failure (no Fix Feature). + let status: ValidationResult["status"]; + if (sawInconclusive && !allPassed) { + status = "inconclusive"; + } else if (allPassed) { + status = "pass"; + } else { + status = "fail"; + } + + const summary = status === "pass" + ? judgeResult.summary + : status === "inconclusive" + ? `Behavioral verification inconclusive: ${inconclusiveReason ?? "verification could not conclude"}` + : "One or more behavioral assertions were not confirmed by verification."; + + return { + status, + assertions: newAssertionResults, + summary, + blockedReason: status === "inconclusive" ? (inconclusiveReason ?? "verification inconclusive") : judgeResult.blockedReason, + }; + } + + /** + * Resolve the trusted revision (integration SHA) whose disposable checkout the + * verification run executes against. The live task worktree is pruned before + * the done-transition that triggers validation, so it cannot be used. + * + * In this unit we read it from the linked task when available; callers that do + * not supply a resolvable SHA cause the verification run to resolve to + * inconclusive (fail-closed). A richer derivation is owned by a later unit. + */ + private async resolveIntegrationSha(feature: MissionFeature): Promise { + if (!feature.taskId) return undefined; + try { + const task = await this.taskStore.getTask(feature.taskId); + const candidate = (task as { integrationSha?: string; baseCommit?: string } | undefined); + return candidate?.integrationSha ?? candidate?.baseCommit ?? undefined; + } catch { + return undefined; + } + } + private resolveValidationSessionModel( task: Awaited> | null, settings: Partial | undefined, diff --git a/packages/engine/src/mission-verification.ts b/packages/engine/src/mission-verification.ts new file mode 100644 index 0000000000..4f8863d9f4 --- /dev/null +++ b/packages/engine/src/mission-verification.ts @@ -0,0 +1,515 @@ +/** + * Mission behavioral-verification capability (U3). + * + * The Validator Run's read-only AI judge cannot run code, so its "pass" on a + * *behavioral* assertion is advisory only (U2). This module supplies the + * authoritative, NON-MUTATING verification step that confirms a behavioral/bug + * assertion by exercising the implemented code. + * + * Channels: + * - **test-execution** (this unit): run the project's scoped test suite / an + * agent-supplied regression test against a disposable checkout at a trusted + * revision, through an explicit isolating sandbox backend. + * - **app-driving** (later unit U5/U8): drive a running app instance. Not + * implemented here — the capability surface is structured so it can be added + * without reshaping callers. + * + * Safety invariants enforced here (the boundary, not a convention): + * - R18: execute under an *isolating* sandbox backend (bubblewrap / sandbox-exec) + * with a scrubbed env allowlist; FAIL CLOSED to a non-pass when no isolating + * backend is available — never fall through to the unrestricted native backend. + * - R19: the command is built from a fixed, system-owned template into which only + * a validated test-file path is substituted; shell metacharacters are rejected. + * - R11/R17: verification runs against a disposable checkout at the integration + * SHA (never the pruned live worktree, never the repo root); the source tree + * that feeds diff/merge is asserted git-clean after a run. + * - R5/AE5: agent-supplied proof must FAIL on a second disposable checkout at + * `git merge-base` (a revision the agent does not control) and PASS on the + * implementation; a test that passes on both is rejected. + * - R9: inconclusive / timeout / setup failure resolves to a non-pass. + * - R10: no board / mission writes happen here. + */ + +import { promises as fs } from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { exec } from "node:child_process"; +import { promisify } from "node:util"; +import type { TaskStore } from "@fusion/core"; +import type { SandboxCapabilities } from "./sandbox/index.js"; +import { __setSandboxBackendForTests, resolveSandboxBackend } from "./sandbox/index.js"; +import type { SandboxBackend } from "./sandbox/index.js"; +import { detectBwrap } from "./sandbox/bubblewrap-detect.js"; +import { detectSandboxExec } from "./sandbox/sandbox-exec-detect.js"; +import { runVerificationCommand } from "./verification-utils.js"; +import { createLogger } from "./logger.js"; + +const execAsync = promisify(exec); +const verifyLog = createLogger("mission-verify"); + +// ── Verdict types ─────────────────────────────────────────────────────────── + +/** + * Outcome of a verification run for a single behavioral assertion. + * + * - `pass`: behavior confirmed by execution. + * - `fail`: behavior observed wrong (the defect still reproduces / proof rejected). + * - `inconclusive`: verification could not run or conclude (no isolating backend, + * timeout, setup failure, rejected/invalid proof input). First-class and + * distinct from `fail`: it must NOT spawn remediation (handled by later units), + * but in this unit it never resolves to a default pass either. + */ +export type VerificationVerdict = "pass" | "fail" | "inconclusive"; + +/** Why a verification run reached its verdict (for durable observability later). */ +export interface VerificationOutcome { + verdict: VerificationVerdict; + /** Human-readable reason, suitable for surfacing in a failure record. */ + reason: string; + /** The assertion this outcome corresponds to. */ + assertionId: string; + /** Optional summarized command output for the failure record. */ + detail?: string; +} + +/** Shape of agent-supplied executable proof (a regression test). */ +export interface VerificationProof { + /** + * Path to the regression test file, relative to the checkout root. Validated + * to reject shell metacharacters and path escapes before use (R19). + */ + testFilePath: string; +} + +/** Input describing a single behavioral assertion to verify. */ +export interface VerificationRequest { + assertionId: string; + /** The assertion text (for logging / reason building). */ + assertion: string; + /** Board task id associated with the feature, used for verification-command logging. */ + taskId?: string; + /** + * The trusted revision (integration SHA) whose checkout the implementation is + * verified against. When absent, verification is inconclusive (cannot + * materialize a trusted checkout). + */ + integrationSha?: string; + /** + * The `git merge-base` revision (feature branch vs base branch) used as the + * pre-fix baseline for agent-supplied proof. Not agent-controlled. + */ + mergeBaseSha?: string; + /** Optional agent-supplied executable proof. */ + proof?: VerificationProof; + /** Abort signal to bound the run. */ + signal?: AbortSignal; +} + +/** + * Injected verification capability. Mirrors the `createFnAgent` injection + * pattern so MissionExecutionLoop can swap a real implementation for a mock in + * tests. Optional on the loop: when absent, behavioral assertions resolve to a + * non-pass without invoking any execution (preserving existing behavior for + * call sites that do not inject a capability). + */ +export interface VerificationCapability { + verifyBehavioralAssertion(request: VerificationRequest): Promise; +} + +// ── Command-template safety (R19) ───────────────────────────────────────────── + +/** + * Characters that could break out of the fixed command template or inject + * additional shell behavior. Agent-supplied test paths containing any of these + * are rejected before execution. + */ +const SHELL_METACHARACTERS = /[;&|`$(){}<>!*?\[\]\\"'\n\r\t\0]/; + +/** + * Validate an agent-supplied test-file path. Returns the normalized path when + * safe, or `null` when it must be rejected (R19). + * + * Rejects: empty, absolute paths, parent-dir escapes, shell metacharacters, + * and leading dashes (which could be read as command flags). + */ +export function validateTestPath(rawPath: unknown): string | null { + if (typeof rawPath !== "string") return null; + const path = rawPath.trim(); + if (path.length === 0) return null; + if (SHELL_METACHARACTERS.test(path)) return null; + if (path.startsWith("/")) return null; // must be relative to the checkout + if (path.startsWith("-")) return null; // could be parsed as a flag + // Reject parent-dir escapes (any `..` segment). + const segments = path.split("/"); + if (segments.some((seg) => seg === "..")) return null; + return path; +} + +/** + * Build the verification command from the fixed system-owned template. Only a + * pre-validated test path may be substituted (R19). Callers MUST pass a path + * already run through {@link validateTestPath}; this function re-checks and + * throws on violation as a defense-in-depth guard. + */ +export function buildVerificationCommand(template: string, validatedTestPath?: string): string { + if (validatedTestPath !== undefined) { + if (validateTestPath(validatedTestPath) === null) { + throw new Error(`Refusing to build verification command: invalid test path ${JSON.stringify(validatedTestPath)}`); + } + if (!template.includes("{testPath}")) { + throw new Error("Verification command template must contain a {testPath} placeholder when a test path is supplied"); + } + return template.replace("{testPath}", validatedTestPath); + } + // Whole-suite invocation: the template must not reference a test path. + return template.replace("{testPath}", "").trimEnd(); +} + +// ── Isolating-backend selection (R18, fail-closed) ──────────────────────────── + +/** + * Result of selecting an isolating sandbox backend for verification. + */ +export interface IsolatingBackendSelection { + /** The backend id to request from `resolveSandboxBackend`, or null if none. */ + backendId: SandboxCapabilities["id"] | null; + /** Why no isolating backend is available (when backendId is null). */ + reason?: string; +} + +/** + * Describes the detected availability of isolating backends on this host. + * Injectable for tests so we don't shell out to detect bwrap/sandbox-exec. + */ +export interface IsolatingBackendProbe { + platform: NodeJS.Platform; + bubblewrapAvailable: boolean; + sandboxExecAvailable: boolean; +} + +/** + * Choose an isolating backend, failing closed. Returns `backendId: null` (a + * non-pass signal) when no isolating backend is available — verification must + * NEVER fall through to the unrestricted native backend (R18). + */ +export function selectIsolatingBackend(probe: IsolatingBackendProbe): IsolatingBackendSelection { + if (probe.platform === "linux" && probe.bubblewrapAvailable) { + return { backendId: "bubblewrap" }; + } + if (probe.platform === "darwin" && probe.sandboxExecAvailable) { + return { backendId: "sandbox-exec" }; + } + return { + backendId: null, + reason: `no isolating sandbox backend available (platform=${probe.platform}, bwrap=${probe.bubblewrapAvailable}, sandbox-exec=${probe.sandboxExecAvailable})`, + }; +} + +// ── Environment scrubbing (R18) ─────────────────────────────────────────────── + +/** + * Environment variables permitted into the verification child process. Anything + * not on the allowlist (API keys, auth tokens, DB credentials, agent logs) is + * dropped so agent-authored code executes with a minimal environment. + */ +export const VERIFICATION_ENV_ALLOWLIST = [ + "PATH", + "HOME", + "SHELL", + "LANG", + "LC_ALL", + "TMPDIR", + "TERM", + "NODE_ENV", + // pnpm / corepack need these to resolve the package manager in the checkout. + "PNPM_HOME", + "COREPACK_HOME", + "npm_config_registry", +] as const; + +/** + * Produce a scrubbed environment containing only allowlisted keys from the + * source environment, with `CI=1` forced for deterministic test runs. + */ +export function scrubEnv(source: NodeJS.ProcessEnv = process.env): NodeJS.ProcessEnv { + const scrubbed: NodeJS.ProcessEnv = {}; + for (const key of VERIFICATION_ENV_ALLOWLIST) { + const value = source[key]; + if (value !== undefined) { + scrubbed[key] = value; + } + } + // Force deterministic, non-interactive execution. + scrubbed.CI = "1"; + return scrubbed; +} + +// ── Disposable checkout materialization (R11/R17) ───────────────────────────── + +/** A disposable checkout the verification run can execute against. */ +export interface DisposableCheckout { + /** Absolute path to the checkout root (under a run-unique tmpdir). */ + dir: string; + /** Tear the checkout down unconditionally (idempotent). */ + dispose(): Promise; +} + +/** + * Materializes disposable checkouts at a trusted revision. Injectable so tests + * can supply a fixture checkout without invoking git. + */ +export interface CheckoutMaterializer { + /** + * Create a disposable checkout of `rootDir` at `revision` under a run-unique + * tmpdir. The implementation MUST NOT mutate the source tree at `rootDir`. + */ + materialize(rootDir: string, revision: string): Promise; + /** + * Assert that the source tree feeding diff/merge is git-clean (byte-identical) + * — the R17 post-condition. Throws if dirty. + */ + assertSourceClean(rootDir: string): Promise; +} + +/** + * Default git-backed materializer: `git worktree add --detach ` + * produces an isolated checkout without touching the source working tree, and + * `git status --porcelain` on the source confirms cleanliness afterwards. + */ +export class GitCheckoutMaterializer implements CheckoutMaterializer { + async materialize(rootDir: string, revision: string): Promise { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), "fn-verify-")); + // `git worktree add --detach` checks out the revision into a throwaway dir + // without modifying the source working tree. + await execAsync(`git worktree add --detach ${JSON.stringify(dir)} ${JSON.stringify(revision)}`, { + cwd: rootDir, + timeout: 60_000, + maxBuffer: 8 * 1024 * 1024, + }); + return { + dir, + dispose: async () => { + try { + await execAsync(`git worktree remove --force ${JSON.stringify(dir)}`, { + cwd: rootDir, + timeout: 30_000, + }); + } catch (err) { + verifyLog.warn(`Failed to remove verification worktree ${dir}:`, err); + } + await fs.rm(dir, { recursive: true, force: true }).catch(() => {}); + }, + }; + } + + async assertSourceClean(rootDir: string): Promise { + const { stdout } = await execAsync("git status --porcelain", { + cwd: rootDir, + timeout: 30_000, + maxBuffer: 8 * 1024 * 1024, + }); + if (stdout.trim().length > 0) { + throw new Error(`Source tree is not git-clean after verification run:\n${stdout.trim()}`); + } + } +} + +/** Probe the host for isolating-backend availability (cached by the detectors). */ +async function probeIsolatingBackends(): Promise { + const [bwrap, sandboxExec] = await Promise.all([ + detectBwrap().catch(() => ({ available: false })), + detectSandboxExec().catch(() => ({ available: false })), + ]); + return { + platform: process.platform, + bubblewrapAvailable: bwrap.available, + sandboxExecAvailable: sandboxExec.available, + }; +} + +// ── Test-execution verification capability ──────────────────────────────────── + +export interface TestExecutionVerificationOptions { + /** Task store, reused by runVerificationCommand for command logging. */ + store: TaskStore; + /** Repo root whose source tree must remain git-clean. */ + rootDir: string; + /** + * Fixed, system-owned command template. Must contain `{testPath}` when an + * agent-supplied proof path is used. Example: `pnpm vitest run {testPath}`. + */ + commandTemplate: string; + /** Injectable checkout materializer (defaults to git-backed). */ + materializer?: CheckoutMaterializer; + /** Injectable backend probe (defaults to host detection). */ + probeBackends?: () => Promise; + /** + * Injectable factory for the isolating sandbox backend, given the selected + * backend id. Defaults to `resolveSandboxBackend({ backendId })`. Injectable so + * tests can supply a scripted backend without mutating global sandbox state. + */ + backendFactory?: (backendId: SandboxCapabilities["id"]) => SandboxBackend; + /** Injectable env source (defaults to process.env). */ + envSource?: NodeJS.ProcessEnv; +} + +/** + * The test-execution channel of the verification run. Confirms a behavioral + * assertion by running the suite / an agent-supplied regression test against a + * disposable checkout at the integration SHA, under an isolating sandbox + * backend with a scrubbed env. Fails closed to a non-pass on any setup failure. + * + * App-driving is NOT handled here; a later unit dispatches UI/bug assertions to + * an app-driving channel. This class is the canonical pattern that channel will + * mirror. + */ +export class TestExecutionVerificationCapability implements VerificationCapability { + private readonly store: TaskStore; + private readonly rootDir: string; + private readonly commandTemplate: string; + private readonly materializer: CheckoutMaterializer; + private readonly probeBackends: () => Promise; + private readonly backendFactory: (backendId: SandboxCapabilities["id"]) => SandboxBackend; + private readonly envSource: NodeJS.ProcessEnv; + + constructor(options: TestExecutionVerificationOptions) { + this.store = options.store; + this.rootDir = options.rootDir; + this.commandTemplate = options.commandTemplate; + this.materializer = options.materializer ?? new GitCheckoutMaterializer(); + this.probeBackends = options.probeBackends ?? probeIsolatingBackends; + this.backendFactory = options.backendFactory ?? ((backendId) => resolveSandboxBackend({ backendId })); + this.envSource = options.envSource ?? process.env; + } + + async verifyBehavioralAssertion(request: VerificationRequest): Promise { + const { assertionId } = request; + + // R11: a trusted revision is required to materialize a disposable checkout. + if (!request.integrationSha) { + return this.inconclusive(assertionId, "no integration SHA available to materialize a trusted checkout"); + } + + // R19: validate any agent-supplied proof path BEFORE doing any work. + let validatedTestPath: string | undefined; + if (request.proof) { + const safe = validateTestPath(request.proof.testFilePath); + if (safe === null) { + return this.inconclusive( + assertionId, + `agent-supplied test path rejected (invalid or contains shell metacharacters): ${JSON.stringify(request.proof.testFilePath)}`, + ); + } + validatedTestPath = safe; + } + + // R18: select an isolating backend, fail closed when none is available. + const probe = await this.probeBackends(); + const selection = selectIsolatingBackend(probe); + if (selection.backendId === null) { + return this.inconclusive(assertionId, selection.reason ?? "no isolating sandbox backend available"); + } + + const command = buildVerificationCommand(this.commandTemplate, validatedTestPath); + const scrubbedEnv = scrubEnv(this.envSource); + const logTaskId = request.taskId ?? `verify-${assertionId}`; + + // Route runVerificationCommand through the explicitly-selected isolating + // backend rather than the no-arg native fallback (R18). runVerificationCommand + // resolves its backend via the no-arg resolveSandboxBackend(), so we pin the + // selected isolating backend via the test-override hook for the duration of + // the run and unconditionally restore afterwards. + const isolating = this.backendFactory(selection.backendId); + const restoreBackend = () => __setSandboxBackendForTests(null); + __setSandboxBackendForTests(isolating); + + let implCheckout: DisposableCheckout | undefined; + let baselineCheckout: DisposableCheckout | undefined; + try { + implCheckout = await this.materializer.materialize(this.rootDir, request.integrationSha); + + const implResult = await runVerificationCommand( + this.store, + implCheckout.dir, + logTaskId, + command, + "test", + request.signal, + verifyLog, + "reviewer", + scrubbedEnv, + ); + + // R5/AE5: agent-supplied proof must fail on the merge-base baseline and + // pass on the implementation. A test that passes on both is not exercising + // the defect — reject it. + if (validatedTestPath) { + if (!request.mergeBaseSha) { + return this.inconclusive(assertionId, "no merge-base SHA available to validate agent-supplied proof"); + } + baselineCheckout = await this.materializer.materialize(this.rootDir, request.mergeBaseSha); + const baselineResult = await runVerificationCommand( + this.store, + baselineCheckout.dir, + logTaskId, + command, + "test", + request.signal, + verifyLog, + "reviewer", + scrubbedEnv, + ); + + if (baselineResult.success && implResult.success) { + return { + verdict: "fail", + assertionId, + reason: "agent-supplied proof passes on both the pre-fix baseline and the implementation; it does not exercise the defect", + detail: "pass-on-both rejected (R5/AE5)", + }; + } + if (!baselineResult.success && implResult.success) { + return { verdict: "pass", assertionId, reason: "regression test fails on the pre-fix baseline and passes on the implementation" }; + } + // Fails on the implementation → defect still reproduces. + return { + verdict: "fail", + assertionId, + reason: "regression test does not pass on the implementation; behavior not confirmed", + detail: implResult.stderr || implResult.stdout || undefined, + }; + } + + // Whole-suite channel: pass only when the suite passes. + if (implResult.success) { + return { verdict: "pass", assertionId, reason: "verification suite passed on the implementation checkout" }; + } + return { + verdict: "fail", + assertionId, + reason: "verification suite failed on the implementation checkout; behavior not confirmed", + detail: implResult.stderr || implResult.stdout || undefined, + }; + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + // R9: any setup/exec failure (timeout, abort, materialization error) is a + // non-pass; we route it to inconclusive (infra, not behavioral). + return this.inconclusive(assertionId, `verification run could not complete: ${message}`); + } finally { + restoreBackend(); + await implCheckout?.dispose(); + await baselineCheckout?.dispose(); + // R17: the source tree feeding diff/merge must be byte-clean afterwards. + try { + await this.materializer.assertSourceClean(this.rootDir); + } catch (cleanErr) { + verifyLog.error("Verification post-condition violated (source not git-clean):", cleanErr); + throw cleanErr instanceof Error ? cleanErr : new Error(String(cleanErr)); + } + } + } + + private inconclusive(assertionId: string, reason: string): VerificationOutcome { + return { verdict: "inconclusive", assertionId, reason }; + } +}