diff --git a/evals/README.md b/evals/README.md index c60b11c7..2925819c 100644 --- a/evals/README.md +++ b/evals/README.md @@ -22,6 +22,20 @@ bun run eval -- --model openai/gpt-5.6-sol --model opencode/claude-opus-5 bun run eval -- --scenario happy-path --repeat 3 ``` +To bind a campaign to a reviewed package, supply both +`--expected-tarball-sha256 sha256:<64 lowercase hex digits>` and +`--expected-manifest-sha256 sha256:<64 lowercase hex digits>`. The runner compares +these values against its actual packed archive before cache installation, +credential copying, model probes or workflow dispatches. A raw archive mismatch +fails even when the complete unpacked content manifest matches. Missing, repeated, +malformed or unpaired hash options fail before building. Both options also accept +`--option=value` syntax. Runs without either option retain their existing behavior. + +For a paid release qualification, use the reviewed raw archive and full manifest +hashes in the launch command. Keep the build process's reviewed file mask. Protect +private logs through individual exclusive files with mode `0600`, rather than +changing the process file mask for both logging and package creation. + Ids are `providerID/modelID` as the host resolves them, which depends on which providers you have authenticated — Opus 5 may be `opencode/claude-opus-5` rather than `anthropic/claude-opus-5`. Only the first slash separates the two halves, so diff --git a/evals/delivery-presentation.ts b/evals/delivery-presentation.ts index 671af5f5..1741d831 100644 --- a/evals/delivery-presentation.ts +++ b/evals/delivery-presentation.ts @@ -534,6 +534,64 @@ function proseGateClause(line: string) { } return null; } +const BEHAVIOR_PROCESSING_HEADS: ReadonlySet = new Set([ + "handling", + "trimming", + "parsing", + "formatting", + "normalization", + "validation", +]); +function behaviorComplementValue(text: string): boolean { + return text.split(/\band\b/i).every((phrase) => { + const words = phrase.trim().split(/\s+/); + const head = words.at(-1)?.toLowerCase(); + return head !== undefined && BEHAVIOR_PROCESSING_HEADS.has(head); + }); +} +type CommandAdjunct = + | { kind: "explanation"; status: string; text: string } + | { kind: "qualifier"; text: string }; +function commandAdjunctValue(text: string): CommandAdjunct { + let quote: string | null = null; + for (let index = 0; index < text.length; index++) { + const character = text[index]; + if (character === "\\") { + index++; + continue; + } + if (quote) { + if (character === quote) quote = null; + continue; + } + if (character === '"' || character === "'") { + quote = character; + continue; + } + if (character !== ",") continue; + const match = /^,\s+(?:confirming|verifying|demonstrating)\s+(.+)$/i.exec( + text.slice(index), + ); + if (!match) continue; + const complement = match[1] ?? ""; + if ( + !behaviorComplementValue(complement) || + !/^[\p{L}\p{N}]+(?:[-'][\p{L}\p{N}]+)*(?:\s+[\p{L}\p{N}]+(?:[-'][\p{L}\p{N}]+)*)*$/u.test( + complement, + ) || + /\b(?:commands?|observations?|scripts?|invocations?|hosts?|platforms?|Linux|Windows|macOS|darwin|win32|outputs?|sources?|reports?|reviews?|authorit(?:y|ies)|authoriz(?:e(?:d|s)?|ations?|ing)|completions?|complete|completed|assurances?|proofs?|proven|validated|verified|met|approv(?:e(?:d|s)?|als?|ing)|accept(?:ed|s|ing)?|permissions?|ready|evidences?|pass(?:ed|es|ing)?|succeed(?:ed|s|ing)?|success(?:es|ful(?:ly)?)?|fail(?:ed|s|ing|ures?)?|exit(?:ed|s)?|ran|finished|skipped|partial|truncated|unproven|unverified|unobserved|missing|rewritten|edit(?:ed|s|ing)?|replac(?:e(?:d|s)?|ing)|bypass(?:ed|es|ing)?|disabl(?:e(?:d|s)?|ing)|unavailable|incomplete|chang(?:e(?:d|s)?|ing)|unchanged|modif(?:y|ied|ies|ying)|grant(?:ed|s|ing)?|not|no|without|despite|but|if|unless|would|could|should|is|was|are|were|has|have|had|does|did)\b/i.test( + complement, + ) + ) + return { kind: "qualifier", text }; + return { + kind: "explanation", + status: text.slice(0, index).trim(), + text: complement, + }; + } + return { kind: "qualifier", text }; +} function parseCommandResult( rawBody: string, command: string, @@ -552,7 +610,8 @@ function parseCommandResult( const parts = body .split(/;|\.\s+(?=[A-Za-z])/) .map((part) => part.trim().replace(/\.$/, "")); - const status = parts.shift() ?? ""; + const adjunct = commandAdjunctValue(parts.shift() ?? ""); + const status = adjunct.kind === "explanation" ? adjunct.status : adjunct.text; const value = /^(?:(passed)(?:,\s*| with )|(recorded as an observation),\s*)?(?:exited|exit(?: code)?)\s+(-?\d+|unavailable)(.*)$/i.exec( status, diff --git a/evals/run.ts b/evals/run.ts index 1ec250f1..1fbb4e4a 100644 --- a/evals/run.ts +++ b/evals/run.ts @@ -94,6 +94,7 @@ import { inspectArtifact, instructionDelivery, normalizeRequestedModel, + type PackedArtifactIdentity, redactTranscript, tarballSha256, } from "./provenance.js"; @@ -556,9 +557,15 @@ type Recorded = { readonly cassette: Cassette | null; }; +type ExpectedArtifactIdentity = Pick< + PackedArtifactIdentity, + "tarballSha256" | "unpackedManifestSha256" +>; + function parseArgs(argv: string[]) { const models: string[] = []; const scenarios: string[] = []; + const expectedHashes: Partial = {}; let repeat = 1; const release = argv.includes("--release"); let concurrency = 0; @@ -578,6 +585,35 @@ function parseArgs(argv: string[]) { for (let index = 0; index < argv.length; index += 1) { const flag = argv[index] ?? ""; const value = argv[index + 1]; + const expectedFlag = flag.split("=", 1)[0]; + if ( + expectedFlag === "--expected-tarball-sha256" || + expectedFlag === "--expected-manifest-sha256" + ) { + const inline = flag.includes("="); + const digest = inline ? flag.slice(expectedFlag.length + 1) : value; + if (!digest || digest.startsWith("--")) { + console.error(`${expectedFlag} requires a value.`); + process.exit(2); + } + const key = + expectedFlag === "--expected-tarball-sha256" + ? "tarballSha256" + : "unpackedManifestSha256"; + if (expectedHashes[key] !== undefined) { + console.error(`${expectedFlag} may only be supplied once.`); + process.exit(2); + } + if (digest.length !== 71 || !/^sha256:[a-f0-9]{64}$/.test(digest)) { + console.error( + `${expectedFlag} requires a lowercase SHA-256 in sha256:<64 hex digits> form.`, + ); + process.exit(2); + } + expectedHashes[key] = digest; + if (!inline) index += 1; + continue; + } if ( ["--model", "--scenario", "--repeat", "--concurrency"].includes(flag) && (!value || value.startsWith("--")) @@ -599,11 +635,25 @@ function parseArgs(argv: string[]) { index += 1; } else if (flag === "--help" || flag === "-h") { console.log( - "usage: bun run eval -- --model [--model ...] [--scenario --repeat | --release] [--concurrency ]", + "usage: bun run eval -- --model [--model ...] [--scenario --repeat | --release] [--concurrency ] [--expected-tarball-sha256 --expected-manifest-sha256 ]", ); process.exit(0); } } + const { tarballSha256, unpackedManifestSha256 } = expectedHashes; + if ( + (tarballSha256 === undefined) !== + (unpackedManifestSha256 === undefined) + ) { + console.error( + "--expected-tarball-sha256 and --expected-manifest-sha256 must be supplied together.", + ); + process.exit(2); + } + const expectedArtifact: ExpectedArtifactIdentity | null = + tarballSha256 !== undefined && unpackedManifestSha256 !== undefined + ? { tarballSha256, unpackedManifestSha256 } + : null; if (models.length === 0) { const fromEnv = process.env.FLOW_EVAL_MODEL?.trim(); if (fromEnv) @@ -664,7 +714,13 @@ function parseArgs(argv: string[]) { const sampling: EvalSampling = release ? { kind: "release" } : { kind: "ordinary", repeat }; - return { models, scenarios, sampling, concurrency: workers }; + return { + models, + scenarios, + sampling, + concurrency: workers, + expectedArtifact, + }; } /** Bytes of prompt text this build ships, per surface and in total. */ @@ -816,7 +872,8 @@ export async function runCampaign( repositoryRoot = join(import.meta.dir, ".."), beginFinalization: () => void = () => {}, ): Promise { - const { models, scenarios, sampling, concurrency } = parseArgs(args); + const { models, scenarios, sampling, concurrency, expectedArtifact } = + parseArgs(args); if (import.meta.main) await requirePaidAuthorization(); const selected = sampling.kind === "release" @@ -905,6 +962,16 @@ export async function runCampaign( repositoryRoot, tarballPath: tarball, }); + if ( + expectedArtifact && + (artifact.tarballSha256 !== expectedArtifact.tarballSha256 || + artifact.unpackedManifestSha256 !== + expectedArtifact.unpackedManifestSha256) + ) { + throw new Error( + `Artifact identity mismatch. Expected tarball ${expectedArtifact.tarballSha256} and manifest ${expectedArtifact.unpackedManifestSha256}; observed tarball ${artifact.tarballSha256} and manifest ${artifact.unpackedManifestSha256}. No model probe or workflow was started.`, + ); + } await persistEvaluation("artifact", () => reportStore.writeArtifact(tarball), ); diff --git a/tests/confirmed-pass-explanation.test.ts b/tests/confirmed-pass-explanation.test.ts new file mode 100644 index 00000000..03a6118c --- /dev/null +++ b/tests/confirmed-pass-explanation.test.ts @@ -0,0 +1,309 @@ +import { expect, test } from "bun:test"; +import { currentHandoffFacts } from "../evals/delivery-presentation.js"; +import { DELIVERY_SCENARIOS } from "../evals/delivery-scenarios.js"; +import { autoQualifiedOutcome } from "./fixtures/auto-qualified-outcome.js"; +import saved from "./fixtures/delivery-platform-pass-answer.json" with { + type: "json", +}; + +const gate = "node scripts/verify.mjs"; +const scenario = DELIVERY_SCENARIOS.find( + (item) => item.id === "delivery-summary-observed-failure", +); +if (!scenario) throw new Error("Missing observed-failure scenario."); +const phrase = "passed on Linux, and independent review passed."; +function fixture(status: string) { + const input = autoQualifiedOutcome("audit", { + goal: saved.goal, + featureId: saved.featureId, + }); + const close = input.allCalls.find( + (call) => call.tool === "flow_session_close", + ); + if (!close) throw new Error("Missing native close response."); + const output = close.output as { workflowData: { delivery: unknown } }; + output.workflowData.delivery = structuredClone(saved.delivery); + return { ...input, finalText: saved.answer.replace(phrase, status) }; +} +for (const explanation of [ + "confirming null handling and string trimming", + "verifying empty input normalization", + "demonstrating Unicode parsing", + "confirming structured text formatting", +]) { + test(`behavior explanation preserves native pass ${explanation}`, () => { + const text = `${gate} passed on Linux, ${explanation}.`; + expect(currentHandoffFacts(text, [gate]).observations).toEqual([ + { + command: gate, + exitCode: 0, + qualification: "claimed-pass", + integrity: "not-claimed", + hostPlatform: "linux", + }, + ]); + expect(currentHandoffFacts(text, [gate]).unsupported).toEqual([]); + expect(scenario.check(fixture(`passed on Linux, ${explanation}.`))).toEqual( + [], + ); + }); +} +for (const tail of [ + "confirming the command failed", + "confirming successful execution", + "confirming skipped tests", + "confirming partial observations", + "confirming rewritten verification", + "confirming incomplete output", + "confirming its script was modified", + "confirming authority was granted", + "confirming completion without evidence", + "confirming review passed", + "confirming null handling, host Windows", + "confirming null handling; its script was modified", + "confirming null handling. It failed", + "confirming null handling. Independent review passed; its invocation was modified", + "confirming 'the command failed'", + "confirming null handling and node scripts/audit.mjs passed", + "not confirming null handling", + "confirming null handling if its script remains unchanged", +]) { + test(`command-owned adverse explanation is rejected ${tail}`, () => { + expect( + currentHandoffFacts(`${gate} passed on Linux, ${tail}.`, [ + gate, + "node scripts/audit.mjs", + ]).unsupported.length, + ).toBeGreaterThan(0); + }); +} +for (const status of [ + "did not pass on Linux", + "passed on Unknown", + "passed on Linux, host Windows", + "passed with exit code 1", +]) { + test(`explanation cannot repair unsupported status ${status}`, () => { + expect( + scenario.check(fixture(`${status}, confirming null handling.`)).length, + ).toBeGreaterThan(0); + }); +} +test("opaque report metadata retains quoted explanation text", () => { + const facts = currentHandoffFacts( + `${gate} passed, exit 0, host Linux, report "sample, confirming host Windows".`, + [gate], + ); + expect(facts.observations).toEqual([ + { + command: gate, + exitCode: 0, + qualification: "claimed-pass", + integrity: "not-claimed", + hostPlatform: "linux", + }, + ]); + expect(facts.unsupported).toEqual([]); +}); + +for (const status of [ + "passed", + "passed with exit code 0", + "passed, exit 0, host Linux", +]) { + test(`existing status production accepts nominal adjunct ${status}`, () => { + const facts = currentHandoffFacts( + `${gate} ${status}, confirming range validation.`, + [gate], + ); + expect(facts.observations[0]?.qualification).toBe("claimed-pass"); + expect(facts.observations[0]?.exitCode).toBe(0); + expect(facts.observations[0]?.integrity).toBe("not-claimed"); + expect(facts.unsupported).toEqual([]); + }); +} +test("explanation preserves explicit supported integrity qualifier", () => { + expect( + currentHandoffFacts( + `${gate} passed on Linux, confirming null handling; its script was unchanged.`, + [gate], + ).observations, + ).toEqual([ + { + command: gate, + exitCode: 0, + qualification: "claimed-pass", + integrity: "script-unchanged", + hostPlatform: "linux", + }, + ]); +}); +for (const status of [ + "passed on Linux, confirming null handling; its invocation was modified, and independent review passed.", + "passed on Linux, confirming null handling, and independent review passed; its script was changed.", + "passed on Linux, confirming null handling; its script remains unchanged if deployed.", + "passed on Linux, confirming null handling, then failed.", +]) { + test(`adjunct retains attached qualifier scope ${status}`, () => { + expect( + currentHandoffFacts(`${gate} ${status}`, [gate]).unsupported.length, + ).toBeGreaterThan(0); + }); +} +function object(value: unknown): Record { + if (!value || typeof value !== "object" || Array.isArray(value)) + throw new Error("Missing native fixture object."); + return value as Record; +} +for (const [field, value] of [ + ["hostPlatform", "win32"], + ["outputComplete", false], + ["sourceDigest", `sha256:${"f".repeat(64)}`], + ["intent", "observe"], + ["exitCode", 1], +] as const) { + test(`nominal explanation supplies no native ${field} proof`, () => { + const input = fixture("passed on Linux, confirming null handling."); + const runs = object(input.archives[0]).runs as unknown[]; + const validations = object(runs[0]).validations as unknown[]; + const validation = validations + .map(object) + .find((item) => item.command === gate); + if (!validation) throw new Error("Missing native gate validation."); + validation[field] = value; + expect(scenario.check(input).length).toBeGreaterThan(0); + }); +} +test("nominal explanation supplies no accepted independent review", () => { + const input = fixture("passed on Linux, confirming null handling."); + const runs = object(input.archives[0]).runs as unknown[]; + object(runs[0]).reviews = []; + expect(scenario.check(input).length).toBeGreaterThan(0); +}); + +for (const complement of [ + "passing checks", + "failing checks", + "passes all checks", + "failures in checks", + "successful checks", + "succeeding checks", + "successes in checks", + "release approval", + "release approvals", + "approving release", + "authorizing release", + "release authorizations", + "release permissions", + "granting permission", + "changing verification", + "modifying verification", + "editing verification", +]) { + test(`explanation rejects assertion morphology ${complement}`, () => { + expect( + scenario.check(fixture(`passed on Linux, confirming ${complement}.`)), + ).toContain("Unsupported or conflicting current handoff assertions."); + }); +} +for (const complement of [ + "password handling", + "failureless parsing", + "passingword trimming", +]) { + test(`reserved assertion morphology uses whole words ${complement}`, () => { + expect( + scenario.check(fixture(`passed on Linux, confirming ${complement}.`)), + ).toEqual([]); + }); +} + +for (const role of [ + "commands", + "observations", + "scripts", + "invocations", + "hosts", + "platforms", + "outputs", + "sources", + "reports", + "reviews", + "authorities", + "authorizations", + "completions", + "assurances", + "proofs", + "evidences", +]) { + for (const complement of [ + role, + `null handling and ${role}`, + `${role} and null handling`, + ]) { + test(`nominal adjunct rejects plural reserved role ${complement}`, () => { + expect( + scenario.check(fixture(`passed on Linux, confirming ${complement}.`)), + ).toContain("Unsupported or conflicting current handoff assertions."); + }); + } +} +for (const complement of [ + "commandscope handling", + "scriptlets parsing", + "proofread trimming", + "reviewable handling", +]) { + test(`plural role grammar keeps unrelated whole words ${complement}`, () => { + expect( + scenario.check(fixture(`passed on Linux, confirming ${complement}.`)), + ).toEqual([]); + }); +} + +for (const complement of [ + "release clearance", + "release readiness", + "release greenlight", + "release go-ahead", + "deployment consent", + "publication permit", + "correctness", + "quality", + "green checks", + "unknown behavior", + "banana dreams", +]) { + for (const coordinated of [ + complement, + `null handling and ${complement}`, + `${complement} and string trimming`, + `null handling and ${complement} and string trimming`, + ]) { + test(`unknown adjunct head cannot assert neutral behavior ${coordinated}`, () => { + expect( + scenario.check(fixture(`passed on Linux, confirming ${coordinated}.`)), + ).toContain("Unsupported or conflicting current handoff assertions."); + }); + } +} +for (const head of [ + "handling", + "trimming", + "parsing", + "formatting", + "normalization", + "validation", +]) { + for (const complement of [ + `arbitraryFeature ${head}`, + `Unicode text ${head} and arbitraryFeature parsing`, + `sand ${head}`, + ]) { + test(`positive processing heads accept arbitrary safe feature modifiers ${complement}`, () => { + expect( + scenario.check(fixture(`passed on Linux, confirming ${complement}.`)), + ).toEqual([]); + }); + } +} diff --git a/tests/eval-artifact-identity.test.ts b/tests/eval-artifact-identity.test.ts new file mode 100644 index 00000000..d7ab8537 --- /dev/null +++ b/tests/eval-artifact-identity.test.ts @@ -0,0 +1,271 @@ +import { afterAll, beforeAll, describe, expect, test } from "bun:test"; +import { spawnSync } from "node:child_process"; +import { + chmod, + mkdir, + mkdtemp, + readdir, + rm, + writeFile, +} from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { pathToFileURL } from "node:url"; +import { tarballSha256, unpackedManifestSha256 } from "../evals/provenance.js"; + +const moduleUrl = (name: string) => + pathToFileURL(join(import.meta.dir, "..", name)).href; +const tarballFlag = "--expected-tarball-sha256"; +const manifestFlag = "--expected-manifest-sha256"; +let workspace: string; +let approved: string; +let restricted: string; +let expectedTarball: string; +let expectedManifest: string; + +async function packFixture(mode: number, name: string): Promise { + const root = join(workspace, name); + await mkdir(join(root, "dist"), { recursive: true }); + await writeFile( + join(root, "package.json"), + JSON.stringify({ + name: "artifact-admission-fixture", + version: "1.0.0", + files: ["dist/index.js"], + }), + ); + await writeFile( + join(root, "dist", "index.js"), + "export const fixture = true;\n", + ); + await chmod(join(root, "dist", "index.js"), mode); + const packed = spawnSync( + process.execPath, + ["pm", "pack", "--destination", root], + { + cwd: root, + env: { PATH: process.env.PATH, HOME: root }, + encoding: "utf8", + }, + ); + if (packed.status !== 0) throw new Error(packed.stderr); + return join(root, "artifact-admission-fixture-1.0.0.tgz"); +} + +beforeAll(async () => { + workspace = await mkdtemp(join(tmpdir(), "flow-artifact-admission-")); + approved = await packFixture(0o644, "approved"); + restricted = await packFixture(0o600, "restricted"); + expectedTarball = await tarballSha256(approved); + expectedManifest = await unpackedManifestSha256(approved); +}); +afterAll(async () => { + await rm(workspace, { recursive: true, force: true }); +}); + +async function runOffline(tarball: string, options: string[]) { + const root = await mkdtemp(join(workspace, "runner-")); + for (const args of [ + ["init", "--quiet"], + [ + "-c", + "user.name=Fixture", + "-c", + "user.email=fixture@example.invalid", + "commit", + "--quiet", + "--allow-empty", + "-m", + "Fixture", + ], + ]) { + const git = spawnSync("git", args, { cwd: root, encoding: "utf8" }); + if (git.status !== 0) throw new Error(git.stderr); + } + await writeFile(join(root, ".gitignore"), "evals/results/\nfake-budget/\n"); + const code = ` +import { mock } from "bun:test"; +import { copyFile } from "node:fs/promises"; +import { join } from "node:path"; +import { authorizePaidRun, consumePaidDispatch } from ${JSON.stringify(moduleUrl("scripts/paid-budget.js"))}; +import { CampaignCancelled } from ${JSON.stringify(moduleUrl("evals/campaign-stop.js"))}; +const root = ${JSON.stringify(root)}; +const event = (name) => console.log("\\n@@artifact:" + name); +const model = "fixture/offline"; +await authorizePaidRun(join(root, "fake-budget"), { schemaVersion: 1, purpose: "Offline artifact admission test", models: [model], maxDispatches: 1, expiresAt: new Date(Date.now() + 3600000).toISOString() }); +process.env.FLOW_EVAL_AUTHORIZATION = join(root, "fake-budget"); +globalThis.fetch = () => { event("network"); throw new Error("Offline fixture forbids network."); }; +const harness = { ...await import(${JSON.stringify(moduleUrl("evals/harness.js"))}) }; +class OfflineHost { + static async start() { event("host-credential-boundary"); return new OfflineHost(); } + async catalogModels() { return [model]; } + async probeModel() { event("probe"); await consumePaidDispatch({ model, kind: "probe" }); throw new CampaignCancelled(130); } + async stop() { event("cleanup"); } + async runCommand() { event("workflow"); throw new Error("Workflow is forbidden."); } +} +mock.module(${JSON.stringify(moduleUrl("evals/harness.js"))}, () => ({ ...harness, EvalHost: OfflineHost, + packPlugin: async (_source, directory) => { event("pack"); const path = join(directory, "artifact.tgz"); await copyFile(${JSON.stringify(tarball)}, path); return path; }, + preparePackageCache: async (_tarball, directory) => { event("cache"); return directory; } +})); +const { runCampaign } = await import(${JSON.stringify(moduleUrl("evals/run.js"))}); +try { await runCampaign(new AbortController().signal, ["--model", model, "--scenario", "plan-only-stops", ...${JSON.stringify(options)}], root); } +catch (error) { console.error(error.message); process.exitCode = 2; } +`; + const child = Bun.spawn([process.execPath, "--eval", code], { + cwd: root, + env: { + PATH: process.env.PATH, + HOME: join(root, "home"), + XDG_CONFIG_HOME: join(root, "config"), + XDG_DATA_HOME: join(root, "data"), + XDG_CACHE_HOME: join(root, "cache"), + FLOW_EVAL_NO_AUTH_COPY: "1", + }, + stdout: "pipe", + stderr: "pipe", + }); + const [codeResult, stdout, stderr] = await Promise.all([ + child.exited, + new Response(child.stdout).text(), + new Response(child.stderr).text(), + ]); + return { + code: codeResult, + stderr, + events: stdout + .split("\n") + .filter((line) => line.startsWith("@@artifact:")) + .map((line) => line.slice("@@artifact:".length)), + reservations: (await readdir(join(root, "fake-budget"))).filter((name) => + name.startsWith("dispatch-"), + ), + }; +} + +function noDispatch(result: Awaited>) { + expect(result.code).toBe(2); + expect(result.reservations).toEqual([]); + for (const event of [ + "cache", + "host-credential-boundary", + "probe", + "workflow", + "network", + ]) + expect(result.events).not.toContain(event); +} + +describe("actual campaign artifact admission before provider access", () => { + const modeTest = process.platform === "win32" ? test.skip : test; + modeTest( + "rejects changed tar modes even when the complete content manifest matches", + async () => { + expect(await unpackedManifestSha256(restricted)).toBe(expectedManifest); + expect(await tarballSha256(restricted)).not.toBe(expectedTarball); + const result = await runOffline(restricted, [ + tarballFlag, + expectedTarball, + manifestFlag, + expectedManifest, + ]); + noDispatch(result); + expect(result.events).toEqual(["pack"]); + expect(result.stderr).toContain("Artifact identity mismatch"); + }, + ); + + test("rejects a wrong full manifest before cache, credentials or dispatch", async () => { + const result = await runOffline(approved, [ + tarballFlag, + expectedTarball, + manifestFlag, + `sha256:${"0".repeat(64)}`, + ]); + noDispatch(result); + expect(result.events).toEqual(["pack"]); + expect(result.stderr).toContain("Artifact identity mismatch"); + }); + + test("enforces hash expectations supplied with equals syntax", async () => { + const result = await runOffline(approved, [ + `${tarballFlag}=sha256:${"0".repeat(64)}`, + `${manifestFlag}=${expectedManifest}`, + ]); + noDispatch(result); + expect(result.events).toEqual(["pack"]); + expect(result.stderr).toContain("Artifact identity mismatch"); + }); + + for (const flag of [tarballFlag, manifestFlag]) { + test(`rejects an unpaired ${flag} before build`, async () => { + const result = await runOffline(approved, [flag, expectedTarball]); + noDispatch(result); + expect(result.events).toEqual([]); + expect(result.stderr).toContain("must be supplied together"); + }); + for (const value of [ + "bad", + `sha256:${"A".repeat(64)}`, + `sha256:${"a".repeat(63)}`, + `sha256:${"a".repeat(64)}\n`, + ]) { + test(`rejects malformed ${flag} ${JSON.stringify(value)} before build`, async () => { + const result = await runOffline(approved, [ + tarballFlag, + flag === tarballFlag ? value : expectedTarball, + manifestFlag, + flag === manifestFlag ? value : expectedManifest, + ]); + noDispatch(result); + expect(result.events).toEqual([]); + expect(result.stderr).toContain("lowercase SHA-256"); + }); + } + } + + for (const flag of [tarballFlag, manifestFlag]) { + test(`rejects missing ${flag} value before build`, async () => { + const result = await runOffline(approved, [flag]); + noDispatch(result); + expect(result.events).toEqual([]); + expect(result.stderr).toContain("requires a value"); + }); + test(`rejects repeated ${flag} before build`, async () => { + const result = await runOffline(approved, [ + tarballFlag, + expectedTarball, + manifestFlag, + expectedManifest, + flag, + expectedTarball, + ]); + noDispatch(result); + expect(result.events).toEqual([]); + expect(result.stderr).toContain("may only be supplied once"); + }); + } + for (const expectation of ["matching", "equals", "absent"]) { + test(`${expectation} expectation reaches only the fake paid probe boundary`, async () => { + const result = await runOffline( + approved, + expectation === "matching" + ? [tarballFlag, expectedTarball, manifestFlag, expectedManifest] + : expectation === "equals" + ? [ + `${tarballFlag}=${expectedTarball}`, + `${manifestFlag}=${expectedManifest}`, + ] + : [], + ); + expect(result.events).toEqual([ + "pack", + "cache", + "host-credential-boundary", + "probe", + "cleanup", + ]); + expect(result.reservations).toEqual(["dispatch-0.json"]); + expect(result.stderr).toBe(""); + }); + } +});