From f384845fac9f72a4b3b2af2aac8f7c2f6c476b46 Mon Sep 17 00:00:00 2001 From: RSI smoke Date: Wed, 30 Sep 2026 15:52:22 -0600 Subject: [PATCH 1/6] feat(rsi): add controlled proof demonstration protocol --- docker/OpenShell-RSI.Dockerfile | 6 + docker/OpenShell.Dockerfile | 4 + package.json | 1 + scripts/rsi-proof-demonstration.ts | 214 ++++++++++++++ src/rsi/__tests__/inference-broker.test.ts | 72 ++++- src/rsi/__tests__/openshell.test.ts | 22 +- src/rsi/__tests__/proof-campaign.test.ts | 30 +- .../proof-demonstration-manifest.test.ts | 37 +++ .../proof-demonstration-plan.test.ts | 51 ++++ src/rsi/__tests__/proof-plan-path.test.ts | 19 ++ src/rsi/__tests__/reports.test.ts | 2 +- src/rsi/archive.ts | 5 + src/rsi/benchmark/__tests__/runner.test.ts | 4 +- src/rsi/benchmark/runner.ts | 2 +- src/rsi/config.ts | 10 + src/rsi/controller.ts | 95 +++++- src/rsi/development-benchmark.ts | 19 +- src/rsi/inference-broker.ts | 97 ++++-- .../008_openrouter_returned_provider.sql | 2 + src/rsi/openshell.ts | 3 + src/rsi/postgres-queue.ts | 17 +- src/rsi/proof-campaign.ts | 53 +++- src/rsi/proof-confirmation.ts | 89 ++++++ src/rsi/proof-demonstration-manifest.ts | 73 +++++ src/rsi/proof-demonstration-plan.ts | 168 +++++++++++ src/rsi/proof-demonstration.ts | 275 ++++++++++++++++++ src/rsi/reports.ts | 6 +- src/rsi/types.ts | 11 +- 28 files changed, 1296 insertions(+), 91 deletions(-) create mode 100644 scripts/rsi-proof-demonstration.ts create mode 100644 src/rsi/__tests__/proof-demonstration-manifest.test.ts create mode 100644 src/rsi/__tests__/proof-demonstration-plan.test.ts create mode 100644 src/rsi/__tests__/proof-plan-path.test.ts create mode 100644 src/rsi/migrations/008_openrouter_returned_provider.sql create mode 100644 src/rsi/proof-confirmation.ts create mode 100644 src/rsi/proof-demonstration-manifest.ts create mode 100644 src/rsi/proof-demonstration-plan.ts create mode 100644 src/rsi/proof-demonstration.ts diff --git a/docker/OpenShell-RSI.Dockerfile b/docker/OpenShell-RSI.Dockerfile index 3f971a6..d48809a 100644 --- a/docker/OpenShell-RSI.Dockerfile +++ b/docker/OpenShell-RSI.Dockerfile @@ -12,6 +12,12 @@ RUN apt-get update \ && find /opt/headlesscode/src -type d -name __tests__ -prune -exec rm -rf {} + \ && find /opt/headlesscode/src -type f \( -name '*.test.ts' -o -name '*.spec.ts' \) -delete \ && rm -rf /opt/headlesscode/scripts/eval-suite \ + && rm -rf /opt/headlesscode/scripts/rsi-benchmark \ + && rm -rf /opt/headlesscode/fixtures/rsi-benchmark \ && test ! -e /opt/headlesscode/src/rsi +RUN test ! -e /opt/headlesscode/scripts/rsi-benchmark/task-spec.mjs \ + && test ! -e /opt/headlesscode/fixtures/rsi-benchmark \ + && test ! -e /opt/headlesscode/.git + USER sandbox diff --git a/docker/OpenShell.Dockerfile b/docker/OpenShell.Dockerfile index 1c2eddd..949a810 100644 --- a/docker/OpenShell.Dockerfile +++ b/docker/OpenShell.Dockerfile @@ -19,6 +19,10 @@ COPY src ./src COPY shared ./shared COPY scripts/spawn-parallel-worktrees.sh scripts/run-worker.sh scripts/headlesscode-answer.sh ./scripts/ RUN chmod -R a+rX /opt/headlesscode/src /opt/headlesscode/shared /opt/headlesscode/scripts +RUN test ! -e /opt/headlesscode/scripts/rsi-benchmark/task-spec.mjs \ + && test ! -e /opt/headlesscode/scripts/rsi-benchmark/manual-tests \ + && test ! -e /opt/headlesscode/fixtures/rsi-benchmark \ + && test ! -e /opt/headlesscode/.git RUN printf '%s\n' "$HEADLESSCODE_HARNESS_COMMIT" > /opt/headlesscode/.headlesscode-harness-commit LABEL org.capsize.headlesscode.harness-commit="$HEADLESSCODE_HARNESS_COMMIT" RUN chmod +x scripts/spawn-parallel-worktrees.sh scripts/run-worker.sh scripts/headlesscode-answer.sh diff --git a/package.json b/package.json index a911113..7c3a9e4 100644 --- a/package.json +++ b/package.json @@ -43,6 +43,7 @@ "cli": "tsx src/cli.ts", "monitor:pilot": "tsx scripts/monitor-pilot/run.ts", "rsi:promote-curriculum": "tsx src/rsi/promote-curriculum.ts", + "rsi:proof-demonstration": "tsx scripts/rsi-proof-demonstration.ts", "test": "node scripts/run-tests.mjs", "prepublishOnly": "npm ci && npm test" }, diff --git a/scripts/rsi-proof-demonstration.ts b/scripts/rsi-proof-demonstration.ts new file mode 100644 index 0000000..18bf2a2 --- /dev/null +++ b/scripts/rsi-proof-demonstration.ts @@ -0,0 +1,214 @@ +import { createHash } from "node:crypto" +import * as fs from "node:fs/promises" +import * as path from "node:path" +import { execFileSync } from "node:child_process" +import { fileURLToPath, pathToFileURL } from "node:url" +import { loadBenchmark } from "../src/rsi/benchmark/runner.js" +import { parseRsiArgs } from "../src/rsi/config.js" +import { fetchPinnedOpenRouterEndpointPrice, openRouterAllocatedEnvironment, openRouterBrokerLimitsFromEnvironment, RSI_OPENROUTER_MODEL } from "../src/rsi/inference-broker.js" +import { freezeProofDemonstrationManifest, frozenProofConfigDigest, verifyFrozenProofDemonstrationManifest } from "../src/rsi/proof-demonstration-manifest.js" +import { planProofDemonstration } from "../src/rsi/proof-demonstration-plan.js" +import { workerHarnessRuntimeIdentityFromEnvironment } from "../src/rsi/worker-harness.js" + +interface Options { + command: "estimate" | "plan" | "verify" + output: string + repo: string + operatorCapUsd?: number + seeds: string[] + mutationTask: string +} + +type ProofPlanPreview = Omit[0], "operatorCapUsd"> & { plan: ReturnType } + +function usage(): string { + return `Usage: + npx tsx scripts/rsi-proof-demonstration.ts estimate --seed --campaign-seed --campaign-seed --campaign-seed [--repo ] + npx tsx scripts/rsi-proof-demonstration.ts plan --operator-cap-usd --seed --campaign-seed --campaign-seed --campaign-seed [--out ] [--repo ] + npx tsx scripts/rsi-proof-demonstration.ts verify --plan + +Estimate and plan perform pricing metadata GETs only; they do not submit model inference. +Estimate prints reservations without assuming an operator cap. Plan requires a clean root +checkout, frozen runtime image identity, explicit broker ceilings, four distinct seeds, and +a supplied operator cap covering all four campaign reservations. +` +} + +function parse(argv: string[]): Options { + const command = argv[0] + if (command !== "estimate" && command !== "plan" && command !== "verify") throw new Error("expected estimate, plan or verify") + const options: Options = { command, output: path.resolve(".headlesscode/rsi-proof/plan.json"), repo: process.cwd(), seeds: [], mutationTask: "Improve headlesscode agent reliability on the frozen development benchmark while preserving existing behavior." } + for (let index = 1; index < argv.length; index++) { + const arg = argv[index]! + if (arg === "--help" || arg === "-h") throw new Error(usage()) + const value = argv[index + 1] + if (!value || value.startsWith("--")) throw new Error(`${arg} requires a value`) + if (arg === "--out" || arg === "--plan") options.output = path.resolve(value) + else if (arg === "--repo") options.repo = path.resolve(value) + else if (arg === "--operator-cap-usd") options.operatorCapUsd = Number(value) + else if (arg === "--seed" || arg === "--campaign-seed") options.seeds.push(value) + else if (arg === "--mutation-task") options.mutationTask = value + else throw new Error(`unknown option ${arg}`) + index++ + } + return options +} + +export async function assertSupervisorOnlyPlanPath(repoRoot: string, output: string): Promise { + const repo = await fs.realpath(repoRoot) + const absolute = path.resolve(output) + const segments = absolute.slice(path.parse(absolute).root.length).split(path.sep).filter(Boolean) + let current = path.parse(absolute).root + for (const segment of segments.slice(0, -1)) { + current = path.join(current, segment) + try { if ((await fs.lstat(current)).isSymbolicLink()) throw new Error("frozen plan path may not pass through a symlink") } + catch (error) { if ((error as NodeJS.ErrnoException).code !== "ENOENT") throw error } + } + const relative = path.relative(repo, absolute) + if (relative && !relative.startsWith(`..${path.sep}`) && relative !== "..") { + if (!relative.startsWith(`${path.join(".headlesscode", "rsi-proof")}${path.sep}`)) throw new Error("frozen plan must be outside the source repository or under ignored .headlesscode/rsi-proof") + try { execFileSync("git", ["check-ignore", "--quiet", "--", relative], { cwd: repo, stdio: "ignore" }) } + catch { throw new Error("frozen plan path is not Git-ignored") } + } + return absolute +} + +function sha256(value: string | Buffer): string { return createHash("sha256").update(value).digest("hex") } +function canonicalHash(value: unknown): string { return sha256(JSON.stringify(value)) } + +function campaignId(seed: string): string { return `rsi-proof-${sha256(seed).slice(0, 20)}` } + +async function analysisDigest(repo: string): Promise { + const sources = [ + "src/rsi/development-proof.ts", "src/rsi/proof-campaign.ts", "src/rsi/proof-confirmation.ts", + "src/rsi/proof-demonstration.ts", "src/rsi/proof-demonstration-plan.ts", "src/rsi/proof-demonstration-manifest.ts", + "src/rsi/controller.ts", "scripts/rsi-proof-demonstration.ts", + ] + const hash = createHash("sha256") + for (const relative of sources) hash.update(relative).update("\0").update(await fs.readFile(path.join(repo, relative))).update("\0") + return hash.digest("hex") +} + +async function ensureCleanRoot(repo: string): Promise { + const status = execFileSync("git", ["status", "--porcelain=v1", "--untracked-files=all"], { cwd: repo, encoding: "utf8" }) + if (status.trim()) throw new Error("plan requires a clean root checkout; commit or remove tracked and untracked changes first") + return execFileSync("git", ["rev-parse", "HEAD"], { cwd: repo, encoding: "utf8" }).trim() +} + +function parseUsd(value: number | undefined): number { + if (value === undefined || !Number.isFinite(value) || value <= 0) throw new Error("--operator-cap-usd must be a positive finite amount") + return Math.floor(value * 1_000_000) / 1_000_000 +} + +function formatPlan(preview: ProofPlanPreview, operatorCapUsd?: number, digestSha256?: string): string { + const { plan } = preview + return [ + `${digestSha256 ? `Frozen proof plan ${digestSha256}` : "Read-only proof plan estimate (not frozen)"}`, + `Root commit: ${preview.rootCommit}`, + `Benchmark manifest: ${preview.benchmarkManifestSha256}`, + `Model: ${preview.model.requestedId} (endpoint: ${preview.endpointPrice.provider})`, + `Endpoint rate observed ${preview.endpointPrice.observedAt}: input $${preview.endpointPrice.inputPricePerMillionUsd}/M, output $${preview.endpointPrice.outputPricePerMillionUsd}/M`, + `Configured maximum price ceilings: input $${preview.priceCeilingsPerMillionUsd.input}/M, output $${preview.priceCeilingsPerMillionUsd.output}/M`, + `Worst-case price-ceiling-derived reserve: $${plan.campaign.rootPriceDerivedSpendCeilingUsd.toFixed(6)} per campaign root; all four roots $${plan.totalRootPriceDerivedSpendCeilingUsd.toFixed(6)}`, + `Broker reservations: ${plan.totalCalls} calls, ${plan.totalTokens} tokens, $${plan.totalRootSpendCeilingUsd.toFixed(6)} root spend; operator cap ${operatorCapUsd === undefined ? "not supplied" : `$${operatorCapUsd.toFixed(6)}`}`, + `Schedule: engineering ${plan.engineering.developmentProofCells} development cells; each campaign ${plan.campaign.developmentProofCells} development + ${plan.campaign.confirmationScheduleCells} confirmation cells (${plan.campaign.confirmationNotSelectedCells} not-selected, ${plan.campaign.paidConfirmationCells} final, ${plan.campaign.rootControlCells} root-control)`, + `Worst-case serial task wall bound: ${(plan.totalTaskWallMs / 3_600_000).toFixed(1)} hours`, + `Seeds: ${[preview.seeds.engineering, ...preview.seeds.campaigns].join(", ")}`, + `Configured ceilings bound the estimate. No inference request was made.`, + ].join("\n") +} + +export async function buildProofPlanPreview(options: Options, env: NodeJS.ProcessEnv = process.env, fetcher: typeof fetch = fetch): Promise { + if (options.command !== "estimate" && options.command !== "plan") throw new Error("plan options are required") + if (options.seeds.length !== 4) throw new Error("provide one --seed for engineering and exactly three --campaign-seed values") + if (new Set(options.seeds).size !== 4) throw new Error("engineering and campaign seeds must be distinct") + const repo = path.resolve(options.repo) + const rootCommit = await ensureCleanRoot(repo) + const runtime = workerHarnessRuntimeIdentityFromEnvironment(env) + try { execFileSync("git", ["cat-file", "-e", `${runtime.harnessCommit}^{commit}`], { cwd: repo, stdio: "ignore" }) } + catch { throw new Error("fixed-control runtime harness commit must be present in the root repository") } + const parsed = parseRsiArgs(["--repo", repo, "--population", "2", "--generations", "3", "--seed", options.seeds[0]!, "--base-ref", rootCommit, "--max-runtime-ms", String(72 * 60 * 60_000), "--mutation-task", options.mutationTask]) + if (!parsed.config) throw new Error(parsed.error ?? "could not construct proof configuration") + const baseConfig = { ...parsed.config, model: RSI_OPENROUTER_MODEL, roles: { worker: { provider: "openrouter" as const, model: RSI_OPENROUTER_MODEL } }, mutationTask: options.mutationTask } + const configs = options.seeds.map((seed) => ({ ...baseConfig, seed })) as [typeof baseConfig, typeof baseConfig, typeof baseConfig, typeof baseConfig] + const config = configs[0] + const loaded = await loadBenchmark(path.join(repo, "fixtures/rsi-benchmark"), repo) + const modelKey = env.HEADLESSCODE_OPENROUTER_API_KEY?.trim() ?? "" + const price = await fetchPinnedOpenRouterEndpointPrice(modelKey, fetcher) + const mutationEnv = openRouterAllocatedEnvironment(env, "mutation") + const benchmarkEnv = openRouterAllocatedEnvironment(env, "benchmark") + const mutationLimits = openRouterBrokerLimitsFromEnvironment(mutationEnv, "plan-mutation", { maxCalls: config.maxIterations, maxDurationMs: config.commandTimeoutMs }) + const benchmarkLimits = openRouterBrokerLimitsFromEnvironment(benchmarkEnv, "plan-benchmark") + if (mutationLimits.maxInputPricePerMillionUsd < price.inputPricePerMillionUsd || mutationLimits.maxOutputPricePerMillionUsd < price.outputPricePerMillionUsd || benchmarkLimits.maxInputPricePerMillionUsd < price.inputPricePerMillionUsd || benchmarkLimits.maxOutputPricePerMillionUsd < price.outputPricePerMillionUsd) throw new Error("configured broker price ceiling is below the live pinned endpoint rate") + const plan = planProofDemonstration({ config, manifest: loaded.manifest, mutationLimits, benchmarkLimits }) + return { + createdAt: new Date().toISOString(), rootCommit, benchmarkManifestSha256: loaded.manifestDigest, + model: { provider: "openshell-openrouter", requestedId: RSI_OPENROUTER_MODEL, sampling: { temperature: 0, think: false, seed: "provider-default" } }, + endpointPrice: price, + priceCeilingsPerMillionUsd: { input: benchmarkLimits.maxInputPricePerMillionUsd, output: benchmarkLimits.maxOutputPricePerMillionUsd }, + runtime, + execution: { + configs, configSha256: frozenProofConfigDigest(configs), analysisSourceSha256: await analysisDigest(repo), + inferenceLimits: { mutation: mutationLimits, benchmark: benchmarkLimits }, + taskIdentities: loaded.manifest.tasks.map((task) => ({ id: task.id, split: task.split, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, knownGoodDigest: task.knownGoodDigest, knownBadDigest: task.knownBadDigest, ceilings: task.ceilings })), + }, + seeds: { engineering: options.seeds[0]!, campaigns: [options.seeds[1]!, options.seeds[2]!, options.seeds[3]!] }, + campaignIds: { engineering: campaignId(options.seeds[0]!), campaigns: [campaignId(options.seeds[1]!), campaignId(options.seeds[2]!), campaignId(options.seeds[3]!)] }, + plan, + } +} + +export async function buildFrozenProofPlan(options: Options, env: NodeJS.ProcessEnv = process.env, fetcher: typeof fetch = fetch): Promise> { + if (options.command !== "plan") throw new Error("plan options are required") + const operatorCapUsd = parseUsd(options.operatorCapUsd) + const preview = await buildProofPlanPreview(options, env, fetcher) + if (operatorCapUsd < preview.plan.totalRootSpendCeilingUsd) throw new Error(`operator cap $${operatorCapUsd.toFixed(6)} is below required root reservations $${preview.plan.totalRootSpendCeilingUsd.toFixed(6)}`) + return freezeProofDemonstrationManifest({ ...preview, operatorCapUsd }) +} + +export async function verifyLiveFrozenProofPlan(manifest: ReturnType, env: NodeJS.ProcessEnv = process.env, fetcher: typeof fetch = fetch): Promise { + const repo = path.resolve(manifest.execution.configs[0].repoRoot) + if (await ensureCleanRoot(repo) !== manifest.rootCommit) throw new Error("plan root commit or working tree differs from the frozen plan") + const loaded = await loadBenchmark(path.join(repo, "fixtures/rsi-benchmark"), repo) + if (loaded.manifestDigest !== manifest.benchmarkManifestSha256) throw new Error("benchmark manifest differs from the frozen plan") + if (canonicalHash(await analysisDigest(repo)) !== canonicalHash(manifest.execution.analysisSourceSha256)) throw new Error("analysis or launcher source differs from the frozen plan") + const runtime = workerHarnessRuntimeIdentityFromEnvironment(env) + if (JSON.stringify(runtime) !== JSON.stringify(manifest.runtime)) throw new Error("fixed-control runtime identity differs from the frozen plan") + const config = manifest.execution.configs[0] + const mutation = openRouterBrokerLimitsFromEnvironment(openRouterAllocatedEnvironment(env, "mutation"), "plan-mutation", { maxCalls: config.maxIterations, ...(config.maxTokens ? { maxOutputTokensPerCall: config.maxTokens } : {}), maxDurationMs: config.commandTimeoutMs }) + const benchmark = openRouterBrokerLimitsFromEnvironment(openRouterAllocatedEnvironment(env, "benchmark"), "plan-benchmark") + if (canonicalHash(mutation) !== canonicalHash(manifest.execution.inferenceLimits.mutation) || canonicalHash(benchmark) !== canonicalHash(manifest.execution.inferenceLimits.benchmark)) throw new Error("broker limits or allocations differ from the frozen plan") + const key = env.HEADLESSCODE_OPENROUTER_API_KEY?.trim() ?? "" + const price = await fetchPinnedOpenRouterEndpointPrice(key, fetcher) + if (price.model !== manifest.model.requestedId || price.provider !== manifest.endpointPrice.provider || price.inputPricePerMillionUsd > manifest.priceCeilingsPerMillionUsd.input || price.outputPricePerMillionUsd > manifest.priceCeilingsPerMillionUsd.output) throw new Error("live pinned endpoint price exceeds or differs from frozen plan ceilings") +} + +export async function proofDemonstrationPlanMain(argv = process.argv.slice(2)): Promise { + if (argv.includes("--help") || argv.includes("-h")) { process.stdout.write(usage()); return 0 } + try { + const options = parse(argv) + if (options.command === "estimate") { + process.stdout.write(`${formatPlan(await buildProofPlanPreview(options))}\n`) + return 0 + } + if (options.command === "verify") { + const safePath = await assertSupervisorOnlyPlanPath(options.repo, options.output) + const parsed = JSON.parse(await fs.readFile(safePath, "utf8")) as unknown + const manifest = verifyFrozenProofDemonstrationManifest(parsed) + await verifyLiveFrozenProofPlan(manifest) + process.stdout.write(`Verified frozen plan ${manifest.digestSha256}; operator cap $${manifest.operatorCapUsd.toFixed(6)}; paid provider inference has not been started.\n`) + return 0 + } + const manifest = await buildFrozenProofPlan(options) + const safePath = await assertSupervisorOnlyPlanPath(options.repo, options.output) + await fs.mkdir(path.dirname(safePath), { recursive: true, mode: 0o700 }) + await fs.writeFile(safePath, `${JSON.stringify(manifest, null, 2)}\n`, { mode: 0o600, flag: "wx" }) + process.stdout.write(`${formatPlan(manifest, manifest.operatorCapUsd, manifest.digestSha256)}\nPlan file: ${safePath}\n`) + return 0 + } catch (error) { + process.stderr.write(`rsi-proof-demonstration: ${error instanceof Error ? error.message : String(error)}\n`) + return 2 + } +} + +if (process.argv[1] && import.meta.url === pathToFileURL(path.resolve(process.argv[1])).href) process.exitCode = await proofDemonstrationPlanMain() diff --git a/src/rsi/__tests__/inference-broker.test.ts b/src/rsi/__tests__/inference-broker.test.ts index 5778dac..40dd681 100644 --- a/src/rsi/__tests__/inference-broker.test.ts +++ b/src/rsi/__tests__/inference-broker.test.ts @@ -3,7 +3,7 @@ import * as fs from "node:fs/promises" import * as os from "node:os" import * as path from "node:path" import test from "node:test" -import { FileOpenRouterInferenceLedger, openRouterCampaignBudgetAllocation, RSI_OPENROUTER_MODEL, startOpenRouterBroker, validateOpenRouterBrokerLimits } from "../inference-broker.js" +import { fetchPinnedOpenRouterEndpointPrice, FileOpenRouterInferenceLedger, openRouterCampaignBudgetAllocation, RSI_OPENROUTER_ENDPOINT_PROVIDER, RSI_OPENROUTER_ENDPOINT_TAG, RSI_OPENROUTER_MODEL, startOpenRouterBroker, validateOpenRouterBrokerLimits } from "../inference-broker.js" import type { OpenRouterBrokerLimits } from "../inference-broker.js" function limits(campaignId: string, overrides: Partial = {}): OpenRouterBrokerLimits { @@ -15,23 +15,32 @@ function limits(campaignId: string, overrides: Partial = } } -function fakeFetch(options: { usage?: Record; model?: string } = {}) { +function fakeFetch(options: { usage?: Record; model?: string; returnedProvider?: string | null } = {}) { const requests: Array<{ url: string; init?: RequestInit }> = [] const fetcher = (async (input: string | URL, init?: RequestInit) => { const url = String(input) requests.push({ url, init }) - if (url.includes("/models/")) return new Response(JSON.stringify({ data: { endpoints: [{ provider_slug: "deepseek", provider_name: "DeepSeek", pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200 }) + if (url.includes("/models/")) return new Response(JSON.stringify({ data: { endpoints: [endpointMetadata()] } }), { status: 200 }) const body = JSON.parse(String(init?.body)) as Record assert.equal(body.model, RSI_OPENROUTER_MODEL) - assert.deepEqual(body.provider, { order: ["deepseek"], allow_fallbacks: false }) - assert.equal(body.max_completion_tokens, 64) + assert.deepEqual(body.provider, { only: [RSI_OPENROUTER_ENDPOINT_PROVIDER], allow_fallbacks: false }) + assert.equal(body.max_tokens, 64) + assert.equal((init?.headers as Record)["X-OpenRouter-Metadata"], "enabled") assert.equal((init?.headers as Record).authorization, "Bearer host-secret") - const response = { id: "or-test-1", model: options.model ?? RSI_OPENROUTER_MODEL, choices: [], usage: options.usage ?? { prompt_tokens: 20, completion_tokens: 10, cost: 0.0002 } } + const response = routedResponse({ id: "or-test-1", model: options.model ?? RSI_OPENROUTER_MODEL, choices: [], usage: options.usage ?? { prompt_tokens: 20, completion_tokens: 10, cost: 0.0002 } }, options.returnedProvider) return new Response(JSON.stringify(response), { status: 200, headers: { "content-type": "application/json" } }) }) as typeof fetch return { fetcher, requests } } +function endpointMetadata(overrides: Record = {}) { + return { provider_name: "DeepInfra", tag: RSI_OPENROUTER_ENDPOINT_TAG, quantization: "fp8", status: 0, pricing: { prompt: "0.00000006", completion: "0.00000018" }, supported_parameters: ["tools", "tool_choice", "temperature", "seed", "max_tokens"], supports_tool_choice: { auto: true, required: true, function: true }, ...overrides } +} + +function routedResponse(response: Record, selectedProvider: string | null = RSI_OPENROUTER_ENDPOINT_PROVIDER): Record { + return selectedProvider === null ? response : { ...response, openrouter_metadata: { endpoints: { available: [{ provider: selectedProvider, selected: true }] } } } +} + const requestBody = (extra: Record = {}) => JSON.stringify({ model: RSI_OPENROUTER_MODEL, messages: [{ role: "user", content: "Fix the small function and run its tests." }], @@ -39,6 +48,32 @@ const requestBody = (extra: Record = {}) => JSON.stringify({ ...extra, }) +test("pinned endpoint pricing is a timestamped metadata GET with no inference submission", async () => { + const requests: Array<{ url: string; method?: string }> = [] + const fetcher = (async (input: string | URL, init?: RequestInit) => { + requests.push({ url: String(input), method: init?.method }) + return new Response(JSON.stringify({ data: { endpoints: [endpointMetadata({ pricing: { prompt: "0.000002", completion: "0.00003" } })] } }), { status: 200 }) + }) as typeof fetch + const price = await fetchPinnedOpenRouterEndpointPrice("metadata-test-key", fetcher, () => 1_800_000_000_000) + assert.equal(price.model, RSI_OPENROUTER_MODEL) + assert.equal(price.provider, "deepinfra") + assert.equal(price.endpointTag, "deepinfra/fp8") + assert.equal(price.quantization, "fp8") + assert.equal(price.inputPricePerMillionUsd, 2) + assert.equal(price.outputPricePerMillionUsd, 30) + assert.equal(price.observedAt, "2027-01-15T08:00:00.000Z") + assert.equal(requests.length, 1) + assert.equal(requests[0]?.method, "GET") + assert.match(requests[0]?.url ?? "", /\/models\/deepseek\/deepseek-v4-flash-0731\/endpoints$/) +}) + +test("pinned endpoint preflight rejects unavailable, retagged, or parameter-incomplete routes", async () => { + for (const endpoint of [endpointMetadata({ status: -2 }), endpointMetadata({ tag: "deepinfra/other" }), endpointMetadata({ supported_parameters: ["tools"] }), endpointMetadata({ supports_tool_choice: { auto: false, required: false } })]) { + const fetcher = (async () => new Response(JSON.stringify({ data: { endpoints: [endpoint] } }), { status: 200 })) as typeof fetch + await assert.rejects(fetchPinnedOpenRouterEndpointPrice("", fetcher), /pinned DeepInfra fp8 endpoint/) + } +}) + async function setup(t: { after(fn: () => void | Promise): void }, overrides: { usage?: Record; model?: string; limits?: OpenRouterBrokerLimits; upstreamFetch?: typeof fetch } = {}) { const root = await fs.mkdtemp(path.join(os.tmpdir(), "rsi-inference-broker-")) t.after(() => fs.rm(root, { recursive: true, force: true })) @@ -72,6 +107,21 @@ test("OpenRouter broker pins model/provider, accounts usage, and survives broker finally { await restarted.close() } }) +test("OpenRouter broker leaves missing or mismatched returned endpoint identity unqualified", async (t) => { + for (const returnedProvider of ["openai", null] as const) { + const { broker } = await setup(t, { upstreamFetch: fakeFetch({ returnedProvider }).fetcher }) + const response = await fetch(`${broker.localBaseUrl}/api/v1/chat/completions`, { + method: "POST", headers: { authorization: `Bearer ${broker.capability}`, "content-type": "application/json", "x-rsi-call-id": `call-${returnedProvider ?? "missing"}-provider-123` }, body: requestBody(), + }) + assert.equal(response.status, 502) + const receipt = await broker.receipt() + assert.equal(receipt.qualified, false) + assert.equal(receipt.callReceipts[0]?.authoritative, false) + assert.deepEqual(receipt.returnedProviders, returnedProvider ? [returnedProvider] : []) + await broker.close() + } +}) + test("file ledger preserves the job deadline across broker restart and rejects a late response", async (t) => { const root = await fs.mkdtemp(path.join(os.tmpdir(), "rsi-inference-deadline-")) t.after(() => fs.rm(root, { recursive: true, force: true })) @@ -84,7 +134,7 @@ test("file ledger preserves the job deadline across broker restart and rejects a let dispatched!: () => void const wasDispatched = new Promise((resolve) => { dispatched = resolve }) const upstreamFetch = (async (input: string | URL) => { - if (String(input).includes("/models/")) return new Response(JSON.stringify({ data: { endpoints: [{ provider_slug: "deepseek", pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200 }) + if (String(input).includes("/models/")) return new Response(JSON.stringify({ data: { endpoints: [endpointMetadata({ pricing: { prompt: "0.000001", completion: "0.000001" } })] } }), { status: 200 }) dispatched() return await new Promise((resolve) => { releaseUpstream = resolve }) }) as typeof fetch @@ -106,7 +156,7 @@ test("file ledger preserves the job deadline across broker restart and rejects a }) await wasDispatched nowMs = startedAt + 1_001 - releaseUpstream(new Response(JSON.stringify({ id: "late", model: RSI_OPENROUTER_MODEL, choices: [], usage: { prompt_tokens: 20, completion_tokens: 10, cost: 0.0002 } }), { status: 200, headers: { "content-type": "application/json" } })) + releaseUpstream(new Response(JSON.stringify(routedResponse({ id: "late", model: RSI_OPENROUTER_MODEL, choices: [], usage: { prompt_tokens: 20, completion_tokens: 10, cost: 0.0002 } })), { status: 200, headers: { "content-type": "application/json" } })) const response = await post assert.equal(response.status, 429, "a response arriving after the persisted job deadline is not returned to the guest") assert.match((await response.json() as { error: { message: string } }).error.message, /duration limit exhausted/) @@ -183,7 +233,7 @@ test("OpenRouter broker can expose status-only access counters without request d test("OpenRouter broker marks timeouts unmeasured and does not automatically spend again", async (t) => { let dispatched = 0 const upstream = (async (input: string | URL, init?: RequestInit) => { - if (String(input).includes("/models/")) return new Response(JSON.stringify({ data: { endpoints: [{ provider_slug: "deepseek", pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200 }) + if (String(input).includes("/models/")) return new Response(JSON.stringify({ data: { endpoints: [endpointMetadata({ pricing: { prompt: "0.000001", completion: "0.000001" } })] } }), { status: 200 }) dispatched += 1 return await new Promise((_resolve, reject) => init?.signal?.addEventListener("abort", () => reject(new DOMException("timed out", "AbortError")), { once: true })) }) as typeof fetch @@ -201,8 +251,8 @@ test("OpenRouter broker marks timeouts unmeasured and does not automatically spe test("OpenRouter broker parses final SSE usage when streaming is enabled", async (t) => { const sseFetch = (async (input: string | URL) => { - if (String(input).includes("/models/")) return new Response(JSON.stringify({ data: { endpoints: [{ provider_slug: "deepseek", pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200 }) - return new Response(`data: ${JSON.stringify({ id: "stream-id", model: RSI_OPENROUTER_MODEL, choices: [{ delta: { content: "done" } }] })}\n\ndata: ${JSON.stringify({ id: "stream-id", model: RSI_OPENROUTER_MODEL, usage: { prompt_tokens: 8, completion_tokens: 2, cost: 0.00001 } })}\n\ndata: [DONE]\n\n`, { status: 200, headers: { "content-type": "text/event-stream" } }) + if (String(input).includes("/models/")) return new Response(JSON.stringify({ data: { endpoints: [endpointMetadata({ pricing: { prompt: "0.000001", completion: "0.000001" } })] } }), { status: 200 }) + return new Response(`data: ${JSON.stringify({ id: "stream-id", model: RSI_OPENROUTER_MODEL, choices: [{ delta: { content: "done" } }] })}\n\ndata: ${JSON.stringify(routedResponse({ id: "stream-id", model: RSI_OPENROUTER_MODEL, usage: { prompt_tokens: 8, completion_tokens: 2, cost: 0.00001 } }))}\n\ndata: [DONE]\n\n`, { status: 200, headers: { "content-type": "text/event-stream" } }) }) as typeof fetch const { broker } = await setup(t, { upstreamFetch: sseFetch }) const response = await fetch(`${broker.localBaseUrl}/api/v1/chat/completions`, { diff --git a/src/rsi/__tests__/openshell.test.ts b/src/rsi/__tests__/openshell.test.ts index 15fcd4c..cda4880 100644 --- a/src/rsi/__tests__/openshell.test.ts +++ b/src/rsi/__tests__/openshell.test.ts @@ -26,6 +26,7 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker fs.mkdirSync(path.join(repo, "src", "engine"), { recursive: true }) fs.mkdirSync(path.join(repo, "src", "rsi"), { recursive: true }) fs.mkdirSync(path.join(repo, "scripts", "eval-suite"), { recursive: true }) + fs.mkdirSync(path.join(repo, "scripts", "rsi-benchmark", "manual-tests"), { recursive: true }) fs.mkdirSync(path.join(repo, "test"), { recursive: true }) fs.mkdirSync(path.join(repo, "fixtures", "rsi-benchmark"), { recursive: true }) fs.mkdirSync(path.join(repo, "__rsi_benchmark_verifier__"), { recursive: true }) @@ -38,6 +39,8 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker fs.writeFileSync(path.join(repo, "package.json"), "{\"name\":\"fixture\"}\n") fs.writeFileSync(path.join(repo, "src", "rsi", "evaluator.ts"), "export const hiddenScoring = 'secret'\n") fs.writeFileSync(path.join(repo, "scripts", "eval-suite", "hidden.sh"), "secret\n") + fs.writeFileSync(path.join(repo, "scripts", "rsi-benchmark", "task-spec.mjs"), 'export const confirmationFix = "private-known-good-sha"\n') + fs.writeFileSync(path.join(repo, "scripts", "rsi-benchmark", "manual-tests", "confirmation.mjs"), 'throw new Error("private confirmation oracle")\n') fs.writeFileSync(path.join(repo, "test", "engine.test.ts"), "test secret\n") fs.writeFileSync(path.join(repo, ".env"), "TOKEN=secret\n") execFileSync("git", ["add", "."], { cwd: repo }) @@ -116,6 +119,8 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker sessionEnv = request.env ?? {} assert.equal(path.relative(repo, workspace).startsWith(".."), false) assert.equal(fs.existsSync(path.join(workspace, "fixtures", "rsi-benchmark", "confirmation.json")), false, "mutation guest cannot read confirmation fixtures") + assert.equal(fs.existsSync(path.join(workspace, "scripts", "rsi-benchmark", "task-spec.mjs")), false, "mutation guest cannot read confirmation task descriptions or known-good SHAs") + assert.equal(fs.existsSync(path.join(workspace, "scripts", "rsi-benchmark", "manual-tests", "confirmation.mjs")), false, "mutation guest cannot read confirmation verifier code") assert.equal(fs.existsSync(path.join(workspace, "__rsi_benchmark_verifier__", "verify.mjs")), false, "mutation guest cannot read verifier source") assert.equal(fs.existsSync(path.join(workspace, "node_modules", "private.js")), false, "candidate dependencies are excluded from the guest snapshot") assert.throws(() => fs.readFileSync(path.join(workspace, "__rsi_benchmark_verifier__", "verify.mjs")), /ENOENT/, "an attempted verifier read has no guest-visible file") @@ -138,6 +143,8 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker assert.match(command, /test -r \/workspace\/src\/cli\.ts/, "guest preflight must reject a missing self-hosted CLI") assert.match(command, /readlink \/workspace\/node_modules/) assert.match(command, /test \! -e \/workspace\/__rsi_benchmark_verifier__/) + assert.match(command, /test \! -e \/opt\/headlesscode\/scripts\/rsi-benchmark\/task-spec\.mjs/) + assert.match(command, /test \! -e \/opt\/headlesscode\/fixtures\/rsi-benchmark/) if (["self-hosted-missing-cli", "self-hosted-corrupt-cli", "dependency-escape"].includes(mode)) { assert.match(command, /\/workspace\/src\/cli\.ts/) assert.match(command, /test "\$\(readlink \/workspace\/node_modules\)" = \/opt\/headlesscode\/node_modules/) @@ -319,7 +326,7 @@ await run("dependency-escape") execFileSync("git", ["init", "--quiet", "--initial-branch=main"], { cwd: repo }) execFileSync("git", ["config", "user.name", "Test"], { cwd: repo }) execFileSync("git", ["config", "user.email", "test@example.invalid"], { cwd: repo }) - for (const [relative, body] of [["src/app.ts", "source\n"], ["test/app.test.ts", "visible test\n"], ["scripts/eval-suite/hidden.js", "hidden suite\n"], ["src/rsi/fitness.ts", "hidden scorer\n"], [".headlesscode/archive.json", "operator archive\n"], ["fixtures/rsi-curriculum/generalization.json", "{}\n"], ["fixtures/rsi-benchmark/confirmation.json", "private proof task\n"]]) { + for (const [relative, body] of [["src/app.ts", "source\n"], ["test/app.test.ts", "visible test\n"], ["scripts/eval-suite/hidden.js", "hidden suite\n"], ["src/rsi/fitness.ts", "hidden scorer\n"], [".headlesscode/archive.json", "operator archive\n"], [".headlesscode/rsi-proof/plan.json", "sealed confirmation metadata\n"], ["fixtures/rsi-curriculum/generalization.json", "{}\n"], ["fixtures/rsi-benchmark/confirmation.json", "private proof task\n"]]) { const target = path.join(repo, relative) fs.mkdirSync(path.dirname(target), { recursive: true }) fs.writeFileSync(target, body) @@ -341,6 +348,7 @@ await run("dependency-escape") assert.equal(fs.existsSync(path.join(clonePath, ".headlesscode/archive.json")), false) assert.equal(fs.existsSync(path.join(clonePath, "fixtures/rsi-curriculum/generalization.json")), true, "registered curriculum fixtures remain available to evaluation guests") assert.equal(fs.existsSync(path.join(clonePath, "fixtures/rsi-benchmark/confirmation.json")), false, "benchmark task and verifier data stays outside evaluation guests") + assert.equal(fs.existsSync(path.join(clonePath, ".headlesscode/rsi-proof/plan.json")), false, "frozen operator plan and confirmation metadata stay outside evaluation guests") assert.equal(fs.lstatSync(path.join(clonePath, "test/app.test.ts")).isSymbolicLink(), false, "bundle transfer excludes harness-only dependency symlinks") const mutation = createMutationSnapshotBundle(repo, head, { protectedPaths: [], repoRoot: repo } as unknown as RsiConfig) const mutationBundle = path.join(unpack, "mutation.bundle") @@ -349,12 +357,22 @@ await run("dependency-escape") execFileSync("git", ["clone", "--quiet", mutationBundle, mutationClone], { cwd: unpack }) assert.equal(fs.existsSync(path.join(mutationClone, "fixtures/rsi-curriculum/generalization.json")), false, "mutation guests never receive evaluator-owned fixture assets") assert.equal(fs.existsSync(path.join(mutationClone, "fixtures/rsi-benchmark/confirmation.json")), false, "mutation guests never receive benchmark task and verifier data") + assert.equal(fs.existsSync(path.join(mutationClone, ".headlesscode/rsi-proof/plan.json")), false, "mutation guests never receive the frozen proof plan") } finally { fs.rmSync(repo, { recursive: true, force: true }) fs.rmSync(unpack, { recursive: true, force: true }) } } -console.log("RSI OpenShell isolation assertions passed") + const runtimeDockerfile = fs.readFileSync(path.join(process.cwd(), "docker", "OpenShell.Dockerfile"), "utf8") + const rsiDockerfile = fs.readFileSync(path.join(process.cwd(), "docker", "OpenShell-RSI.Dockerfile"), "utf8") + const dockerIgnore = fs.readFileSync(path.join(process.cwd(), ".dockerignore"), "utf8") + assert.match(dockerIgnore, /^scripts\/$/m, "benchmark corpus source is excluded from Docker build context") + assert.match(runtimeDockerfile, /COPY scripts\/spawn-parallel-worktrees\.sh scripts\/run-worker\.sh scripts\/headlesscode-answer\.sh \.\/scripts\//, "runtime image copies only the named operational scripts") + assert.doesNotMatch(runtimeDockerfile, /COPY (?:\.\/)?scripts\s/, "runtime image does not copy the scripts tree") + assert.match(runtimeDockerfile, /test ! -e \/opt\/headlesscode\/scripts\/rsi-benchmark\/task-spec\.mjs/, "runtime image build fails if task descriptions and known-good SHAs enter the image") + assert.match(rsiDockerfile, /test ! -e \/opt\/headlesscode\/fixtures\/rsi-benchmark/, "RSI image build fails if confirmation fixtures enter the image") + assert.match(rsiDockerfile, /test ! -e \/opt\/headlesscode\/\.git/, "RSI guest image cannot expose harness history containing known fixes") + console.log("RSI OpenShell isolation assertions passed") { const repo = fs.mkdtempSync(path.join(os.tmpdir(), "hc-rsi-fleet-patch-")) diff --git a/src/rsi/__tests__/proof-campaign.test.ts b/src/rsi/__tests__/proof-campaign.test.ts index 46261c0..c8a94cc 100644 --- a/src/rsi/__tests__/proof-campaign.test.ts +++ b/src/rsi/__tests__/proof-campaign.test.ts @@ -98,13 +98,13 @@ test("controller freezes paired broker allocations as separate campaigns under o const oldEnv = { ...process.env } try { Object.assign(process.env, { - HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD: "1", - HEADLESSCODE_RSI_OPENROUTER_MUTATION_BUDGET_USD: "0.4", - HEADLESSCODE_RSI_OPENROUTER_BENCHMARK_BUDGET_USD: "0.2", - HEADLESSCODE_RSI_OPENROUTER_MAX_INPUT_PRICE_PER_MILLION_USD: "2", - HEADLESSCODE_RSI_OPENROUTER_MAX_OUTPUT_PRICE_PER_MILLION_USD: "4", - HEADLESSCODE_RSI_OPENROUTER_MAX_CALL_SPEND_USD: "0.1", - HEADLESSCODE_RSI_OPENROUTER_MAX_JOB_SPEND_USD: "0.2", + HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD: "0.32", + HEADLESSCODE_RSI_OPENROUTER_MUTATION_BUDGET_USD: "0.04", + HEADLESSCODE_RSI_OPENROUTER_BENCHMARK_BUDGET_USD: "0.28", + HEADLESSCODE_RSI_OPENROUTER_MAX_INPUT_PRICE_PER_MILLION_USD: "0.0001", + HEADLESSCODE_RSI_OPENROUTER_MAX_OUTPUT_PRICE_PER_MILLION_USD: "0.0001", + HEADLESSCODE_RSI_OPENROUTER_MAX_CALL_SPEND_USD: "0.01", + HEADLESSCODE_RSI_OPENROUTER_MAX_JOB_SPEND_USD: "0.01", }) const task = { id: "proof-task", family: "controller-integration", split: "development" as const, task: "change a real repository behavior", @@ -135,10 +135,19 @@ test("controller freezes paired broker allocations as separate campaigns under o assert.deepEqual(frozen.armAssignment, { fixedControl: "fixed-control", selfHosted: "self-hosted" }) assert.equal(frozen.benchmarkDigests.manifest, "4".repeat(64), "the full loaded manifest digest, including runner/version metadata, is frozen") assert.deepEqual(frozen.settings.inferenceCampaignIds, { mutation: `${campaignId}:mutation`, benchmark: `${campaignId}:benchmark` }) - assert.deepEqual(frozen.settings.inferenceAllocationsUsd, { mutation: 0.4, benchmark: 0.2 }) + assert.deepEqual(frozen.settings.inferenceAllocationsUsd, { mutation: 0.04, benchmark: 0.28 }) assert.equal(frozen.schedule.filter((cell) => cell.armId === "fixed-control").length, 14) assert.equal(frozen.schedule.filter((cell) => cell.armId === "self-hosted").length, 14) - assert.equal(frozen.schedule.filter((cell) => cell.armId === "fixed-control" && cell.phase === "confirmation").length, 2, "final candidate and root-control confirmation cells are frozen") + assert.equal(frozen.schedule.filter((cell) => cell.armId === "fixed-control" && cell.phase === "confirmation").length, 5, "all generation candidate slots and root-control confirmation cells are frozen") + assert.equal(frozen.budgets.root.maxCalls, 40, "root calls sum mutation calls, all three development proof roles, and selected-final/root confirmation grants for both arms") + assert.equal(frozen.budgets.root.maxTokens, 645_120, "root tokens sum effective task-specific grants rather than the base per-call defaults") + assert.equal(frozen.budgets.root.maxSpendUsd, 0.2, "root spend is the sum of every frozen provider-job reservation") + assert.equal(frozen.settings.taskWallBoundMs, 140_000, "the runtime plan sums mutation, required evaluation, development proof, and selected-final/root confirmation deadlines") + assert.throws(() => createProofCampaignManifest({ config: { ...config, maxRuntimeMs: 139_999 }, campaignId: `${campaignId}-short`, baseCommit: "f".repeat(40), model: RSI_OPENROUTER_MODEL, provider: "openshell-openrouter", runtime: { imageName: "rsi:test", imageDigest: `sha256:${"9".repeat(64)}`, harnessCommit: "8".repeat(40), dependencyLockDigest: "7".repeat(64) }, developmentManifest, developmentManifestDigest: "4".repeat(64), plannedGenerations: 2, proofProtocol: protocol }), /below the summed frozen task-time ceiling/) + process.env.HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD = "0.19" + process.env.HEADLESSCODE_RSI_OPENROUTER_MUTATION_BUDGET_USD = "0.04" + process.env.HEADLESSCODE_RSI_OPENROUTER_BENCHMARK_BUDGET_USD = "0.15" + assert.throws(() => createProofCampaignManifest({ config, campaignId: `${campaignId}-underfunded`, baseCommit: "f".repeat(40), model: RSI_OPENROUTER_MODEL, provider: "openshell-openrouter", runtime: { imageName: "rsi:test", imageDigest: `sha256:${"9".repeat(64)}`, harnessCommit: "8".repeat(40), dependencyLockDigest: "7".repeat(64) }, developmentManifest, developmentManifestDigest: "4".repeat(64), plannedGenerations: 2, proofProtocol: protocol }), /below development plus selected-final\/root confirmation job reservations/) const fixed = await campaign.acquire(frozen, "fixed-control", "run-fixed", "coordinator-fixed") const selfHosted = await campaign.acquire(frozen, "self-hosted", "run-self", "coordinator-self") assert.equal(fixed.manifestSha256, selfHosted.manifestSha256) @@ -211,6 +220,7 @@ test("development completion preserves frozen confirmation cells for fenced post assert.equal(confirmationLease.epoch, lease.epoch + 1) assert.equal(confirmationLease.phase, "confirmation") await assert.rejects(() => campaign.finishArm(confirmationLease, ["candidate-a"]), /missing, incomplete, or uncertain frozen task cell/) + await campaign.selectFinalHarness(confirmationLease, { generation: 0, candidateId: "candidate-a", commit: "2".repeat(40) }) await campaign.markNotSelected(confirmationLease, { attemptKey: "not-selected-b", generation: 0, candidateId: "candidate-b", taskId: "hidden", seed: "seed-1", reason: "slot was not selected by the development decision" }) await assert.rejects(() => campaign.markNotSelected(confirmationLease, { attemptKey: "not-selected-a", generation: 0, candidateId: "candidate-a", taskId: "hidden", seed: "seed-1", reason: "selected slots cannot be skipped" }), /selected or undecided/) for (const cell of [schedule[2]!, schedule[4]!]) { @@ -471,7 +481,7 @@ test("runner replays a saved benchmark file through settlement after a crash and const patch = execFileSync("git", ["diff", "--binary", "--no-ext-diff", `${base}...HEAD`], { cwd: candidate, encoding: "buffer" }) return { validity: "valid", patch, stdout: "bounded mock response", stderr: "", calls: 1, tokens: 22, wallTimeMs: 10, stopReason: "mock-complete", - inferenceReceipt: { provider: "openrouter", jobId: request.attemptId, campaignId: frozen.campaignId, model: request.model.id, returnedModels: [request.model.id], calls: 1, promptTokens: 10, completionTokens: 12, totalTokens: 22, costUsd: 0.001, accounting: "supervisor-enforced", qualified: true, callReceipts: [], denials: [] }, + inferenceReceipt: { provider: "openrouter", endpointProvider: "deepinfra", jobId: request.attemptId, campaignId: frozen.campaignId, model: request.model.id, returnedModels: [request.model.id], returnedProviders: ["deepinfra"], calls: 1, promptTokens: 10, completionTokens: 12, totalTokens: 22, costUsd: 0.001, accounting: "supervisor-enforced", qualified: true, callReceipts: [], denials: [] }, } } finally { await fs.rm(candidate, { recursive: true, force: true }) } }, diff --git a/src/rsi/__tests__/proof-demonstration-manifest.test.ts b/src/rsi/__tests__/proof-demonstration-manifest.test.ts new file mode 100644 index 0000000..cc283b6 --- /dev/null +++ b/src/rsi/__tests__/proof-demonstration-manifest.test.ts @@ -0,0 +1,37 @@ +import assert from "node:assert/strict" +import { test } from "node:test" +import { freezeProofDemonstrationManifest, frozenProofConfigDigest, verifyFrozenProofDemonstrationManifest } from "../proof-demonstration-manifest.js" +import type { ProofDemonstrationPlan } from "../proof-demonstration-plan.js" +import type { RsiConfig, WorkerHarnessRuntimeIdentity } from "../types.js" + +const config = { repoRoot: "/work", seed: "eng-seed", protectedPaths: ["src/rsi/"], mutationTask: "fixed objective" } as RsiConfig +const plan = { totalRootSpendCeilingUsd: 4, totalCalls: 10, totalTokens: 100, totalTaskWallMs: 1000 } as ProofDemonstrationPlan +const runtime: WorkerHarnessRuntimeIdentity = { imageName: "headlesscode-openshell-rsi:release", imageDigest: `sha256:${"a".repeat(64)}`, harnessCommit: "b".repeat(40), dependencyLockDigest: "c".repeat(64) } +const limits = { campaignId: "plan-test", maxCalls: 2, maxInputTokensPerCall: 100, maxOutputTokensPerCall: 50, maxTotalTokens: 300, maxDurationMs: 1000, maxRequestBytes: 1000, maxResponseBytes: 1000, maxInputPricePerMillionUsd: 0.2, maxOutputPricePerMillionUsd: 0.3, maxCallSpendUsd: 0.01, maxJobSpendUsd: 0.02, maxCampaignSpendUsd: 1 } +const taskIdentities = [{ id: "task-one", split: "development", family: "engine", seed: 1, startDigest: "d".repeat(64), verifierDigest: "e".repeat(64), knownGoodDigest: "f".repeat(64), knownBadDigest: "1".repeat(64), ceilings: { timeoutMs: 1000, maxCalls: 3, maxOutputTokensPerCall: 100, maxPatchBytes: 1000 } }] + +function frozen(operatorCapUsd = 5, priceCeilingsPerMillionUsd = { input: 0.2, output: 0.3 }) { + return freezeProofDemonstrationManifest({ + createdAt: "2026-09-30T00:00:00.000Z", rootCommit: "2".repeat(40), benchmarkManifestSha256: "3".repeat(64), + model: { provider: "openshell-openrouter", requestedId: "deepseek/deepseek-v4-flash-0731", sampling: { temperature: 0, think: false, seed: "provider-default" } }, + endpointPrice: { model: "deepseek/deepseek-v4-flash-0731", provider: "deepinfra", endpointTag: "deepinfra/fp8", quantization: "fp8", supportedParameters: ["max_tokens", "temperature", "tools", "tool_choice"], inputPricePerMillionUsd: 0.1, outputPricePerMillionUsd: 0.2, observedAt: "2026-09-30T00:00:00.000Z", responseSha256: "4".repeat(64) }, + priceCeilingsPerMillionUsd: priceCeilingsPerMillionUsd, runtime, + execution: { configs: [config, config, config, config], configSha256: frozenProofConfigDigest([config, config, config, config]), analysisSourceSha256: "5".repeat(64), inferenceLimits: { mutation: limits, benchmark: limits }, taskIdentities }, + seeds: { engineering: "eng-seed", campaigns: ["campaign-a", "campaign-b", "campaign-c"] }, + campaignIds: { engineering: "eng-id", campaigns: ["a-id", "b-id", "c-id"] }, operatorCapUsd, plan, + }) +} + +test("frozen demonstration manifest verifies exact config, schedule, endpoint price and separate runtime identity", () => { + const body = frozen() + assert.equal(verifyFrozenProofDemonstrationManifest(body).runtime.harnessCommit, runtime.harnessCommit) + assert.equal(body.rootCommit, "2".repeat(40)) + const changed = structuredClone(body) + changed.execution.configs[2]!.mutationTask = "changed prompt" + assert.throws(() => verifyFrozenProofDemonstrationManifest(changed), /manifest digest mismatch/) +}) + +test("frozen plan rejects a cap below reservations and price ceilings below observed endpoint rates", () => { + assert.throws(() => frozen(3), /below the frozen root reservation/) + assert.throws(() => frozen(5, { input: 0.01, output: 0.01 }), /price ceilings do not cover/) +}) diff --git a/src/rsi/__tests__/proof-demonstration-plan.test.ts b/src/rsi/__tests__/proof-demonstration-plan.test.ts new file mode 100644 index 0000000..ebe8f0e --- /dev/null +++ b/src/rsi/__tests__/proof-demonstration-plan.test.ts @@ -0,0 +1,51 @@ +import assert from "node:assert/strict" +import { readFile } from "node:fs/promises" +import { planProofDemonstration } from "../proof-demonstration-plan.js" +import type { BenchmarkManifest } from "../benchmark/types.js" +import type { RsiConfig } from "../types.js" +import type { ProofRoleLimits } from "../proof-demonstration-plan.js" + +const manifest = { + schemaVersion: 1, runnerVersion: "test", createdFromCommit: "a".repeat(40), tasks: [ + { id: "dev-a", family: "a", split: "development", task: "dev a", startDigest: "b".repeat(64), verifierPath: "v", verifierDigest: "c".repeat(64), seed: 1, ceilings: { timeoutMs: 100, maxCalls: 2, maxOutputTokensPerCall: 100, maxPatchBytes: 1000 } }, + { id: "dev-b", family: "b", split: "development", task: "dev b", startDigest: "d".repeat(64), verifierPath: "v", verifierDigest: "e".repeat(64), seed: 2, ceilings: { timeoutMs: 100, maxCalls: 2, maxOutputTokensPerCall: 100, maxPatchBytes: 1000 } }, + { id: "confirm-a", family: "c", split: "confirmation", task: "confirm a", startDigest: "f".repeat(64), verifierPath: "v", verifierDigest: "1".repeat(64), seed: 3, ceilings: { timeoutMs: 200, maxCalls: 3, maxOutputTokensPerCall: 100, maxPatchBytes: 1000 } }, + ], +} as unknown as BenchmarkManifest +const config = { + population: 2, evalCommands: [], commandTimeoutMs: 100, maxIterations: 4, +} as unknown as RsiConfig +const limits = (campaignId: string): ProofRoleLimits => ({ + campaignId, maxCalls: 4, maxInputTokensPerCall: 1000, maxOutputTokensPerCall: 100, + maxTotalTokens: 16_000, maxDurationMs: 10_000, maxRequestBytes: 100_000, maxResponseBytes: 100_000, + maxInputPricePerMillionUsd: 0.1, maxOutputPricePerMillionUsd: 0.2, + maxCallSpendUsd: 0.01, maxJobSpendUsd: 0.2, maxCampaignSpendUsd: 100, +}) +const plan = planProofDemonstration({ config, manifest, mutationLimits: limits("mutation"), benchmarkLimits: limits("benchmark") }) +assert.deepEqual({ mutation: plan.engineering.mutationJobs, verification: plan.engineering.verificationJobs, dev: plan.engineering.developmentProofCells, confirmationSchedule: plan.engineering.confirmationScheduleCells }, { mutation: 8, verification: 16, dev: 48, confirmationSchedule: 10 }) +assert.deepEqual({ candidateConfirm: plan.campaign.confirmationScheduleCells, notSelected: plan.campaign.confirmationNotSelectedCells, final: plan.campaign.paidConfirmationCells, root: plan.campaign.rootControlCells }, { candidateConfirm: 14, notSelected: 10, final: 2, root: 2 }) +assert.equal(plan.engineering.maximumCalls, 128) +assert.equal(plan.campaign.maximumCalls, 204) +assert.equal(plan.totalReservedSpendUsd, plan.engineering.reservedSpendUsd + 3 * plan.campaign.reservedSpendUsd) +assert.equal(plan.totalCalls, plan.engineering.maximumCalls + 3 * plan.campaign.maximumCalls) +assert.throws(() => planProofDemonstration({ config, manifest, mutationLimits: { ...limits("mutation"), maxCallSpendUsd: 0.00001 }, benchmarkLimits: limits("benchmark") }), /per-call spend cap is below/) + +const corpus = JSON.parse(await readFile("fixtures/rsi-benchmark/manifest.json", "utf8")) as BenchmarkManifest +const corpusConfig = { ...config, generations: 3, maxIterations: 40, commandTimeoutMs: 900_000, evalCommands: ["npm test"], maxRuntimeMs: 72 * 60 * 60_000 } +const corpusMutationLimits = { ...limits("mutation"), maxCalls: 40, maxInputTokensPerCall: 16_000, maxOutputTokensPerCall: 4096, maxTotalTokens: 160_768, maxInputPricePerMillionUsd: 0.05, maxOutputPricePerMillionUsd: 0.16, maxCallSpendUsd: 0.01, maxJobSpendUsd: 0.04 } +const corpusBenchmarkLimits = { ...corpusMutationLimits, campaignId: "benchmark", maxCalls: 8, maxOutputTokensPerCall: 4096 } +const corpusPlan = planProofDemonstration({ config: corpusConfig, manifest: corpus, mutationLimits: corpusMutationLimits, benchmarkLimits: corpusBenchmarkLimits }) +assert.equal(corpusPlan.engineering.developmentProofCells, 384) +assert.equal(corpusPlan.engineering.confirmationDisposition, "pre-registered-deferred") +assert.equal(corpusPlan.engineering.paidConfirmationCells, 0) +assert.equal(corpusPlan.engineering.deferredConfirmationReservationCells, 64) +assert.equal(corpusPlan.campaign.developmentProofCells, 576) +assert.equal(corpusPlan.campaign.confirmationScheduleCells, 224) +assert.equal(corpusPlan.campaign.confirmationNotSelectedCells, 160) +assert.equal(corpusPlan.campaign.paidConfirmationCells, 32) +assert.equal(corpusPlan.campaign.rootControlCells, 32) +assert.equal(corpusPlan.campaign.maximumCalls, 19_680) +assert.equal(corpusPlan.totalCalls, 70_880) +assert.equal(corpusPlan.totalReservedSpendUsd, 93.92) +assert.equal(corpusPlan.totalRootSpendCeilingUsd, 96.48, "operator cap includes the engineering root's frozen, deferred final/root confirmation reservation") +console.log("RSI proof demonstration schedule and ceiling assertions passed") diff --git a/src/rsi/__tests__/proof-plan-path.test.ts b/src/rsi/__tests__/proof-plan-path.test.ts new file mode 100644 index 0000000..38f4757 --- /dev/null +++ b/src/rsi/__tests__/proof-plan-path.test.ts @@ -0,0 +1,19 @@ +import assert from "node:assert/strict" +import * as fs from "node:fs/promises" +import * as os from "node:os" +import * as path from "node:path" +import { execFileSync } from "node:child_process" +import test from "node:test" +import { assertSupervisorOnlyPlanPath } from "../../../scripts/rsi-proof-demonstration.js" + +test("frozen plan paths are external or ignored supervisor-only paths and reject symlink parents", async (t) => { + const repo = await fs.mkdtemp(path.join(os.tmpdir(), "rsi-plan-path-")) + t.after(() => fs.rm(repo, { recursive: true, force: true })) + execFileSync("git", ["init", "--quiet"], { cwd: repo }) + await fs.writeFile(path.join(repo, ".gitignore"), "/.headlesscode/\n") + assert.equal(await assertSupervisorOnlyPlanPath(repo, path.join(repo, ".headlesscode/rsi-proof/plan.json")), path.join(repo, ".headlesscode/rsi-proof/plan.json")) + assert.equal(await assertSupervisorOnlyPlanPath(repo, path.join(os.tmpdir(), "outside-plan.json")), path.join(os.tmpdir(), "outside-plan.json")) + await assert.rejects(assertSupervisorOnlyPlanPath(repo, path.join(repo, "src/leaked-plan.json")), /outside the source repository or under ignored/) + await fs.symlink(path.join(repo, "src"), path.join(repo, ".headlesscode")) + await assert.rejects(assertSupervisorOnlyPlanPath(repo, path.join(repo, ".headlesscode/rsi-proof/plan.json")), /symlink/) +}) diff --git a/src/rsi/__tests__/reports.test.ts b/src/rsi/__tests__/reports.test.ts index 7797611..b82b40d 100644 --- a/src/rsi/__tests__/reports.test.ts +++ b/src/rsi/__tests__/reports.test.ts @@ -7,7 +7,7 @@ test("candidate report includes authenticated broker-denial summary from its rec const candidate = { id: "candidate-1", status: "accepted", model: "deepseek/deepseek-v4-flash-0731", modelProvider: "openshell-openrouter", inferenceUsage: { - provider: "openrouter", model: "deepseek/deepseek-v4-flash-0731", returnedModels: ["deepseek/deepseek-v4-flash-0731"], + provider: "openrouter", endpointProvider: "deepinfra", model: "deepseek/deepseek-v4-flash-0731", returnedModels: ["deepseek/deepseek-v4-flash-0731"], calls: 1, promptTokens: 10, completionTokens: 2, totalTokens: 12, costUsd: 0.000012, accounting: "supervisor-enforced", qualified: true, jobId: "job-1", campaignId: "campaign-1", denials: [{ statusCode: 429, reason: "budget-or-fence", count: 2 }], diff --git a/src/rsi/archive.ts b/src/rsi/archive.ts index 12c18b6..b94733d 100644 --- a/src/rsi/archive.ts +++ b/src/rsi/archive.ts @@ -121,6 +121,11 @@ export async function findActiveRun(archiveDir: string, runId: string): Promise< return (await readArchive(archiveDir)).activeRuns.find((run) => run.runId === runId) } +export async function findRun(archiveDir: string, runId: string): Promise { + const archive = await readArchive(archiveDir) + return archive.activeRuns.find((run) => run.runId === runId) ?? archive.runs.find((run) => run.runId === runId) +} + export function archiveCandidate(archive: RsiArchive, candidate: CandidateRecord): RsiArchive { const index = archive.candidates.findIndex((entry) => entry.id === candidate.id) if (index >= 0) archive.candidates[index] = candidate diff --git a/src/rsi/benchmark/__tests__/runner.test.ts b/src/rsi/benchmark/__tests__/runner.test.ts index 6adfe82..93f9cb9 100644 --- a/src/rsi/benchmark/__tests__/runner.test.ts +++ b/src/rsi/benchmark/__tests__/runner.test.ts @@ -173,8 +173,8 @@ try { return { validity: "agent-error", stdout: "", stderr: "", calls: 1, tokens: 10, wallTimeMs: 1, stopReason: "fake", inferenceReceipt: { - provider: "openrouter", jobId: request.attemptId, campaignId: request.campaignId, model: "deepseek/deepseek-v4-flash-0731", - returnedModels: ["deepseek/deepseek-v4-flash-0731"], calls: 1, promptTokens: 5, completionTokens: 5, totalTokens: 10, + provider: "openrouter", endpointProvider: "deepinfra", jobId: request.attemptId, campaignId: request.campaignId, model: "deepseek/deepseek-v4-flash-0731", + returnedModels: ["deepseek/deepseek-v4-flash-0731"], returnedProviders: ["deepinfra"], calls: 1, promptTokens: 5, completionTokens: 5, totalTokens: 10, costUsd: 0.001, accounting: "supervisor-enforced", qualified: true, callReceipts: [], denials: [], }, } diff --git a/src/rsi/benchmark/runner.ts b/src/rsi/benchmark/runner.ts index 83e7200..04d1ea3 100644 --- a/src/rsi/benchmark/runner.ts +++ b/src/rsi/benchmark/runner.ts @@ -451,7 +451,7 @@ export async function runBenchmark(options: RunBenchmarkOptions & { replayFile?: harnessCommit: options.harnessCommit, agentImage, model: { provider: options.agent.provider, id: options.modelId, sampling }, seed: task.seed, family: task.family, startDigest: task.startDigest, ceilings: task.ceilings, verifiedSuccess, - proofQualified: Boolean(result.inferenceReceipt?.qualified && result.inferenceReceipt.provider === "openrouter" && result.inferenceReceipt.jobId === attemptId && result.inferenceReceipt.campaignId === campaignId && result.inferenceReceipt.model === options.modelId && result.inferenceReceipt.returnedModels.length === 1 && result.inferenceReceipt.returnedModels[0] === options.modelId), + proofQualified: Boolean(result.inferenceReceipt?.qualified && result.inferenceReceipt.provider === "openrouter" && result.inferenceReceipt.endpointProvider === "deepinfra" && result.inferenceReceipt.jobId === attemptId && result.inferenceReceipt.campaignId === campaignId && result.inferenceReceipt.model === options.modelId && result.inferenceReceipt.returnedModels.length === 1 && result.inferenceReceipt.returnedModels[0] === options.modelId), resourceAccounting: result.inferenceReceipt?.qualified ? "supervisor-enforced" : result.inferenceReceipt ? "incomplete" : result.calls !== null && result.tokens !== null ? "guest-reported" : "incomplete", ...(proofRequestSha256 ? { proofRequestSha256 } : {}), ...(result.inferenceReceipt ? { inferenceReceipt: result.inferenceReceipt } : {}), diff --git a/src/rsi/config.ts b/src/rsi/config.ts index 308e573..befe9ce 100644 --- a/src/rsi/config.ts +++ b/src/rsi/config.ts @@ -75,6 +75,7 @@ Options: --model-candidate-id model identity used for combination tracking --worker-harness-mode fixed-control | self-hosted (generation zero is always fixed-control) --worker-harness-pair-id shared identifier for matching fixed-control and self-hosted runs + --worker-harness-phase development | confirmation (default: development) --base-ref git ref to branch from (default: HEAD) --dry-run print the planned loop without changing files --keep-worktrees retain candidate worktrees after evaluation @@ -123,6 +124,7 @@ export function parseRsiArgs(argv: string[], cwd = process.cwd()): ParsedRsiArgs let resumeRunId: string | undefined let modelCandidateId: string | undefined let workerHarnessMode: WorkerHarnessMode | undefined + let workerHarnessPhase: "development" | "confirmation" = "development" let workerHarnessPairId: string | undefined let proofMinimumNetWins = DEFAULT_DEVELOPMENT_PROOF_RULE.minimumNetWins let proofRetentionTaskIds = [...DEFAULT_DEVELOPMENT_PROOF_RULE.retentionTaskIds] @@ -289,6 +291,13 @@ export function parseRsiArgs(argv: string[], cwd = process.cwd()): ParsedRsiArgs index = next break } + case "--worker-harness-phase": { + const [value, next] = take(index, arg) + if (value !== "development" && value !== "confirmation") throw new Error(`${arg} must be development or confirmation`) + workerHarnessPhase = value + index = next + break + } case "--proof-min-net-wins": { const [value, next] = take(index, arg) proofMinimumNetWins = positiveInteger(value, arg) @@ -417,6 +426,7 @@ export function parseRsiArgs(argv: string[], cwd = process.cwd()): ParsedRsiArgs resumeRunId, modelCandidateId, ...(workerHarnessMode ? { workerHarnessMode } : {}), + ...(workerHarnessMode ? { workerHarnessPhase } : {}), ...(workerHarnessPairId ? { workerHarnessPairId } : {}), }, } diff --git a/src/rsi/controller.ts b/src/rsi/controller.ts index 6d1bf53..2e4d8d6 100644 --- a/src/rsi/controller.ts +++ b/src/rsi/controller.ts @@ -4,7 +4,7 @@ import * as fs from "node:fs/promises" import * as path from "node:path" import { setTimeout as delay } from "node:timers/promises" import { fileURLToPath } from "node:url" -import { appendRun, checkpointRun, findActiveRun, readArchive } from "./archive.js" +import { appendRun, checkpointRun, findActiveRun, findRun, readArchive } from "./archive.js" import { parseRsiArgs, rsiHelp } from "./config.js" import { generateCurriculumProposals, validateCurriculumProposals, writeCurriculumProposals } from "./curriculum.js" import { evaluateCandidate } from "./evaluator.js" @@ -41,6 +41,7 @@ import { runDevelopmentBenchmarkForCommit } from "./development-benchmark.js" import { workerHarnessConfigDigest, workerHarnessRuntimeIdentityFromEnvironment, validWorkerHarnessRuntimeIdentity, effectiveWorkerHarnessMode } from "./worker-harness.js" import { workerHarnessCliPath } from "./worker-harness.js" import { PostgresProofCampaign, proofCampaignManifestDigest, type ProofCampaignArmLease, type ProofCampaignManifest } from "./proof-campaign.js" +import { runProofConfirmation } from "./proof-confirmation.js" function log(hooks: RsiHooks, line: string): void { (hooks.log ?? ((message) => process.stdout.write(`${message}\n`)))(line) @@ -51,6 +52,18 @@ function isTerminal(candidate: CandidateRecord): boolean { } function proofCandidateSlot(generation: number, index: number): string { return `g${generation}-candidate-${index}` } +export const RSI_CONFIRMATION_RETENTION_TASK_IDS = ["redirect-escape", "qa-baseline-counts", "strict-tool-schema", "dashboard-git-owner-api"] as const + +export function proofCampaignTaskWallBoundMs(config: RsiConfig, plannedGenerations: number, developmentTasks: BenchmarkManifest["tasks"], confirmationTasks: BenchmarkManifest["tasks"], includeConfirmation = true): number { + const slotsPerArm = plannedGenerations * config.population + const candidateCountBothArms = 2 * slotsPerArm + const candidateTaskMs = config.commandTimeoutMs * (3 + config.evalCommands.length) + + 3 * developmentTasks.reduce((sum, task) => sum + task.ceilings.timeoutMs, 0) + const confirmationTaskMs = includeConfirmation ? 4 * confirmationTasks.reduce((sum, task) => sum + task.ceilings.timeoutMs, 0) : 0 + const bound = candidateCountBothArms * candidateTaskMs + confirmationTaskMs + if (!Number.isSafeInteger(bound) || bound <= 0) throw new Error("proof campaign task-time ceiling exceeds safe integer accounting") + return bound +} export function createProofCampaignManifest(args: { config: RsiConfig @@ -69,6 +82,7 @@ export function createProofCampaignManifest(args: { const benchmarkTasks = args.developmentManifest.tasks.filter((task) => task.split === "development") const confirmationTasks = args.developmentManifest.tasks.filter((task) => task.split === "confirmation") if (!benchmarkTasks.length || !confirmationTasks.length) throw new Error("proof campaign requires nonempty development and confirmation benchmark splits") + if (RSI_CONFIRMATION_RETENTION_TASK_IDS.some((taskId) => !confirmationTasks.some((task) => task.id === taskId))) throw new Error("frozen benchmark manifest is missing a preregistered confirmation retention task") const taskDigest = (tasks: typeof args.developmentManifest.tasks) => createHash("sha256").update(JSON.stringify(tasks)).digest("hex") const mutationCampaignId = proofBrokerCampaignId(args.campaignId, "mutation") const benchmarkCampaignId = proofBrokerCampaignId(args.campaignId, "benchmark") @@ -76,6 +90,17 @@ export function createProofCampaignManifest(args: { const benchmarkEnv = openRouterAllocatedEnvironment(process.env, "benchmark") const benchmarkLimits = openRouterBrokerLimitsFromEnvironment(benchmarkEnv, benchmarkCampaignId, { maxDurationMs: Math.min(...benchmarkTasks.map((task) => task.ceilings.timeoutMs)) }) const confirmationLimits = openRouterBrokerLimitsFromEnvironment(benchmarkEnv, benchmarkCampaignId, { maxDurationMs: Math.min(...confirmationTasks.map((task) => task.ceilings.timeoutMs)) }) + const jobCostCeiling = (limits: typeof mutationLimits, maxCalls: number, maxOutputTokensPerCall: number): number => { + const perCallTokens = Math.min(limits.maxTotalTokens, limits.maxInputTokensPerCall + maxOutputTokensPerCall) + const jobTokens = Math.min(limits.maxTotalTokens, maxCalls * (limits.maxInputTokensPerCall + maxOutputTokensPerCall)) + const unitPrice = Math.max(limits.maxInputPricePerMillionUsd, limits.maxOutputPricePerMillionUsd) + const micros = (tokens: number) => Math.ceil(tokens * unitPrice) + if (micros(perCallTokens) > limits.maxCallSpendUsd * 1_000_000) throw new Error("OpenRouter per-call spend cap is below the price ceiling for the admitted token ceiling") + return micros(jobTokens) / 1_000_000 + } + if (jobCostCeiling(mutationLimits, mutationLimits.maxCalls, mutationLimits.maxOutputTokensPerCall) > mutationLimits.maxJobSpendUsd) throw new Error("OpenRouter mutation job spend cap is below its admitted token and price ceilings") + for (const task of benchmarkTasks) if (jobCostCeiling(benchmarkLimits, task.ceilings.maxCalls, task.ceilings.maxOutputTokensPerCall) > benchmarkLimits.maxJobSpendUsd) throw new Error(`OpenRouter benchmark job spend cap is below the frozen token and price ceiling for ${task.id}`) + for (const task of confirmationTasks) if (jobCostCeiling(confirmationLimits, task.ceilings.maxCalls, task.ceilings.maxOutputTokensPerCall) > confirmationLimits.maxJobSpendUsd) throw new Error(`OpenRouter confirmation job spend cap is below the frozen token and price ceiling for ${task.id}`) const candidateCounts = Array.from({ length: args.plannedGenerations }, (_, generation) => args.config.computePolicy === "adaptive-independent" && generation > 0 ? 1 : args.config.population) const schedule: ProofCampaignManifest["schedule"] = [] for (const armId of ["fixed-control", "self-hosted"]) for (let generation = 0; generation < args.plannedGenerations; generation++) { @@ -89,18 +114,35 @@ export function createProofCampaignManifest(args: { } } const finalGeneration = args.plannedGenerations - 1 - const finalSlots = candidateCounts[finalGeneration]! for (const armId of ["fixed-control", "self-hosted"]) { - for (let index = 0; index < finalSlots; index++) for (const task of confirmationTasks) schedule.push({ armId, generation: finalGeneration, candidateId: proofCandidateSlot(finalGeneration, index), taskId: task.id, seed: String(task.seed), attemptKind: "confirmation", phase: "confirmation" }) + for (let generation = 0; generation < args.plannedGenerations; generation++) { + for (let index = 0; index < candidateCounts[generation]!; index++) for (const task of confirmationTasks) schedule.push({ armId, generation, candidateId: proofCandidateSlot(generation, index), taskId: task.id, seed: String(task.seed), attemptKind: "confirmation", phase: "confirmation" }) + } for (const task of confirmationTasks) schedule.push({ armId, generation: finalGeneration, candidateId: "root-control", taskId: task.id, seed: String(task.seed), attemptKind: "fixed-control", phase: "confirmation" }) } - const rootCalls = 2 * candidateCounts.reduce((sum, count, generation) => sum + count * (mutationLimits.maxCalls + 3 * benchmarkTasks.length * benchmarkLimits.maxCalls) + (generation === finalGeneration ? (finalSlots + 1) * confirmationTasks.length * confirmationLimits.maxCalls : 0), 0) - const rootTokens = 2 * candidateCounts.reduce((sum, count, generation) => sum + count * (mutationLimits.maxTotalTokens + 3 * benchmarkTasks.length * benchmarkLimits.maxTotalTokens) + (generation === finalGeneration ? (finalSlots + 1) * confirmationTasks.length * confirmationLimits.maxTotalTokens : 0), 0) + const developmentCallsPerProofRole = benchmarkTasks.reduce((sum, task) => sum + task.ceilings.maxCalls, 0) + const confirmationCalls = 4 * confirmationTasks.reduce((sum, task) => sum + task.ceilings.maxCalls, 0) // selected final and root-control, for both arms + const rootCalls = 2 * candidateCounts.reduce((sum, count) => sum + count * (mutationLimits.maxCalls + 3 * developmentCallsPerProofRole), 0) + confirmationCalls + const developmentTokensPerProofRole = benchmarkTasks.reduce((sum, task) => sum + Math.min(benchmarkLimits.maxTotalTokens, task.ceilings.maxCalls * (benchmarkLimits.maxInputTokensPerCall + task.ceilings.maxOutputTokensPerCall)), 0) + const confirmationTokensPerHarness = confirmationTasks.reduce((sum, task) => sum + Math.min(confirmationLimits.maxTotalTokens, task.ceilings.maxCalls * (confirmationLimits.maxInputTokensPerCall + task.ceilings.maxOutputTokensPerCall)), 0) + const mutationTokens = Math.min(mutationLimits.maxTotalTokens, mutationLimits.maxCalls * (mutationLimits.maxInputTokensPerCall + mutationLimits.maxOutputTokensPerCall)) + const rootTokens = 2 * candidateCounts.reduce((sum, count) => sum + count * (mutationTokens + 3 * developmentTokensPerProofRole), 0) + 4 * confirmationTokensPerHarness if (!Number.isSafeInteger(rootCalls) || !Number.isSafeInteger(rootTokens)) throw new Error("proof campaign root call or token ceiling exceeds safe integer accounting") const spendAllocation = Number(process.env.HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD) if (!Number.isFinite(spendAllocation) || spendAllocation <= 0) throw new Error("HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD is required for a bounded proof campaign") const totalBudgetUsd = Number(process.env.HEADLESSCODE_RSI_OPENROUTER_MUTATION_BUDGET_USD) + Number(process.env.HEADLESSCODE_RSI_OPENROUTER_BENCHMARK_BUDGET_USD) if (!Number.isFinite(totalBudgetUsd) || totalBudgetUsd <= 0 || totalBudgetUsd > spendAllocation) throw new Error("proof campaign mutation and benchmark allocations must fit the operator spend ceiling") + const slotsPerArm = candidateCounts.reduce((sum, count) => sum + count, 0) + const mutationReservationsUsd = 2 * slotsPerArm * mutationLimits.maxJobSpendUsd + const developmentReservationsUsd = 2 * slotsPerArm * 3 * benchmarkTasks.length * benchmarkLimits.maxJobSpendUsd + const confirmationReservationsUsd = 4 * confirmationTasks.length * confirmationLimits.maxJobSpendUsd + const requiredBenchmarkUsd = developmentReservationsUsd + confirmationReservationsUsd + if (openRouterBrokerLimitsFromEnvironment(openRouterAllocatedEnvironment(process.env, "mutation"), mutationCampaignId).maxCampaignSpendUsd < mutationReservationsUsd) throw new Error("mutation broker campaign allocation is below the sum of frozen mutation job reservations") + if (benchmarkLimits.maxCampaignSpendUsd < requiredBenchmarkUsd) throw new Error("benchmark broker campaign allocation is below development plus selected-final/root confirmation job reservations") + if (totalBudgetUsd < mutationReservationsUsd + requiredBenchmarkUsd) throw new Error("proof campaign role allocations cannot fund every frozen provider job reservation") + const taskWallBoundMs = proofCampaignTaskWallBoundMs(args.config, args.plannedGenerations, benchmarkTasks, confirmationTasks) + const maxRuntimeMs = args.config.maxRuntimeMs ?? 60 * 60_000 + if (maxRuntimeMs < taskWallBoundMs) throw new Error(`proof campaign runtime ceiling ${maxRuntimeMs}ms is below the summed frozen task-time ceiling ${taskWallBoundMs}ms`) return { schemaVersion: 1, campaignId: args.campaignId, rootCommit: args.baseCommit, benchmarkDigests: { manifest: args.developmentManifestDigest, development: taskDigest(benchmarkTasks), confirmation: taskDigest(confirmationTasks) }, @@ -109,10 +151,10 @@ export function createProofCampaignManifest(args: { seed: args.config.seed, taskOrder: ["mutation", "regression", "typecheck", ...args.config.evalCommands.map((_, index) => `visible-${index}`), ...benchmarkTasks.map((task) => task.id), ...confirmationTasks.map((task) => task.id)], schedule, budgets: { attempt: { maxCalls: Math.max(mutationLimits.maxCalls, benchmarkLimits.maxCalls, confirmationLimits.maxCalls), maxTokens: Math.max(mutationLimits.maxTotalTokens, benchmarkLimits.maxTotalTokens, confirmationLimits.maxTotalTokens), maxSpendUsd: Math.max(mutationLimits.maxJobSpendUsd, benchmarkLimits.maxJobSpendUsd, confirmationLimits.maxJobSpendUsd), maxDurationMs: Math.max(args.config.commandTimeoutMs, ...benchmarkTasks.map((task) => task.ceilings.timeoutMs), ...confirmationTasks.map((task) => task.ceilings.timeoutMs)) }, - root: { maxCalls: rootCalls, maxTokens: rootTokens, maxSpendUsd: Math.min(spendAllocation, totalBudgetUsd), maxRuntimeMs: args.config.maxRuntimeMs ?? 60 * 60_000 }, + root: { maxCalls: rootCalls, maxTokens: rootTokens, maxSpendUsd: Math.min(spendAllocation, totalBudgetUsd, mutationReservationsUsd + requiredBenchmarkUsd), maxRuntimeMs }, }, - acceptanceRule: { ...args.proofProtocol.rule }, runtime: { imageDigest: args.runtime.imageDigest, harnessCommit: args.runtime.harnessCommit, dependencyLockDigest: args.runtime.dependencyLockDigest }, - analysisVersion: args.proofProtocol.schemaVersion.toString(), settings: { configDigest: workerHarnessConfigDigest(args.config, { baseCommit: args.baseCommit, model: args.model, provider: args.provider, proofProtocolDigest: args.proofProtocol.protocolDigest }), candidateCounts, evaluationCells: ["regression", "typecheck", ...args.config.evalCommands.map((_, index) => `visible-${index}`)], inferenceCampaignIds: { mutation: mutationCampaignId, benchmark: benchmarkCampaignId }, inferenceAllocationsUsd: { mutation: mutationLimits.maxCampaignSpendUsd, benchmark: benchmarkLimits.maxCampaignSpendUsd } }, + acceptanceRule: { ...args.proofProtocol.rule, confirmationRetentionTaskIds: [...RSI_CONFIRMATION_RETENTION_TASK_IDS], minimumSelfHostedCampaigns: 2, requiredSuccessiveTransitions: 2, clusterConfidence: 0.9 }, runtime: { imageDigest: args.runtime.imageDigest, harnessCommit: args.runtime.harnessCommit, dependencyLockDigest: args.runtime.dependencyLockDigest }, + analysisVersion: args.proofProtocol.schemaVersion.toString(), settings: { configDigest: workerHarnessConfigDigest(args.config, { baseCommit: args.baseCommit, model: args.model, provider: args.provider, proofProtocolDigest: args.proofProtocol.protocolDigest }), candidateCounts, taskWallBoundMs, evaluationCells: ["regression", "typecheck", ...args.config.evalCommands.map((_, index) => `visible-${index}`)], developmentTaskIdentities: benchmarkTasks.map((task) => ({ id: task.id, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings })), confirmationTaskIdentities: confirmationTasks.map((task) => ({ id: task.id, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings })), inferenceCeilings: { maxInputTokensPerCall: benchmarkLimits.maxInputTokensPerCall, maxOutputTokensPerCall: benchmarkLimits.maxOutputTokensPerCall, maxCallsPerAttempt: benchmarkLimits.maxCalls, maxTotalTokens: benchmarkLimits.maxTotalTokens, maxCallSpendUsd: benchmarkLimits.maxCallSpendUsd, maxJobSpendUsd: benchmarkLimits.maxJobSpendUsd, maxInputPricePerMillionUsd: benchmarkLimits.maxInputPricePerMillionUsd, maxOutputPricePerMillionUsd: benchmarkLimits.maxOutputPricePerMillionUsd }, inferenceCampaignIds: { mutation: mutationCampaignId, benchmark: benchmarkCampaignId }, inferenceAllocationsUsd: { mutation: mutationLimits.maxCampaignSpendUsd, benchmark: benchmarkLimits.maxCampaignSpendUsd } }, } } @@ -315,14 +357,14 @@ function recordFleetInferenceUsage(candidate: CandidateRecord, resultValue: unkn if (!resultValue || typeof resultValue !== "object" || !payload.inferenceLimits) throw new Error("OpenRouter fleet mutation result is missing its broker identity") const result = resultValue as { provider?: unknown; model?: unknown; inferenceReceipt?: Record } const receipt = result.inferenceReceipt - if (result.provider !== "openrouter" || result.model !== payload.model || !receipt || receipt.provider !== "openrouter" || receipt.jobId !== jobId || receipt.campaignId !== payload.inferenceLimits.campaignId || receipt.model !== payload.model || receipt.accounting !== "supervisor-enforced" || receipt.qualified !== true || !Array.isArray(receipt.returnedModels) || receipt.returnedModels.length !== 1 || receipt.returnedModels[0] !== payload.model) { + if (result.provider !== "openrouter" || result.model !== payload.model || !receipt || receipt.provider !== "openrouter" || receipt.endpointProvider !== "deepinfra" || receipt.jobId !== jobId || receipt.campaignId !== payload.inferenceLimits.campaignId || receipt.model !== payload.model || receipt.accounting !== "supervisor-enforced" || receipt.qualified !== true || !Array.isArray(receipt.returnedModels) || receipt.returnedModels.length !== 1 || receipt.returnedModels[0] !== payload.model || !Array.isArray(receipt.returnedProviders) || receipt.returnedProviders.length !== 1 || receipt.returnedProviders[0] !== "deepinfra") { throw new Error("OpenRouter fleet mutation result lacks a qualified receipt bound to the admitted provider, model, job and campaign") } const values = [receipt.calls, receipt.promptTokens, receipt.completionTokens, receipt.totalTokens, receipt.costUsd] if (!values.every((value) => typeof value === "number" && Number.isFinite(value) && value >= 0) || !Number.isSafeInteger(receipt.calls) || !Number.isSafeInteger(receipt.promptTokens) || !Number.isSafeInteger(receipt.completionTokens) || !Number.isSafeInteger(receipt.totalTokens)) throw new Error("OpenRouter fleet mutation receipt contains malformed resource totals") if (!Array.isArray(receipt.denials) || !receipt.denials.every((entry) => entry && typeof entry === "object" && Number.isSafeInteger((entry as OpenRouterDenialSummary).statusCode) && (entry as OpenRouterDenialSummary).statusCode >= 400 && (entry as OpenRouterDenialSummary).statusCode <= 599 && Number.isSafeInteger((entry as OpenRouterDenialSummary).count) && (entry as OpenRouterDenialSummary).count >= 1 && /^[a-z-]+$/.test((entry as OpenRouterDenialSummary).reason))) throw new Error("OpenRouter fleet mutation receipt contains malformed denial summaries") candidate.inferenceUsage = { - provider: "openrouter", model: payload.model, returnedModels: receipt.returnedModels as string[], calls: receipt.calls as number, + provider: "openrouter", endpointProvider: "deepinfra", model: payload.model, returnedModels: receipt.returnedModels as string[], returnedProviders: receipt.returnedProviders as string[], calls: receipt.calls as number, promptTokens: receipt.promptTokens as number, completionTokens: receipt.completionTokens as number, totalTokens: receipt.totalTokens as number, costUsd: receipt.costUsd as number, accounting: "supervisor-enforced", qualified: true, jobId, campaignId: payload.inferenceLimits.campaignId, @@ -683,12 +725,13 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverr ? config.workerHarnessRuntime ?? workerHarnessRuntimeIdentityFromEnvironment(process.env) : undefined if (workerHarnessRuntime && !validWorkerHarnessRuntimeIdentity(workerHarnessRuntime)) throw new Error("worker harness runtime identity is malformed") - const resumed = config.resumeRunId ? await findActiveRun(config.archiveDir, config.resumeRunId) : undefined + const requestedPhase = config.workerHarnessPhase ?? "development" + const resumed = config.resumeRunId ? requestedPhase === "confirmation" ? await findRun(config.archiveDir, config.resumeRunId) : await findActiveRun(config.archiveDir, config.resumeRunId) : undefined if (config.resumeRunId && !resumed) throw new Error(`no active RSI run found for --resume ${config.resumeRunId}`) const baseCommit = resumed?.baseCommit ?? (await resolveBaseCommit(config)) const run = resumed ?? initialRun({ ...config, model: workerModel, roles }, baseCommit, now()) - const stageCount = config.computePolicy === "adaptive-independent" ? 2 : config.generations - run.generations = stageCount + const stageCount = requestedPhase === "confirmation" ? 0 : config.computePolicy === "adaptive-independent" ? 2 : config.generations + if (requestedPhase !== "confirmation") run.generations = stageCount run.model = workerModel run.parentSelectionPolicy = config.parentSelectionPolicy ?? run.parentSelectionPolicy ?? "champion-specialist-novelty" if (resumed && (run.workerHarnessMode ?? "fixed-control") !== workerHarnessMode) throw new Error("worker harness mode differs from the frozen run manifest") @@ -759,7 +802,8 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverr if (resumed && !resumed.developmentProofProtocol) throw new Error("cannot resume an RSI run without its frozen development proof protocol") if (resumed && resumed.developmentProofProtocol?.protocolDigest !== proofProtocol.protocolDigest) throw new Error("development proof protocol differs from the frozen protocol on this run") run.developmentProofProtocol ??= proofProtocol - run.confirmationEvidence = "unmeasured" + if (requestedPhase !== "confirmation") run.confirmationEvidence = "unmeasured" + run.confirmationAttempts ??= [] const proofAttemptCache = new Map>() const runDevelopmentProof = async (harnessCommit: string, slotId: string, proofRole: "child" | "parent" | "base", sourceCandidateId = slotId): Promise => { const evidenceRunId = `${run.runId}:${slotId}:${proofRole}` @@ -856,10 +900,11 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverr if (priorArm?.state === "finished") { return rebuildFinishedProofCampaignArchive({ run, config, snapshot: priorSnapshot!, manifest, armId: workerHarnessMode, now }) } - if (priorArm?.phase === "confirmation") { + if (priorArm?.phase === "confirmation" && requestedPhase !== "confirmation") { return rebuildEvaluatedProofCampaignArchive({ run, config, snapshot: priorSnapshot!, manifest, armId: workerHarnessMode, now }) } campaignLease = await campaignStore.acquire(manifest, workerHarnessMode, run.runId, `${process.pid}:${randomUUID()}`) + if (campaignLease.phase !== requestedPhase) throw new Error(`proof campaign lease is in ${campaignLease.phase}; requested ${requestedPhase}`) run.proofCampaign = { manifest, manifestSha256: campaignLease.manifestSha256, campaignId: campaignLease.campaignId, armId: campaignLease.armId, epoch: campaignLease.epoch, state: "running" } campaignHeartbeat = setInterval(() => { if (heartbeatInFlight || !campaignStore || !campaignLease) return @@ -872,6 +917,26 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverr await checkpointRun(config.archiveDir, run) } if (resumed && !fleetQueue) recoverInterruptedCandidates(run, now()) + if (requestedPhase === "confirmation") { + if (!campaignStore || !campaignLease || campaignLease.phase !== "confirmation" || !config.workerHarnessPairId || config.dryRun) throw new Error("confirmation execution requires a resumed, frozen, supervised paired proof campaign") + const confirmation = await runProofConfirmation({ run, config, manifest: run.proofCampaign!.manifest, benchmark: loadedProofBenchmark.manifest, fixturesRoot: proofFixturesRoot, store: campaignStore, lease: campaignLease, assertLease: assertCampaignLease, model: workerModel, provider: "openshell-openrouter" }) + run.confirmationAttempts = [...(run.confirmationAttempts ?? []), ...confirmation.attempts] + run.proofConfirmation = { + selectedHarness: { candidateId: confirmation.selectedHarnessCandidate, slotId: confirmation.selectedCandidateId, generation: confirmation.selectedGeneration, commit: confirmation.selectedCommit, attemptIds: confirmation.selectedAttempts.map((attempt) => attempt.attemptId) }, + rootHarness: { commit: run.baseCommit, attemptIds: confirmation.rootAttempts.map((attempt) => attempt.attemptId) }, + } + run.confirmationEvidence = confirmation.attempts.length > 0 && confirmation.attempts.every((attempt) => attempt.validity === "valid" && attempt.proofQualified && attempt.inferenceReceipt?.qualified) ? "complete" : "incomplete" + await assertCampaignLease() + await campaignStore.finishArm(campaignLease, [confirmation.selectedCandidateId]) + run.proofCampaign = { ...run.proofCampaign!, state: "finished", ledgerSnapshot: await campaignStore.snapshot(campaignLease.campaignId) } + run.finishedAt = now() + const confirmationReportPath = `${config.archiveDir}/${run.runId}.md` + await fs.mkdir(config.archiveDir, { recursive: true }) + await fs.writeFile(confirmationReportPath, formatRunReport(run, config), "utf8") + run.reports = [...new Set([...run.reports, confirmationReportPath])] + await appendRun(config.archiveDir, run) + return run + } if (!config.dryRun && !run.baselineEvaluation) { run.baselineEvaluation = fleetQueue ? await evaluateBaselineFleet(runConfig, baseCommit, fleetQueue, run.runId) diff --git a/src/rsi/development-benchmark.ts b/src/rsi/development-benchmark.ts index b8ed2fc..cbba559 100644 --- a/src/rsi/development-benchmark.ts +++ b/src/rsi/development-benchmark.ts @@ -4,8 +4,8 @@ import * as os from "node:os" import * as path from "node:path" import { fileURLToPath } from "node:url" import { OpenShellBenchmarkAgent, OpenShellBenchmarkVerifier } from "./benchmark/openshell-agent.js" -import { loadBenchmark, runBenchmark } from "./benchmark/runner.js" -import type { BenchmarkAttempt, BenchmarkImageIdentity } from "./benchmark/types.js" +import { freezeBenchmarkProtocol, loadBenchmark, runBenchmark } from "./benchmark/runner.js" +import type { BenchmarkAttempt, BenchmarkImageIdentity, BenchmarkSplit } from "./benchmark/types.js" import type { RunBenchmarkOptions } from "./benchmark/runner.js" const moduleRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..") @@ -60,6 +60,7 @@ export async function runDevelopmentBenchmarkForCommit(args: { campaignId?: string inferenceLedgerPath?: string attemptNamespace?: string + split?: BenchmarkSplit attemptLifecycle?: RunBenchmarkOptions["attemptLifecycle"] }): Promise { const repoRoot = path.resolve(args.repoRoot) @@ -70,6 +71,15 @@ export async function runDevelopmentBenchmarkForCommit(args: { return await withCommitWorktree(repoRoot, args.harnessCommit, async (harnessRoot) => { const agent = new OpenShellBenchmarkAgent(harnessRoot, undefined, agentImage.name, args.harnessCommit, undefined, args.provider ?? "openshell-ollama") const verifier = new OpenShellBenchmarkVerifier(repoRoot, undefined, verifierImage.name) + const split = args.split ?? "development" + const outputRoot = path.join(args.outputRoot, "harness", args.harnessCommit) + if (split === "confirmation") { + try { await fs.access(path.join(outputRoot, "frozen-protocol.json")) } + catch (error) { + if ((error as NodeJS.ErrnoException).code !== "ENOENT") throw error + await freezeBenchmarkProtocol({ loaded, harnessCommit: args.harnessCommit, agentImageIdentity: agentImage, modelId: args.model, outputRoot, sampling: { temperature: 0, think: false, seed: "provider-default" }, provider: agent.provider }) + } + } return await runBenchmark({ loaded, agent, @@ -78,8 +88,9 @@ export async function runDevelopmentBenchmarkForCommit(args: { agentImageIdentity: agentImage, modelId: args.model, sampling: { temperature: 0, think: false, seed: "provider-default" }, - split: "development", - outputRoot: path.join(args.outputRoot, "harness", args.harnessCommit), + split, + ...(split === "confirmation" ? { confirmProtocol: true } : {}), + outputRoot, ...(args.campaignId ? { inferenceCampaignId: args.campaignId } : {}), inferenceLedgerPath: args.inferenceLedgerPath ?? path.join(args.outputRoot, "openrouter-inference-ledger.json"), ...(args.attemptNamespace ? { attemptNamespace: args.attemptNamespace } : {}), diff --git a/src/rsi/inference-broker.ts b/src/rsi/inference-broker.ts index 7f6f18c..d11e9fd 100644 --- a/src/rsi/inference-broker.ts +++ b/src/rsi/inference-broker.ts @@ -5,6 +5,10 @@ import { mkdir, open, readFile, rename, rm, stat } from "node:fs/promises" import * as path from "node:path" export const RSI_OPENROUTER_MODEL = "deepseek/deepseek-v4-flash-0731" +export const RSI_OPENROUTER_ENDPOINT_PROVIDER = "deepinfra" +export const RSI_OPENROUTER_ENDPOINT_TAG = "deepinfra/fp8" +export const RSI_OPENROUTER_ENDPOINT_QUANTIZATION = "fp8" +const REQUIRED_ENDPOINT_PARAMETERS = ["max_tokens", "temperature", "tools", "tool_choice"] as const export interface OpenRouterBrokerLimits { campaignId: string @@ -44,6 +48,7 @@ export interface OpenRouterCallSettlement { state: "complete" | "unmeasured" qualified?: boolean returnedModel?: string + returnedProvider?: string providerRequestId?: string promptTokens?: number completionTokens?: number @@ -67,10 +72,12 @@ export interface OpenRouterDenialSummary { statusCode: number; reason: OpenRoute export interface OpenRouterBrokerReceipt { provider: "openrouter" + endpointProvider: typeof RSI_OPENROUTER_ENDPOINT_PROVIDER jobId: string campaignId: string model: string returnedModels: string[] + returnedProviders: string[] calls: number promptTokens: number | null completionTokens: number | null @@ -231,14 +238,14 @@ export class FileOpenRouterInferenceLedger implements OpenRouterInferenceLedger if (!job || !campaign || !call || job.capabilitySha256 !== sha256(grant.capability) || job.leaseSha256 !== sha256(grant.leaseToken) || call.state !== "reserved") throw new Error("OpenRouter file ledger settlement does not match a pending call") Object.assign(call, settlement) const completedAtMs = Date.parse(settlement.completedAt) - const measurable = settlement.state === "complete" && Number.isFinite(completedAtMs) && completedAtMs <= job.deadlineAtMs && settlement.returnedModel !== undefined && settlement.promptTokens !== undefined && settlement.completionTokens !== undefined && settlement.costUsd !== undefined && settlement.responseSha256 !== undefined + const measurable = settlement.state === "complete" && Number.isFinite(completedAtMs) && completedAtMs <= job.deadlineAtMs && settlement.returnedModel !== undefined && settlement.returnedProvider !== undefined && settlement.promptTokens !== undefined && settlement.completionTokens !== undefined && settlement.costUsd !== undefined && settlement.responseSha256 !== undefined if (!measurable) { call.state = "unmeasured" call.authoritative = false job.blocked = true return } - call.authoritative = settlement.qualified !== false && settlement.returnedModel === grant.model + call.authoritative = settlement.qualified !== false && settlement.returnedModel === grant.model && settlement.returnedProvider === RSI_OPENROUTER_ENDPOINT_PROVIDER && settlement.promptTokens !== undefined && settlement.promptTokens <= grant.limits.maxInputTokensPerCall && settlement.completionTokens !== undefined && settlement.completionTokens <= grant.limits.maxOutputTokensPerCall && settlement.costUsd !== undefined && settlement.costUsd <= call.reservedUsd @@ -260,8 +267,9 @@ export class FileOpenRouterInferenceLedger implements OpenRouterInferenceLedger const promptTokens = qualified ? calls.reduce((sum, call) => sum + call.promptTokens!, 0) : null const completionTokens = qualified ? calls.reduce((sum, call) => sum + call.completionTokens!, 0) : null const costUsd = qualified ? calls.reduce((sum, call) => sum + call.costUsd!, 0) : null - return { provider: "openrouter", jobId: grant.jobId, campaignId: grant.limits.campaignId, model: grant.model, + return { provider: "openrouter", endpointProvider: RSI_OPENROUTER_ENDPOINT_PROVIDER, jobId: grant.jobId, campaignId: grant.limits.campaignId, model: grant.model, returnedModels: [...new Set(calls.map((call) => call.returnedModel).filter((model): model is string => Boolean(model)))], + returnedProviders: [...new Set(calls.map((call) => call.returnedProvider).filter((provider): provider is string => Boolean(provider)))], calls: calls.length, promptTokens, completionTokens, totalTokens: promptTokens === null || completionTokens === null ? null : promptTokens + completionTokens, costUsd, accounting: qualified ? "supervisor-enforced" : "incomplete", qualified, callReceipts: structuredClone(calls), denials: denialSummaries(job.denials) } }) @@ -279,8 +287,9 @@ export class FileOpenRouterInferenceLedger implements OpenRouterInferenceLedger const promptTokens = qualified ? calls.reduce((sum, call) => sum + call.promptTokens!, 0) : null const completionTokens = qualified ? calls.reduce((sum, call) => sum + call.completionTokens!, 0) : null const costUsd = qualified ? calls.reduce((sum, call) => sum + call.costUsd!, 0) : null - return { provider: "openrouter", jobId, campaignId, model: job.grant.model, + return { provider: "openrouter", endpointProvider: RSI_OPENROUTER_ENDPOINT_PROVIDER, jobId, campaignId, model: job.grant.model, returnedModels: [...new Set(calls.map((call) => call.returnedModel).filter((model): model is string => Boolean(model)))], + returnedProviders: [...new Set(calls.map((call) => call.returnedProvider).filter((provider): provider is string => Boolean(provider)))], calls: calls.length, promptTokens, completionTokens, totalTokens: promptTokens === null || completionTokens === null ? null : promptTokens + completionTokens, costUsd, accounting: qualified ? "supervisor-enforced" : "incomplete", qualified, callReceipts: structuredClone(calls), denials: denialSummaries(job.denials) } }) @@ -427,7 +436,7 @@ async function readResponse(response: Response, maximumBytes: number): Promise> = [] if (contentType.includes("text/event-stream")) { for (const line of body.toString("utf8").split(/\r?\n/)) { @@ -440,6 +449,7 @@ function parseUsage(body: Buffer, contentType: string): { model?: string; id?: s try { const item = JSON.parse(body.toString("utf8")) as unknown; if (item && typeof item === "object") values = [item as Record] } catch { return {} } } let model: string | undefined + let provider: string | undefined let id: string | undefined let promptTokens: number | undefined let completionTokens: number | undefined @@ -447,6 +457,15 @@ function parseUsage(body: Buffer, contentType: string): { model?: string; id?: s for (const item of values) { if (typeof item.model === "string") model = item.model if (typeof item.id === "string") id = item.id + const metadata = item.openrouter_metadata + if (metadata && typeof metadata === "object" && !Array.isArray(metadata)) { + const endpoints = (metadata as Record).endpoints + const available = endpoints && typeof endpoints === "object" && !Array.isArray(endpoints) ? (endpoints as Record).available : undefined + if (Array.isArray(available)) { + const selected = available.filter((endpoint) => endpoint && typeof endpoint === "object" && (endpoint as Record).selected === true) + if (selected.length === 1 && typeof (selected[0] as Record).provider === "string") provider = String((selected[0] as Record).provider).toLowerCase() + } + } const usage = item.usage if (!usage || typeof usage !== "object") continue const row = usage as Record @@ -454,7 +473,7 @@ function parseUsage(body: Buffer, contentType: string): { model?: string; id?: s if (Number.isSafeInteger(row.completion_tokens) && Number(row.completion_tokens) >= 0) completionTokens = Number(row.completion_tokens) if (typeof row.cost === "number" && Number.isFinite(row.cost) && row.cost >= 0) costUsd = row.cost } - return { ...(model ? { model } : {}), ...(id ? { id } : {}), ...(promptTokens !== undefined ? { promptTokens } : {}), ...(completionTokens !== undefined ? { completionTokens } : {}), ...(costUsd !== undefined ? { costUsd } : {}) } + return { ...(model ? { model } : {}), ...(provider ? { provider } : {}), ...(id ? { id } : {}), ...(promptTokens !== undefined ? { promptTokens } : {}), ...(completionTokens !== undefined ? { completionTokens } : {}), ...(costUsd !== undefined ? { costUsd } : {}) } } function sendJson(response: ServerResponse, status: number, body: unknown): void { @@ -476,7 +495,7 @@ function validateRequestBody(body: Buffer, limits: OpenRouterBrokerLimits, model try { parsed = JSON.parse(body.toString("utf8")) } catch { throw new Error("request JSON is malformed") } if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) throw new Error("request must be a JSON object") const value = parsed as Record - const allowed = new Set(["model", "messages", "tools", "temperature", "max_tokens", "max_completion_tokens", "stream", "stream_options", "provider", "include_reasoning", "reasoning"]) + const allowed = new Set(["model", "messages", "tools", "tool_choice", "temperature", "max_tokens", "max_completion_tokens", "stream", "stream_options", "provider", "include_reasoning", "reasoning"]) for (const key of Object.keys(value)) if (!allowed.has(key)) throw new Error(`request field is not admitted: ${key}`) if (value.model !== model) throw new Error("request model is not admitted for this job") if (!Array.isArray(value.messages) || value.messages.length < 1) throw new Error("request messages are malformed") @@ -494,37 +513,63 @@ function validateRequestBody(body: Buffer, limits: OpenRouterBrokerLimits, model if (value.provider !== undefined) { if (!value.provider || typeof value.provider !== "object" || Array.isArray(value.provider)) throw new Error("request provider routing is malformed") const provider = value.provider as Record - if (provider.order && JSON.stringify(provider.order) !== JSON.stringify(["deepseek"])) throw new Error("request provider routing does not match the pinned DeepSeek provider") + if (provider.only && JSON.stringify(provider.only) !== JSON.stringify([RSI_OPENROUTER_ENDPOINT_PROVIDER])) throw new Error("request provider routing does not match the pinned endpoint provider") if (provider.allow_fallbacks !== undefined && provider.allow_fallbacks !== false) throw new Error("request provider fallbacks are disabled") - for (const key of Object.keys(provider)) if (!["order", "allow_fallbacks"].includes(key)) throw new Error(`request provider field is not admitted: ${key}`) + for (const key of Object.keys(provider)) if (!["only", "allow_fallbacks"].includes(key)) throw new Error(`request provider field is not admitted: ${key}`) } if (value.stream_options !== undefined) { if (!value.stream_options || typeof value.stream_options !== "object" || Array.isArray(value.stream_options)) throw new Error("request stream options are malformed") for (const key of Object.keys(value.stream_options as Record)) if (key !== "include_usage") throw new Error(`request stream option is not admitted: ${key}`) } - value.provider = { order: ["deepseek"], allow_fallbacks: false } + value.provider = { only: [RSI_OPENROUTER_ENDPOINT_PROVIDER], allow_fallbacks: false } delete value.max_tokens - value.max_completion_tokens = outputLimit + value.max_tokens = outputLimit value.stream_options = { ...(value.stream_options && typeof value.stream_options === "object" ? value.stream_options as Record : {}), include_usage: true } const encodedBody = Buffer.from(JSON.stringify(value)) if (encodedBody.byteLength > limits.maxRequestBytes) throw new Error("request body exceeds broker limit") return { encodedBody, outputLimit } } -async function verifyPinnedPriceCeiling(fetcher: typeof fetch, apiKey: string, limits: OpenRouterBrokerLimits): Promise { +export interface OpenRouterEndpointPriceSnapshot { + model: string + provider: typeof RSI_OPENROUTER_ENDPOINT_PROVIDER + endpointTag: string + quantization: typeof RSI_OPENROUTER_ENDPOINT_QUANTIZATION + supportedParameters: string[] + inputPricePerMillionUsd: number + outputPricePerMillionUsd: number + observedAt: string + responseSha256: string +} + +/** Read current pinned endpoint pricing. This is a metadata GET; it does not submit inference. */ +export async function fetchPinnedOpenRouterEndpointPrice(apiKey: string, fetcher: typeof fetch = fetch, now: () => number = Date.now): Promise { const url = `https://openrouter.ai/api/v1/models/${RSI_OPENROUTER_MODEL}/endpoints` - const response = await fetcher(url, { method: "GET", redirect: "error", signal: AbortSignal.timeout(Math.min(limits.maxDurationMs, 15_000)), headers: { authorization: `Bearer ${apiKey}` } }) + const response = await fetcher(url, { method: "GET", redirect: "error", signal: AbortSignal.timeout(15_000), headers: apiKey.trim() ? { authorization: `Bearer ${apiKey}` } : {} }) if (!response.ok) throw new Error(`OpenRouter price preflight failed with HTTP ${response.status}`) const bytes = await readResponse(response, 1024 * 1024) - let data: { data?: { endpoints?: Array<{ provider_name?: unknown; provider_slug?: unknown; pricing?: { prompt?: unknown; completion?: unknown } }> } } + let data: { data?: { endpoints?: Array<{ provider_name?: unknown; provider_slug?: unknown; tag?: unknown; quantization?: unknown; status?: unknown; supported_parameters?: unknown; supports_tool_choice?: unknown; pricing?: { prompt?: unknown; completion?: unknown } }> } } try { data = JSON.parse(bytes.toString("utf8")) as typeof data } catch { throw new Error("OpenRouter price preflight returned malformed JSON") } const rows = data.data?.endpoints ?? [] - const endpoint = rows.find((row) => row.provider_slug === "deepseek" || (typeof row.provider_name === "string" && row.provider_name.toLowerCase() === "deepseek")) + const endpoint = rows.find((row) => row.provider_name === "DeepInfra" && row.tag === RSI_OPENROUTER_ENDPOINT_TAG && row.quantization === RSI_OPENROUTER_ENDPOINT_QUANTIZATION) const prompt = Number(endpoint?.pricing?.prompt) const completion = Number(endpoint?.pricing?.completion) - if (!Number.isFinite(prompt) || prompt <= 0 || !Number.isFinite(completion) || completion <= 0) throw new Error("OpenRouter price preflight did not return the pinned DeepSeek endpoint rates") - const inputPricePerMillion = prompt * 1_000_000 - const outputPricePerMillion = completion * 1_000_000 + const supportedParameters = Array.isArray(endpoint?.supported_parameters) ? endpoint.supported_parameters.filter((parameter): parameter is string => typeof parameter === "string") : [] + const supportsToolChoice = endpoint?.supports_tool_choice && typeof endpoint.supports_tool_choice === "object" ? endpoint.supports_tool_choice as Record : {} + if (!Number.isFinite(prompt) || prompt <= 0 || !Number.isFinite(completion) || completion <= 0 || endpoint?.status !== 0 || !REQUIRED_ENDPOINT_PARAMETERS.every((parameter) => supportedParameters.includes(parameter)) || supportsToolChoice.auto !== true || supportsToolChoice.required !== true) throw new Error("OpenRouter price preflight did not return the available pinned DeepInfra fp8 endpoint with required agent parameters") + return { + model: RSI_OPENROUTER_MODEL, provider: RSI_OPENROUTER_ENDPOINT_PROVIDER, endpointTag: RSI_OPENROUTER_ENDPOINT_TAG, quantization: RSI_OPENROUTER_ENDPOINT_QUANTIZATION, supportedParameters, + inputPricePerMillionUsd: prompt * 1_000_000, + outputPricePerMillionUsd: completion * 1_000_000, + observedAt: new Date(now()).toISOString(), + responseSha256: sha256(bytes), + } +} + +async function verifyPinnedPriceCeiling(fetcher: typeof fetch, apiKey: string, limits: OpenRouterBrokerLimits): Promise { + const price = await fetchPinnedOpenRouterEndpointPrice(apiKey, fetcher) + const inputPricePerMillion = price.inputPricePerMillionUsd + const outputPricePerMillion = price.outputPricePerMillionUsd if (limits.maxInputPricePerMillionUsd < inputPricePerMillion || limits.maxOutputPricePerMillionUsd < outputPricePerMillion) throw new Error("configured OpenRouter price ceiling is below the current pinned endpoint rate") const worstCase = (limits.maxInputTokensPerCall * limits.maxInputPricePerMillionUsd + limits.maxOutputTokensPerCall * limits.maxOutputPricePerMillionUsd) / 1_000_000 if (limits.maxCallSpendUsd + 0.000001 < worstCase) throw new Error("per-call spend reservation is below the configured worst-case token cost") @@ -604,7 +649,7 @@ export async function startOpenRouterBroker(options: OpenRouterBrokerOptions): P try { upstream = await upstreamFetch("https://openrouter.ai/api/v1/chat/completions", { method: "POST", redirect: "error", signal: controller.signal, - headers: { authorization: `Bearer ${options.apiKey}`, "content-type": "application/json" }, body: checked.encodedBody, + headers: { authorization: `Bearer ${options.apiKey}`, "content-type": "application/json", "X-OpenRouter-Metadata": "enabled" }, body: checked.encodedBody, }) upstreamBody = await readResponse(upstream, limits.maxResponseBytes) } catch { @@ -618,23 +663,23 @@ export async function startOpenRouterBroker(options: OpenRouterBrokerOptions): P } const contentType = upstream.headers.get("content-type") ?? "application/json" const usage = upstream.ok ? parseUsage(upstreamBody, contentType) : {} - const measurable = usage.promptTokens !== undefined && usage.completionTokens !== undefined && usage.costUsd !== undefined + const measurable = usage.promptTokens !== undefined && usage.completionTokens !== undefined && usage.costUsd !== undefined && usage.provider !== undefined if (!upstream.ok || !measurable) { - await options.ledger.settle(grant, { callId, state: "unmeasured", ...(usage.model ? { returnedModel: usage.model } : {}), ...(usage.id ? { providerRequestId: usage.id } : {}), ...(usage.promptTokens !== undefined ? { promptTokens: usage.promptTokens } : {}), ...(usage.completionTokens !== undefined ? { completionTokens: usage.completionTokens } : {}), ...(usage.costUsd !== undefined ? { costUsd: usage.costUsd } : {}), responseSha256: sha256(upstreamBody), completedAt: new Date(now()).toISOString() }).catch(() => undefined) - await deny(502, "usage-unverifiable", "upstream response lacks verifiable usage or cost; request is unqualified"); return + await options.ledger.settle(grant, { callId, state: "unmeasured", ...(usage.model ? { returnedModel: usage.model } : {}), ...(usage.provider ? { returnedProvider: usage.provider } : {}), ...(usage.id ? { providerRequestId: usage.id } : {}), ...(usage.promptTokens !== undefined ? { promptTokens: usage.promptTokens } : {}), ...(usage.completionTokens !== undefined ? { completionTokens: usage.completionTokens } : {}), ...(usage.costUsd !== undefined ? { costUsd: usage.costUsd } : {}), responseSha256: sha256(upstreamBody), completedAt: new Date(now()).toISOString() }).catch(() => undefined) + await deny(502, "usage-unverifiable", "upstream response lacks verifiable usage or selected endpoint metadata; request is unqualified"); return } const promptTokens = usage.promptTokens! const completionTokens = usage.completionTokens! const costUsd = usage.costUsd! const usageWithinBounds = promptTokens <= limits.maxInputTokensPerCall && completionTokens <= checked.outputLimit && costUsd <= limits.maxCallSpendUsd - if (usage.model !== model || !usageWithinBounds) { - await options.ledger.settle(grant, { callId, state: "complete", qualified: false, returnedModel: usage.model ?? "", ...(usage.id ? { providerRequestId: usage.id } : {}), promptTokens: usage.promptTokens, completionTokens: usage.completionTokens, costUsd: usage.costUsd, responseSha256: sha256(upstreamBody), completedAt: new Date(now()).toISOString() }).catch(() => undefined) - const message = usage.model !== model ? "upstream returned a different model" : "upstream usage exceeded admitted broker limits" + if (usage.model !== model || usage.provider !== RSI_OPENROUTER_ENDPOINT_PROVIDER || !usageWithinBounds) { + await options.ledger.settle(grant, { callId, state: "complete", qualified: false, returnedModel: usage.model ?? "", returnedProvider: usage.provider, ...(usage.id ? { providerRequestId: usage.id } : {}), promptTokens: usage.promptTokens, completionTokens: usage.completionTokens, costUsd: usage.costUsd, responseSha256: sha256(upstreamBody), completedAt: new Date(now()).toISOString() }).catch(() => undefined) + const message = usage.model !== model ? "upstream returned a different model" : usage.provider !== RSI_OPENROUTER_ENDPOINT_PROVIDER ? "upstream selected an endpoint outside the pinned provider" : "upstream usage exceeded admitted broker limits" await deny(502, "usage-out-of-bounds", `${message}; request is unqualified`); return } try { await options.ledger.settle(grant, { - callId, state: "complete", qualified: true, returnedModel: usage.model, providerRequestId: usage.id, + callId, state: "complete", qualified: true, returnedModel: usage.model, returnedProvider: usage.provider, providerRequestId: usage.id, promptTokens: usage.promptTokens, completionTokens: usage.completionTokens, costUsd: usage.costUsd, responseSha256: sha256(upstreamBody), completedAt: new Date(now()).toISOString(), }) diff --git a/src/rsi/migrations/008_openrouter_returned_provider.sql b/src/rsi/migrations/008_openrouter_returned_provider.sql new file mode 100644 index 0000000..d645dc1 --- /dev/null +++ b/src/rsi/migrations/008_openrouter_returned_provider.sql @@ -0,0 +1,2 @@ +ALTER TABLE headlesscode_rsi_inference_calls + ADD COLUMN IF NOT EXISTS returned_provider text; diff --git a/src/rsi/openshell.ts b/src/rsi/openshell.ts index 38e8f2d..6c58c0a 100644 --- a/src/rsi/openshell.ts +++ b/src/rsi/openshell.ts @@ -654,6 +654,9 @@ export async function runOpenShellMutation(candidate: CandidateRecord, config: R "test ! -e /workspace/.headlesscode", "test ! -e /opt/headlesscode/src/rsi", "test ! -e /opt/headlesscode/scripts/eval-suite", + "test ! -e /opt/headlesscode/scripts/rsi-benchmark/task-spec.mjs", + "test ! -e /opt/headlesscode/fixtures/rsi-benchmark", + "test ! -e /opt/headlesscode/.git", ...(config.benchmarkHarnessCommit ? ["test -f /opt/headlesscode/.headlesscode-harness-commit", `test "$(cat /opt/headlesscode/.headlesscode-harness-commit)" = ${shellQuote(config.benchmarkHarnessCommit)}`] : []), config.benchmarkHarnessCommit ? "test -z \"$(find /opt/headlesscode/src -type f \\( -path '*/__tests__/*' -o -name '*.test.*' -o -name '*.spec.*' \\) -print -quit)\"" diff --git a/src/rsi/postgres-queue.ts b/src/rsi/postgres-queue.ts index 088bd29..5daa1ef 100644 --- a/src/rsi/postgres-queue.ts +++ b/src/rsi/postgres-queue.ts @@ -3,7 +3,7 @@ import { readFileSync } from "node:fs" import { Pool, type PoolClient } from "pg" import type { ResourceClass, WorkerHarnessMode, WorkerHarnessRuntimeIdentity } from "./types.js" import { createRsiArtifactStore, type RsiArtifactReference, type RsiArtifactStore } from "./artifact-store.js" -import type { OpenRouterBrokerGrant, OpenRouterBrokerReceipt, OpenRouterCallReceipt, OpenRouterCallReservation, OpenRouterCallSettlement, OpenRouterDenialReason, OpenRouterInferenceLedger } from "./inference-broker.js" +import { RSI_OPENROUTER_ENDPOINT_PROVIDER, type OpenRouterBrokerGrant, type OpenRouterBrokerReceipt, type OpenRouterCallReceipt, type OpenRouterCallReservation, type OpenRouterCallSettlement, type OpenRouterDenialReason, type OpenRouterInferenceLedger } from "./inference-broker.js" import type { OpenRouterBrokerLimits } from "./inference-broker.js" import { PostgresProofCampaign, type ProofCampaignArmLease, type ProofCampaignManifest } from "./proof-campaign.js" @@ -15,6 +15,7 @@ const MIGRATIONS = [ { version: 5, sql: readFileSync(new URL("./migrations/005_openrouter_denial_summaries.sql", import.meta.url), "utf8") }, { version: 6, sql: readFileSync(new URL("./migrations/006_proof_campaigns.sql", import.meta.url), "utf8") }, { version: 7, sql: readFileSync(new URL("./migrations/007_proof_campaign_phases.sql", import.meta.url), "utf8") }, + { version: 8, sql: readFileSync(new URL("./migrations/008_openrouter_returned_provider.sql", import.meta.url), "utf8") }, ] function sha256Text(value: string): string { return createHash("sha256").update(value).digest("hex") } @@ -551,7 +552,7 @@ export class PostgresRsiJobQueue { const callResult = await client.query("SELECT * FROM headlesscode_rsi_inference_calls WHERE job_id=$1 AND call_id=$2 FOR UPDATE", [grant.jobId, settlement.callId]) const call = callResult.rows[0] if (!job || !call || call.capability_sha256 !== capabilityHash || call.state !== "reserved") throw new Error("OpenRouter settlement does not match a pending broker call") - if (settlement.state !== "complete" || settlement.promptTokens === undefined || settlement.completionTokens === undefined || settlement.costUsd === undefined || settlement.returnedModel === undefined || !settlement.responseSha256) { + if (settlement.state !== "complete" || settlement.promptTokens === undefined || settlement.completionTokens === undefined || settlement.costUsd === undefined || settlement.returnedModel === undefined || settlement.returnedProvider === undefined || !settlement.responseSha256) { await client.query("UPDATE headlesscode_rsi_inference_calls SET state='unmeasured',completed_at=$3 WHERE job_id=$1 AND call_id=$2", [grant.jobId, settlement.callId, settlement.completedAt]) await client.query("UPDATE headlesscode_rsi_inference_jobs SET blocked=true WHERE job_id=$1", [grant.jobId]) return @@ -561,11 +562,11 @@ export class PostgresRsiJobQueue { const currentLease = job.queue_status === "leased" && job.current_lease_token === grant.leaseToken && job.lease_valid === true const overLimit = job.duration_valid !== true || settlement.qualified === false || costMicros > Number(call.reserved_spend_microusd) || costMicros > usdCapToMicros(grant.limits.maxCallSpendUsd) || settlement.promptTokens > grant.limits.maxInputTokensPerCall || settlement.completionTokens > grant.limits.maxOutputTokensPerCall - || actualTokens > Number(call.reserved_tokens) || settlement.returnedModel !== grant.model || !currentLease + || actualTokens > Number(call.reserved_tokens) || settlement.returnedModel !== grant.model || settlement.returnedProvider !== RSI_OPENROUTER_ENDPOINT_PROVIDER || !currentLease await client.query( `UPDATE headlesscode_rsi_inference_calls SET state='complete',actual_spend_microusd=$3,prompt_tokens=$4,completion_tokens=$5, - returned_model=$6,provider_request_id=$7,response_sha256=$8,completed_at=$9,qualified=$10 WHERE job_id=$1 AND call_id=$2`, - [grant.jobId, settlement.callId, costMicros, settlement.promptTokens, settlement.completionTokens, settlement.returnedModel, settlement.providerRequestId ?? null, settlement.responseSha256, settlement.completedAt, !overLimit], + returned_model=$6,returned_provider=$7,provider_request_id=$8,response_sha256=$9,completed_at=$10,qualified=$11 WHERE job_id=$1 AND call_id=$2`, + [grant.jobId, settlement.callId, costMicros, settlement.promptTokens, settlement.completionTokens, settlement.returnedModel, settlement.returnedProvider, settlement.providerRequestId ?? null, settlement.responseSha256, settlement.completedAt, !overLimit], ) await client.query( `UPDATE headlesscode_rsi_inference_jobs SET reserved_spend_microusd=reserved_spend_microusd-$2,spent_microusd=spent_microusd+$3, @@ -603,13 +604,14 @@ export class PostgresRsiJobQueue { reservedTokens: Number(row.reserved_tokens), startedAt: new Date(String(row.started_at)).toISOString(), state: row.state as OpenRouterCallSettlement["state"], ...(row.returned_model ? { returnedModel: String(row.returned_model) } : {}), + ...(row.returned_provider ? { returnedProvider: String(row.returned_provider) } : {}), ...(row.provider_request_id ? { providerRequestId: String(row.provider_request_id) } : {}), ...(row.prompt_tokens !== null ? { promptTokens: Number(row.prompt_tokens) } : {}), ...(row.completion_tokens !== null ? { completionTokens: Number(row.completion_tokens) } : {}), ...(row.actual_spend_microusd !== null ? { costUsd: microsToUsd(row.actual_spend_microusd) } : {}), ...(row.response_sha256 ? { responseSha256: String(row.response_sha256) } : {}), ...(row.completed_at ? { completedAt: new Date(String(row.completed_at)).toISOString() } : {}), - authoritative: row.state === "complete" && row.qualified === true && row.returned_model === grant.model && row.prompt_tokens !== null && row.completion_tokens !== null && row.actual_spend_microusd !== null, + authoritative: row.state === "complete" && row.qualified === true && row.returned_model === grant.model && row.returned_provider === RSI_OPENROUTER_ENDPOINT_PROVIDER && row.prompt_tokens !== null && row.completion_tokens !== null && row.actual_spend_microusd !== null, })) const allMeasured = calls.length > 0 && calls.every((call) => call.authoritative) const promptTokens = allMeasured ? calls.reduce((sum, call) => sum + (call.promptTokens ?? 0), 0) : null @@ -622,8 +624,9 @@ export class PostgresRsiJobQueue { && Number(job.spent_microusd) <= Number(job.max_spend_microusd) && Number(job.reserved_spend_microusd) === 0 && Number(job.reserved_tokens) === 0 && !job.blocked && campaignResult.rows[0]?.blocked !== true && currentLease && grantMatches return { - provider: "openrouter", jobId: grant.jobId, campaignId: String(job.campaign_id), model: grant.model, + provider: "openrouter", endpointProvider: RSI_OPENROUTER_ENDPOINT_PROVIDER, jobId: grant.jobId, campaignId: String(job.campaign_id), model: grant.model, returnedModels: [...new Set(calls.map((call) => call.returnedModel).filter((model): model is string => Boolean(model)))], + returnedProviders: [...new Set(calls.map((call) => call.returnedProvider).filter((provider): provider is string => Boolean(provider)))], calls: calls.length, promptTokens, completionTokens, totalTokens: promptTokens === null || completionTokens === null ? null : promptTokens + completionTokens, costUsd, accounting: allMeasured && withinLimits ? "supervisor-enforced" : "incomplete", qualified: allMeasured && withinLimits, callReceipts: calls, denials, diff --git a/src/rsi/proof-campaign.ts b/src/rsi/proof-campaign.ts index e521088..9f3eb75 100644 --- a/src/rsi/proof-campaign.ts +++ b/src/rsi/proof-campaign.ts @@ -290,10 +290,8 @@ export class PostgresProofCampaign { const manifest = campaign.manifest as ProofCampaignManifest const cell = manifest.schedule.find((entry) => entry.armId === lease.armId && (entry.phase ?? "development") === "confirmation" && entry.generation === input.generation && entry.candidateId === input.candidateId && entry.taskId === input.taskId && entry.seed === input.seed && entry.attemptKind === "confirmation") if (!cell) throw new Error("not-selected disposition is outside the frozen confirmation schedule") - const finalGeneration = Math.max(...manifest.schedule.filter((entry) => entry.armId === lease.armId && (entry.phase ?? "development") === "confirmation").map((entry) => entry.generation)) - if (input.generation !== finalGeneration) throw new Error("not-selected disposition is only valid for final confirmation slots") - const selection = await client.query("SELECT payload FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='generation-selected' AND payload->>'generation'=$3 FOR UPDATE", [lease.campaignId, lease.armId, String(finalGeneration)]) - if (selection.rowCount !== 1 || (selection.rows[0].payload as { selectedCandidateIds?: string[] }).selectedCandidateIds?.includes(input.candidateId)) throw new Error("selected or undecided confirmation slot cannot be marked not-selected") + const finalSelection = await client.query("SELECT payload FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='final-harness-selected' FOR UPDATE", [lease.campaignId, lease.armId]) + if (finalSelection.rowCount !== 1 || (finalSelection.rows[0].payload as { candidateId?: string }).candidateId === input.candidateId) throw new Error("selected or undecided confirmation slot cannot be marked not-selected") const identity = { generation: input.generation, candidateId: input.candidateId, taskId: input.taskId, seed: input.seed, attemptKind: "confirmation" as const } const identitySha256 = digest({ ...identity, sourceCandidateId: input.candidateId }) const requestSha256 = digest({ identity, disposition: "not-selected", reason: input.reason }) @@ -309,6 +307,42 @@ export class PostgresProofCampaign { }) } + /** Freeze the one final harness whose result will be compared with root on confirmation tasks. */ + async selectFinalHarness(lease: ProofCampaignArmLease, selection: { generation: number; candidateId: string; commit: string }): Promise { + if (!Number.isSafeInteger(selection.generation) || selection.generation < 0 || !selection.candidateId.trim() || !/^[a-f0-9]{40}$/.test(selection.commit)) throw new Error("final confirmation harness identity is malformed") + await transaction(this.pool, async (client) => { + const campaign = await this.lockLease(client, lease) + if (lease.phase !== "confirmation" || campaign.arm_phase !== "confirmation") throw new Error("final harness selection requires the confirmation phase") + const manifest = campaign.manifest as ProofCampaignManifest + const confirmationCells = manifest.schedule.filter((cell) => cell.armId === lease.armId && (cell.phase ?? "development") === "confirmation") + const finalGeneration = Math.max(...confirmationCells.map((cell) => cell.generation)) + if (selection.generation > finalGeneration) throw new Error("final harness selection is outside the frozen generation schedule") + const generationSelections = await client.query("SELECT payload FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='generation-selected' ORDER BY (payload->>'generation')::int", [lease.campaignId, lease.armId]) + const selectedGenerations = generationSelections.rows.filter((entry) => Array.isArray((entry.payload as { selectedCandidateIds?: unknown }).selectedCandidateIds) && ((entry.payload as { selectedCandidateIds: string[] }).selectedCandidateIds.length > 0)) + const latestSelected = selectedGenerations.at(-1) + if (selection.candidateId === "root-control") { + if (latestSelected || selection.generation !== finalGeneration || selection.commit !== manifest.rootCommit) throw new Error("root may be selected as final harness only when no campaign generation selected an accepted candidate") + } else { + const latestGeneration = Number((latestSelected?.payload as { generation?: unknown } | undefined)?.generation) + const latestSelectedIds = (latestSelected?.payload as { selectedCandidateIds?: unknown } | undefined)?.selectedCandidateIds + if (!latestSelected || latestGeneration !== selection.generation || !Array.isArray(latestSelectedIds) || !latestSelectedIds.includes(selection.candidateId)) throw new Error("final confirmation harness must come from the latest nonempty durable generation selection") + const eligible = await client.query(`SELECT 1 FROM headlesscode_rsi_proof_campaign_decisions AS decision + JOIN headlesscode_rsi_proof_campaign_events AS selected ON selected.campaign_id=decision.campaign_id AND selected.arm_id=decision.arm_id AND selected.event_type='generation-selected' + AND selected.payload->>'generation'=$4 AND selected.payload->'selectedCandidateIds' ? decision.candidate_id + WHERE decision.campaign_id=$1 AND decision.arm_id=$2 AND decision.generation=$3 AND decision.candidate_id=$5 AND decision.child_commit=$6 AND decision.decision='accepted'`, + [lease.campaignId, lease.armId, selection.generation, String(selection.generation), selection.candidateId, selection.commit]) + if (eligible.rowCount !== 1) throw new Error("final confirmation harness must be an accepted candidate selected by its generation") + } + const payload = { generation: selection.generation, candidateId: selection.candidateId, commit: selection.commit } + const existing = await client.query("SELECT payload FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='final-harness-selected' FOR UPDATE", [lease.campaignId, lease.armId]) + if (existing.rowCount) { + if (existing.rowCount !== 1 || stableJson(existing.rows[0].payload) !== stableJson(payload)) throw new Error("final confirmation harness conflicts with an earlier durable selection") + return + } + await this.appendEvent(client, lease.campaignId, lease.armId, lease.epoch, "final-harness-selected", payload) + }) + } + async settle(lease: ProofCampaignArmLease, attemptKey: string, settlement: ProofAttemptSettlement): Promise { await transaction(this.pool, async (client) => { await this.lockLease(client, lease) @@ -472,15 +506,14 @@ export class PostgresProofCampaign { } const confirmation = scheduled.filter((cell) => (cell.phase ?? "development") === "confirmation") if (confirmation.length) { - const finalGeneration = Math.max(...confirmation.map((cell) => cell.generation)) - const selectedEvent = await client.query("SELECT payload FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='generation-selected' AND payload->>'generation'=$3", [lease.campaignId, lease.armId, String(finalGeneration)]) - if (selectedEvent.rowCount !== 1) throw new Error("cannot finish proof campaign without final-generation selection") - const selected = (selectedEvent.rows[0].payload as { selectedCandidateIds: string[] }).selectedCandidateIds + const selectedEvent = await client.query("SELECT payload FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='final-harness-selected'", [lease.campaignId, lease.armId]) + if (selectedEvent.rowCount !== 1) throw new Error("cannot finish proof campaign without final-harness selection") + const selected = selectedEvent.rows[0].payload as { candidateId: string; generation: number; commit: string } const recordedSelection = [...selectedCandidateIds].sort((a, b) => Buffer.compare(Buffer.from(a), Buffer.from(b))) - if (stableJson(recordedSelection) !== stableJson(selected)) throw new Error("final arm selection does not match its frozen campaign selection") + if (stableJson(recordedSelection) !== stableJson([selected.candidateId])) throw new Error("final arm selection does not match its frozen campaign selection") for (const cell of confirmation) { const attempt = await client.query("SELECT status FROM headlesscode_rsi_proof_campaign_attempts WHERE campaign_id=$1 AND arm_id=$2 AND generation=$3 AND candidate_id=$4 AND task_id=$5 AND seed=$6 AND attempt_kind=$7", [lease.campaignId, lease.armId, cell.generation, cell.candidateId, cell.taskId, cell.seed, cell.attemptKind]) - const required = cell.candidateId === "root-control" || selected.includes(cell.candidateId) + const required = cell.candidateId === "root-control" || (cell.candidateId === selected.candidateId && cell.generation === selected.generation) const expectedStatus = required ? "completed" : "not-selected" if (attempt.rowCount !== 1 || attempt.rows[0].status !== expectedStatus) throw new Error(`final confirmation cell ${cell.candidateId}/${cell.taskId} must be ${expectedStatus}`) } diff --git a/src/rsi/proof-confirmation.ts b/src/rsi/proof-confirmation.ts new file mode 100644 index 0000000..601f076 --- /dev/null +++ b/src/rsi/proof-confirmation.ts @@ -0,0 +1,89 @@ +import { createHash } from "node:crypto" +import * as path from "node:path" +import type { RsiConfig, RsiRunRecord } from "./types.js" +import type { BenchmarkAttempt, BenchmarkManifest } from "./benchmark/types.js" +import { openRouterAllocatedEnvironment, openRouterBrokerLimitsFromEnvironment } from "./inference-broker.js" +import { runDevelopmentBenchmarkForCommit } from "./development-benchmark.js" +import type { PostgresProofCampaign, ProofCampaignArmLease, ProofCampaignManifest } from "./proof-campaign.js" + +function candidateSlot(generation: number, index: number): string { return "g" + generation + "-candidate-" + index } + +/** Execute only the preselected final harness and root-control confirmation cells under the PG campaign lease. */ +export async function runProofConfirmation(args: { + run: RsiRunRecord + config: RsiConfig + manifest: ProofCampaignManifest + benchmark: BenchmarkManifest + fixturesRoot: string + store: PostgresProofCampaign + lease: ProofCampaignArmLease + assertLease: () => Promise + model: string + provider: "openshell-openrouter" +}): Promise<{ attempts: BenchmarkAttempt[]; selectedCandidateId: string; selectedGeneration: number; selectedCommit: string; selectedHarnessCandidate: string; selectedAttempts: BenchmarkAttempt[]; rootAttempts: BenchmarkAttempt[] }> { + if (args.lease.phase !== "confirmation") throw new Error("confirmation requires a confirmation-phase campaign lease") + const tasks = args.benchmark.tasks.filter((task) => task.split === "confirmation") + if (!tasks.length) throw new Error("frozen benchmark manifest has no confirmation tasks") + const candidate = args.run.selected ? args.run.candidates.find((entry) => entry.id === args.run.selected && entry.status === "accepted" && entry.developmentProof?.accepted === true) : undefined + const candidatesAtGeneration = candidate ? args.run.candidates.filter((entry) => entry.generation === candidate.generation) : [] + const candidateIndex = candidate ? candidatesAtGeneration.indexOf(candidate) : -1 + const selectedCandidateId = candidate && candidateIndex >= 0 ? candidateSlot(candidate.generation, candidateIndex) : "root-control" + const selectedGeneration = candidate?.generation ?? Math.max(...args.manifest.schedule.filter((cell) => cell.armId === args.lease.armId && cell.phase === "confirmation").map((cell) => cell.generation)) + const selectedCommit = candidate?.commits[0] ?? args.run.baseCommit + await args.store.selectFinalHarness(args.lease, { generation: selectedGeneration, candidateId: selectedCandidateId, commit: selectedCommit }) + for (const cell of args.manifest.schedule.filter((entry) => entry.armId === args.lease.armId && entry.phase === "confirmation" && entry.attemptKind === "confirmation")) { + if (cell.candidateId === selectedCandidateId && cell.generation === selectedGeneration) continue + await args.store.markNotSelected(args.lease, { + attemptKey: "not-selected:" + args.lease.armId + ":" + cell.candidateId + ":" + cell.taskId, + generation: cell.generation, candidateId: cell.candidateId, taskId: cell.taskId, seed: cell.seed, + reason: "candidate is not the final selected harness for this campaign arm", + }) + } + const attempts: BenchmarkAttempt[] = [] + const campaignId = args.lease.campaignId + ":benchmark" + const ledgerPath = path.join(args.config.archiveDir, "development-proof", args.lease.campaignId, "benchmark-openrouter-inference-ledger.json") + const runHarness = async (candidateId: string, generation: number, commit: string, kind: "confirmation" | "fixed-control", sourceCandidateId: string): Promise => { + await args.assertLease() + const outputRoot = path.join(args.config.archiveDir, "confirmation", args.lease.campaignId, args.lease.armId, candidateId, kind) + const result = await runDevelopmentBenchmarkForCommit({ + repoRoot: args.config.repoRoot, fixturesRoot: args.fixturesRoot, outputRoot, harnessCommit: commit, model: args.model, + provider: args.provider, campaignId, inferenceLedgerPath: ledgerPath, + attemptNamespace: args.lease.armId + ":confirmation:" + candidateId, split: "confirmation", + attemptLifecycle: { + beforeAttempt: async ({ task, attemptId, campaignId: requestCampaign, agentImage, model, sampling }) => { + await args.assertLease() + const remaining = Math.max(1, Date.parse(args.lease.deadlineAt) - Date.now()) + const limits = openRouterBrokerLimitsFromEnvironment(openRouterAllocatedEnvironment(process.env, "benchmark"), campaignId, { maxCalls: task.ceilings.maxCalls, maxOutputTokensPerCall: task.ceilings.maxOutputTokensPerCall, maxDurationMs: Math.max(1, Math.min(task.ceilings.timeoutMs, remaining)) }) + const request = { phase: "confirmation", taskId: task.id, attemptId, campaignId: requestCampaign, candidateId, kind, sourceCandidateId, harnessCommit: commit, agentImage, model, sampling, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings } + const requestSha256 = createHash("sha256").update(JSON.stringify(request)).digest("hex") + const attemptKey = "confirmation:" + args.lease.armId + ":" + candidateId + ":" + attemptId + const admitted = await args.store.admit(args.lease, { attemptKey, brokerJobId: attemptId, kind, identity: { generation, candidateId, taskId: task.id, seed: String(task.seed), attemptKind: kind }, sourceCandidateId, requestSha256, calls: limits.maxCalls, tokens: limits.maxTotalTokens, spendUsd: limits.maxJobSpendUsd }) + if (admitted.status !== "admitted") throw new Error("confirmation cell is already dispatched or settled without a verified artifact; refusing duplicate provider call") + await args.store.markDispatched(args.lease, attemptKey) + return requestSha256 + }, + settledAttempt: async ({ attempt }) => { + if (!attempt.proofRequestSha256) throw new Error("confirmation attempt is missing its campaign request digest") + const limits = openRouterBrokerLimitsFromEnvironment(openRouterAllocatedEnvironment(process.env, "benchmark"), campaignId, { maxCalls: attempt.ceilings.maxCalls, maxOutputTokensPerCall: attempt.ceilings.maxOutputTokensPerCall, maxDurationMs: Math.max(1, Math.min(attempt.ceilings.timeoutMs, Date.parse(args.lease.deadlineAt) - Date.now())) }) + const attemptKey = "confirmation:" + args.lease.armId + ":" + candidateId + ":" + attempt.attemptId + const admitted = await args.store.admit(args.lease, { attemptKey, brokerJobId: attempt.attemptId, kind, identity: { generation, candidateId, taskId: attempt.taskId, seed: String(attempt.seed), attemptKind: kind }, sourceCandidateId, requestSha256: attempt.proofRequestSha256, calls: limits.maxCalls, tokens: limits.maxTotalTokens, spendUsd: limits.maxJobSpendUsd }) + if (admitted.status === "admitted" || admitted.status === "dispatched") await args.store.markDispatched(args.lease, attemptKey) + else if (admitted.status !== "completed") throw new Error("confirmation attempt cannot be reconciled under the current campaign epoch") + const receipt = attempt.inferenceReceipt + if (!receipt?.qualified) await args.store.settle(args.lease, attemptKey, { status: "uncertain", error: "confirmation broker receipt is missing or unqualified" }) + else await args.store.settle(args.lease, attemptKey, { + status: "completed", resultSha256: createHash("sha256").update(JSON.stringify(attempt)).digest("hex"), + artifactRefs: Object.fromEntries(Object.entries(attempt.artifactRefs).map(([key, value]) => [key, { sha256: value }])), + actualCalls: receipt.calls, actualTokens: receipt.totalTokens ?? 0, actualSpendUsd: receipt.costUsd ?? 0, + }) + }, + }, + }) + attempts.push(...result) + return result + } + const selectedAttempts = candidate ? await runHarness(selectedCandidateId, selectedGeneration, selectedCommit, "confirmation", candidate.id) : [] + const rootGeneration = Math.max(...args.manifest.schedule.filter((cell) => cell.armId === args.lease.armId && cell.phase === "confirmation").map((cell) => cell.generation)) + const rootAttempts = await runHarness("root-control", rootGeneration, args.run.baseCommit, "fixed-control", "root-control") + return { attempts, selectedCandidateId, selectedGeneration, selectedCommit, selectedHarnessCandidate: candidate?.id ?? "baseline", selectedAttempts: candidate ? selectedAttempts : rootAttempts, rootAttempts } +} diff --git a/src/rsi/proof-demonstration-manifest.ts b/src/rsi/proof-demonstration-manifest.ts new file mode 100644 index 0000000..a1d06cf --- /dev/null +++ b/src/rsi/proof-demonstration-manifest.ts @@ -0,0 +1,73 @@ +import { createHash } from "node:crypto" +import { RSI_OPENROUTER_ENDPOINT_PROVIDER, RSI_OPENROUTER_ENDPOINT_QUANTIZATION, RSI_OPENROUTER_ENDPOINT_TAG, type OpenRouterEndpointPriceSnapshot } from "./inference-broker.js" +import type { OpenRouterBrokerLimits } from "./inference-broker.js" +import type { ProofDemonstrationPlan } from "./proof-demonstration-plan.js" +import type { RsiConfig } from "./types.js" +import type { WorkerHarnessRuntimeIdentity } from "./types.js" + +export interface FrozenProofDemonstrationManifest { + schemaVersion: 1 + createdAt: string + rootCommit: string + benchmarkManifestSha256: string + model: { provider: "openshell-openrouter"; requestedId: string; sampling: { temperature: 0; think: false; seed: "provider-default" } } + endpointPrice: OpenRouterEndpointPriceSnapshot + priceCeilingsPerMillionUsd: { input: number; output: number } + runtime: WorkerHarnessRuntimeIdentity + execution: { + configs: [RsiConfig, RsiConfig, RsiConfig, RsiConfig] + configSha256: string + analysisSourceSha256: string + inferenceLimits: { mutation: OpenRouterBrokerLimits; benchmark: OpenRouterBrokerLimits } + taskIdentities: Array<{ id: string; split: string; family: string; seed: number; startDigest: string; verifierDigest: string; knownGoodDigest: string; knownBadDigest: string; ceilings: { timeoutMs: number; maxCalls: number; maxOutputTokensPerCall: number; maxPatchBytes: number } }> + } + seeds: { engineering: string; campaigns: [string, string, string] } + campaignIds: { engineering: string; campaigns: [string, string, string] } + operatorCapUsd: number + plan: ProofDemonstrationPlan + digestSha256: string +} + +function canonical(value: unknown): unknown { + if (Array.isArray(value)) return value.map(canonical) + if (value && typeof value === "object") return Object.fromEntries(Object.entries(value as Record) + .filter(([, item]) => item !== undefined) + .sort(([a], [b]) => Buffer.compare(Buffer.from(a), Buffer.from(b))) + .map(([key, item]) => [key, canonical(item)])) + return value +} + +export function frozenProofDemonstrationDigest(manifest: Omit): string { + return createHash("sha256").update(JSON.stringify(canonical(manifest))).digest("hex") +} + + +export function frozenProofConfigDigest(config: unknown): string { + return createHash("sha256").update(JSON.stringify(canonical(config))).digest("hex") +} + +export function freezeProofDemonstrationManifest(args: Omit): FrozenProofDemonstrationManifest { + if (!/^[a-f0-9]{40}$/.test(args.rootCommit)) throw new Error("proof demonstration root commit must be a full commit SHA") + if (!/^[a-f0-9]{64}$/.test(args.benchmarkManifestSha256)) throw new Error("proof demonstration benchmark manifest digest is malformed") + if (!Number.isFinite(args.operatorCapUsd) || args.operatorCapUsd <= 0) throw new Error("proof demonstration requires a positive finite operator cap") + if (new Set([args.seeds.engineering, ...args.seeds.campaigns]).size !== 4 || [args.seeds.engineering, ...args.seeds.campaigns].some((seed) => !/^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/.test(seed))) throw new Error("proof demonstration requires four distinct explicit seeds") + if (args.endpointPrice.model !== args.model.requestedId || args.endpointPrice.provider !== RSI_OPENROUTER_ENDPOINT_PROVIDER || args.endpointPrice.endpointTag !== RSI_OPENROUTER_ENDPOINT_TAG || args.endpointPrice.quantization !== RSI_OPENROUTER_ENDPOINT_QUANTIZATION) throw new Error("endpoint price identity differs from the frozen DeepInfra fp8 model route") + if (![args.priceCeilingsPerMillionUsd.input, args.priceCeilingsPerMillionUsd.output].every((value) => Number.isFinite(value) && value > 0) || args.priceCeilingsPerMillionUsd.input < args.endpointPrice.inputPricePerMillionUsd || args.priceCeilingsPerMillionUsd.output < args.endpointPrice.outputPricePerMillionUsd) throw new Error("configured price ceilings do not cover the observed endpoint rates") + if (!/^[a-f0-9]{64}$/.test(args.execution.configSha256) || args.execution.configSha256 !== frozenProofConfigDigest(args.execution.configs) || !/^[a-f0-9]{64}$/.test(args.execution.analysisSourceSha256)) throw new Error("proof demonstration config or analysis-source digest is malformed") + if (!args.execution.taskIdentities.length || args.execution.taskIdentities.some((task) => !/^[a-f0-9]{64}$/.test(task.verifierDigest) || !/^[a-f0-9]{64}$/.test(task.knownGoodDigest) || !/^[a-f0-9]{64}$/.test(task.knownBadDigest))) throw new Error("proof demonstration task identities are incomplete") + if (args.operatorCapUsd + 1e-9 < args.plan.totalRootSpendCeilingUsd) throw new Error(`operator cap $${args.operatorCapUsd.toFixed(6)} is below the frozen root reservation $${args.plan.totalRootSpendCeilingUsd.toFixed(6)}`) + const body = { schemaVersion: 1 as const, ...args } + return { ...body, digestSha256: frozenProofDemonstrationDigest(body) } +} + +export function verifyFrozenProofDemonstrationManifest(value: unknown): FrozenProofDemonstrationManifest { + if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("frozen proof demonstration manifest is malformed") + const manifest = value as FrozenProofDemonstrationManifest + if (manifest.schemaVersion !== 1 || !/^[a-f0-9]{64}$/.test(manifest.digestSha256)) throw new Error("frozen proof demonstration manifest schema or digest is malformed") + const { digestSha256, ...body } = manifest + if (frozenProofDemonstrationDigest(body) !== digestSha256) throw new Error("frozen proof demonstration manifest digest mismatch") + if (manifest.endpointPrice.model !== manifest.model.requestedId || manifest.endpointPrice.provider !== RSI_OPENROUTER_ENDPOINT_PROVIDER || manifest.endpointPrice.endpointTag !== RSI_OPENROUTER_ENDPOINT_TAG || manifest.endpointPrice.quantization !== RSI_OPENROUTER_ENDPOINT_QUANTIZATION || !["max_tokens", "temperature", "tools", "tool_choice"].every((parameter) => manifest.endpointPrice.supportedParameters.includes(parameter))) throw new Error("frozen proof demonstration endpoint is not the pinned DeepInfra fp8 route with required parameters") + if (manifest.operatorCapUsd + 1e-9 < manifest.plan.totalRootSpendCeilingUsd) throw new Error("frozen proof demonstration operator cap is below root reservations") + if (frozenProofConfigDigest(manifest.execution.configs) !== manifest.execution.configSha256) throw new Error("frozen proof demonstration config digest mismatch") + return manifest +} diff --git a/src/rsi/proof-demonstration-plan.ts b/src/rsi/proof-demonstration-plan.ts new file mode 100644 index 0000000..1dae229 --- /dev/null +++ b/src/rsi/proof-demonstration-plan.ts @@ -0,0 +1,168 @@ +import type { BenchmarkManifest, BenchmarkTask } from "./benchmark/types.js" +import type { OpenRouterBrokerLimits } from "./inference-broker.js" +import { proofCampaignTaskWallBoundMs } from "./controller.js" +import type { RsiConfig } from "./types.js" + +export interface ProofRoleLimits extends OpenRouterBrokerLimits {} +export interface ProofDemonstrationPlan { + engineering: ProofDemonstrationArmPlan + campaign: ProofDemonstrationArmPlan + totalReservedSpendUsd: number + totalRootSpendCeilingUsd: number + totalPriceDerivedSpendCeilingUsd: number + totalRootPriceDerivedSpendCeilingUsd: number + totalCalls: number + totalTokens: number + totalTaskWallMs: number +} +export interface ProofDemonstrationArmPlan { + phase: "engineering" | "campaign" + generations: number + population: number + mutationJobs: number + verificationJobs: number + developmentProofCells: number + confirmationScheduleCells: number + confirmationDisposition: "pre-registered-deferred" | "executed-after-development" + confirmationNotSelectedCells: number + paidConfirmationCells: number + rootControlCells: number + deferredConfirmationReservationCells: number + maximumCalls: number + maximumTokens: number + reservedSpendUsd: number + rootSpendCeilingUsd: number + priceDerivedSpendCeilingUsd: number + rootPriceDerivedSpendCeilingUsd: number + taskWallBoundMs: number + developmentTaskIds: string[] + confirmationTaskIds: string[] +} + +function microsCeil(usd: number): number { return Math.ceil(usd * 1_000_000) } +function fromMicros(value: number): number { return value / 1_000_000 } + +function worstTaskTokens(task: BenchmarkTask, limits: ProofRoleLimits): number { + return Math.min(limits.maxTotalTokens, task.ceilings.maxCalls * (limits.maxInputTokensPerCall + task.ceilings.maxOutputTokensPerCall)) +} + +function worstJobPriceMicros(maxCalls: number, maxOutputTokens: number, limits: ProofRoleLimits): number { + const inputPrice = limits.maxInputPricePerMillionUsd + const outputPrice = limits.maxOutputPricePerMillionUsd + const inputCapacity = maxCalls * limits.maxInputTokensPerCall + const outputCapacity = maxCalls * maxOutputTokens + let inputTokens: number + let outputTokens: number + if (inputPrice >= outputPrice) { + inputTokens = Math.min(limits.maxTotalTokens, inputCapacity) + outputTokens = Math.min(outputCapacity, Math.max(0, limits.maxTotalTokens - inputTokens)) + } else { + outputTokens = Math.min(limits.maxTotalTokens, outputCapacity) + inputTokens = Math.min(inputCapacity, Math.max(0, limits.maxTotalTokens - outputTokens)) + } + return Math.ceil(inputTokens * inputPrice + outputTokens * outputPrice) +} + +function assertPriceFeasible(label: string, calls: number, outputTokens: number, limits: ProofRoleLimits): number { + const perCall = worstJobPriceMicros(1, outputTokens, limits) + const perJob = worstJobPriceMicros(calls, outputTokens, limits) + if (perCall > microsCeil(limits.maxCallSpendUsd)) throw new Error(`${label} per-call spend cap is below the frozen price/token ceiling`) + if (perJob > microsCeil(limits.maxJobSpendUsd)) throw new Error(`${label} per-job spend cap is below the frozen price/token ceiling`) + return perJob +} + +export function planProofDemonstrationArm(args: { + phase: "engineering" | "campaign" + config: RsiConfig + manifest: BenchmarkManifest + mutationLimits: ProofRoleLimits + benchmarkLimits: ProofRoleLimits + generations: number + population: number +}): ProofDemonstrationArmPlan { + if (!Number.isSafeInteger(args.generations) || args.generations < 1 || !Number.isSafeInteger(args.population) || args.population < 1) throw new Error("proof plan generations and population must be positive integers") + const development = args.manifest.tasks.filter((task) => task.split === "development") + const confirmation = args.manifest.tasks.filter((task) => task.split === "confirmation") + if (!development.length || !confirmation.length) throw new Error("proof plan requires development and confirmation task splits") + const slotsPerArm = args.generations * args.population + const candidateSlotsBothArms = 2 * slotsPerArm + const mutationJobs = candidateSlotsBothArms + const verificationJobs = candidateSlotsBothArms * (2 + args.config.evalCommands.length) + const developmentProofCells = candidateSlotsBothArms * 3 * development.length + const candidateConfirmationCells = candidateSlotsBothArms * confirmation.length + const paidConfirmationCells = args.phase === "campaign" ? 2 * confirmation.length : 0 + const rootControlCells = args.phase === "campaign" ? 2 * confirmation.length : 0 + const deferredConfirmationReservationCells = args.phase === "engineering" ? 4 * confirmation.length : 0 + const confirmationNotSelectedCells = args.phase === "campaign" ? candidateConfirmationCells - paidConfirmationCells : 0 + const confirmationScheduleCells = candidateConfirmationCells + 2 * confirmation.length + const mutationCalls = mutationJobs * args.mutationLimits.maxCalls + const mutationTokensPerJob = Math.min(args.mutationLimits.maxTotalTokens, args.mutationLimits.maxCalls * (args.mutationLimits.maxInputTokensPerCall + args.mutationLimits.maxOutputTokensPerCall)) + const mutationPriceMicros = assertPriceFeasible("mutation", args.mutationLimits.maxCalls, args.mutationLimits.maxOutputTokensPerCall, args.mutationLimits) + let maximumCalls = mutationCalls + let maximumTokens = mutationJobs * mutationTokensPerJob + let reservedSpendMicros = mutationJobs * microsCeil(args.mutationLimits.maxJobSpendUsd) + let priceDerivedSpendMicros = mutationJobs * mutationPriceMicros + let rootSpendMicros = reservedSpendMicros + let rootPriceSpendMicros = priceDerivedSpendMicros + const proofRoles = candidateSlotsBothArms * 3 + const taskCalls = (task: BenchmarkTask) => task.ceilings.maxCalls + const taskPrice = (task: BenchmarkTask) => assertPriceFeasible(`benchmark task ${task.id}`, task.ceilings.maxCalls, task.ceilings.maxOutputTokensPerCall, args.benchmarkLimits) + const devCallsPerRole = development.reduce((sum, task) => sum + taskCalls(task), 0) + const devTokensPerRole = development.reduce((sum, task) => sum + worstTaskTokens(task, args.benchmarkLimits), 0) + const devPriceMicrosPerRole = development.reduce((sum, task) => sum + taskPrice(task), 0) + maximumCalls += proofRoles * devCallsPerRole + maximumTokens += proofRoles * devTokensPerRole + reservedSpendMicros += developmentProofCells * microsCeil(args.benchmarkLimits.maxJobSpendUsd) + priceDerivedSpendMicros += proofRoles * devPriceMicrosPerRole + rootSpendMicros = reservedSpendMicros + rootPriceSpendMicros = priceDerivedSpendMicros + if (args.phase === "engineering") { + const deferredFinalAndRoot = 4 * confirmation.length + rootSpendMicros += deferredFinalAndRoot * microsCeil(args.benchmarkLimits.maxJobSpendUsd) + rootPriceSpendMicros += confirmation.reduce((sum, task) => sum + taskPrice(task), 0) * 4 + } + if (args.phase === "campaign") { + const finalAndRoot = 4 // selected final and root control, in both arms + const confirmationCalls = confirmation.reduce((sum, task) => sum + taskCalls(task), 0) * finalAndRoot + const confirmationTokens = confirmation.reduce((sum, task) => sum + worstTaskTokens(task, args.benchmarkLimits), 0) * finalAndRoot + const confirmationPrice = confirmation.reduce((sum, task) => sum + taskPrice(task), 0) * finalAndRoot + maximumCalls += confirmationCalls + maximumTokens += confirmationTokens + reservedSpendMicros += paidConfirmationCells + rootControlCells ? (paidConfirmationCells + rootControlCells) * microsCeil(args.benchmarkLimits.maxJobSpendUsd) : 0 + priceDerivedSpendMicros += confirmationPrice + rootSpendMicros = reservedSpendMicros + rootPriceSpendMicros = priceDerivedSpendMicros + } + if (![maximumCalls, maximumTokens, reservedSpendMicros, rootSpendMicros, priceDerivedSpendMicros, rootPriceSpendMicros].every(Number.isSafeInteger)) throw new Error("proof plan exceeds safe integer accounting") + const taskWallBoundMs = proofCampaignTaskWallBoundMs(args.config, args.generations, development, confirmation, args.phase === "campaign") + return { + phase: args.phase, generations: args.generations, population: args.population, + mutationJobs, verificationJobs, developmentProofCells, confirmationScheduleCells, + confirmationDisposition: args.phase === "campaign" ? "executed-after-development" : "pre-registered-deferred", + confirmationNotSelectedCells, paidConfirmationCells, rootControlCells, deferredConfirmationReservationCells, + maximumCalls, maximumTokens, reservedSpendUsd: fromMicros(reservedSpendMicros), + rootSpendCeilingUsd: fromMicros(rootSpendMicros), priceDerivedSpendCeilingUsd: fromMicros(priceDerivedSpendMicros), + rootPriceDerivedSpendCeilingUsd: fromMicros(rootPriceSpendMicros), taskWallBoundMs, + developmentTaskIds: development.map((task) => task.id), confirmationTaskIds: confirmation.map((task) => task.id), + } +} + +export function planProofDemonstration(args: { + config: RsiConfig + manifest: BenchmarkManifest + mutationLimits: ProofRoleLimits + benchmarkLimits: ProofRoleLimits +}): ProofDemonstrationPlan { + const engineering = planProofDemonstrationArm({ ...args, phase: "engineering", generations: 2, population: 2 }) + const campaign = planProofDemonstrationArm({ ...args, phase: "campaign", generations: 3, population: 2 }) + return { + engineering, campaign, + totalReservedSpendUsd: fromMicros(microsCeil(engineering.reservedSpendUsd) + 3 * microsCeil(campaign.reservedSpendUsd)), + totalRootSpendCeilingUsd: fromMicros(microsCeil(engineering.rootSpendCeilingUsd) + 3 * microsCeil(campaign.rootSpendCeilingUsd)), + totalPriceDerivedSpendCeilingUsd: fromMicros(microsCeil(engineering.priceDerivedSpendCeilingUsd) + 3 * microsCeil(campaign.priceDerivedSpendCeilingUsd)), + totalRootPriceDerivedSpendCeilingUsd: fromMicros(microsCeil(engineering.rootPriceDerivedSpendCeilingUsd) + 3 * microsCeil(campaign.rootPriceDerivedSpendCeilingUsd)), + totalCalls: engineering.maximumCalls + 3 * campaign.maximumCalls, + totalTokens: engineering.maximumTokens + 3 * campaign.maximumTokens, + totalTaskWallMs: engineering.taskWallBoundMs + 3 * campaign.taskWallBoundMs, + } +} diff --git a/src/rsi/proof-demonstration.ts b/src/rsi/proof-demonstration.ts new file mode 100644 index 0000000..bb6fccf --- /dev/null +++ b/src/rsi/proof-demonstration.ts @@ -0,0 +1,275 @@ +import { createHash } from "node:crypto" +import type { BenchmarkAttempt } from "./benchmark/types.js" +import type { RsiRunRecord } from "./types.js" +import { proofCampaignManifestDigest } from "./proof-campaign.js" +import { RSI_OPENROUTER_MODEL } from "./inference-broker.js" + +export type ProofDemonstrationVerdict = "demonstrated" | "inconclusive" | "not-demonstrated" +export interface ProofDemonstrationReport { + schemaVersion: 1 + analysisVersion: "task-cluster-bootstrap-90-v1" + verdict: ProofDemonstrationVerdict + verdictReasons: string[] + campaigns: Array<{ + campaignId: string + selfHosted: { runId: string; finalCommit: string; selectedCandidateId: string; adjacentAcceptedDepth: number; confirmation: Array>; root: Array>; netWinsVsRoot: number; retentionLosses: string[]; resources: Record; complete: boolean } + fixedControl: { runId: string; finalCommit: string; selectedCandidateId: string; confirmation: Array>; root: Array>; resources: Record; complete: boolean } + }> + pooled: { taskIds: string[]; campaignCount: number; selfHostedSuccesses: number; fixedControlSuccesses: number; difference: number; cluster90PercentInterval: [number, number] | null; bootstrapReplicates: number; bootstrapSeed: string } + denominators: { campaigns: number; pairedConfirmationTasks: number; missingCells: string[]; excludedAttempts: string[] } + limitations: string[] +} + +function sha256(value: string): string { return createHash("sha256").update(value).digest("hex") } +function sorted(items: T[]): T[] { return [...items].sort((a, b) => Buffer.compare(Buffer.from(a), Buffer.from(b))) } +function percentile(values: number[], fraction: number): number { + if (!values.length) throw new Error("cannot calculate percentile from no bootstrap samples") + return values[Math.min(values.length - 1, Math.max(0, Math.ceil(fraction * values.length) - 1))]! +} +function seededRandom(seed: string): () => number { + let state = Number.parseInt(sha256(seed).slice(0, 8), 16) >>> 0 + if (!state) state = 0x9e3779b9 + return () => { state ^= state << 13; state ^= state >>> 17; state ^= state << 5; return (state >>> 0) / 0x1_0000_0000 } +} +function campaignSlot(run: RsiRunRecord, candidateId: string): string | undefined { + const candidate = run.candidates.find((entry) => entry.id === candidateId) + if (!candidate) return undefined + const generation = run.candidates.filter((entry) => entry.generation === candidate.generation) + const index = generation.indexOf(candidate) + return index < 0 ? undefined : "g" + candidate.generation + "-candidate-" + index +} +function durablySelected(run: RsiRunRecord, candidate: RsiRunRecord["candidates"][number]): boolean { + const snapshot = run.proofCampaign?.ledgerSnapshot + const slot = campaignSlot(run, candidate.id) + if (!snapshot || !slot) return false + const events = (snapshot.events as Array> | undefined) ?? [] + const decisions = (snapshot.decisions as Array> | undefined) ?? [] + const selected = events.some((event) => event.arm_id === run.workerHarnessMode && event.event_type === "generation-selected" && Number((event.payload as { generation?: unknown })?.generation) === candidate.generation && ((event.payload as { selectedCandidateIds?: unknown })?.selectedCandidateIds as unknown[] | undefined)?.includes(slot)) + const decision = decisions.find((entry) => entry.arm_id === run.workerHarnessMode && Number(entry.generation) === candidate.generation && entry.candidate_id === slot) + return selected && decision?.decision === "accepted" && decision.child_commit === candidate.commits[0] && decision.parent_commit === candidate.parentCommit && decision.parent_id === (candidate.parent === "baseline" ? "baseline" : campaignSlot(run, candidate.parent ?? "")) +} +function chainDepth(run: RsiRunRecord): number { + const selected = run.proofConfirmation?.selectedHarness + let current = selected ? run.candidates.find((candidate) => candidate.id === selected.candidateId) : undefined + if (!current || current.status !== "accepted" || current.developmentProof?.accepted !== true || !durablySelected(run, current)) return 0 + let depth = 0 + for (;;) { + if (!current.parent || current.parent === "baseline") { + if (current.generation === 0 && current.baseCommit === run.baseCommit && current.parentCommit === run.baseCommit && current.developmentProof?.immediateParent.comparable && current.developmentProof.immediateParent.eligible && durablySelected(run, current)) depth++ + return depth + } + const parent = run.candidates.find((candidate) => candidate.id === current!.parent) + if (!parent || parent.status !== "accepted" || parent.developmentProof?.accepted !== true || !durablySelected(run, parent) || current.generation !== parent.generation + 1 || current.baseCommit !== parent.commits[0] || current.parentCommit !== parent.commits[0] || !current.developmentProof?.immediateParent.comparable || !current.developmentProof.immediateParent.eligible || current.workerHarnessExecution?.parentCommit !== parent.commits[0]) return depth + depth++ + current = parent + } +} + +function attemptSummary(attempt: BenchmarkAttempt): Record { + return { + attemptId: attempt.attemptId, taskId: attempt.taskId, harnessCommit: attempt.harnessCommit, + verifiedSuccess: attempt.verifiedSuccess, validity: attempt.validity, proofQualified: attempt.proofQualified, + resourceAccounting: attempt.resourceAccounting, calls: attempt.calls, tokens: attempt.tokens, + wallTimeMs: attempt.wallTimeMs, costUsd: attempt.inferenceReceipt?.costUsd ?? null, + model: attempt.model, seed: attempt.seed, startDigest: attempt.startDigest, + evaluatorDigest: attempt.evaluatorDigest, patchDigest: attempt.patchDigest, + artifactRefs: attempt.artifactRefs, stopReason: attempt.stopReason, + } +} +function ledgerAttempt(run: RsiRunRecord, attempt: BenchmarkAttempt): Record | undefined { + const entries = (run.proofCampaign?.ledgerSnapshot?.attempts as Array> | undefined) ?? [] + return entries.find((entry) => entry.broker_job_id === attempt.attemptId) +} +function validateScheduledConfirmation(run: RsiRunRecord, missing: string[]): boolean { + const manifest = run.proofCampaign?.manifest + const snapshot = run.proofCampaign?.ledgerSnapshot + const rows = snapshot?.attempts as Array> | undefined + if (!manifest || !rows) { missing.push(run.runId + ":missing-campaign-snapshot"); return false } + const selected = run.proofConfirmation?.selectedHarness + let complete = true + for (const cell of manifest.schedule.filter((entry) => entry.armId === run.proofCampaign?.armId && entry.phase === "confirmation")) { + const expected = cell.candidateId === "root-control" || (cell.candidateId === selected?.slotId && cell.generation === selected.generation) ? "completed" : "not-selected" + const matches = rows.filter((row) => row.arm_id === run.proofCampaign?.armId && Number(row.generation) === cell.generation && row.candidate_id === cell.candidateId && row.task_id === cell.taskId && row.seed === cell.seed && row.attempt_kind === cell.attemptKind) + if (matches.length !== 1 || matches[0]?.status !== expected) { + missing.push(run.runId + "/" + cell.candidateId + "/" + cell.taskId + ":expected-" + expected) + complete = false + } + } + return complete +} +type FrozenTaskIdentity = { id: string; seed: number; startDigest: string; verifierDigest: string; ceilings: BenchmarkAttempt["ceilings"] } +function validateAttempt(run: RsiRunRecord, attempt: BenchmarkAttempt, expectedCommit: string, expectedTask: FrozenTaskIdentity | undefined, missing: string[], excluded: string[]): boolean { + const expectedTaskId = expectedTask?.id ?? "unknown-task" + const label = run.runId + "/" + expectedTaskId + if (!expectedTask || attempt.taskId !== expectedTaskId || attempt.harnessCommit !== expectedCommit || attempt.split !== "confirmation" || attempt.seed !== expectedTask.seed || attempt.startDigest !== expectedTask.startDigest || attempt.evaluatorDigest !== expectedTask.verifierDigest || JSON.stringify(attempt.ceilings) !== JSON.stringify(expectedTask.ceilings)) { missing.push(label + ":identity"); return false } + const expectedModel = run.proofCampaign?.manifest.model + if (!expectedModel || attempt.model.provider !== expectedModel.provider || attempt.model.id !== expectedModel.id || JSON.stringify(attempt.model.sampling) !== JSON.stringify(expectedModel.settings.sampling) || attempt.inferenceReceipt?.model !== expectedModel.id || attempt.inferenceReceipt.returnedModels.length !== 1 || attempt.inferenceReceipt.returnedModels[0] !== expectedModel.id) { missing.push(label + ":model-identity"); return false } + const limits = run.proofCampaign?.manifest.settings.inferenceCeilings as { maxInputTokensPerCall?: unknown; maxTotalTokens?: unknown; maxJobSpendUsd?: unknown } | undefined + const maxTokens = Math.min(Number(limits?.maxTotalTokens), expectedTask.ceilings.maxCalls * (Number(limits?.maxInputTokensPerCall) + expectedTask.ceilings.maxOutputTokensPerCall)) + if (!Number.isSafeInteger(maxTokens) || attempt.wallTimeMs > expectedTask.ceilings.timeoutMs || attempt.calls === null || attempt.calls > expectedTask.ceilings.maxCalls || attempt.tokens === null || attempt.tokens > maxTokens || attempt.inferenceReceipt?.costUsd === null || attempt.inferenceReceipt?.costUsd === undefined || attempt.inferenceReceipt.costUsd > Number(limits?.maxJobSpendUsd) || attempt.inferenceReceipt.calls !== attempt.calls || attempt.inferenceReceipt.totalTokens !== attempt.tokens) { missing.push(label + ":resource-ceiling"); return false } + const row = ledgerAttempt(run, attempt) + if (!row || row.status !== "completed" || row.result_sha256 !== sha256(JSON.stringify(attempt))) { missing.push(label + ":ledger-result"); return false } + const savedRefs = row.artifact_refs as Record | undefined + if (!savedRefs || Object.entries(attempt.artifactRefs).some(([key, value]) => savedRefs[key]?.sha256 !== value)) { missing.push(label + ":artifact-reference-mismatch"); return false } + if (attempt.validity !== "valid" || !attempt.proofQualified || attempt.resourceAccounting !== "supervisor-enforced" || !attempt.inferenceReceipt?.qualified) { + excluded.push(label + ":" + attempt.validity + "/" + attempt.resourceAccounting) + return false + } + return true +} +function matchAttempts(run: RsiRunRecord, attemptIds: string[], commit: string, taskIds: string[], taskIdentities: Map, missing: string[], excluded: string[]): Map { + const all = new Map((run.confirmationAttempts ?? []).map((attempt) => [attempt.attemptId, attempt])) + const selected = attemptIds.map((id) => all.get(id)).filter((entry): entry is BenchmarkAttempt => Boolean(entry)) + if (selected.length !== attemptIds.length) missing.push(run.runId + ":attempt-artifact") + const byTask = new Map() + for (const taskId of taskIds) { + const matches = selected.filter((attempt) => attempt.taskId === taskId) + if (matches.length !== 1) { missing.push(run.runId + "/" + taskId + ":expected-one-attempt"); continue } + if (validateAttempt(run, matches[0]!, commit, taskIdentities.get(taskId), missing, excluded)) byTask.set(taskId, matches[0]!) + } + return byTask +} + +/** Recompute the preregistered verdict from immutable attempt artifacts and durable campaign snapshots. */ +export function analyzeProofDemonstration(runs: RsiRunRecord[], options: { expectedCampaignIds: string[]; retentionTaskIds?: string[]; bootstrapSeed: string; bootstrapReplicates?: number }): ProofDemonstrationReport { + const missing: string[] = [] + const excluded: string[] = [] + const campaignIds = sorted([...new Set(options.expectedCampaignIds)]) + if (campaignIds.length !== 3 || options.expectedCampaignIds.length !== 3) throw new Error("analysis requires exactly three distinct preregistered campaign IDs") + const confirmationTasks = sorted([...new Set(runs.flatMap((run) => run.proofCampaign?.manifest.schedule.filter((cell) => cell.phase === "confirmation" && cell.candidateId === "root-control").map((cell) => cell.taskId) ?? []))]) + if (!confirmationTasks.length) throw new Error("confirmation task schedule is empty") + const frozenRetentionTaskIds = (runs.find((run) => run.proofCampaign)?.proofCampaign?.manifest.acceptanceRule.confirmationRetentionTaskIds as string[] | undefined) ?? [] + if (!frozenRetentionTaskIds.length || frozenRetentionTaskIds.some((id) => !confirmationTasks.includes(id))) throw new Error("frozen confirmation retention set is missing or outside the confirmation split") + if (options.retentionTaskIds && JSON.stringify(sorted(options.retentionTaskIds)) !== JSON.stringify(sorted(frozenRetentionTaskIds))) throw new Error("analysis retention set differs from the frozen confirmation rule") + const campaigns: ProofDemonstrationReport["campaigns"] = [] + const benchmarkManifestDigests = new Set() + const providerModels = new Set() + const protocolIdentities = new Set() + const campaignSeeds = new Set() + const paired: Array<{ taskId: string; self: number; fixed: number }> = [] + let selfSuccesses = 0 + let fixedSuccesses = 0 + for (const campaignId of campaignIds) { + const arms = runs.filter((run) => run.workerHarnessPairId === campaignId) + const selfRun = arms.find((run) => run.workerHarnessMode === "self-hosted") + const fixedRun = arms.find((run) => run.workerHarnessMode === "fixed-control") + if (!selfRun || !fixedRun) { missing.push(campaignId + ":missing-arm"); continue } + if (selfRun.proofCampaign?.manifestSha256 !== fixedRun.proofCampaign?.manifestSha256 || selfRun.workerHarnessConfigDigest !== fixedRun.workerHarnessConfigDigest || selfRun.workerHarnessSamplingSeed !== fixedRun.workerHarnessSamplingSeed) missing.push(campaignId + ":paired-manifest-or-seed-mismatch") + for (const run of [selfRun, fixedRun]) { + const frozen = run.proofCampaign?.manifest + if (frozen) { + benchmarkManifestDigests.add(frozen.benchmarkDigests.manifest) + providerModels.add(frozen.model.provider + "/" + frozen.model.id) + campaignSeeds.add(frozen.seed) + const frozenSettings = Object.fromEntries(Object.entries(frozen.settings).filter(([key]) => key !== "configDigest" && key !== "inferenceCampaignIds")) + protocolIdentities.add(JSON.stringify({ rootCommit: frozen.rootCommit, benchmarkDigests: frozen.benchmarkDigests, provider: frozen.model.provider, model: frozen.model.id, modelSettings: frozen.model.settings, settings: frozenSettings, acceptanceRule: frozen.acceptanceRule, runtime: frozen.runtime, budgets: frozen.budgets, taskOrder: frozen.taskOrder, analysisVersion: frozen.analysisVersion })) + for (const armId of [frozen.armAssignment.fixedControl, frozen.armAssignment.selfHosted]) { + const rootTaskIds = sorted(frozen.schedule.filter((cell) => cell.armId === armId && cell.phase === "confirmation" && cell.candidateId === "root-control").map((cell) => cell.taskId)) + if (JSON.stringify(rootTaskIds) !== JSON.stringify(confirmationTasks)) missing.push(campaignId + ":confirmation-task-schedule-mismatch:" + armId) + } + } + } + const selfMeta = selfRun.proofConfirmation + const fixedMeta = fixedRun.proofConfirmation + if (!selfMeta || !fixedMeta) { missing.push(campaignId + ":missing-final-harness-selection"); continue } + const identities = ((selfRun.proofCampaign?.manifest.settings.confirmationTaskIdentities as FrozenTaskIdentity[] | undefined) ?? []).map((task) => [task.id, task] as const) + const taskIdentities = new Map(identities) + if (confirmationTasks.some((taskId) => !taskIdentities.has(taskId))) missing.push(campaignId + ":missing-frozen-task-identity") + const selfFinal = matchAttempts(selfRun, selfMeta.selectedHarness.attemptIds, selfMeta.selectedHarness.commit, confirmationTasks, taskIdentities, missing, excluded) + const selfRoot = matchAttempts(selfRun, selfMeta.rootHarness.attemptIds, selfMeta.rootHarness.commit, confirmationTasks, taskIdentities, missing, excluded) + const fixedFinal = matchAttempts(fixedRun, fixedMeta.selectedHarness.attemptIds, fixedMeta.selectedHarness.commit, confirmationTasks, taskIdentities, missing, excluded) + const fixedRoot = matchAttempts(fixedRun, fixedMeta.rootHarness.attemptIds, fixedMeta.rootHarness.commit, confirmationTasks, taskIdentities, missing, excluded) + const wins = confirmationTasks.filter((id) => selfFinal.get(id)?.verifiedSuccess && !selfRoot.get(id)?.verifiedSuccess).length + const losses = confirmationTasks.filter((id) => !selfFinal.get(id)?.verifiedSuccess && selfRoot.get(id)?.verifiedSuccess).length + const retentionLosses = frozenRetentionTaskIds.filter((id) => selfFinal.has(id) && selfRoot.has(id) && !selfFinal.get(id)?.verifiedSuccess && Boolean(selfRoot.get(id)?.verifiedSuccess)) + const depth = chainDepth(selfRun) + const selfComplete = selfFinal.size === confirmationTasks.length && selfRoot.size === confirmationTasks.length && selfRun.confirmationEvidence === "complete" && validateScheduledConfirmation(selfRun, missing) + const fixedComplete = fixedFinal.size === confirmationTasks.length && fixedRoot.size === confirmationTasks.length && fixedRun.confirmationEvidence === "complete" && validateScheduledConfirmation(fixedRun, missing) + const latestSnapshot = [selfRun, fixedRun].map((armRun) => armRun.proofCampaign?.ledgerSnapshot).filter((snapshot): snapshot is NonNullable["ledgerSnapshot"] => Boolean(snapshot)).sort((a, b) => Number((b?.campaign as Record | undefined)?.sequence ?? 0) - Number((a?.campaign as Record | undefined)?.sequence ?? 0))[0] + const ledger = latestSnapshot?.campaign as Record | undefined + const ledgerArms = latestSnapshot?.arms as Array> | undefined + const manifest = selfRun.proofCampaign?.manifest + if (!ledger || !manifest || ledger.state !== "finished" || String(ledger.manifest_sha256).trim() !== proofCampaignManifestDigest(manifest) || manifest.campaignId !== campaignId) missing.push(campaignId + ":campaign-ledger-not-finished-or-manifest-corrupt") + if (!ledgerArms || ledgerArms.length !== 2 || ledgerArms.some((arm) => arm.state !== "finished")) missing.push(campaignId + ":campaign-arm-not-finished") + if (ledger && manifest) { + const usedCalls = Number(ledger.used_calls) + const usedTokens = Number(ledger.used_tokens) + const usedSpend = Number(ledger.used_spend_microusd) + const reservedCalls = Number(ledger.reserved_calls) + const reservedTokens = Number(ledger.reserved_tokens) + const reservedSpend = Number(ledger.reserved_spend_microusd) + if (![usedCalls, usedTokens, usedSpend, reservedCalls, reservedTokens, reservedSpend].every(Number.isSafeInteger) || usedCalls > manifest.budgets.root.maxCalls || usedTokens > manifest.budgets.root.maxTokens || usedSpend > Math.floor(manifest.budgets.root.maxSpendUsd * 1_000_000) || reservedCalls !== 0 || reservedTokens !== 0 || reservedSpend !== 0) missing.push(campaignId + ":root-budget-accounting-invalid") + const deadline = Date.parse(String(ledger.deadline_at)) + if (!Number.isFinite(deadline) || [selfRun, fixedRun].some((armRun) => !armRun.finishedAt || Date.parse(armRun.finishedAt) > deadline)) missing.push(campaignId + ":root-runtime-deadline-exceeded") + } + const retentionIds = manifest?.acceptanceRule.confirmationRetentionTaskIds + if (!Array.isArray(retentionIds) || retentionIds.length === 0 || retentionIds.some((id) => typeof id !== "string" || !confirmationTasks.includes(id)) || JSON.stringify(sorted(retentionIds as string[])) !== JSON.stringify(sorted(frozenRetentionTaskIds))) missing.push(campaignId + ":confirmation-retention-set-invalid") + if (!selfComplete || !fixedComplete) missing.push(campaignId + ":incomplete-confirmation-cells") + for (const taskId of confirmationTasks) { + const left = selfFinal.get(taskId) + const right = fixedFinal.get(taskId) + if (left && right) { + const selfValue = left.verifiedSuccess ? 1 : 0 + const fixedValue = right.verifiedSuccess ? 1 : 0 + paired.push({ taskId, self: selfValue, fixed: fixedValue }) + selfSuccesses += selfValue + fixedSuccesses += fixedValue + } + } + const sumResources = (attempts: BenchmarkAttempt[]): Record => ({ + calls: attempts.every((entry) => entry.calls !== null) ? attempts.reduce((sum, entry) => sum + (entry.calls ?? 0), 0) : null, + tokens: attempts.every((entry) => entry.tokens !== null) ? attempts.reduce((sum, entry) => sum + (entry.tokens ?? 0), 0) : null, + wallTimeMs: attempts.reduce((sum, entry) => sum + entry.wallTimeMs, 0), + spendUsd: attempts.every((entry) => entry.inferenceReceipt?.costUsd !== null && entry.inferenceReceipt?.costUsd !== undefined) ? attempts.reduce((sum, entry) => sum + (entry.inferenceReceipt?.costUsd ?? 0), 0) : null, + }) + campaigns.push({ + campaignId, + selfHosted: { runId: selfRun.runId, finalCommit: selfMeta.selectedHarness.commit, selectedCandidateId: selfMeta.selectedHarness.candidateId, adjacentAcceptedDepth: depth, confirmation: [...selfFinal.values()].map(attemptSummary), root: [...selfRoot.values()].map(attemptSummary), netWinsVsRoot: wins - losses, retentionLosses, resources: sumResources([...selfFinal.values(), ...selfRoot.values()]), complete: selfComplete }, + fixedControl: { runId: fixedRun.runId, finalCommit: fixedMeta.selectedHarness.commit, selectedCandidateId: fixedMeta.selectedHarness.candidateId, confirmation: [...fixedFinal.values()].map(attemptSummary), root: [...fixedRoot.values()].map(attemptSummary), resources: sumResources([...fixedFinal.values(), ...fixedRoot.values()]), complete: fixedComplete }, + }) + } + if (benchmarkManifestDigests.size !== 1) missing.push("paired:benchmark-manifest-digest-mismatch") + if (providerModels.size !== 1) missing.push("paired:provider-model-mismatch") + if (providerModels.size === 1 && !providerModels.has("openshell-openrouter/" + RSI_OPENROUTER_MODEL)) missing.push("paired:model-is-not-the-pinned-OpenRouter-endpoint") + if (protocolIdentities.size !== 1) missing.push("paired:frozen-protocol-runtime-budget-or-selection-mismatch") + if (campaignSeeds.size !== 3) missing.push("paired:campaign-seeds-are-not-three-independent-values") + const taskIds = sorted([...new Set(paired.map((entry) => entry.taskId))]) + const replicates = options.bootstrapReplicates ?? 10_000 + if (!Number.isSafeInteger(replicates) || replicates < 1) throw new Error("bootstrap replicates must be a positive safe integer") + let interval: [number, number] | null = null + let difference = 0 + if (taskIds.length === confirmationTasks.length && paired.length === confirmationTasks.length * campaignIds.length) { + const clusterMeans = taskIds.map((taskId) => { + const cluster = paired.filter((entry) => entry.taskId === taskId) + return cluster.reduce((sum, entry) => sum + entry.self - entry.fixed, 0) / cluster.length + }) + difference = clusterMeans.reduce((sum, value) => sum + value, 0) / clusterMeans.length + const random = seededRandom(options.bootstrapSeed) + const samples: number[] = [] + for (let sample = 0; sample < replicates; sample++) { + let total = 0 + for (let index = 0; index < clusterMeans.length; index++) total += clusterMeans[Math.floor(random() * clusterMeans.length)]! + samples.push(total / clusterMeans.length) + } + samples.sort((a, b) => a - b) + interval = [percentile(samples, 0.05), percentile(samples, 0.95)] + } else missing.push("pooled:incomplete-task-cluster-matrix") + const allComplete = missing.length === 0 && excluded.length === 0 && campaigns.length === 3 && campaigns.every((entry) => entry.selfHosted.complete && entry.fixedControl.complete) + const winningCampaigns = campaigns.filter((entry) => entry.selfHosted.adjacentAcceptedDepth >= 2 && entry.selfHosted.netWinsVsRoot > 0 && entry.selfHosted.retentionLosses.length === 0).length + const intervalPass = interval !== null && interval[0] > 0 + const demonstrated = allComplete && winningCampaigns >= 2 && intervalPass + const reasons: string[] = [] + if (!allComplete) reasons.push("a frozen confirmation cell or durable artifact is missing, unqualified, or invalid") + if (winningCampaigns < 2) reasons.push("fewer than two self-hosted campaigns have two successive immediate-parent transitions, a net confirmation win, and no retention loss") + if (!intervalPass) reasons.push("the task-cluster 90% interval lower bound is not above zero") + if (!demonstrated && allComplete) reasons.push("all scheduled cells completed, but one or more preregistered improvement gates did not pass") + return { + schemaVersion: 1, analysisVersion: "task-cluster-bootstrap-90-v1", + verdict: demonstrated ? "demonstrated" : allComplete ? "not-demonstrated" : "inconclusive", + verdictReasons: demonstrated ? ["all frozen campaign, lineage, confirmation, retention, and task-cluster interval gates passed"] : reasons, + campaigns, + pooled: { taskIds, campaignCount: campaignIds.length, selfHostedSuccesses: selfSuccesses, fixedControlSuccesses: fixedSuccesses, difference, cluster90PercentInterval: interval, bootstrapReplicates: replicates, bootstrapSeed: options.bootstrapSeed }, + denominators: { campaigns: campaignIds.length, pairedConfirmationTasks: taskIds.length, missingCells: sorted(missing), excludedAttempts: sorted(excluded) }, + limitations: ["This analysis covers only the registered tasks, runtime, model endpoint, prompts, and budgets.", "A result does not establish model-weight improvement or broad coding ability.", "Input and output artifacts must remain available for independent digest and verifier replay."], + } +} diff --git a/src/rsi/reports.ts b/src/rsi/reports.ts index a6fe2b7..2aad465 100644 --- a/src/rsi/reports.ts +++ b/src/rsi/reports.ts @@ -14,7 +14,7 @@ export function formatCandidate(candidate: CandidateRecord): string { ? `, development=${candidate.developmentProof.accepted ? "eligible" : "ineligible"} (${candidate.developmentProof.reason}); ${describeProofArm("immediate-parent", candidate.developmentProof.immediateParent)}; ${describeProofArm("original-base", candidate.developmentProof.originalBase)}; protocol=${candidate.developmentProof.protocolDigest}; attemptRefs child/parent/root=${candidate.developmentProof.childAttemptIds.length}/${candidate.developmentProof.parentAttemptIds.length}/${candidate.developmentProof.originalBaseAttemptIds.length}` : candidate.developmentProof ? `, development=ineligible (${candidate.developmentProof.reason || "incomplete proof record"})` : ", development=unmeasured (no benchmark proof recorded)" const formatDenials = (denials: NonNullable) => denials.map((entry) => `${entry.statusCode}:${entry.reason}=${entry.count}`).join(",") || "none" - const model = candidate.modelProvider ? `, worker=${candidate.modelProvider}/${candidate.model}${candidate.inferenceUsage ? ` returned=${candidate.inferenceUsage.returnedModels.join(",")} calls=${candidate.inferenceUsage.calls} tokens=${candidate.inferenceUsage.totalTokens} costUsd=${candidate.inferenceUsage.costUsd} denials=${formatDenials(candidate.inferenceUsage.denials)}` : ` usage=unmeasured${candidate.inferenceDenials ? ` denials=${formatDenials(candidate.inferenceDenials)}` : ""}`}` : "" + const model = candidate.modelProvider ? `, worker=${candidate.modelProvider}/${candidate.model}${candidate.inferenceUsage ? ` endpoint=${candidate.inferenceUsage.endpointProvider} returned=${candidate.inferenceUsage.returnedModels.join(",")} calls=${candidate.inferenceUsage.calls} tokens=${candidate.inferenceUsage.totalTokens} costUsd=${candidate.inferenceUsage.costUsd} denials=${formatDenials(candidate.inferenceUsage.denials)}` : ` usage=unmeasured${candidate.inferenceDenials ? ` denials=${formatDenials(candidate.inferenceDenials)}` : ""}`}` : "" const workerHarness = candidate.workerHarnessExecution ? `, harness=${candidate.workerHarnessExecution.mode} parent=${candidate.workerHarnessExecution.parentCommit} snapshot=${candidate.workerHarnessExecution.snapshotCommit}/${candidate.workerHarnessExecution.snapshotTreeDigest} image=${candidate.workerHarnessExecution.runtime.imageDigest} commit=${candidate.workerHarnessExecution.runtime.harnessCommit} lock=${candidate.workerHarnessExecution.runtime.dependencyLockDigest} cli=${candidate.workerHarnessExecution.executedCliPath}` : "" @@ -30,6 +30,10 @@ export function formatRunReport(run: RsiRunRecord, config: RsiConfig): string { `- Worker harness: mode=${run.workerHarnessMode ?? "fixed-control"}; pair=${run.workerHarnessPairId ?? "unpaired"}; plannedGenerations=${run.workerHarnessPlannedGenerations ?? run.generations}; configDigest=${run.workerHarnessConfigDigest ?? "unavailable"}; samplingSeed=${run.workerHarnessSamplingSeed ?? "unavailable"}${run.workerHarnessRuntime ? `; image=${run.workerHarnessRuntime.imageDigest}; harnessCommit=${run.workerHarnessRuntime.harnessCommit}; dependencyLock=${run.workerHarnessRuntime.dependencyLockDigest}` : ""}`, `- Development proof: ${run.developmentProofProtocol ? `frozen protocol ${run.developmentProofProtocol.protocolDigest}; ${run.developmentProofProtocol.taskIds.length} development tasks; minimum net wins ${run.developmentProofProtocol.rule.minimumNetWins}; retention tasks ${run.developmentProofProtocol.rule.retentionTaskIds.join(", ")}; maximum mean-token increase ratio ${run.developmentProofProtocol.rule.maximumMeanTokenIncreaseRatio}` : "unavailable"}`, `- Confirmation evidence: ${run.confirmationEvidence ?? "unmeasured"}`, + ...(run.proofConfirmation ? [ + `- Confirmation harness: ${run.proofConfirmation.selectedHarness.candidateId} (${run.proofConfirmation.selectedHarness.commit}; generation ${run.proofConfirmation.selectedHarness.generation}; attempt IDs ${run.proofConfirmation.selectedHarness.attemptIds.join(",") || "none"})`, + `- Root control: ${run.proofConfirmation.rootHarness.commit}; attempt IDs ${run.proofConfirmation.rootHarness.attemptIds.join(",") || "none"}`, + ] : []), `- Generations: ${run.generations}`, `- Started: ${run.startedAt}`, `- Finished: ${run.finishedAt ?? "in progress"}`, diff --git a/src/rsi/types.ts b/src/rsi/types.ts index 13fb0d3..4531782 100644 --- a/src/rsi/types.ts +++ b/src/rsi/types.ts @@ -7,6 +7,7 @@ export type CandidateStatus = "planned" | "mutating" | "evaluating" | "accepted" export type MutationKind = "corrective" | "architectural" | "search-policy" | "curriculum" | "model-adaptation" export type ParentSelectionPolicy = "champion-specialist-novelty" | "pareto-front" | "all-eligible" export type WorkerHarnessMode = "fixed-control" | "self-hosted" +export type WorkerHarnessPhase = "development" | "confirmation" export interface WorkerHarnessRuntimeIdentity { imageName: string @@ -243,6 +244,7 @@ export interface RsiConfig { maxTokens?: number benchmarkHarnessCommit?: string workerHarnessMode?: WorkerHarnessMode + workerHarnessPhase?: WorkerHarnessPhase workerHarnessPairId?: string workerHarnessRuntime?: WorkerHarnessRuntimeIdentity workerHarnessParentCommit?: string @@ -281,8 +283,10 @@ export interface CandidateRecord { modelProvider?: string inferenceUsage?: { provider: "openrouter" + endpointProvider: "deepinfra" model: string returnedModels: string[] + returnedProviders: string[] calls: number promptTokens: number completionTokens: number @@ -451,7 +455,12 @@ export interface RsiRunRecord { workerHarnessPlannedGenerations?: number workerHarnessSamplingSeed?: string workerHarnessRuntime?: WorkerHarnessRuntimeIdentity - confirmationEvidence?: "unmeasured" + confirmationEvidence?: "unmeasured" | "complete" | "incomplete" + confirmationAttempts?: BenchmarkAttempt[] + proofConfirmation?: { + selectedHarness: { candidateId: string; slotId: string; generation: number; commit: string; attemptIds: string[] } + rootHarness: { commit: string; attemptIds: string[] } + } proofCampaign?: { manifest: import("./proof-campaign.js").ProofCampaignManifest manifestSha256: string From 2cef48fa657769946fe02823300c269e29a049e6 Mon Sep 17 00:00:00 2001 From: RSI smoke Date: Wed, 30 Sep 2026 17:23:39 -0600 Subject: [PATCH 2/6] Add controlled RSI proof demonstration workflow --- docs/recursive-self-improvement.md | 74 ++- .../mock-openrouter-guest-smoke.ts | 5 +- .../self-hosted-harness-smoke.ts | 5 +- scripts/rsi-proof-demonstration.ts | 429 ++++++++++++++++-- src/budget/__tests__/cost.test.ts | 5 +- src/budget/cost.ts | 24 +- src/llm/__tests__/openrouter.test.ts | 10 +- src/llm/openrouter.ts | 34 +- src/rsi/__tests__/config.test.ts | 8 + src/rsi/__tests__/postgres-queue.test.ts | 4 +- src/rsi/__tests__/proof-campaign.test.ts | 83 +++- .../proof-demonstration-launcher.test.ts | 123 +++++ .../proof-demonstration-manifest.test.ts | 12 +- .../proof-demonstration-plan.test.ts | 13 +- src/rsi/__tests__/proof-demonstration.test.ts | 188 ++++++++ .../worker-harness-controller.test.ts | 10 + .../__tests__/openshell-integration.test.ts | 4 +- src/rsi/config.ts | 7 + src/rsi/controller.ts | 109 ++++- src/rsi/proof-campaign.ts | 23 +- src/rsi/proof-demonstration-manifest.ts | 13 +- src/rsi/proof-demonstration-plan.ts | 33 +- src/rsi/proof-demonstration.ts | 155 ++++++- src/rsi/types.ts | 4 + 24 files changed, 1231 insertions(+), 144 deletions(-) create mode 100644 src/rsi/__tests__/proof-demonstration-launcher.test.ts create mode 100644 src/rsi/__tests__/proof-demonstration.test.ts diff --git a/docs/recursive-self-improvement.md b/docs/recursive-self-improvement.md index 9733c30..3d6f82f 100644 --- a/docs/recursive-self-improvement.md +++ b/docs/recursive-self-improvement.md @@ -246,7 +246,10 @@ OpenRouter is available for RSI fleet mutations and the development benchmark when the worker role is configured with `HEADLESSCODE_RSI_ROLE_WORKER_PROVIDER=openrouter` and `HEADLESSCODE_RSI_ROLE_WORKER_MODEL=deepseek/deepseek-v4-flash-0731`. The model -is pinned to DeepSeek's OpenRouter endpoint with fallback routing disabled. +is routed through OpenRouter's available DeepInfra fp8 endpoint +(`deepinfra/fp8`) with fallback routing disabled. The broker checks endpoint +metadata and requires the returned selected-provider metadata to identify +DeepInfra before qualifying usage. Include `REMOTE_API` in `HEADLESSCODE_RSI_WORKER_RESOURCE_CLASSES` for workers that claim brokered calls; Ollama workers continue to claim `LOCAL_GPU`. Workers need the host-side `HEADLESSCODE_OPENROUTER_API_KEY` and @@ -311,15 +314,66 @@ benchmark inference use distinct broker campaign IDs derived from the shared proof campaign ID, so their allocation caps remain separate while their reservations count against the one PostgreSQL proof-campaign root ceiling. -The current controller completes only the development phase. It records the -arm as `evaluated` and releases its lease without finishing the proof campaign. -A later confirmation phase can reacquire the same arm and use only the -predeclared cells. Every selected final candidate and each root-control -comparison must complete; each unused final candidate slot must be recorded -with the zero-spend `not-selected` disposition. Only then may `finishArm` -mark that arm terminal. Confirmation execution is not yet wired into the -controller; until that work is added, these campaigns remain development-only -and are not complete proof results. +The demonstration launcher separates engineering from holdout planning. The +engineering run freezes development-only cells and runs both arms for two +generations. It must record an accepted generation-zero self-hosted candidate +and an accepted generation-one child whose worker execution used that exact +parent commit. The fixed-control arm must also reach a terminal generation-one +decision. Only after these checks pass may the operator create the final plan; +that plan binds the engineering evidence and freezes the three confirmation +campaigns. Engineering-only manifests contain no confirmation task IDs or +confirmation schedule cells. + +Use the staged commands below from a clean checkout with the exact fixed and +self-hosted harness image identities configured. Estimate and plan fetch only +the pinned endpoint's read-only price metadata. Execution requires PostgreSQL, +OpenShell, the broker, the model key on the supervisor, finite per-role +allocations, and a separately reviewed operator cap. No production inference +has been run as part of this implementation. + +```sh +# Review the engineering-only schedule and live price before choosing a cap. +npm run rsi:proof-demonstration -- engineering --repo "$PWD" --seed "$ENGINEERING_SEED" + +# Run the two-generation engineering pair. The cap must cover this root's +# frozen reservation and fit the configured mutation + benchmark allocations. +npm run rsi:proof-demonstration -- engineering --repo "$PWD" --seed "$ENGINEERING_SEED" \ + --operator-cap-usd "$ENGINEERING_CAP_USD" --execute + +# Freeze three new campaign seeds only after the engineering evidence passes. +npm run rsi:proof-demonstration -- plan --repo "$PWD" \ + --engineering-evidence .headlesscode/rsi-proof/engineering-evidence.json \ + --operator-cap-usd "$CAMPAIGN_CAP_USD" --seed "$ENGINEERING_SEED" \ + --campaign-seed "$CAMPAIGN_SEED_A" --campaign-seed "$CAMPAIGN_SEED_B" \ + --campaign-seed "$CAMPAIGN_SEED_C" --out .headlesscode/rsi-proof/frozen-plan.json + +# Verify the frozen identities and cap, execute all three paired development +# and confirmation campaigns, then write JSON and Markdown analysis reports. +npm run rsi:proof-demonstration -- verify --repo "$PWD" --plan .headlesscode/rsi-proof/frozen-plan.json +npm run rsi:proof-demonstration -- execute --repo "$PWD" --plan .headlesscode/rsi-proof/frozen-plan.json \ + --operator-cap-usd "$CAMPAIGN_CAP_USD" --execute +npm run rsi:proof-demonstration -- analyze --repo "$PWD" --plan .headlesscode/rsi-proof/frozen-plan.json \ + --archive .headlesscode/rsi +``` + +The engineering command binds each campaign arm to a deterministic +campaign-and-arm run ID. A restart reuses a finished arm's verified archive +record or resumes the same active identity; an archived incomplete arm is +rejected. Final campaign execution likewise reuses terminal arm records and +will not dispatch a second attempt for an already completed cell. A crash +after broker dispatch but before a verified result artifact remains uncertain +and fails closed under the PostgreSQL ledger; it is not retried automatically. +Each final campaign's cap is checked against its frozen root reservation, and +the command's supplied cap must exactly match the value in the plan. The +engineering cap is separate from the final plan's cap for the three campaigns. + +Analysis verifies the three paired terminal campaign records, frozen +manifest/config/model/runtime identities, each confirmation schedule cell, +the selected-parent lineage, broker receipts, artifact digests, and PostgreSQL +root accounting. Missing, unqualified, or altered evidence yields +`inconclusive`; a completed set that misses the frozen improvement gates yields +`not-demonstrated`. A demonstration verdict is not a release or promotion +action. To exercise the worker image and guest network boundary without a production provider credential or paid request, build the exact current harness image diff --git a/scripts/rsi-benchmark/mock-openrouter-guest-smoke.ts b/scripts/rsi-benchmark/mock-openrouter-guest-smoke.ts index 47cbd49..ac67ab9 100644 --- a/scripts/rsi-benchmark/mock-openrouter-guest-smoke.ts +++ b/scripts/rsi-benchmark/mock-openrouter-guest-smoke.ts @@ -5,7 +5,7 @@ import * as os from "node:os" import * as path from "node:path" import { createServer } from "node:http" import { OpenShellBenchmarkAgent, initializeTaskRepository } from "../../src/rsi/benchmark/openshell-agent.js" -import { RSI_OPENROUTER_MODEL, startOpenRouterBroker } from "../../src/rsi/inference-broker.js" +import { RSI_OPENROUTER_ENDPOINT_PROVIDER, RSI_OPENROUTER_ENDPOINT_TAG, RSI_OPENROUTER_MODEL, startOpenRouterBroker } from "../../src/rsi/inference-broker.js" import { OpenShellSessionProvider } from "../../src/cloud/openshell-provider.js" const repoRoot = path.resolve(process.argv[2] ?? process.cwd()) @@ -60,7 +60,7 @@ try { upstreamFetch: async (input, init) => { const url = String(input) if (url.endsWith(`/models/${RSI_OPENROUTER_MODEL}/endpoints`)) { - return new Response(JSON.stringify({ data: { endpoints: [{ provider_slug: "deepseek", pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200, headers: { "content-type": "application/json" } }) + return new Response(JSON.stringify({ data: { endpoints: [{ provider_name: "DeepInfra", tag: RSI_OPENROUTER_ENDPOINT_TAG, quantization: "fp8", status: 0, supported_parameters: ["tools", "tool_choice", "temperature", "seed", "max_tokens"], supports_tool_choice: { auto: true, required: true, function: true }, pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200, headers: { "content-type": "application/json" } }) } assert.equal(url, "https://openrouter.ai/api/v1/chat/completions") assert.equal(new Headers(init?.headers).get("authorization"), `Bearer ${key}`) @@ -76,6 +76,7 @@ try { assert.ok(body.messages?.length) return new Response(JSON.stringify({ id: `mock-response-${upstreamCalls}`, model: RSI_OPENROUTER_MODEL, + openrouter_metadata: { endpoints: { available: [{ provider: RSI_OPENROUTER_ENDPOINT_PROVIDER, selected: true }] } }, choices: [{ message, finish_reason: "tool_calls" }], usage: { prompt_tokens: 400, completion_tokens: 24, cost: 0.000424 }, }), { status: 200, headers: { "content-type": "application/json" } }) diff --git a/scripts/rsi-benchmark/self-hosted-harness-smoke.ts b/scripts/rsi-benchmark/self-hosted-harness-smoke.ts index c297059..5628c52 100644 --- a/scripts/rsi-benchmark/self-hosted-harness-smoke.ts +++ b/scripts/rsi-benchmark/self-hosted-harness-smoke.ts @@ -4,7 +4,7 @@ import { createHash } from "node:crypto" import * as fs from "node:fs/promises" import * as os from "node:os" import * as path from "node:path" -import { FileOpenRouterInferenceLedger, RSI_OPENROUTER_MODEL, startOpenRouterBroker, type OpenRouterBrokerLimits } from "../../src/rsi/inference-broker.js" +import { FileOpenRouterInferenceLedger, RSI_OPENROUTER_ENDPOINT_PROVIDER, RSI_OPENROUTER_ENDPOINT_TAG, RSI_OPENROUTER_MODEL, startOpenRouterBroker, type OpenRouterBrokerLimits } from "../../src/rsi/inference-broker.js" import { createMutationSnapshotBundle, runOpenShellMutation } from "../../src/rsi/openshell.js" import { workerHarnessSentinelDiff } from "../../src/rsi/worker-harness-sentinel.js" import type { CandidateRecord, RsiConfig, WorkerHarnessMode, WorkerHarnessRuntimeIdentity } from "../../src/rsi/types.js" @@ -54,7 +54,7 @@ function mockUpstream(stage: "seed" | "bridge" | "sentinel", observed: { sentine return async (input: RequestInfo | URL, init?: RequestInit): Promise => { const url = String(input) if (url.endsWith(`/models/${RSI_OPENROUTER_MODEL}/endpoints`)) { - return new Response(JSON.stringify({ data: { endpoints: [{ provider_slug: "deepseek", pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200, headers: { "content-type": "application/json" } }) + return new Response(JSON.stringify({ data: { endpoints: [{ provider_name: "DeepInfra", tag: RSI_OPENROUTER_ENDPOINT_TAG, quantization: "fp8", status: 0, supported_parameters: ["tools", "tool_choice", "temperature", "seed", "max_tokens"], supports_tool_choice: { auto: true, required: true, function: true }, pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200, headers: { "content-type": "application/json" } }) } assert.equal(url, "https://openrouter.ai/api/v1/chat/completions") assert.equal(new Headers(init?.headers).get("authorization"), "Bearer mock-only-not-a-provider-key") @@ -81,6 +81,7 @@ function mockUpstream(stage: "seed" | "bridge" | "sentinel", observed: { sentine return new Response(JSON.stringify({ id: `mock-self-hosted-${stage}-${calls}`, model: RSI_OPENROUTER_MODEL, + openrouter_metadata: { endpoints: { available: [{ provider: RSI_OPENROUTER_ENDPOINT_PROVIDER, selected: true }] } }, choices: [{ message, finish_reason: "tool_calls" }], usage: { prompt_tokens: 10_000, completion_tokens: 200, cost: 0.0102 }, }), { status: 200, headers: { "content-type": "application/json" } }) diff --git a/scripts/rsi-proof-demonstration.ts b/scripts/rsi-proof-demonstration.ts index 18bf2a2..2bb781d 100644 --- a/scripts/rsi-proof-demonstration.ts +++ b/scripts/rsi-proof-demonstration.ts @@ -4,46 +4,76 @@ import * as path from "node:path" import { execFileSync } from "node:child_process" import { fileURLToPath, pathToFileURL } from "node:url" import { loadBenchmark } from "../src/rsi/benchmark/runner.js" +import { readArchive } from "../src/rsi/archive.js" import { parseRsiArgs } from "../src/rsi/config.js" +import { createProofCampaignManifest, RSI_CONFIRMATION_RETENTION_TASK_IDS, runRsi } from "../src/rsi/controller.js" import { fetchPinnedOpenRouterEndpointPrice, openRouterAllocatedEnvironment, openRouterBrokerLimitsFromEnvironment, RSI_OPENROUTER_MODEL } from "../src/rsi/inference-broker.js" -import { freezeProofDemonstrationManifest, frozenProofConfigDigest, verifyFrozenProofDemonstrationManifest } from "../src/rsi/proof-demonstration-manifest.js" -import { planProofDemonstration } from "../src/rsi/proof-demonstration-plan.js" +import { freezeProofDemonstrationManifest, frozenProofConfigDigest, verifyFrozenProofDemonstrationManifest, type FrozenProofDemonstrationManifest } from "../src/rsi/proof-demonstration-manifest.js" +import { assertFrozenProofDemonstrationPlanMatches, planProofDemonstration } from "../src/rsi/proof-demonstration-plan.js" +import { analyzeProofDemonstration, verifyProofDemonstrationArtifactBytes } from "../src/rsi/proof-demonstration.js" import { workerHarnessRuntimeIdentityFromEnvironment } from "../src/rsi/worker-harness.js" +import { proofCampaignManifestDigest } from "../src/rsi/proof-campaign.js" +import type { CandidateRecord, RsiConfig, RsiRunRecord } from "../src/rsi/types.js" interface Options { - command: "estimate" | "plan" | "verify" + command: "estimate" | "plan" | "verify" | "analyze" | "engineering" | "execute" output: string + planPath?: string + archiveDir?: string + outputExplicit?: boolean + execute?: boolean repo: string operatorCapUsd?: number seeds: string[] mutationTask: string + engineeringEvidencePath?: string } -type ProofPlanPreview = Omit[0], "operatorCapUsd"> & { plan: ReturnType } +type ProofPlanPreview = Omit[0], "operatorCapUsd" | "engineeringEvidence"> & { plan: ReturnType } +interface EngineeringEvidenceRecord { + schemaVersion: 1 + campaignId: string + rootCommit: string + manifestSha256: string + snapshotSha256: string + archiveDir: string + arms: Array<{ runId: string; armId: "fixed-control" | "self-hosted"; runSha256: string; selectedCommit: string }> + digestSha256: string +} function usage(): string { return `Usage: npx tsx scripts/rsi-proof-demonstration.ts estimate --seed --campaign-seed --campaign-seed --campaign-seed [--repo ] - npx tsx scripts/rsi-proof-demonstration.ts plan --operator-cap-usd --seed --campaign-seed --campaign-seed --campaign-seed [--out ] [--repo ] + npx tsx scripts/rsi-proof-demonstration.ts engineering --seed --operator-cap-usd [--execute] [--repo ] [--out ] + npx tsx scripts/rsi-proof-demonstration.ts plan --engineering-evidence --operator-cap-usd --seed --campaign-seed --campaign-seed --campaign-seed [--out ] [--repo ] + npx tsx scripts/rsi-proof-demonstration.ts execute --plan --operator-cap-usd --execute npx tsx scripts/rsi-proof-demonstration.ts verify --plan + npx tsx scripts/rsi-proof-demonstration.ts analyze --plan --archive [--out ] -Estimate and plan perform pricing metadata GETs only; they do not submit model inference. -Estimate prints reservations without assuming an operator cap. Plan requires a clean root -checkout, frozen runtime image identity, explicit broker ceilings, four distinct seeds, and -a supplied operator cap covering all four campaign reservations. +Estimate, engineering estimates, plan, and verify perform pricing metadata GETs only; they do not submit model inference. +Engineering prints a bounded estimate unless --execute is supplied. Paid execution requires an +explicit operator cap, broker/PostgreSQL/OpenShell configuration, and a clean checkout. Its +development-only manifest contains no confirmation cells. Final plan requires passing +engineering evidence and freezes the three confirmation campaigns afterward. Execute resumes +the frozen campaigns; analyze independently checks their archive and artifacts. ` } function parse(argv: string[]): Options { const command = argv[0] - if (command !== "estimate" && command !== "plan" && command !== "verify") throw new Error("expected estimate, plan or verify") - const options: Options = { command, output: path.resolve(".headlesscode/rsi-proof/plan.json"), repo: process.cwd(), seeds: [], mutationTask: "Improve headlesscode agent reliability on the frozen development benchmark while preserving existing behavior." } + if (command !== "estimate" && command !== "plan" && command !== "verify" && command !== "analyze" && command !== "engineering" && command !== "execute") throw new Error("expected estimate, engineering, plan, execute, verify or analyze") + const defaultOutput = command === "analyze" ? ".headlesscode/rsi-proof/report.json" : command === "engineering" ? ".headlesscode/rsi-proof/engineering-evidence.json" : ".headlesscode/rsi-proof/plan.json" + const options: Options = { command, output: path.resolve(defaultOutput), repo: process.cwd(), seeds: [], mutationTask: "Improve headlesscode agent reliability on the frozen development benchmark while preserving existing behavior." } for (let index = 1; index < argv.length; index++) { const arg = argv[index]! + if (arg === "--execute") { options.execute = true; continue } if (arg === "--help" || arg === "-h") throw new Error(usage()) const value = argv[index + 1] if (!value || value.startsWith("--")) throw new Error(`${arg} requires a value`) - if (arg === "--out" || arg === "--plan") options.output = path.resolve(value) + if (arg === "--out") { options.output = path.resolve(value); options.outputExplicit = true } + else if (arg === "--plan") options.planPath = path.resolve(value) + else if (arg === "--archive") options.archiveDir = path.resolve(value) + else if (arg === "--engineering-evidence") options.engineeringEvidencePath = path.resolve(value) else if (arg === "--repo") options.repo = path.resolve(value) else if (arg === "--operator-cap-usd") options.operatorCapUsd = Number(value) else if (arg === "--seed" || arg === "--campaign-seed") options.seeds.push(value) @@ -77,6 +107,77 @@ function sha256(value: string | Buffer): string { return createHash("sha256").up function canonicalHash(value: unknown): string { return sha256(JSON.stringify(value)) } function campaignId(seed: string): string { return `rsi-proof-${sha256(seed).slice(0, 20)}` } +function engineeringRunId(campaign: string, arm: "fixed-control" | "self-hosted"): string { + const match = /^rsi-proof-([a-f0-9]{20})$/.exec(campaign) + if (!match) throw new Error("engineering campaign ID is malformed") + return `rsi-proof-${match[1]}-${arm === "fixed-control" ? "fixed" : "self"}` +} + +export function assertEngineeringTransition(fixed: RsiRunRecord, self: RsiRunRecord): void { + const selfSnapshot = self.proofCampaign?.ledgerSnapshot + const selfEvents = (selfSnapshot?.events as Array> | undefined) ?? [] + const selfDecisions = (selfSnapshot?.decisions as Array> | undefined) ?? [] + const selectedEvent = selfEvents.find((event) => event.arm_id === "self-hosted" && event.event_type === "generation-selected" && Number((event.payload as Record).generation) === 0) + const slots = ((selectedEvent?.payload as Record | undefined)?.selectedCandidateIds as string[] | undefined) ?? [] + const generationZero = self.candidates.filter((candidate) => candidate.generation === 0) + const selectedZero = slots.map((slot) => generationZero[Number(/candidate-(\d+)$/.exec(slot)?.[1] ?? -1)]) + .filter((candidate): candidate is CandidateRecord => Boolean(candidate && candidate.status === "accepted" && candidate.developmentProof?.accepted === true && candidate.commits[0])) + if (!selectedZero.length) throw new Error("engineering evidence lacks a durably selected accepted generation-zero self-hosted parent") + const generationOne = self.candidates.filter((candidate) => candidate.generation === 1) + const transition = generationOne.some((child) => selectedZero.some((parent) => { + const slot = `g1-candidate-${generationOne.indexOf(child)}` + return child.status === "accepted" && child.developmentProof?.accepted === true && child.parent === parent.id && child.baseCommit === parent.commits[0] && child.parentCommit === parent.commits[0] && child.workerHarnessExecution?.parentCommit === parent.commits[0] && Boolean(child.commits[0]) && selfDecisions.some((entry) => entry.arm_id === "self-hosted" && Number(entry.generation) === 1 && entry.candidate_id === slot && entry.decision === "accepted" && entry.parent_commit === parent.commits[0]) + })) + if (!transition) throw new Error("engineering evidence lacks an accepted generation-one self-hosted mutation from a durable selected generation-zero parent") + const fixedSchedule = fixed.proofCampaign?.manifest.schedule.filter((cell) => cell.armId === "fixed-control" && cell.phase === "development" && cell.generation === 1) ?? [] + const fixedRows = (fixed.proofCampaign?.ledgerSnapshot?.attempts as Array> | undefined) ?? [] + const fixedDecisions = (fixed.proofCampaign?.ledgerSnapshot?.decisions as Array> | undefined) ?? [] + const fixedCandidates = [...new Set(fixedSchedule.map((cell) => cell.candidateId))] + if (!fixedSchedule.length || fixedSchedule.some((cell) => { + const rows = fixedRows.filter((row) => row.arm_id === "fixed-control" && Number(row.generation) === 1 && row.candidate_id === cell.candidateId && row.task_id === cell.taskId && row.seed === cell.seed && row.attempt_kind === cell.attemptKind) + return rows.length !== 1 || !["completed", "skipped-after-hard-gate"].includes(String(rows[0]?.status)) + }) || fixedCandidates.some((candidateId) => { + const rows = fixedDecisions.filter((row) => row.arm_id === "fixed-control" && Number(row.generation) === 1 && row.candidate_id === candidateId) + return rows.length !== 1 || !["accepted", "rejected", "failed"].includes(String(rows[0]?.decision)) + })) throw new Error("engineering evidence lacks terminal fixed-control generation-one cells and candidate decisions") +} + +async function verifyEngineeringEvidenceFile(evidencePath: string, repo: string, expectedRoot?: string): Promise<{ record: EngineeringEvidenceRecord; fileSha256: string }> { + const safePath = await assertSupervisorOnlyPlanPath(repo, evidencePath) + const bytes = await fs.readFile(safePath) + const record = JSON.parse(bytes.toString("utf8")) as EngineeringEvidenceRecord + if (!record || record.schemaVersion !== 1 || !Array.isArray(record.arms) || record.arms.length !== 2) throw new Error("engineering evidence record is malformed") + const { digestSha256, ...body } = record + if (!/^[a-f0-9]{64}$/.test(digestSha256) || canonicalHash(body) !== digestSha256) throw new Error("engineering evidence digest mismatch") + if (expectedRoot && record.rootCommit !== expectedRoot) throw new Error("engineering evidence belongs to a different clean root commit") + const archive = await readArchive(record.archiveDir) + const snapshot: Record | undefined = archive.runs.concat(archive.activeRuns).map((run) => run.proofCampaign?.ledgerSnapshot).find((entry) => { + const campaign = (entry as Record | undefined)?.campaign as Record | undefined + return campaign?.campaign_id === record.campaignId && campaign.manifest_sha256?.toString().trim() === record.manifestSha256 && canonicalHash(entry) === record.snapshotSha256 + }) + if (!snapshot) throw new Error("engineering evidence durable PostgreSQL snapshot is missing or changed") + const campaign = snapshot.campaign as Record + const arms = snapshot.arms as Array> + const events = snapshot.events as Array> + if (campaign.state !== "finished" || String(campaign.manifest_sha256).trim() !== record.manifestSha256 || arms.length !== 2 || arms.some((arm) => arm.state !== "finished") || events.filter((event) => event.event_type === "engineering-development-finished").length !== 2) throw new Error("engineering evidence does not have two finished development-only ledger arms") + const uniqueArms = new Set() + for (const armRef of record.arms) { + if (!/^[a-f0-9]{64}$/.test(armRef.runSha256) || !/^[a-f0-9]{40}$/.test(armRef.selectedCommit) || uniqueArms.has(armRef.armId)) throw new Error("engineering evidence arm reference is malformed or duplicated") + uniqueArms.add(armRef.armId) + const run = archive.runs.concat(archive.activeRuns).find((entry) => entry.runId === armRef.runId) + if (!run || canonicalHash(run) !== armRef.runSha256 || run.baseCommit !== record.rootCommit || run.finishedAt === undefined || run.workerHarnessPairId !== record.campaignId || run.workerHarnessMode !== armRef.armId || !run.proofCampaign || run.proofCampaign.manifest.phaseMode !== "engineering-only" || run.proofCampaign.manifestSha256 !== record.manifestSha256 || proofCampaignManifestDigest(run.proofCampaign.manifest) !== record.manifestSha256 || run.proofCampaign.state !== "finished" || run.proofConfirmation || (run.confirmationAttempts?.length ?? 0) > 0 || run.proofCampaign.manifest.schedule.some((cell) => cell.phase === "confirmation") || (run.proofCampaign.manifest.settings.confirmationTaskIdentities as unknown[] | undefined)?.length) throw new Error(`engineering evidence run ${armRef.runId} is incomplete or includes confirmation inputs`) + if (run.candidates.some((candidate) => !["accepted", "rejected", "failed"].includes(candidate.status))) throw new Error(`engineering evidence run ${armRef.runId} has incomplete candidate outcomes`) + if (armRef.armId === "self-hosted" && (!run.selected || !run.candidates.some((candidate) => candidate.id === run.selected && candidate.status === "accepted" && candidate.developmentProof?.accepted === true && candidate.commits.length > 0 && candidate.commits[0] === armRef.selectedCommit))) throw new Error("engineering self-hosted arm did not select an accepted development-proof candidate") + if (armRef.armId === "fixed-control" && armRef.selectedCommit !== run.baseCommit && !run.candidates.some((candidate) => candidate.id === run.selected && candidate.commits[0] === armRef.selectedCommit)) throw new Error("engineering fixed-control selected commit is not in its recorded run") + } + if (!uniqueArms.has("fixed-control") || !uniqueArms.has("self-hosted")) throw new Error("engineering evidence must contain one run for each paired arm") + const [fixed, self] = record.arms + const fixedRun = archive.runs.concat(archive.activeRuns).find((run) => run.runId === fixed!.runId)! + const selfRun = archive.runs.concat(archive.activeRuns).find((run) => run.runId === self!.runId)! + if (fixedRun.workerHarnessConfigDigest !== selfRun.workerHarnessConfigDigest || fixedRun.workerHarnessSamplingSeed !== selfRun.workerHarnessSamplingSeed || fixedRun.workerHarnessRuntime?.harnessCommit !== selfRun.workerHarnessRuntime?.harnessCommit) throw new Error("engineering pair has mismatched frozen configuration, seed, or runtime") + assertEngineeringTransition(fixedRun, selfRun) + return { record, fileSha256: sha256(bytes) } +} async function analysisDigest(repo: string): Promise { const sources = [ @@ -109,28 +210,65 @@ function formatPlan(preview: ProofPlanPreview, operatorCapUsd?: number, digestSh `Model: ${preview.model.requestedId} (endpoint: ${preview.endpointPrice.provider})`, `Endpoint rate observed ${preview.endpointPrice.observedAt}: input $${preview.endpointPrice.inputPricePerMillionUsd}/M, output $${preview.endpointPrice.outputPricePerMillionUsd}/M`, `Configured maximum price ceilings: input $${preview.priceCeilingsPerMillionUsd.input}/M, output $${preview.priceCeilingsPerMillionUsd.output}/M`, - `Worst-case price-ceiling-derived reserve: $${plan.campaign.rootPriceDerivedSpendCeilingUsd.toFixed(6)} per campaign root; all four roots $${plan.totalRootPriceDerivedSpendCeilingUsd.toFixed(6)}`, - `Broker reservations: ${plan.totalCalls} calls, ${plan.totalTokens} tokens, $${plan.totalRootSpendCeilingUsd.toFixed(6)} root spend; operator cap ${operatorCapUsd === undefined ? "not supplied" : `$${operatorCapUsd.toFixed(6)}`}`, - `Schedule: engineering ${plan.engineering.developmentProofCells} development cells; each campaign ${plan.campaign.developmentProofCells} development + ${plan.campaign.confirmationScheduleCells} confirmation cells (${plan.campaign.confirmationNotSelectedCells} not-selected, ${plan.campaign.paidConfirmationCells} final, ${plan.campaign.rootControlCells} root-control)`, + `Worst-case price-ceiling-derived reserve: engineering root $${plan.engineering.rootPriceDerivedSpendCeilingUsd.toFixed(6)}; three future campaign roots $${plan.campaignRootsSpendCeilingUsd.toFixed(6)} total ($${plan.campaign.rootPriceDerivedSpendCeilingUsd.toFixed(6)} each)`, + `Broker reservations across engineering + campaigns: ${plan.totalCalls} calls, ${plan.totalTokens} tokens, $${plan.totalReservedSpendUsd.toFixed(6)} reserved; final-plan operator cap applies to the three future campaigns: ${operatorCapUsd === undefined ? "not supplied" : `$${operatorCapUsd.toFixed(6)}`}`, + `Schedule: engineering ${plan.engineering.developmentProofCells} development cells; confirmation is not frozen until it passes. Each final campaign has ${plan.campaign.developmentProofCells} development + ${plan.campaign.confirmationScheduleCells} confirmation cells (${plan.campaign.confirmationNotSelectedCells} not-selected, ${plan.campaign.paidConfirmationCells} selected-final, ${plan.campaign.rootControlCells} root-control)`, `Worst-case serial task wall bound: ${(plan.totalTaskWallMs / 3_600_000).toFixed(1)} hours`, `Seeds: ${[preview.seeds.engineering, ...preview.seeds.campaigns].join(", ")}`, `Configured ceilings bound the estimate. No inference request was made.`, ].join("\n") } +export function formatAnalysisReport(report: ReturnType, planDigest: string): string { + const lines = [ + "# RSI proof demonstration analysis", "", `Frozen plan: ${planDigest}`, `Verdict: **${report.verdict}**`, "", + "## Paired campaign outcomes", "", + "Campaign | Self-hosted lineage (generation: parent → commit) | Final commit | Selected lineage depth | Net confirmation wins vs root | Retention losses | Self-hosted calls / tokens / wall ms / spend USD | Fixed-control final commit | Fixed-control calls / tokens / wall ms / spend USD", + "---|---|---|---:|---:|---|---|---|---", + ] + const resource = (value: Record) => `${value.calls ?? "unmeasured"} / ${value.tokens ?? "unmeasured"} / ${value.wallTimeMs ?? "unmeasured"} / ${value.spendUsd ?? "unmeasured"}` + for (const campaign of report.campaigns) { + const lineage = campaign.selfHosted.lineage.map((edge) => `g${edge.generation}: ${edge.parentId}@${edge.parentCommit} → ${edge.candidateId}@${edge.commit}`).join("; ") || "root (no accepted mutation)" + lines.push(`${campaign.campaignId} | ${lineage} | ${campaign.selfHosted.finalCommit} | ${campaign.selfHosted.adjacentAcceptedDepth} | ${campaign.selfHosted.netWinsVsRoot} | ${campaign.selfHosted.retentionLosses.join(", ") || "none"} | ${resource(campaign.selfHosted.resources)} | ${campaign.fixedControl.finalCommit} | ${resource(campaign.fixedControl.resources)}`) + } + lines.push("", "## Development denominators", "") + for (const campaign of report.campaigns) { + for (const [armName, arm] of [["self-hosted", campaign.selfHosted], ["fixed-control", campaign.fixedControl]] as const) { + const data = arm.development as { scheduledCandidateSlots?: number; candidateCounts?: Record; scheduledDevelopmentProofCells?: number; completedDevelopmentProofCells?: number; candidates?: Array<{ id: string; generation: number; status: string; proofAccepted: boolean; reason: string }>; otherDevelopmentCells?: Array<{ generation: number; candidateId: string; taskId: string; attemptKind: string; status: unknown }> } + lines.push(`- ${campaign.campaignId}/${armName}: ${data.candidateCounts?.accepted ?? 0} accepted, ${data.candidateCounts?.rejected ?? 0} rejected, ${data.candidateCounts?.failed ?? 0} failed, ${data.candidateCounts?.incomplete ?? 0} incomplete of ${data.scheduledCandidateSlots ?? 0} scheduled candidate slots; ${data.completedDevelopmentProofCells ?? 0}/${data.scheduledDevelopmentProofCells ?? 0} development-proof cells completed.`) + for (const candidate of data.candidates ?? []) if (candidate.status !== "accepted") lines.push(` - ${candidate.id} (generation ${candidate.generation}, ${candidate.status}): ${candidate.reason}`) + for (const cell of data.otherDevelopmentCells ?? []) lines.push(` - ${cell.candidateId}/${cell.taskId}/${cell.attemptKind}: ${String(cell.status)}`) + } + } + lines.push("", "## Confirmation denominators", "") + for (const campaign of report.campaigns) { + for (const [armName, arm] of [["self-hosted", campaign.selfHosted], ["fixed-control", campaign.fixedControl]] as const) { + const valid = (attempts: Array>) => attempts.filter((attempt) => attempt.validity === "valid" && attempt.proofQualified === true).length + lines.push(`- ${campaign.campaignId}/${armName}: selected-final ${valid(arm.confirmation)}/${arm.confirmation.length} valid; root ${valid(arm.root)}/${arm.root.length} valid; complete=${arm.complete}.`) + } + } + lines.push("", "## Aggregate results", "", `Task clusters: ${report.pooled.taskIds.length}; campaigns: ${report.pooled.campaignCount}; self-hosted successful cells: ${report.pooled.selfHostedSuccesses}; fixed-control successful cells: ${report.pooled.fixedControlSuccesses}.`, `Mean paired difference: ${report.pooled.difference}; task-cluster 90% interval: ${report.pooled.cluster90PercentInterval?.join(" to ") ?? "unavailable"}.`, `Bootstrap: ${report.pooled.bootstrapReplicates} replicates; frozen seed ${report.pooled.bootstrapSeed}.`, "", "## Denominators and exclusions", "", `Confirmation task denominator: ${report.denominators.pairedConfirmationTasks}; missing cells: ${report.denominators.missingCells.length}; excluded attempts: ${report.denominators.excludedAttempts.length}.`) + if (report.denominators.missingCells.length) lines.push("", "Missing cells:", ...report.denominators.missingCells.map((cell) => `- ${cell}`)) + if (report.denominators.excludedAttempts.length) lines.push("", "Excluded attempts:", ...report.denominators.excludedAttempts.map((attempt) => `- ${attempt}`)) + lines.push("", "## Verdict gates", "", ...report.verdictReasons.map((reason) => `- ${reason}`), "", "## Limitations", "", ...report.limitations.map((limitation) => `- ${limitation}`), "") + return lines.join("\n") +} + export async function buildProofPlanPreview(options: Options, env: NodeJS.ProcessEnv = process.env, fetcher: typeof fetch = fetch): Promise { - if (options.command !== "estimate" && options.command !== "plan") throw new Error("plan options are required") - if (options.seeds.length !== 4) throw new Error("provide one --seed for engineering and exactly three --campaign-seed values") - if (new Set(options.seeds).size !== 4) throw new Error("engineering and campaign seeds must be distinct") + if (options.command !== "estimate" && options.command !== "plan" && options.command !== "engineering") throw new Error("plan options are required") + if (options.command === "engineering" && options.seeds.length !== 1) throw new Error("engineering requires exactly one explicit --seed") + if (options.command !== "engineering" && options.seeds.length !== 4) throw new Error("provide one --seed for engineering and exactly three --campaign-seed values") + const seeds = options.command === "engineering" ? [options.seeds[0]!, "preview-a-" + sha256(options.seeds[0]!).slice(0, 12), "preview-b-" + sha256(options.seeds[0]!).slice(12, 24), "preview-c-" + sha256(options.seeds[0]!).slice(24, 36)] : options.seeds + if (new Set(seeds).size !== 4) throw new Error("engineering and campaign seeds must be distinct") const repo = path.resolve(options.repo) const rootCommit = await ensureCleanRoot(repo) const runtime = workerHarnessRuntimeIdentityFromEnvironment(env) try { execFileSync("git", ["cat-file", "-e", `${runtime.harnessCommit}^{commit}`], { cwd: repo, stdio: "ignore" }) } catch { throw new Error("fixed-control runtime harness commit must be present in the root repository") } - const parsed = parseRsiArgs(["--repo", repo, "--population", "2", "--generations", "3", "--seed", options.seeds[0]!, "--base-ref", rootCommit, "--max-runtime-ms", String(72 * 60 * 60_000), "--mutation-task", options.mutationTask]) + const parsed = parseRsiArgs(["--repo", repo, "--population", "2", "--generations", "3", "--seed", seeds[0]!, "--base-ref", rootCommit, "--max-runtime-ms", String(72 * 60 * 60_000), "--mutation-task", options.mutationTask]) if (!parsed.config) throw new Error(parsed.error ?? "could not construct proof configuration") - const baseConfig = { ...parsed.config, model: RSI_OPENROUTER_MODEL, roles: { worker: { provider: "openrouter" as const, model: RSI_OPENROUTER_MODEL } }, mutationTask: options.mutationTask } - const configs = options.seeds.map((seed) => ({ ...baseConfig, seed })) as [typeof baseConfig, typeof baseConfig, typeof baseConfig, typeof baseConfig] + const baseConfig = { ...parsed.config, archiveDir: options.archiveDir ?? parsed.config.archiveDir, model: RSI_OPENROUTER_MODEL, roles: { worker: { provider: "openrouter" as const, model: RSI_OPENROUTER_MODEL } }, mutationTask: options.mutationTask, workerHarnessRuntime: runtime } + const configs = seeds.map((seed, index) => ({ ...baseConfig, generations: index === 0 ? 2 : 3, seed })) as [typeof baseConfig, typeof baseConfig, typeof baseConfig, typeof baseConfig] const config = configs[0] const loaded = await loadBenchmark(path.join(repo, "fixtures/rsi-benchmark"), repo) const modelKey = env.HEADLESSCODE_OPENROUTER_API_KEY?.trim() ?? "" @@ -152,18 +290,123 @@ export async function buildProofPlanPreview(options: Options, env: NodeJS.Proces inferenceLimits: { mutation: mutationLimits, benchmark: benchmarkLimits }, taskIdentities: loaded.manifest.tasks.map((task) => ({ id: task.id, split: task.split, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, knownGoodDigest: task.knownGoodDigest, knownBadDigest: task.knownBadDigest, ceilings: task.ceilings })), }, - seeds: { engineering: options.seeds[0]!, campaigns: [options.seeds[1]!, options.seeds[2]!, options.seeds[3]!] }, - campaignIds: { engineering: campaignId(options.seeds[0]!), campaigns: [campaignId(options.seeds[1]!), campaignId(options.seeds[2]!), campaignId(options.seeds[3]!)] }, + seeds: { engineering: seeds[0]!, campaigns: [seeds[1]!, seeds[2]!, seeds[3]!] }, + campaignIds: { engineering: campaignId(seeds[0]!), campaigns: [campaignId(seeds[1]!), campaignId(seeds[2]!), campaignId(seeds[3]!)] }, + analysis: { bootstrapSeed: "rsi-proof-task-cluster-bootstrap-v1", bootstrapReplicates: 10_000, confirmationRetentionTaskIds: [...RSI_CONFIRMATION_RETENTION_TASK_IDS] }, plan, } } +export async function runEngineeringArm(config: RsiConfig, campaign: string, arm: "fixed-control" | "self-hosted", rootCommit: string, seed: string, runtime: NonNullable, runner: typeof runRsi = runRsi): Promise { + const before = await readArchive(config.archiveDir) + const prior = before.runs.concat(before.activeRuns).filter((entry) => entry.workerHarnessPairId === campaign && entry.workerHarnessMode === arm) + if (prior.length > 1) throw new Error(`engineering replay found duplicate archived ${arm} runs`) + const existing = prior[0] + const plannedRunId = engineeringRunId(campaign, arm) + if (existing) { + if (existing.runId !== plannedRunId || existing.baseCommit !== rootCommit || existing.workerHarnessSamplingSeed !== seed || JSON.stringify(existing.workerHarnessRuntime) !== JSON.stringify(runtime) || existing.proofCampaign?.manifest.phaseMode !== "engineering-only") throw new Error(`engineering replay found a ${arm} run with changed frozen identity`) + if (existing.finishedAt && existing.proofCampaign.state === "finished") { + const arms = existing.proofCampaign.ledgerSnapshot?.arms as Array> | undefined + if (!arms?.some((entry) => entry.arm_id === arm && entry.state === "finished") || existing.proofCampaign.manifest.schedule.some((cell) => cell.phase === "confirmation")) throw new Error(`engineering replay found an invalid terminal ${arm} checkpoint`) + return existing + } + if (before.runs.some((entry) => entry.runId === existing.runId)) throw new Error(`engineering ${arm} run is archived without a terminal campaign finish; refusing to create a second identity`) + } + return runner({ ...config, dryRun: false, workerHarnessRunId: plannedRunId, ...(existing ? { resumeRunId: existing.runId } : {}), workerHarnessMode: arm, workerHarnessPhase: "development", workerHarnessPairId: campaign, workerHarnessEngineeringOnly: true, workerHarnessRuntime: runtime }) +} + +export async function executeEngineeringPair(options: Options, env: NodeJS.ProcessEnv = process.env, runner: typeof runRsi = runRsi): Promise { + if (options.command !== "engineering") throw new Error("engineering command is required") + const preview = await buildProofPlanPreview(options, env) + const required = preview.plan.engineering.rootSpendCeilingUsd + if (!options.execute) { + if (options.operatorCapUsd !== undefined && parseUsd(options.operatorCapUsd) + 1e-9 < required) throw new Error(`operator cap $${options.operatorCapUsd.toFixed(6)} is below engineering-only root reservation $${required.toFixed(6)}`) + process.stdout.write(`${formatPlan(preview, options.operatorCapUsd)}\nEngineering evidence output: ${options.output}\nNo inference request or campaign write was made. Use --execute only after provisioning PostgreSQL/OpenShell and reviewing the cap.\n`) + return undefined + } + if (env !== process.env) throw new Error("engineering execution must use the validated process environment; injected environments are allowed only for read-only estimates") + const cap = parseUsd(options.operatorCapUsd) + if (cap + 1e-9 < required) throw new Error(`operator cap $${cap.toFixed(6)} is below engineering-only root reservation $${required.toFixed(6)}`) + const mutationAllocation = preview.execution.inferenceLimits.mutation.maxCampaignSpendUsd + const benchmarkAllocation = preview.execution.inferenceLimits.benchmark.maxCampaignSpendUsd + if (mutationAllocation + benchmarkAllocation > cap + 1e-9 || mutationAllocation + benchmarkAllocation + 1e-9 < required) throw new Error("configured mutation and benchmark allocations must fit the operator cap and fund the engineering-only root reservation") + if (await ensureCleanRoot(options.repo) !== preview.rootCommit) throw new Error("engineering execution root changed after its estimate") + const campaign = campaignId(preview.seeds.engineering) + const config = preview.execution.configs[0] + const runs: RsiRunRecord[] = [] + for (const arm of ["fixed-control", "self-hosted"] as const) { + runs.push(await runEngineeringArm(config, campaign, arm, preview.rootCommit, preview.seeds.engineering, preview.runtime, runner)) + } + const archive = await readArchive(config.archiveDir) + const savedRuns = runs.map((run) => archive.runs.concat(archive.activeRuns).filter((entry) => entry.runId === run.runId)) + if (savedRuns.some((entries) => entries.length !== 1)) throw new Error("engineering runner did not persist exactly one archive record per paired arm") + const records = savedRuns.map((entries) => entries[0]!) + const fixed = records.find((run) => run.workerHarnessMode === "fixed-control") + const self = records.find((run) => run.workerHarnessMode === "self-hosted") + if (!fixed || !self) throw new Error("engineering runner did not persist both fixed-control and self-hosted arms") + for (const run of [fixed, self]) { + const manifest = run.proofCampaign?.manifest + if (!run.finishedAt || run.proofCampaign?.state !== "finished" || manifest?.phaseMode !== "engineering-only" || manifest.schedule.some((cell) => cell.phase === "confirmation") || (manifest.settings.confirmationTaskIdentities as unknown[] | undefined)?.length || run.proofConfirmation || (run.confirmationAttempts?.length ?? 0) > 0 || run.candidates.some((candidate) => !["accepted", "rejected", "failed"].includes(candidate.status))) throw new Error(`engineering arm ${run.runId} did not finish without confirmation work`) + } + const selectedSelf = self.candidates.find((candidate) => candidate.id === self.selected) + if (!selectedSelf || selectedSelf.status !== "accepted" || selectedSelf.developmentProof?.accepted !== true || !selectedSelf.commits[0]) throw new Error("engineering gate failed: self-hosted arm has no accepted development-proof candidate") + const selfSnapshot = self.proofCampaign!.ledgerSnapshot! + const selfEvents = selfSnapshot.events as Array> + const selfDecisions = selfSnapshot.decisions as Array> + const generationZeroSelection = selfEvents.find((event) => event.arm_id === "self-hosted" && event.event_type === "generation-selected" && Number((event.payload as Record).generation) === 0) + const selectedZeroSlots = ((generationZeroSelection?.payload as Record | undefined)?.selectedCandidateIds as string[] | undefined) ?? [] + const generationZero = self.candidates.filter((candidate) => candidate.generation === 0) + const selectedZero = selectedZeroSlots.map((slot) => { + const selectedIndex = Number(/candidate-(\d+)$/.exec(slot)?.[1] ?? -1) + return generationZero[selectedIndex] + }).filter((candidate): candidate is CandidateRecord => Boolean(candidate && candidate.status === "accepted" && candidate.developmentProof?.accepted === true && candidate.commits[0])) + if (!selectedZero.length) throw new Error("engineering gate failed: generation 0 has no durably selected accepted self-hosted parent") + const generationOneChildren = self.candidates.filter((candidate) => candidate.generation === 1 && selectedZero.some((parent) => candidate.status === "accepted" && candidate.developmentProof?.accepted === true && candidate.parent === parent.id && candidate.baseCommit === parent.commits[0] && candidate.parentCommit === parent.commits[0] && candidate.workerHarnessExecution?.parentCommit === parent.commits[0] && candidate.commits.length > 0 && selfDecisions.some((entry) => entry.arm_id === "self-hosted" && Number(entry.generation) === 1 && entry.candidate_id === `g1-candidate-${self.candidates.filter((item) => item.generation === 1).indexOf(candidate)}` && entry.decision === "accepted" && entry.parent_commit === parent.commits[0]))) + if (generationOneChildren.length === 0) throw new Error("engineering gate failed: no accepted generation-one self-hosted mutation ran from a durably selected generation-zero harness") + const fixedSnapshot = fixed.proofCampaign?.ledgerSnapshot + const fixedSchedule = fixed.proofCampaign!.manifest.schedule.filter((cell) => cell.armId === "fixed-control" && cell.phase === "development" && cell.generation === 1) + const fixedRows = (fixedSnapshot?.attempts as Array> | undefined) ?? [] + const fixedDecisions = (fixedSnapshot?.decisions as Array> | undefined) ?? [] + const fixedCandidates = [...new Set(fixedSchedule.map((cell) => cell.candidateId))] + if (!fixedSchedule.length || fixedSchedule.some((cell) => { + const rows = fixedRows.filter((row) => row.arm_id === "fixed-control" && Number(row.generation) === 1 && row.candidate_id === cell.candidateId && row.task_id === cell.taskId && row.seed === cell.seed && row.attempt_kind === cell.attemptKind) + return rows.length !== 1 || !["completed", "skipped-after-hard-gate"].includes(String(rows[0]?.status)) + }) || fixedCandidates.some((candidateId) => { + const rows = fixedDecisions.filter((row) => row.arm_id === "fixed-control" && Number(row.generation) === 1 && row.candidate_id === candidateId) + return rows.length !== 1 || !["accepted", "rejected", "failed"].includes(String(rows[0]?.decision)) + })) throw new Error("engineering gate failed: fixed-control generation-one cells and candidate decisions are not durably terminal") + const snapshot = self.proofCampaign?.ledgerSnapshot + if (!snapshot || !fixed.proofCampaign?.ledgerSnapshot || (fixed.proofCampaign.ledgerSnapshot.events as Array>).filter((event) => event.event_type === "engineering-development-finished" && event.arm_id === "fixed-control").length !== 1) throw new Error("engineering pair lacks both durable arm finish records") + const campaignRow = snapshot.campaign as Record + if (campaignRow.state !== "finished" || String(campaignRow.manifest_sha256).trim() !== self.proofCampaign?.manifestSha256) throw new Error("engineering campaign ledger is not finished or has a mismatched manifest") + const body = { + schemaVersion: 1 as const, campaignId: campaign, rootCommit: preview.rootCommit, + manifestSha256: self.proofCampaign.manifestSha256, snapshotSha256: canonicalHash(snapshot), archiveDir: config.archiveDir, + arms: ([fixed, self] as const).map((run) => ({ runId: run.runId, armId: run.workerHarnessMode!, runSha256: canonicalHash(run), selectedCommit: run.workerHarnessMode === "self-hosted" ? selectedSelf.commits[0]! : run.candidates.find((candidate) => candidate.id === run.selected)?.commits[0] ?? run.baseCommit })), + } + const evidence: EngineeringEvidenceRecord = { ...body, digestSha256: canonicalHash(body) } + const outputPath = await assertSupervisorOnlyPlanPath(options.repo, options.output) + await fs.mkdir(path.dirname(outputPath), { recursive: true, mode: 0o700 }) + try { + const existing = JSON.parse(await fs.readFile(outputPath, "utf8")) as EngineeringEvidenceRecord + const { digestSha256, ...existingBody } = existing + if (digestSha256 !== evidence.digestSha256 || canonicalHash(existingBody) !== evidence.digestSha256) throw new Error("existing engineering evidence differs from the verified completed pair") + } catch (error) { + if ((error as NodeJS.ErrnoException).code !== "ENOENT") throw error + await fs.writeFile(outputPath, `${JSON.stringify(evidence, null, 2)}\n`, { mode: 0o600, flag: "wx" }) + } + return evidence +} + export async function buildFrozenProofPlan(options: Options, env: NodeJS.ProcessEnv = process.env, fetcher: typeof fetch = fetch): Promise> { if (options.command !== "plan") throw new Error("plan options are required") const operatorCapUsd = parseUsd(options.operatorCapUsd) + if (!options.engineeringEvidencePath) throw new Error("final campaign plan requires --engineering-evidence from a completed development-only engineering pair") const preview = await buildProofPlanPreview(options, env, fetcher) - if (operatorCapUsd < preview.plan.totalRootSpendCeilingUsd) throw new Error(`operator cap $${operatorCapUsd.toFixed(6)} is below required root reservations $${preview.plan.totalRootSpendCeilingUsd.toFixed(6)}`) - return freezeProofDemonstrationManifest({ ...preview, operatorCapUsd }) + if (operatorCapUsd < preview.plan.campaignRootsSpendCeilingUsd) throw new Error(`operator cap $${operatorCapUsd.toFixed(6)} is below required three-campaign root reservations $${preview.plan.campaignRootsSpendCeilingUsd.toFixed(6)}`) + const evidence = await verifyEngineeringEvidenceFile(options.engineeringEvidencePath, options.repo, preview.rootCommit) + if (evidence.record.campaignId !== preview.campaignIds.engineering) throw new Error("engineering evidence seed/campaign identity differs from the final plan") + return freezeProofDemonstrationManifest({ ...preview, engineeringEvidence: { path: options.engineeringEvidencePath, sha256: evidence.fileSha256, campaignId: evidence.record.campaignId, rootCommit: evidence.record.rootCommit, manifestSha256: evidence.record.manifestSha256, snapshotSha256: evidence.record.snapshotSha256 }, operatorCapUsd }) } export async function verifyLiveFrozenProofPlan(manifest: ReturnType, env: NodeJS.ProcessEnv = process.env, fetcher: typeof fetch = fetch): Promise { @@ -171,6 +414,8 @@ export async function verifyLiveFrozenProofPlan(manifest: ReturnType manifest.priceCeilingsPerMillionUsd.input || price.outputPricePerMillionUsd > manifest.priceCeilingsPerMillionUsd.output) throw new Error("live pinned endpoint price exceeds or differs from frozen plan ceilings") } +export async function runFrozenCampaignArm( + manifest: ReturnType, + campaignIndex: number, + arm: "fixed-control" | "self-hosted", + phase: "development" | "confirmation", + runner: typeof runRsi = runRsi, +): Promise { + const campaign = manifest.campaignIds.campaigns[campaignIndex] + const config = manifest.execution.configs[campaignIndex + 1] + if (!campaign || !config) throw new Error("frozen campaign index is outside the three-campaign plan") + const runId = engineeringRunId(campaign, arm) + const archive = await readArchive(config.archiveDir) + const matches = archive.runs.concat(archive.activeRuns).filter((entry) => entry.workerHarnessPairId === campaign && entry.workerHarnessMode === arm) + if (matches.length > 1) throw new Error(`campaign ${campaign} has duplicate ${arm} archive identities`) + const existing = matches[0] + if (existing && (existing.runId !== runId || existing.baseCommit !== manifest.rootCommit || existing.workerHarnessSamplingSeed !== config.seed || existing.workerHarnessConfigDigest === undefined || existing.workerHarnessRuntime?.harnessCommit !== manifest.runtime.harnessCommit)) throw new Error(`campaign ${campaign}/${arm} differs from its frozen run identity`) + if (existing && (existing.workerHarnessPlannedGenerations !== config.generations || existing.model !== config.model || JSON.stringify(existing.workerHarnessRuntime) !== JSON.stringify(manifest.runtime))) throw new Error(`campaign ${campaign}/${arm} runtime, model, or generation count differs from the frozen plan`) + if (existing) { + const frozen = existing.proofCampaign?.manifest + const expectedRuntime = { imageDigest: manifest.runtime.imageDigest, harnessCommit: manifest.runtime.harnessCommit, dependencyLockDigest: manifest.runtime.dependencyLockDigest } + const internal = new Set(["workerHarnessMode", "workerHarnessPhase", "workerHarnessEngineeringOnly", "workerHarnessPairId", "workerHarnessRunId", "resumeRunId", "candidateIdNamespace", "workerHarnessParentCommit", "workerHarnessSnapshotTreeDigest"]) + const expectedConfig = Object.fromEntries(Object.entries(config).filter(([key]) => !internal.has(key))) + if (!frozen || frozen.phaseMode !== "paired-confirmation" || frozen.campaignId !== campaign || frozen.seed !== config.seed || frozen.rootCommit !== manifest.rootCommit || frozen.benchmarkDigests.manifest !== manifest.benchmarkManifestSha256 || existing.proofCampaign?.manifestSha256 !== proofCampaignManifestDigest(frozen) || JSON.stringify(frozen.runtime) !== JSON.stringify(expectedRuntime) || canonicalHash(frozen.settings.effectiveConfig) !== canonicalHash(expectedConfig) || frozen.model.id !== manifest.model.requestedId || frozen.model.provider !== manifest.model.provider) throw new Error(`campaign ${campaign}/${arm} frozen ledger manifest differs from the plan`) + } + const armRow = (existing?.proofCampaign?.ledgerSnapshot?.arms as Array> | undefined)?.find((row) => row.arm_id === arm) + if (phase === "development" && existing && armRow?.phase === "confirmation" && ["evaluated", "finished"].includes(existing.proofCampaign?.state ?? "")) return existing + if (phase === "confirmation" && existing && isFullyConfirmedRun(existing, arm)) return existing + if (existing && archive.runs.some((entry) => entry.runId === existing.runId) && phase === "development" && armRow?.phase !== "confirmation") throw new Error(`campaign ${campaign}/${arm} is archived before development completion; refusing duplicate execution`) + const result = await runner({ + ...config, dryRun: false, workerHarnessRunId: runId, + ...(existing ? { resumeRunId: existing.runId } : {}), + workerHarnessMode: arm, workerHarnessPhase: phase, workerHarnessPairId: campaign, + workerHarnessRuntime: manifest.runtime, + }) + const after = await readArchive(config.archiveDir) + const saved = after.runs.concat(after.activeRuns).filter((entry) => entry.runId === runId) + if (saved.length !== 1 || saved[0]?.runId !== result.runId) throw new Error(`campaign ${campaign}/${arm}/${phase} did not persist its deterministic run identity`) + if (phase === "confirmation" && !isFullyConfirmedRun(saved[0]!, arm)) throw new Error(`campaign ${campaign}/${arm} did not persist a complete terminal confirmation record`) + return saved[0]! +} + +export function isFullyConfirmedRun(run: RsiRunRecord, armId: "fixed-control" | "self-hosted"): boolean { + const campaign = run.proofCampaign + const snapshot = campaign?.ledgerSnapshot + const manifest = campaign?.manifest + const arm = (snapshot?.arms as Array> | undefined)?.find((entry) => entry.arm_id === armId) + const campaignRow = snapshot?.campaign as Record | undefined + const scheduled = manifest?.schedule.filter((cell) => cell.armId === armId && cell.phase === "confirmation") ?? [] + if (!snapshot || manifest?.phaseMode !== "paired-confirmation" || campaign?.state !== "finished" || campaignRow?.state !== "finished" || arm?.state !== "finished" || arm.phase !== "confirmation" || run.confirmationEvidence !== "complete" || !run.proofConfirmation || scheduled.length === 0) return false + const selected = run.proofConfirmation.selectedHarness + const attempts = snapshot.attempts as Array> | undefined + if (!attempts || !run.confirmationAttempts?.length || !selected.attemptIds.length || !run.proofConfirmation.rootHarness.attemptIds.length) return false + for (const cell of scheduled) { + const rows = attempts.filter((entry) => entry.arm_id === armId && Number(entry.generation) === cell.generation && entry.candidate_id === cell.candidateId && entry.task_id === cell.taskId && entry.seed === cell.seed && entry.attempt_kind === cell.attemptKind) + if (rows.length !== 1) return false + const required = cell.candidateId === "root-control" || (cell.candidateId === selected.slotId && cell.generation === selected.generation) + if (rows[0]?.status !== (required ? "completed" : "not-selected")) return false + if (required) { + const brokerJobId = String(rows[0]?.broker_job_id ?? "") + const matching = run.confirmationAttempts.filter((attempt) => attempt.attemptId === brokerJobId) + if (matching.length !== 1 || rows[0]?.result_sha256 !== sha256(JSON.stringify(matching[0]))) return false + const attempt = matching[0]! + const expectedCommit = cell.candidateId === "root-control" ? run.proofConfirmation.rootHarness.commit : selected.commit + if (attempt.taskId !== cell.taskId || attempt.seed !== Number(cell.seed) || attempt.harnessCommit !== expectedCommit || attempt.validity !== "valid" || attempt.proofQualified !== true || attempt.resourceAccounting !== "supervisor-enforced" || attempt.inferenceReceipt?.qualified !== true || attempt.inferenceReceipt.endpointProvider !== "deepinfra" || attempt.inferenceReceipt.returnedProviders?.length !== 1 || attempt.inferenceReceipt.returnedProviders[0] !== "deepinfra") return false + const savedRefs = rows[0]?.artifact_refs as Record | undefined + if (!savedRefs || Object.entries(attempt.artifactRefs).some(([name, digest]) => savedRefs[name]?.sha256 !== digest)) return false + } + } + const attemptIds = new Set(run.confirmationAttempts.map((attempt) => attempt.attemptId)) + const requiredIds = [...selected.attemptIds, ...run.proofConfirmation.rootHarness.attemptIds] + if (!requiredIds.every((id) => attemptIds.has(id))) return false + const artifacts = new Map(run.confirmationAttempts.map((attempt) => [attempt.attemptId, attempt])) + return requiredIds.every((id) => { + const attempt = artifacts.get(id) + return attempt?.validity === "valid" && attempt.proofQualified === true && attempt.inferenceReceipt?.qualified === true && attempt.inferenceReceipt.returnedProviders?.every((provider) => provider === "deepinfra") === true + }) +} + +export async function executeFrozenCampaigns(options: Options, env: NodeJS.ProcessEnv = process.env, runner: typeof runRsi = runRsi): Promise { + if (options.command !== "execute" || !options.execute) throw new Error("campaign execution requires execute command and explicit --execute") + if (env !== process.env || runner !== runRsi) throw new Error("paid campaign execution only accepts the validated process environment and production runner") + if (!options.planPath) throw new Error("--plan is required for campaign execution") + const safePath = await assertSupervisorOnlyPlanPath(options.repo, options.planPath) + const manifest = verifyFrozenProofDemonstrationManifest(JSON.parse(await fs.readFile(safePath, "utf8")) as unknown) + const cap = parseUsd(options.operatorCapUsd) + if (Math.abs(cap - manifest.operatorCapUsd) > 0.000001) throw new Error("execution operator cap must exactly match the frozen plan cap") + await verifyLiveFrozenProofPlan(manifest, env) + for (let campaignIndex = 0; campaignIndex < 3; campaignIndex++) { + for (const arm of ["fixed-control", "self-hosted"] as const) await runFrozenCampaignArm(manifest, campaignIndex, arm, "development") + for (const arm of ["fixed-control", "self-hosted"] as const) await runFrozenCampaignArm(manifest, campaignIndex, arm, "confirmation") + } + process.stdout.write(`All three frozen paired campaigns reached terminal confirmation state. No analysis verdict has been issued; run analyze against the frozen plan and archive.\n`) +} + export async function proofDemonstrationPlanMain(argv = process.argv.slice(2)): Promise { if (argv.includes("--help") || argv.includes("-h")) { process.stdout.write(usage()); return 0 } try { @@ -191,12 +531,43 @@ export async function proofDemonstrationPlanMain(argv = process.argv.slice(2)): process.stdout.write(`${formatPlan(await buildProofPlanPreview(options))}\n`) return 0 } - if (options.command === "verify") { - const safePath = await assertSupervisorOnlyPlanPath(options.repo, options.output) + if (options.command === "engineering") { + const evidence = await executeEngineeringPair(options) + if (evidence) process.stdout.write(`Engineering pair passed development gate: campaign=${evidence.campaignId}; evidence=${options.output}; sha256=${evidence.digestSha256}\n`) + return 0 + } + if (options.command === "execute") { + await executeFrozenCampaigns(options) + return 0 + } + if (options.command === "verify" || options.command === "analyze") { + const requestedPlanPath = options.planPath ?? (options.command === "verify" ? options.output : undefined) + if (!requestedPlanPath) throw new Error("--plan is required") + const safePath = await assertSupervisorOnlyPlanPath(options.repo, requestedPlanPath) const parsed = JSON.parse(await fs.readFile(safePath, "utf8")) as unknown const manifest = verifyFrozenProofDemonstrationManifest(parsed) - await verifyLiveFrozenProofPlan(manifest) - process.stdout.write(`Verified frozen plan ${manifest.digestSha256}; operator cap $${manifest.operatorCapUsd.toFixed(6)}; paid provider inference has not been started.\n`) + if (options.command === "verify") { + await verifyLiveFrozenProofPlan(manifest) + process.stdout.write(`Verified frozen plan ${manifest.digestSha256}; operator cap $${manifest.operatorCapUsd.toFixed(6)}; paid provider inference has not been started.\n`) + return 0 + } + if (!options.archiveDir) throw new Error("--archive is required") + const engineering = await verifyEngineeringEvidenceFile(manifest.engineeringEvidence.path, options.repo, manifest.rootCommit) + if (engineering.fileSha256 !== manifest.engineeringEvidence.sha256 || engineering.record.campaignId !== manifest.engineeringEvidence.campaignId || engineering.record.manifestSha256 !== manifest.engineeringEvidence.manifestSha256 || engineering.record.snapshotSha256 !== manifest.engineeringEvidence.snapshotSha256) throw new Error("engineering gate evidence no longer matches the frozen final plan") + if (canonicalHash(await analysisDigest(options.repo)) !== canonicalHash(manifest.execution.analysisSourceSha256)) throw new Error("analysis source differs from the frozen manifest") + const archive = await readArchive(options.archiveDir) + const campaignIds = new Set(manifest.campaignIds.campaigns) + const runs = [...archive.runs, ...archive.activeRuns].filter((run) => run.workerHarnessPairId && campaignIds.has(run.workerHarnessPairId)) + await verifyProofDemonstrationArtifactBytes(runs, options.archiveDir) + const report = analyzeProofDemonstration(runs, { frozenPlan: manifest }) + const defaultReportPath = path.resolve(".headlesscode/rsi-proof", `analysis-${manifest.digestSha256}.json`) + const reportPath = await assertSupervisorOnlyPlanPath(options.repo, options.outputExplicit ? options.output : defaultReportPath) + const markdownPath = path.join(path.dirname(reportPath), path.basename(reportPath, path.extname(reportPath)) + ".md") + await assertSupervisorOnlyPlanPath(options.repo, markdownPath) + await fs.mkdir(path.dirname(reportPath), { recursive: true, mode: 0o700 }) + await fs.writeFile(reportPath, `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600, flag: "wx" }) + await fs.writeFile(markdownPath, formatAnalysisReport(report, manifest.digestSha256), { mode: 0o600, flag: "wx" }) + process.stdout.write(`${JSON.stringify({ verdict: report.verdict, reportPath, markdownPath, missingCells: report.denominators.missingCells.length, excludedAttempts: report.denominators.excludedAttempts.length })}\n`) return 0 } const manifest = await buildFrozenProofPlan(options) diff --git a/src/budget/__tests__/cost.test.ts b/src/budget/__tests__/cost.test.ts index b10899b..092400c 100644 --- a/src/budget/__tests__/cost.test.ts +++ b/src/budget/__tests__/cost.test.ts @@ -60,8 +60,9 @@ async function testDeepseekV4Flash0731Priced(): Promise { const price = DEFAULT_PRICING_TABLE["deepseek/deepseek-v4-flash-0731"] assert.ok(price, "deepseek/deepseek-v4-flash-0731 must be a real entry, not fall through to the fallback") assert.notDeepEqual(price, FALLBACK_MODEL_PRICE, "must not accidentally equal the unknown-model fallback price") - assert.ok(price.input < 1, `input price should be well under $1/1M (real rate is ~$0.14), got $${price.input}`) - assert.ok(price.cacheRead !== undefined && price.cacheRead < price.input, "cacheRead must be a real discount, not left unset") + assert.equal(price.input, 0.06, "the pinned DeepInfra fp8 endpoint metadata reports $0.06/1M input") + assert.equal(price.output, 0.18, "the pinned DeepInfra fp8 endpoint metadata reports $0.18/1M output") + assert.equal(price.cacheRead, undefined, "without a verified cache price, cached input uses the full input rate") } async function testDefaultTableShape(): Promise { diff --git a/src/budget/cost.ts b/src/budget/cost.ts index 01ab4c4..21cb99f 100644 --- a/src/budget/cost.ts +++ b/src/budget/cost.ts @@ -55,21 +55,13 @@ export type PricingTable = Record /** * Default pricing table (OpenRouter list rates, USD per 1M tokens). * - * - `deepseek/deepseek-v4-flash-0731` — the harness default (Phase 1/2/5 - * workers; see `DEFAULT_MODEL` in src/llm/openrouter.ts). Priced the same - * as the prior default, `deepseek/deepseek-v4-flash` (kept below for - * override compat) — same model line, dated point-release id. The rate is - * the OFFICIAL DeepSeek provider's own rate specifically (verified - * against OpenRouter's own - * `/api/v1/models/deepseek/deepseek-v4-flash/endpoints` on 2026-08-01, - * cross-checked against OpenRouter's own pricing UI). This price is only - * actually correct because `src/llm/openrouter.ts` pins routing to this - * exact provider (`provider: { order: ["deepseek"], allow_fallbacks: - * false }`) for deepseek/* models — OTHER routed endpoints behind the - * same model id (DeepInfra, Baidu, Mancer, ...) have meaningfully - * different prices, especially for cache reads, so this number would be - * wrong/unverifiable without that pin. If the pin is ever removed, this - * price must be revisited, not left as a stale guess. + * - `deepseek/deepseek-v4-flash-0731` — the RSI proof model. It is routed + * exclusively to OpenRouter's available DeepInfra fp8 endpoint by + * `src/llm/openrouter.ts`. The 2026-09 endpoint metadata reports + * $0.06/1M input and $0.18/1M output tokens; cached input is conservatively + * charged at the full input rate because this endpoint metadata provides + * no verified cache-read rate. The RSI broker independently fetches and + * pins endpoint pricing before admitting work. * Missing entirely from this table before was the actual root cause of a * real incident: sessions silently fell through to `FALLBACK_MODEL_PRICE` * (7x+ this model's real input rate), producing wildly inflated internal @@ -99,7 +91,7 @@ export type PricingTable = Record export const DEFAULT_PRICING_TABLE: PricingTable = { "deepseek/deepseek-chat": { input: 0.27, output: 1.1 }, "deepseek/deepseek-reasoner": { input: 0.55, output: 2.19 }, - "deepseek/deepseek-v4-flash-0731": { input: 0.14, output: 0.28, cacheRead: 0.0028 }, + "deepseek/deepseek-v4-flash-0731": { input: 0.06, output: 0.18 }, "deepseek/deepseek-v4-flash": { input: 0.14, output: 0.28, cacheRead: 0.0028 }, "anthropic/claude-3.5-sonnet": { input: 3.0, output: 15.0 }, "qwen/qwen3-embedding-4b": { input: 0.15, output: 0.15 }, diff --git a/src/llm/__tests__/openrouter.test.ts b/src/llm/__tests__/openrouter.test.ts index 84cfd1a..e953577 100644 --- a/src/llm/__tests__/openrouter.test.ts +++ b/src/llm/__tests__/openrouter.test.ts @@ -34,6 +34,13 @@ async function testDeepseekReasonerAlsoPinned(): Promise { assert.deepEqual(body.provider, { order: ["deepseek"], allow_fallbacks: false }) } +async function testRsiProofModelUsesPinnedDeepinfraRoute(): Promise { + const body = buildRequestBody({ ...baseRequest, reasoningEffort: "high" }, "deepseek/deepseek-v4-flash-0731") + assert.deepEqual(body.provider, { only: ["deepinfra"], allow_fallbacks: false }) + assert.equal(body.include_reasoning, undefined, "the selected DeepInfra endpoint does not advertise include_reasoning") + assert.equal(body.reasoning, undefined, "the selected DeepInfra endpoint does not advertise reasoning effort") +} + async function testNonDeepseekModelsAreNotPinned(): Promise { const body = buildRequestBody(baseRequest, "anthropic/claude-3.5-sonnet") assert.equal(body.provider, undefined, "pinning to the deepseek provider for a non-deepseek model would be wrong") @@ -608,8 +615,9 @@ async function testIsRetryableOpenRouterErrorRejectsDeterministicFailures(): Pro const tests: Array<[string, () => Promise]> = [ ["isRetryableOpenRouterError matches network/429/5xx/no-allowed-providers shapes", testIsRetryableOpenRouterErrorMatchesTransientShapes], ["isRetryableOpenRouterError rejects deterministic 4xx and non-OpenRouterError values", testIsRetryableOpenRouterErrorRejectsDeterministicFailures], - ["deepseek/* models are pinned to the official DeepSeek provider", testDeepseekModelsArePinnedToOfficialProvider], + ["legacy deepseek/* models remain pinned to the official DeepSeek provider", testDeepseekModelsArePinnedToOfficialProvider], ["the pin applies to any deepseek/* model, not just one id", testDeepseekReasonerAlsoPinned], + ["the RSI proof model uses the pinned DeepInfra endpoint and advertised parameters", testRsiProofModelUsesPinnedDeepinfraRoute], ["non-deepseek models are never pinned to the deepseek provider", testNonDeepseekModelsAreNotPinned], ["buildRequestBody emits max_tokens when set, omits it when unset", testRequestBodyEmitsMaxTokensWhenSet], ["deepseek/* requests include include_reasoning (load-bearing)", testDeepseekRequestIncludesReasoning], diff --git a/src/llm/openrouter.ts b/src/llm/openrouter.ts index 9dfce39..fdb76b0 100644 --- a/src/llm/openrouter.ts +++ b/src/llm/openrouter.ts @@ -800,7 +800,7 @@ export function buildRequestBody(request: LlmRequest, model: string): Record item.status === "fulfilled").length, 1, "campaign and job spend locks admit only one call under contention") const callId = admissions[0]?.status === "fulfilled" ? "call-concurrent-0001" : "call-concurrent-0002" - await queue.settle(grant, { callId, state: "complete", qualified: true, returnedModel: grant.model, promptTokens: 12, completionTokens: 8, costUsd: 0.001, responseSha256: "b".repeat(64), completedAt: new Date().toISOString() }) + await queue.settle(grant, { callId, state: "complete", qualified: true, returnedModel: grant.model, returnedProvider: "deepinfra", promptTokens: 12, completionTokens: 8, costUsd: 0.001, responseSha256: "b".repeat(64), completedAt: new Date().toISOString() }) const receipt = await queue.receipt(grant) assert.equal(receipt.qualified, true) assert.equal(receipt.totalTokens, 20) @@ -179,7 +179,7 @@ test("PostgreSQL OpenRouter admission persists reservations, blocks overspend, a const lateReservation = reservation("call-late-12345678") await queue.reserve(lateGrant, lateReservation) await queue.pool.query("UPDATE headlesscode_rsi_inference_jobs SET started_at=clock_timestamp()-interval '61 seconds' WHERE job_id=$1", [lateJob.jobId]) - await queue.settle(lateGrant, { callId: lateReservation.callId, state: "complete", qualified: true, returnedModel: lateGrant.model, promptTokens: 12, completionTokens: 8, costUsd: 0.001, responseSha256: "c".repeat(64), completedAt: new Date().toISOString() }) + await queue.settle(lateGrant, { callId: lateReservation.callId, state: "complete", qualified: true, returnedModel: lateGrant.model, returnedProvider: "deepinfra", promptTokens: 12, completionTokens: 8, costUsd: 0.001, responseSha256: "c".repeat(64), completedAt: new Date().toISOString() }) assert.equal((await queue.receipt(lateGrant)).qualified, false, "a late settlement is persisted as non-qualifying") } finally { await queue.close().catch(() => undefined) diff --git a/src/rsi/__tests__/proof-campaign.test.ts b/src/rsi/__tests__/proof-campaign.test.ts index c8a94cc..53e884d 100644 --- a/src/rsi/__tests__/proof-campaign.test.ts +++ b/src/rsi/__tests__/proof-campaign.test.ts @@ -13,7 +13,7 @@ import { openRouterAllocatedEnvironment, openRouterBrokerLimitsFromEnvironment, import { readArchive, writeArchive } from "../archive.js" import { createProofCampaignManifest, proofBrokerCampaignId, reconcileProofCampaignJob, rebuildEvaluatedProofCampaignArchive, rebuildFinishedProofCampaignArchive } from "../controller.js" import { loadBenchmark, runBenchmark, type BenchmarkVerifier } from "../benchmark/runner.js" -import type { BenchmarkAgent, BenchmarkImageIdentity, BenchmarkTask } from "../benchmark/types.js" +import type { BenchmarkAgent, BenchmarkAttempt, BenchmarkImageIdentity, BenchmarkTask } from "../benchmark/types.js" import type { CandidateRecord, RsiConfig, RsiRunRecord } from "../types.js" const databaseUrl = process.env.HEADLESSCODE_RSI_TEST_DATABASE_URL @@ -73,6 +73,34 @@ test("proof campaign retains uncertain dispatched spend and enforces one key per }) }) +test("engineering-only proof pair finishes after development and freezes no confirmation cells", { skip: !databaseUrl }, async () => { + await withCampaign(async (campaign) => { + const campaignId = `engineering-${randomUUID()}` + const schedule: ProofCampaignManifest["schedule"] = ["fixed", "self"].map((armId) => ({ armId, generation: 0, candidateId: "g0-candidate-0", taskId: "mutation", seed: "seed-0", attemptKind: "mutation", phase: "development" })) + const frozen = manifest(campaignId, schedule) + frozen.phaseMode = "engineering-only" + frozen.benchmarkDigests.confirmation = createHash("sha256").update("[]").digest("hex") + frozen.settings = { phaseMode: "engineering-only", confirmationTaskIdentities: [] } + for (const armId of ["fixed", "self"]) { + const lease = await campaign.acquire(frozen, armId, `run-${armId}`, `coordinator-${armId}`) + const cell = schedule.find((entry) => entry.armId === armId)! + const attemptKey = `engineering-${armId}` + await admit(campaign, lease, cell, attemptKey) + await campaign.markDispatched(lease, attemptKey) + await campaign.settle(lease, attemptKey, { status: "completed", resultSha256: "9".repeat(64), actualCalls: 1, actualTokens: 10, actualSpendUsd: 0.001 }) + await campaign.recordDecision(lease, { generation: 0, candidateId: cell.candidateId, parentId: "baseline", status: "accepted", childCommit: "b".repeat(40), parentCommit: frozen.rootCommit, proof: { developmentAccepted: true }, attemptKeys: [attemptKey] }) + await campaign.selectGeneration(lease, 0, [cell.candidateId]) + await campaign.finishDevelopmentArm(lease) + } + const snapshot = await campaign.snapshot(campaignId) + assert.equal((snapshot.campaign as Record).state, "finished") + assert.ok((snapshot.arms as Array>).every((arm) => arm.state === "finished" && arm.phase === "development")) + assert.equal((snapshot.attempts as Array>).some((attempt) => attempt.attempt_kind === "confirmation"), false) + assert.equal((snapshot.events as Array>).filter((event) => event.event_type === "engineering-development-finished").length, 2) + await assert.rejects(() => campaign.acquire(frozen, "fixed", "run-fixed", "coordinator-reopen"), /finished/) + }) +}) + test("proof campaign enforces frozen per-attempt call, token, and spend ceilings", { skip: !databaseUrl }, async () => { await withCampaign(async (campaign) => { const schedule = [{ armId: "fixed", generation: 0, candidateId: "candidate-a", taskId: "mutation", seed: "seed-0", attemptKind: "mutation" as const }] @@ -143,6 +171,16 @@ test("controller freezes paired broker allocations as separate campaigns under o assert.equal(frozen.budgets.root.maxTokens, 645_120, "root tokens sum effective task-specific grants rather than the base per-call defaults") assert.equal(frozen.budgets.root.maxSpendUsd, 0.2, "root spend is the sum of every frozen provider-job reservation") assert.equal(frozen.settings.taskWallBoundMs, 140_000, "the runtime plan sums mutation, required evaluation, development proof, and selected-final/root confirmation deadlines") + const engineering = createProofCampaignManifest({ + config: { ...config, workerHarnessEngineeringOnly: true }, campaignId: `${campaignId}-engineering`, baseCommit: "f".repeat(40), model: RSI_OPENROUTER_MODEL, provider: "openshell-openrouter", + runtime: { imageName: "rsi:test", imageDigest: `sha256:${"9".repeat(64)}`, harnessCommit: "8".repeat(40), dependencyLockDigest: "7".repeat(64) }, + developmentManifest, developmentManifestDigest: "4".repeat(64), plannedGenerations: 2, proofProtocol: protocol, + }) + assert.equal(engineering.phaseMode, "engineering-only") + assert.ok(engineering.schedule.every((cell) => cell.phase === "development"), "engineering-only authority freezes no confirmation cells") + assert.deepEqual(engineering.settings.confirmationTaskIdentities, [], "engineering manifest carries no holdout task identities") + assert.equal(engineering.benchmarkDigests.confirmation, createHash("sha256").update("[]").digest("hex"), "engineering manifest records an empty confirmation schedule, not holdout task identities") + assert.notEqual(proofCampaignManifestDigest(engineering), proofCampaignManifestDigest(frozen), "phase mode and schedule are bound into the campaign identity") assert.throws(() => createProofCampaignManifest({ config: { ...config, maxRuntimeMs: 139_999 }, campaignId: `${campaignId}-short`, baseCommit: "f".repeat(40), model: RSI_OPENROUTER_MODEL, provider: "openshell-openrouter", runtime: { imageName: "rsi:test", imageDigest: `sha256:${"9".repeat(64)}`, harnessCommit: "8".repeat(40), dependencyLockDigest: "7".repeat(64) }, developmentManifest, developmentManifestDigest: "4".repeat(64), plannedGenerations: 2, proofProtocol: protocol }), /below the summed frozen task-time ceiling/) process.env.HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD = "0.19" process.env.HEADLESSCODE_RSI_OPENROUTER_MUTATION_BUDGET_USD = "0.04" @@ -599,14 +637,51 @@ test("finished ledger rebuilds the final report from a reloaded active-run check } const run: RsiRunRecord = { runId: "finished-run", startedAt: "2026-01-01T00:00:00.000Z", model: "test/model", baseRef: "main", baseCommit: "a".repeat(40), - generations: 1, candidates: [candidate], reports: [], + generations: 1, candidates: [candidate], reports: [], confirmationEvidence: "complete", + proofConfirmation: { + selectedHarness: { candidateId: candidate.id, slotId: "g0-candidate-0", generation: 0, commit: "b".repeat(40), attemptIds: ["confirm-selected"] }, + rootHarness: { commit: "a".repeat(40), attemptIds: ["confirm-root"] }, + }, } const config = { archiveDir, regressionCommand: "npm test", evalCommands: [], hiddenEvalCommands: [], protectedPaths: [] } as unknown as RsiConfig - const frozen = manifest("finished-campaign", [{ armId: "fixed", generation: 0, candidateId: "g0-candidate-0", taskId: "mutation", seed: "seed-0", attemptKind: "mutation" }]) + const confirmationSchedule: ProofCampaignManifest["schedule"] = [ + { armId: "fixed", generation: 0, candidateId: "g0-candidate-0", taskId: "confirm", seed: "seed-0", attemptKind: "confirmation", phase: "confirmation" }, + { armId: "fixed", generation: 0, candidateId: "root-control", taskId: "confirm", seed: "seed-0", attemptKind: "fixed-control", phase: "confirmation" }, + ] + const frozen = manifest("finished-campaign", confirmationSchedule) + const makeAttempt = (attemptId: string, candidateId: string, kind: "confirmation" | "fixed-control", commit: string): BenchmarkAttempt => ({ + schemaVersion: 1, manifestDigest: "d".repeat(64), taskId: "confirm", split: "confirmation", attemptId, harnessCommit: commit, + agentImage: null, model: { provider: "mock", id: "test/model", sampling: {} }, seed: 1, family: "confirmation", + startDigest: "e".repeat(64), ceilings: { timeoutMs: 1000, maxCalls: 1, maxOutputTokensPerCall: 10, maxPatchBytes: 1000 }, + verifiedSuccess: true, proofQualified: true, resourceAccounting: "supervisor-enforced", proofRequestSha256: "f".repeat(64), + inferenceReceipt: { provider: "openrouter", endpointProvider: "deepinfra", jobId: attemptId, campaignId: "finished-campaign:benchmark", model: "deepseek/deepseek-v4-flash-0731", returnedModels: ["deepseek/deepseek-v4-flash-0731"], returnedProviders: ["deepinfra"], calls: 1, promptTokens: 1, completionTokens: 1, totalTokens: 2, costUsd: 0.0001, accounting: "supervisor-enforced", qualified: true, callReceipts: [], denials: [] }, + validity: "valid", calls: 1, tokens: 2, wallTimeMs: 10, stopReason: "completed", patchDigest: null, + evaluatorDigest: "a".repeat(64), artifactRefs: { stdout: createHash("sha256").update(attemptId + " stdout").digest("hex"), stderr: createHash("sha256").update(attemptId + " stderr").digest("hex") }, + }) + const selectedAttempt = makeAttempt("confirm-selected", "g0-candidate-0", "confirmation", "b".repeat(40)) + const rootAttempt = makeAttempt("confirm-root", "root-control", "fixed-control", "a".repeat(40)) + const confirmationAttempts = [selectedAttempt, rootAttempt] + for (const [attempt, candidateId, kind] of [[selectedAttempt, "g0-candidate-0", "confirmation"], [rootAttempt, "root-control", "fixed-control"]] as const) { + for (const [name, digest] of Object.entries(attempt.artifactRefs)) { + const bytes = Buffer.from(attempt.attemptId + (name === "stdout" ? " stdout" : " stderr")) + const artifact = path.join(archiveDir, "confirmation", frozen.campaignId, "fixed", candidateId, kind, "harness", attempt.harnessCommit, "artifacts", "sha256", digest.slice(0, 2), digest) + await fs.mkdir(path.dirname(artifact), { recursive: true }) + await fs.writeFile(artifact, bytes) + } + } + run.confirmationAttempts = confirmationAttempts const snapshot = { campaign: { manifest_sha256: proofCampaignManifestDigest(frozen) }, arms: [{ arm_id: "fixed", run_id: run.runId, state: "finished", epoch: 3 }], - events: [{ sequence: 4, arm_id: "fixed", event_type: "generation-selected", payload: { generation: 0, selectedCandidateIds: ["g0-candidate-0"] } }], + events: [ + { sequence: 4, arm_id: "fixed", event_type: "generation-selected", payload: { generation: 0, selectedCandidateIds: ["g0-candidate-0"] } }, + { sequence: 5, arm_id: "fixed", event_type: "final-harness-selected", payload: { generation: 0, candidateId: "g0-candidate-0", commit: "b".repeat(40) } }, + ], + attempts: confirmationAttempts.map((attempt, index) => ({ + arm_id: "fixed", generation: 0, candidate_id: index === 0 ? "g0-candidate-0" : "root-control", task_id: "confirm", seed: "seed-0", + attempt_kind: index === 0 ? "confirmation" : "fixed-control", status: "completed", broker_job_id: attempt.attemptId, + result_sha256: createHash("sha256").update(JSON.stringify(attempt)).digest("hex"), artifact_refs: Object.fromEntries(Object.entries(attempt.artifactRefs).map(([name, digest]) => [name, { sha256: digest }])), + })), } await writeArchive(archiveDir, { schemaVersion: 2, updatedAt: run.startedAt, runs: [], activeRuns: [run], candidates: [candidate], modelCandidates: [], combinations: [], jobs: [], trajectoryRefs: [], curriculumTasks: [] }) const checkpoint = await readArchive(archiveDir) diff --git a/src/rsi/__tests__/proof-demonstration-launcher.test.ts b/src/rsi/__tests__/proof-demonstration-launcher.test.ts new file mode 100644 index 0000000..17c3af6 --- /dev/null +++ b/src/rsi/__tests__/proof-demonstration-launcher.test.ts @@ -0,0 +1,123 @@ +import assert from "node:assert/strict" +import * as fs from "node:fs/promises" +import * as os from "node:os" +import * as path from "node:path" +import test from "node:test" +import { createHash } from "node:crypto" +import { appendRun } from "../archive.js" +import type { RsiConfig, RsiRunRecord, WorkerHarnessRuntimeIdentity } from "../types.js" +import { assertEngineeringTransition, isFullyConfirmedRun, runEngineeringArm, runFrozenCampaignArm } from "../../../scripts/rsi-proof-demonstration.js" +import type { FrozenProofDemonstrationManifest } from "../proof-demonstration-manifest.js" +import { proofCampaignManifestDigest, type ProofCampaignManifest } from "../proof-campaign.js" + +const digest = (value: unknown) => createHash("sha256").update(JSON.stringify(value)).digest("hex") + +test("engineering verifier requires a terminal fixed-control generation-one decision", () => { + const parentCommit = "1".repeat(40) + const childCommit = "2".repeat(40) + const fixedCell = { armId: "fixed-control", phase: "development", generation: 1, candidateId: "g1-candidate-0", taskId: "dev-1", seed: "41", attemptKind: "parent" } + const fixed = { + candidates: [], + proofCampaign: { + manifest: { schedule: [fixedCell] }, + ledgerSnapshot: { + attempts: [{ arm_id: "fixed-control", generation: 1, candidate_id: "g1-candidate-0", task_id: "dev-1", seed: "41", attempt_kind: "parent", status: "completed" }], + decisions: [{ arm_id: "fixed-control", generation: 1, candidate_id: "g1-candidate-0", decision: "rejected" }], + }, + }, + } as unknown as RsiRunRecord + const parent = { id: "g0-candidate-0", generation: 0, status: "accepted", developmentProof: { accepted: true }, commits: [parentCommit] } + const child = { id: "g1-candidate-0", generation: 1, status: "accepted", developmentProof: { accepted: true }, parent: parent.id, baseCommit: parentCommit, parentCommit, workerHarnessExecution: { parentCommit }, commits: [childCommit] } + const self = { + candidates: [parent, child], + proofCampaign: { + ledgerSnapshot: { + events: [{ arm_id: "self-hosted", event_type: "generation-selected", payload: { generation: 0, selectedCandidateIds: ["g0-candidate-0"] } }], + decisions: [{ arm_id: "self-hosted", generation: 1, candidate_id: "g1-candidate-0", decision: "accepted", parent_commit: parentCommit }], + }, + }, + } as unknown as RsiRunRecord + assert.doesNotThrow(() => assertEngineeringTransition(fixed, self)) + const missingDecision = structuredClone(fixed) + missingDecision.proofCampaign!.ledgerSnapshot!.decisions = [] + assert.throws(() => assertEngineeringTransition(missingDecision, self), /fixed-control generation-one cells and candidate decisions/) +}) + +test("engineering replay reuses a finished fixed arm after a crash before self-hosted starts", async () => { + const root = await fs.mkdtemp(path.join(os.tmpdir(), "rsi-engineering-replay-")) + try { + const campaign = `rsi-proof-${"a".repeat(20)}` + const arm = "fixed-control" as const + const runtime: WorkerHarnessRuntimeIdentity = { imageName: "image", imageDigest: `sha256:${"b".repeat(64)}`, harnessCommit: "c".repeat(40), dependencyLockDigest: "d".repeat(64) } + const config = { archiveDir: root, model: "model", generations: 2, seed: "engineering-seed" } as RsiConfig + const finished = { + runId: `${campaign}-fixed`, startedAt: "2026-09-30T00:00:00Z", finishedAt: "2026-09-30T00:01:00Z", + model: "model", baseRef: "main", baseCommit: "e".repeat(40), generations: 2, candidates: [], reports: [], + workerHarnessPairId: campaign, workerHarnessMode: arm, workerHarnessSamplingSeed: "engineering-seed", workerHarnessRuntime: runtime, + proofCampaign: { + manifest: { phaseMode: "engineering-only", schedule: [] }, manifestSha256: "f".repeat(64), campaignId: campaign, armId: arm, epoch: 1, + ledgerSnapshot: { arms: [{ arm_id: arm, state: "finished" }] }, state: "finished", + }, + } as unknown as RsiRunRecord + await appendRun(root, finished) + let runnerCalls = 0 + const resumed = await runEngineeringArm(config, campaign, arm, finished.baseCommit, "engineering-seed", runtime, async () => { + runnerCalls++ + throw new Error("a completed arm must be read-only on recovery") + }) + assert.equal(resumed.runId, finished.runId) + assert.equal(runnerCalls, 0) + } finally { + await fs.rm(root, { recursive: true, force: true }) + } +}) + +test("campaign executor reuses a finished first arm without dispatching it again", async () => { + const root = await fs.mkdtemp(path.join(os.tmpdir(), "rsi-campaign-replay-")) + try { + const campaign = `rsi-proof-${"a".repeat(20)}` + const armId = "fixed-control" as const + const runtime: WorkerHarnessRuntimeIdentity = { imageName: "image", imageDigest: `sha256:${"b".repeat(64)}`, harnessCommit: "c".repeat(40), dependencyLockDigest: "d".repeat(64) } + const config = { archiveDir: root, seed: "campaign-seed", generations: 3, model: "model" } as RsiConfig + const schedule = [ + { armId, generation: 2, candidateId: "candidate-2-0", taskId: "holdout", seed: "17", attemptKind: "confirmation" as const, phase: "confirmation" as const }, + { armId, generation: 2, candidateId: "root-control", taskId: "holdout", seed: "17", attemptKind: "fixed-control" as const, phase: "confirmation" as const }, + ] + const attempts = [ + { attemptId: "final-attempt", taskId: "holdout", seed: 17, harnessCommit: "1".repeat(40), validity: "valid", proofQualified: true, resourceAccounting: "supervisor-enforced", artifactRefs: {}, inferenceReceipt: { qualified: true, endpointProvider: "deepinfra", returnedProviders: ["deepinfra"] } }, + { attemptId: "root-attempt", taskId: "holdout", seed: 17, harnessCommit: "e".repeat(40), validity: "valid", proofQualified: true, resourceAccounting: "supervisor-enforced", artifactRefs: {}, inferenceReceipt: { qualified: true, endpointProvider: "deepinfra", returnedProviders: ["deepinfra"] } }, + ] + const manifestData = { + schemaVersion: 1, phaseMode: "paired-confirmation", campaignId: campaign, rootCommit: "e".repeat(40), + benchmarkDigests: { manifest: "3".repeat(64), development: "4".repeat(64), confirmation: "5".repeat(64) }, + armAssignment: { fixedControl: "fixed-control", selfHosted: "self-hosted" }, + model: { provider: "openshell-openrouter", id: "model", settings: { sampling: { temperature: 0 } } }, seed: "campaign-seed", taskOrder: ["holdout"], schedule, + budgets: { attempt: { maxCalls: 1, maxTokens: 10, maxSpendUsd: 0.1, maxDurationMs: 1000 }, root: { maxCalls: 10, maxTokens: 100, maxSpendUsd: 1, maxRuntimeMs: 10000 } }, + acceptanceRule: {}, runtime: { imageDigest: runtime.imageDigest, harnessCommit: runtime.harnessCommit, dependencyLockDigest: runtime.dependencyLockDigest }, analysisVersion: "test", + settings: { effectiveConfig: config }, + } as unknown as ProofCampaignManifest + const run = { + runId: `${campaign}-fixed`, startedAt: "2026-09-30T00:00:00Z", finishedAt: "2026-09-30T00:01:00Z", + model: "model", baseRef: "main", baseCommit: "e".repeat(40), generations: 3, candidates: [], reports: [], + workerHarnessPairId: campaign, workerHarnessMode: armId, workerHarnessSamplingSeed: "campaign-seed", workerHarnessConfigDigest: "f".repeat(64), workerHarnessPlannedGenerations: 3, workerHarnessRuntime: runtime, + confirmationEvidence: "complete", confirmationAttempts: attempts, + proofConfirmation: { selectedHarness: { candidateId: "candidate-id", slotId: "candidate-2-0", generation: 2, commit: "1".repeat(40), attemptIds: ["final-attempt"] }, rootHarness: { commit: "e".repeat(40), attemptIds: ["root-attempt"] } }, + proofCampaign: { manifest: manifestData, manifestSha256: proofCampaignManifestDigest(manifestData), campaignId: campaign, armId, epoch: 1, state: "finished", ledgerSnapshot: { campaign: { state: "finished" }, arms: [{ arm_id: armId, phase: "confirmation", state: "finished" }], attempts: schedule.map((cell, index) => ({ broker_job_id: index === 0 ? "final-attempt" : "root-attempt", arm_id: armId, generation: cell.generation, candidate_id: cell.candidateId, task_id: cell.taskId, seed: cell.seed, attempt_kind: cell.attemptKind, status: "completed", result_sha256: digest(attempts[index]), artifact_refs: {} })) } }, + } as unknown as RsiRunRecord + assert.equal(isFullyConfirmedRun(run, armId), true) + await appendRun(root, run) + const plan = { campaignIds: { campaigns: [campaign] }, execution: { configs: [config, config, config, config] }, benchmarkManifestSha256: "3".repeat(64), model: { provider: "openshell-openrouter", requestedId: "model" }, rootCommit: run.baseCommit, runtime } as unknown as FrozenProofDemonstrationManifest + let calls = 0 + const reused = await runFrozenCampaignArm(plan, 0, armId, "confirmation", async () => { calls++; throw new Error("finished arm must not replay inference") }) + assert.equal(reused.runId, run.runId) + assert.equal(calls, 0) + const forgedReceipt = structuredClone(run) + forgedReceipt.confirmationAttempts![0]!.inferenceReceipt!.returnedProviders = [] + assert.equal(isFullyConfirmedRun(forgedReceipt, armId), false, "an empty provider list is not authoritative provider evidence") + const damaged = structuredClone(run) + ;(damaged.proofCampaign!.ledgerSnapshot!.attempts as Array>)[0]!.result_sha256 = "f".repeat(64) + assert.equal(isFullyConfirmedRun(damaged, armId), false, "a changed durable result digest prevents terminal replay") + } finally { + await fs.rm(root, { recursive: true, force: true }) + } +}) diff --git a/src/rsi/__tests__/proof-demonstration-manifest.test.ts b/src/rsi/__tests__/proof-demonstration-manifest.test.ts index cc283b6..89af4c6 100644 --- a/src/rsi/__tests__/proof-demonstration-manifest.test.ts +++ b/src/rsi/__tests__/proof-demonstration-manifest.test.ts @@ -5,12 +5,12 @@ import type { ProofDemonstrationPlan } from "../proof-demonstration-plan.js" import type { RsiConfig, WorkerHarnessRuntimeIdentity } from "../types.js" const config = { repoRoot: "/work", seed: "eng-seed", protectedPaths: ["src/rsi/"], mutationTask: "fixed objective" } as RsiConfig -const plan = { totalRootSpendCeilingUsd: 4, totalCalls: 10, totalTokens: 100, totalTaskWallMs: 1000 } as ProofDemonstrationPlan +const plan = { totalRootSpendCeilingUsd: 9, engineeringRootSpendCeilingUsd: 1, campaignRootsSpendCeilingUsd: 8, totalCalls: 10, totalTokens: 100, totalTaskWallMs: 1000 } as ProofDemonstrationPlan const runtime: WorkerHarnessRuntimeIdentity = { imageName: "headlesscode-openshell-rsi:release", imageDigest: `sha256:${"a".repeat(64)}`, harnessCommit: "b".repeat(40), dependencyLockDigest: "c".repeat(64) } const limits = { campaignId: "plan-test", maxCalls: 2, maxInputTokensPerCall: 100, maxOutputTokensPerCall: 50, maxTotalTokens: 300, maxDurationMs: 1000, maxRequestBytes: 1000, maxResponseBytes: 1000, maxInputPricePerMillionUsd: 0.2, maxOutputPricePerMillionUsd: 0.3, maxCallSpendUsd: 0.01, maxJobSpendUsd: 0.02, maxCampaignSpendUsd: 1 } -const taskIdentities = [{ id: "task-one", split: "development", family: "engine", seed: 1, startDigest: "d".repeat(64), verifierDigest: "e".repeat(64), knownGoodDigest: "f".repeat(64), knownBadDigest: "1".repeat(64), ceilings: { timeoutMs: 1000, maxCalls: 3, maxOutputTokensPerCall: 100, maxPatchBytes: 1000 } }] +const taskIdentities = ["development", "confirmation"].map((split, index) => ({ id: index === 0 ? "task-one" : "qa-baseline-counts", split, family: "engine", seed: index + 1, startDigest: "d".repeat(64), verifierDigest: "e".repeat(64), knownGoodDigest: "f".repeat(64), knownBadDigest: "1".repeat(64), ceilings: { timeoutMs: 1000, maxCalls: 3, maxOutputTokensPerCall: 100, maxPatchBytes: 1000 } })) -function frozen(operatorCapUsd = 5, priceCeilingsPerMillionUsd = { input: 0.2, output: 0.3 }) { +function frozen(operatorCapUsd = 9, priceCeilingsPerMillionUsd = { input: 0.2, output: 0.3 }) { return freezeProofDemonstrationManifest({ createdAt: "2026-09-30T00:00:00.000Z", rootCommit: "2".repeat(40), benchmarkManifestSha256: "3".repeat(64), model: { provider: "openshell-openrouter", requestedId: "deepseek/deepseek-v4-flash-0731", sampling: { temperature: 0, think: false, seed: "provider-default" } }, @@ -19,6 +19,8 @@ function frozen(operatorCapUsd = 5, priceCeilingsPerMillionUsd = { input: 0.2, o execution: { configs: [config, config, config, config], configSha256: frozenProofConfigDigest([config, config, config, config]), analysisSourceSha256: "5".repeat(64), inferenceLimits: { mutation: limits, benchmark: limits }, taskIdentities }, seeds: { engineering: "eng-seed", campaigns: ["campaign-a", "campaign-b", "campaign-c"] }, campaignIds: { engineering: "eng-id", campaigns: ["a-id", "b-id", "c-id"] }, operatorCapUsd, plan, + engineeringEvidence: { path: "/work/.headlesscode/rsi-proof/engineering.json", sha256: "6".repeat(64), campaignId: "eng-id", rootCommit: "2".repeat(40), manifestSha256: "7".repeat(64), snapshotSha256: "8".repeat(64) }, + analysis: { bootstrapSeed: "cluster-seed-1", bootstrapReplicates: 1000, confirmationRetentionTaskIds: ["qa-baseline-counts"] }, }) } @@ -32,6 +34,6 @@ test("frozen demonstration manifest verifies exact config, schedule, endpoint pr }) test("frozen plan rejects a cap below reservations and price ceilings below observed endpoint rates", () => { - assert.throws(() => frozen(3), /below the frozen root reservation/) - assert.throws(() => frozen(5, { input: 0.01, output: 0.01 }), /price ceilings do not cover/) + assert.throws(() => frozen(7), /below the three frozen campaign roots reservation/) + assert.throws(() => frozen(9, { input: 0.01, output: 0.01 }), /price ceilings do not cover/) }) diff --git a/src/rsi/__tests__/proof-demonstration-plan.test.ts b/src/rsi/__tests__/proof-demonstration-plan.test.ts index ebe8f0e..3e3141c 100644 --- a/src/rsi/__tests__/proof-demonstration-plan.test.ts +++ b/src/rsi/__tests__/proof-demonstration-plan.test.ts @@ -1,6 +1,6 @@ import assert from "node:assert/strict" import { readFile } from "node:fs/promises" -import { planProofDemonstration } from "../proof-demonstration-plan.js" +import { assertFrozenProofDemonstrationPlanMatches, planProofDemonstration } from "../proof-demonstration-plan.js" import type { BenchmarkManifest } from "../benchmark/types.js" import type { RsiConfig } from "../types.js" import type { ProofRoleLimits } from "../proof-demonstration-plan.js" @@ -22,13 +22,16 @@ const limits = (campaignId: string): ProofRoleLimits => ({ maxCallSpendUsd: 0.01, maxJobSpendUsd: 0.2, maxCampaignSpendUsd: 100, }) const plan = planProofDemonstration({ config, manifest, mutationLimits: limits("mutation"), benchmarkLimits: limits("benchmark") }) -assert.deepEqual({ mutation: plan.engineering.mutationJobs, verification: plan.engineering.verificationJobs, dev: plan.engineering.developmentProofCells, confirmationSchedule: plan.engineering.confirmationScheduleCells }, { mutation: 8, verification: 16, dev: 48, confirmationSchedule: 10 }) +assert.deepEqual({ mutation: plan.engineering.mutationJobs, verification: plan.engineering.verificationJobs, dev: plan.engineering.developmentProofCells, confirmationSchedule: plan.engineering.confirmationScheduleCells }, { mutation: 8, verification: 16, dev: 48, confirmationSchedule: 0 }) assert.deepEqual({ candidateConfirm: plan.campaign.confirmationScheduleCells, notSelected: plan.campaign.confirmationNotSelectedCells, final: plan.campaign.paidConfirmationCells, root: plan.campaign.rootControlCells }, { candidateConfirm: 14, notSelected: 10, final: 2, root: 2 }) assert.equal(plan.engineering.maximumCalls, 128) assert.equal(plan.campaign.maximumCalls, 204) assert.equal(plan.totalReservedSpendUsd, plan.engineering.reservedSpendUsd + 3 * plan.campaign.reservedSpendUsd) assert.equal(plan.totalCalls, plan.engineering.maximumCalls + 3 * plan.campaign.maximumCalls) assert.throws(() => planProofDemonstration({ config, manifest, mutationLimits: { ...limits("mutation"), maxCallSpendUsd: 0.00001 }, benchmarkLimits: limits("benchmark") }), /per-call spend cap is below/) +assert.doesNotThrow(() => assertFrozenProofDemonstrationPlanMatches(plan, { config, manifest, mutationLimits: limits("mutation"), benchmarkLimits: limits("benchmark") })) +assert.throws(() => assertFrozenProofDemonstrationPlanMatches({ ...plan, campaignRootsSpendCeilingUsd: 0.000001, totalRootSpendCeilingUsd: 0.000001 }, { config, manifest, mutationLimits: limits("mutation"), benchmarkLimits: limits("benchmark") }), /frozen proof budget plan differs/) +assert.throws(() => assertFrozenProofDemonstrationPlanMatches(plan, { config: { ...config, evalCommands: ["npm test"] }, manifest, mutationLimits: limits("mutation"), benchmarkLimits: limits("benchmark") }), /frozen proof budget plan differs/) const corpus = JSON.parse(await readFile("fixtures/rsi-benchmark/manifest.json", "utf8")) as BenchmarkManifest const corpusConfig = { ...config, generations: 3, maxIterations: 40, commandTimeoutMs: 900_000, evalCommands: ["npm test"], maxRuntimeMs: 72 * 60 * 60_000 } @@ -36,9 +39,9 @@ const corpusMutationLimits = { ...limits("mutation"), maxCalls: 40, maxInputToke const corpusBenchmarkLimits = { ...corpusMutationLimits, campaignId: "benchmark", maxCalls: 8, maxOutputTokensPerCall: 4096 } const corpusPlan = planProofDemonstration({ config: corpusConfig, manifest: corpus, mutationLimits: corpusMutationLimits, benchmarkLimits: corpusBenchmarkLimits }) assert.equal(corpusPlan.engineering.developmentProofCells, 384) -assert.equal(corpusPlan.engineering.confirmationDisposition, "pre-registered-deferred") +assert.equal(corpusPlan.engineering.confirmationDisposition, "not-frozen-until-engineering-passes") assert.equal(corpusPlan.engineering.paidConfirmationCells, 0) -assert.equal(corpusPlan.engineering.deferredConfirmationReservationCells, 64) +assert.equal(corpusPlan.engineering.deferredConfirmationReservationCells, 0) assert.equal(corpusPlan.campaign.developmentProofCells, 576) assert.equal(corpusPlan.campaign.confirmationScheduleCells, 224) assert.equal(corpusPlan.campaign.confirmationNotSelectedCells, 160) @@ -47,5 +50,5 @@ assert.equal(corpusPlan.campaign.rootControlCells, 32) assert.equal(corpusPlan.campaign.maximumCalls, 19_680) assert.equal(corpusPlan.totalCalls, 70_880) assert.equal(corpusPlan.totalReservedSpendUsd, 93.92) -assert.equal(corpusPlan.totalRootSpendCeilingUsd, 96.48, "operator cap includes the engineering root's frozen, deferred final/root confirmation reservation") +assert.equal(corpusPlan.totalRootSpendCeilingUsd, 93.92, "engineering root cap is reported separately from the three campaign roots") console.log("RSI proof demonstration schedule and ceiling assertions passed") diff --git a/src/rsi/__tests__/proof-demonstration.test.ts b/src/rsi/__tests__/proof-demonstration.test.ts new file mode 100644 index 0000000..c31edf1 --- /dev/null +++ b/src/rsi/__tests__/proof-demonstration.test.ts @@ -0,0 +1,188 @@ +import assert from "node:assert/strict" +import { createHash } from "node:crypto" +import * as fs from "node:fs/promises" +import * as os from "node:os" +import * as path from "node:path" +import test from "node:test" +import { analyzeProofDemonstration, verifyProofDemonstrationArtifactBytes } from "../proof-demonstration.js" +import { freezeProofDemonstrationManifest, frozenProofConfigDigest } from "../proof-demonstration-manifest.js" +import { proofCampaignManifestDigest, type ProofCampaignManifest } from "../proof-campaign.js" +import { RSI_OPENROUTER_MODEL } from "../inference-broker.js" +import type { BenchmarkAttempt } from "../benchmark/types.js" +import type { FrozenProofDemonstrationManifest } from "../proof-demonstration-manifest.js" +import type { CandidateRecord, RsiConfig, RsiRunRecord, WorkerHarnessRuntimeIdentity } from "../types.js" +import { formatAnalysisReport } from "../../../scripts/rsi-proof-demonstration.js" + +const sha256 = (value: Buffer | string) => createHash("sha256").update(value).digest("hex") + +function frozenPlan(): FrozenProofDemonstrationManifest { + const configs = ["engineering", "campaign-a", "campaign-b", "campaign-c"].map((seed) => ({ repoRoot: "/work", seed, mutationTask: "fixed task", protectedPaths: ["src/rsi"] } as RsiConfig)) as unknown as [RsiConfig, RsiConfig, RsiConfig, RsiConfig] + const runtime: WorkerHarnessRuntimeIdentity = { imageName: "headlesscode-openshell-rsi:fixed", imageDigest: `sha256:${"a".repeat(64)}`, harnessCommit: "b".repeat(40), dependencyLockDigest: "c".repeat(64) } + const tasks = ["dev-task", "confirm-task"].map((id, index) => ({ id, split: index ? "confirmation" : "development", family: "engine", seed: index + 1, startDigest: "d".repeat(64), verifierDigest: "e".repeat(64), knownGoodDigest: "f".repeat(64), knownBadDigest: "1".repeat(64), ceilings: { timeoutMs: 1000, maxCalls: 1, maxOutputTokensPerCall: 100, maxPatchBytes: 1000 } })) + const limits = { campaignId: "test", maxCalls: 1, maxInputTokensPerCall: 100, maxOutputTokensPerCall: 100, maxTotalTokens: 200, maxDurationMs: 1000, maxRequestBytes: 1000, maxResponseBytes: 1000, maxInputPricePerMillionUsd: 1, maxOutputPricePerMillionUsd: 1, maxCallSpendUsd: 0.01, maxJobSpendUsd: 0.02, maxCampaignSpendUsd: 1 } + const plan = { totalRootSpendCeilingUsd: 1, engineeringRootSpendCeilingUsd: 0.25, campaignRootsSpendCeilingUsd: 0.75, totalCalls: 1, totalTokens: 1, totalTaskWallMs: 1000 } as never + const args = { + createdAt: new Date(0).toISOString(), rootCommit: "2".repeat(40), benchmarkManifestSha256: "3".repeat(64), + model: { provider: "openshell-openrouter" as const, requestedId: RSI_OPENROUTER_MODEL, sampling: { temperature: 0 as const, think: false as const, seed: "provider-default" as const } }, + endpointPrice: { model: RSI_OPENROUTER_MODEL, provider: "deepinfra" as const, endpointTag: "deepinfra/fp8" as const, quantization: "fp8" as const, supportedParameters: ["max_tokens", "temperature", "tools", "tool_choice"], inputPricePerMillionUsd: 0.06, outputPricePerMillionUsd: 0.18, observedAt: new Date(0).toISOString(), responseSha256: "4".repeat(64) }, + priceCeilingsPerMillionUsd: { input: 1, output: 1 }, runtime, + execution: { configs, configSha256: frozenProofConfigDigest(configs), analysisSourceSha256: "5".repeat(64), inferenceLimits: { mutation: limits, benchmark: limits }, taskIdentities: tasks }, + seeds: { engineering: "engineering", campaigns: ["campaign-a", "campaign-b", "campaign-c"] as [string, string, string] }, + campaignIds: { engineering: "eng-id", campaigns: ["campaign-a-id", "campaign-b-id", "campaign-c-id"] as [string, string, string] }, operatorCapUsd: 2, plan, + engineeringEvidence: { path: "/work/.headlesscode/rsi-proof/engineering.json", sha256: "9".repeat(64), campaignId: "eng-id", rootCommit: "2".repeat(40), manifestSha256: "a".repeat(64), snapshotSha256: "b".repeat(64) }, + analysis: { bootstrapSeed: "fixed-bootstrap", bootstrapReplicates: 100, confirmationRetentionTaskIds: ["confirm-task"] }, + } + return freezeProofDemonstrationManifest(args) +} + +test("analysis is inconclusive for missing campaign records and duplicate arm runs", () => { + const plan = frozenPlan() + const empty = analyzeProofDemonstration([], { frozenPlan: plan }) + assert.equal(empty.verdict, "inconclusive") + assert.ok(empty.denominators.missingCells.some((cell) => cell.includes("missing-arm"))) + assert.match(formatAnalysisReport(empty, plan.digestSha256), /Missing cells:/) + const armManifest = { + campaignId: plan.campaignIds.campaigns[0], rootCommit: plan.rootCommit, + benchmarkDigests: { manifest: plan.benchmarkManifestSha256, development: "6".repeat(64), confirmation: "7".repeat(64) }, + model: { provider: plan.model.provider, id: plan.model.requestedId, settings: plan.model.sampling }, seed: plan.seeds.campaigns[0], taskOrder: ["confirm-task"], + schedule: [{ armId: "self-hosted", generation: 0, candidateId: "root-control", taskId: "confirm-task", seed: "2", attemptKind: "fixed-control", phase: "confirmation" }], + budgets: {}, acceptanceRule: {}, runtime: plan.runtime, analysisVersion: "test", settings: {}, + } + const duplicateArm = (runId: string) => ({ runId, startedAt: new Date(0).toISOString(), model: plan.model.requestedId, baseRef: "main", baseCommit: plan.rootCommit, generations: 1, candidates: [], reports: [], workerHarnessPairId: plan.campaignIds.campaigns[0], workerHarnessMode: "self-hosted", proofCampaign: { manifest: armManifest, manifestSha256: "8".repeat(64), campaignId: armManifest.campaignId, armId: "self-hosted", epoch: 1, state: "finished" } }) as unknown as RsiRunRecord + const duplicate = analyzeProofDemonstration([duplicateArm("dup-1"), duplicateArm("dup-2")], { frozenPlan: plan }) + assert.equal(duplicate.verdict, "inconclusive") + assert.ok(duplicate.denominators.missingCells.includes(plan.campaignIds.campaigns[0] + ":duplicate-arm-runs")) +}) + +function qualifyingRuns(plan: FrozenProofDemonstrationManifest): RsiRunRecord[] { + const runs: RsiRunRecord[] = [] + const task = plan.execution.taskIdentities.find((entry) => entry.split === "confirmation")! + for (let index = 0; index < 3; index++) { + const campaignId = plan.campaignIds.campaigns[index]! + const seed = plan.seeds.campaigns[index]! + const gen0Commit = (10 + index).toString(16).repeat(40).slice(0, 40) + const gen1Commit = (20 + index).toString(16).repeat(40).slice(0, 40) + const gen0Id = `gen0-${index}` + const gen1Id = `gen1-${index}` + const manifestFor = (armId: "self-hosted" | "fixed-control") => { + const schedule = (["self-hosted", "fixed-control"] as const).flatMap((scheduledArm) => [ + { armId: scheduledArm, generation: 2, candidateId: "g2-candidate-0", taskId: task.id, seed: String(task.seed), attemptKind: "confirmation", phase: "confirmation" }, + { armId: scheduledArm, generation: 2, candidateId: "root-control", taskId: task.id, seed: String(task.seed), attemptKind: "fixed-control", phase: "confirmation" }, + ]) + const raw = { + schemaVersion: 1, campaignId, rootCommit: plan.rootCommit, + benchmarkDigests: { manifest: plan.benchmarkManifestSha256, development: "6".repeat(64), confirmation: "7".repeat(64) }, + armAssignment: { fixedControl: "fixed-control", selfHosted: "self-hosted" }, + model: { provider: plan.model.provider, id: plan.model.requestedId, settings: { sampling: plan.model.sampling } }, seed, + taskOrder: [task.id], schedule, budgets: { attempt: { maxCalls: 1, maxTokens: 200, maxSpendUsd: 0.02, maxDurationMs: 1000 }, root: { maxCalls: 100, maxTokens: 10000, maxSpendUsd: 1, maxRuntimeMs: 10_000_000 } }, + acceptanceRule: { confirmationRetentionTaskIds: [task.id] }, runtime: { imageDigest: plan.runtime.imageDigest, harnessCommit: plan.runtime.harnessCommit, dependencyLockDigest: plan.runtime.dependencyLockDigest }, analysisVersion: "task-cluster-bootstrap-90-v1", + phaseMode: "paired-confirmation", settings: { effectiveConfig: plan.execution.configs[index + 1], confirmationTaskIdentities: [{ id: task.id, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings }], candidateCounts: [1, 1, 1], inferenceCeilings: { maxInputTokensPerCall: 100, maxTotalTokens: 200, maxJobSpendUsd: 0.02 } }, + } + return raw as unknown as ProofCampaignManifest + } + for (const armId of ["self-hosted", "fixed-control"] as const) { + const armManifest = manifestFor(armId) + const selectedId = armId === "self-hosted" ? gen1Id : "fixed-final" + const selectedCommit = armId === "self-hosted" ? gen1Commit : "3".repeat(40) + const finalAttemptId = `${campaignId}-${armId}-final` + const rootAttemptId = `${campaignId}-${armId}-root` + const makeAttempt = (attemptId: string, candidateId: string, kind: string, commit: string, success: boolean) => ({ + attemptId, taskId: task.id, split: "confirmation", harnessCommit: commit, model: { provider: plan.model.provider, id: plan.model.requestedId, sampling: plan.model.sampling }, seed: task.seed, + startDigest: task.startDigest, evaluatorDigest: task.verifierDigest, ceilings: task.ceilings, validity: "valid", proofQualified: true, resourceAccounting: "supervisor-enforced", verifiedSuccess: success, + calls: 1, tokens: 20, wallTimeMs: 10, patchDigest: "8".repeat(64), artifactRefs: {}, stopReason: "completed", + inferenceReceipt: { qualified: true, model: plan.model.requestedId, returnedModels: [plan.model.requestedId], endpointProvider: "deepinfra", returnedProviders: ["deepinfra"], calls: 1, totalTokens: 20, costUsd: 0.001 }, + }) + const finalAttempt = makeAttempt(finalAttemptId, armId === "self-hosted" ? "g2-candidate-0" : "g2-candidate-0", "confirmation", selectedCommit, armId === "self-hosted") as unknown as BenchmarkAttempt + const rootAttempt = makeAttempt(rootAttemptId, "root-control", "fixed-control", plan.rootCommit, false) as unknown as BenchmarkAttempt + const attemptRows = [ + { broker_job_id: finalAttemptId, arm_id: armId, generation: 2, candidate_id: "g2-candidate-0", task_id: task.id, seed: String(task.seed), attempt_kind: "confirmation", status: "completed", result_sha256: sha256(JSON.stringify(finalAttempt)), artifact_refs: {} }, + { broker_job_id: rootAttemptId, arm_id: armId, generation: 2, candidate_id: "root-control", task_id: task.id, seed: String(task.seed), attempt_kind: "fixed-control", status: "completed", result_sha256: sha256(JSON.stringify(rootAttempt)), artifact_refs: {} }, + ] + const slot0 = "g0-candidate-0" + const slot1 = "g1-candidate-0" + const selectedSlot = "g2-candidate-0" + const events = [0, 1, 2].map((generation) => ({ arm_id: armId, event_type: "generation-selected", payload: { generation, selectedCandidateIds: [generation === 0 ? slot0 : generation === 1 ? slot1 : selectedSlot] } })) + const decisions = [ + { arm_id: armId, generation: 0, candidate_id: slot0, decision: "accepted", child_commit: gen0Commit, parent_commit: plan.rootCommit, parent_id: "baseline" }, + { arm_id: armId, generation: 1, candidate_id: slot1, decision: "accepted", child_commit: gen1Commit, parent_commit: gen0Commit, parent_id: slot0 }, + { arm_id: armId, generation: 2, candidate_id: selectedSlot, decision: "accepted", child_commit: selectedCommit, parent_commit: gen1Commit, parent_id: slot1 }, + ] + const candidate0 = { id: gen0Id, generation: 0, status: "accepted", parent: "baseline", parentCommit: plan.rootCommit, baseCommit: plan.rootCommit, commits: [gen0Commit], developmentProof: { accepted: true, immediateParent: { comparable: true, eligible: true } } } as unknown as CandidateRecord + const candidate1 = { id: gen1Id, generation: 1, status: "accepted", parent: gen0Id, parentCommit: gen0Commit, baseCommit: gen0Commit, commits: [gen1Commit], workerHarnessExecution: { parentCommit: gen0Commit }, developmentProof: { accepted: true, immediateParent: { comparable: true, eligible: true } } } as unknown as CandidateRecord + const candidate2 = { id: selectedId, generation: 2, status: "accepted", parent: gen1Id, parentCommit: gen1Commit, baseCommit: gen1Commit, commits: [selectedCommit], workerHarnessExecution: { parentCommit: gen1Commit }, developmentProof: { accepted: true, immediateParent: { comparable: true, eligible: true } } } as unknown as CandidateRecord + const confirmationAttemptList = [finalAttempt, rootAttempt] + const run = { + runId: `${campaignId}-${armId}`, startedAt: "2026-09-30T00:00:00Z", finishedAt: "2026-09-30T00:01:00Z", model: plan.model.requestedId, baseRef: "main", baseCommit: plan.rootCommit, generations: 3, selected: selectedId, + workerHarnessMode: armId, workerHarnessPairId: campaignId, workerHarnessSamplingSeed: seed, workerHarnessRuntime: plan.runtime, workerHarnessConfigDigest: "9".repeat(64), confirmationEvidence: "complete", + candidates: [candidate0, candidate1, candidate2], reports: [], confirmationAttempts: confirmationAttemptList, + proofConfirmation: { selectedHarness: { candidateId: selectedId, slotId: selectedSlot, generation: 2, commit: selectedCommit, attemptIds: [finalAttemptId] }, rootHarness: { commit: plan.rootCommit, attemptIds: [rootAttemptId] } }, + proofCampaign: { campaignId, armId, manifest: armManifest, manifestSha256: proofCampaignManifestDigest(armManifest), epoch: 1, state: "finished", ledgerSnapshot: { + campaign: { state: "finished", manifest_sha256: proofCampaignManifestDigest(armManifest), used_calls: 4, used_tokens: 80, used_spend_microusd: 4_000, reserved_calls: 0, reserved_tokens: 0, reserved_spend_microusd: 0, deadline_at: "2026-09-30T01:00:00Z" }, + arms: [{ arm_id: armId, state: "finished" }, { arm_id: armId === "self-hosted" ? "fixed-control" : "self-hosted", state: "finished" }], attempts: attemptRows, events, decisions, + } }, + } as unknown as RsiRunRecord + runs.push(run) + } + } + return runs +} + +test("a complete synthetic paired record can demonstrate, while provider, config, cells, and lineage tampering fail closed", () => { + const plan = frozenPlan() + const records = qualifyingRuns(plan) + const report = analyzeProofDemonstration(records, { frozenPlan: plan }) + assert.equal(report.verdict, "demonstrated") + assert.ok(report.pooled.cluster90PercentInterval?.[0]! > 0) + const wrongProvider = structuredClone(records) + wrongProvider[0]!.confirmationAttempts![0]!.inferenceReceipt!.returnedProviders = ["other"] + assert.notEqual(analyzeProofDemonstration(wrongProvider, { frozenPlan: plan }).verdict, "demonstrated") + const wrongConfig = structuredClone(records) + ;(wrongConfig[0]!.proofCampaign!.manifest.settings.effectiveConfig as Record).mutationTask = "different task" + wrongConfig[0]!.proofCampaign!.manifestSha256 = proofCampaignManifestDigest(wrongConfig[0]!.proofCampaign!.manifest) + wrongConfig[1]!.proofCampaign!.manifest = structuredClone(wrongConfig[0]!.proofCampaign!.manifest) + wrongConfig[1]!.proofCampaign!.manifestSha256 = wrongConfig[0]!.proofCampaign!.manifestSha256 + assert.notEqual(analyzeProofDemonstration(wrongConfig, { frozenPlan: plan }).verdict, "demonstrated") + const missingCell = structuredClone(records) + missingCell[0]!.proofCampaign!.ledgerSnapshot!.attempts = (missingCell[0]!.proofCampaign!.ledgerSnapshot!.attempts as unknown[]).slice(1) + assert.notEqual(analyzeProofDemonstration(missingCell, { frozenPlan: plan }).verdict, "demonstrated") + const wrongLineage = structuredClone(records) + for (const run of wrongLineage.filter((entry) => entry.workerHarnessMode === "self-hosted")) run.candidates.find((candidate) => candidate.generation === 1)!.parentCommit = "f".repeat(40) + assert.notEqual(analyzeProofDemonstration(wrongLineage, { frozenPlan: plan }).verdict, "demonstrated") +}) + +test("confirmation analysis artifact verifier checks referenced bytes and rejects changed files", async () => { + const archiveDir = await fs.mkdtemp(path.join(os.tmpdir(), "proof-demonstration-artifacts-")) + try { + const campaignId = "campaign-artifact-check" + const armId = "self-hosted" + const attemptId = "attempt-confirmation-1" + const commit = "a".repeat(40) + const stdout = Buffer.from("verified output\n") + const stderr = Buffer.from("\u0000binary\u00ff\n", "latin1") + const refs = { stdout: sha256(stdout), stderr: sha256(stderr) } + const attempt = { attemptId, harnessCommit: commit, artifactRefs: refs } as BenchmarkAttempt + const row = { + broker_job_id: attemptId, arm_id: armId, status: "completed", result_sha256: sha256(JSON.stringify(attempt)), + candidate_id: "g0-candidate-0", attempt_kind: "confirmation", + artifact_refs: { stdout: { sha256: refs.stdout }, stderr: { sha256: refs.stderr } }, + } + const outputRoot = path.join(archiveDir, "confirmation", campaignId, armId, "g0-candidate-0", "confirmation", "harness", commit, "artifacts", "sha256") + for (const [name, bytes] of [["stdout", stdout], ["stderr", stderr]] as const) { + const digest = refs[name] + const file = path.join(outputRoot, digest.slice(0, 2), digest) + await fs.mkdir(path.dirname(file), { recursive: true }) + await fs.writeFile(file, bytes) + } + const run = { + runId: "run-artifact-check", startedAt: new Date(0).toISOString(), model: "test/model", baseRef: "main", baseCommit: "b".repeat(40), generations: 1, candidates: [], reports: [], + proofCampaign: { campaignId, armId, manifest: {}, manifestSha256: "c".repeat(64), epoch: 1, state: "finished", ledgerSnapshot: { attempts: [row] } }, + proofConfirmation: { selectedHarness: { attemptIds: [attemptId] }, rootHarness: { attemptIds: [] } }, + confirmationAttempts: [attempt], + } as unknown as RsiRunRecord + await verifyProofDemonstrationArtifactBytes([run], archiveDir) + await fs.writeFile(path.join(outputRoot, refs.stdout.slice(0, 2), refs.stdout), "changed\n") + await assert.rejects(() => verifyProofDemonstrationArtifactBytes([run], archiveDir), /missing or changed/) + } finally { + await fs.rm(archiveDir, { recursive: true, force: true }) + } +}) diff --git a/src/rsi/__tests__/worker-harness-controller.test.ts b/src/rsi/__tests__/worker-harness-controller.test.ts index 3cf5573..36707d6 100644 --- a/src/rsi/__tests__/worker-harness-controller.test.ts +++ b/src/rsi/__tests__/worker-harness-controller.test.ts @@ -40,6 +40,16 @@ async function main(): Promise { workerHarnessMode: "self-hosted", workerHarnessRuntime: runtime, } + await assert.rejects( + runRsi({ ...config, workerHarnessPairId: `rsi-proof-${"a".repeat(20)}`, workerHarnessRunId: `rsi-proof-${"b".repeat(20)}-self` }, { log: () => undefined }), + /deterministic worker harness run ID must match its paired campaign and arm/, + "a deterministic run identity cannot be rebound across campaign IDs", + ) + await assert.rejects( + runRsi({ ...config, workerHarnessPairId: `rsi-proof-${"a".repeat(20)}`, workerHarnessRunId: `rsi-proof-${"a".repeat(20)}-fixed` }, { log: () => undefined }), + /deterministic worker harness run ID must match its paired campaign and arm/, + "a fixed-control run cannot use the self-hosted deterministic suffix", + ) const observed: Array<{ generation: number; mode: string; parentCommit: string; digest: string }> = [] const run = await runRsi(config, { log: () => undefined, diff --git a/src/rsi/benchmark/__tests__/openshell-integration.test.ts b/src/rsi/benchmark/__tests__/openshell-integration.test.ts index 2495158..bc6b5fc 100644 --- a/src/rsi/benchmark/__tests__/openshell-integration.test.ts +++ b/src/rsi/benchmark/__tests__/openshell-integration.test.ts @@ -152,10 +152,10 @@ try { const brokerFactory = async (options: Parameters[0]) => { const broker = await startOpenRouterBroker({ ...options, upstreamFetch: async (input, init) => { const url = String(input) - if (url.endsWith(`/models/${RSI_OPENROUTER_MODEL}/endpoints`)) return new Response(JSON.stringify({ data: { endpoints: [{ provider_slug: "deepseek", pricing: { prompt: "0.000001", completion: "0.000001" } }] } }), { status: 200, headers: { "content-type": "application/json" } }) + if (url.endsWith(`/models/${RSI_OPENROUTER_MODEL}/endpoints`)) return new Response(JSON.stringify({ data: { endpoints: [{ provider_name: "DeepInfra", tag: "deepinfra/fp8", quantization: "fp8", status: 0, supported_parameters: ["max_tokens", "temperature", "tools", "tool_choice"], supports_tool_choice: { "auto": true, "required": true }, pricing: { prompt: "0.00000006", completion: "0.00000018" } }] } }), { status: 200, headers: { "content-type": "application/json" } }) assert.equal(url, "https://openrouter.ai/api/v1/chat/completions") assert.equal(new Headers(init?.headers).get("authorization"), "Bearer host-test-key-never-forwarded-to-guest") - return new Response(JSON.stringify({ id: "mock-provider-response", model: RSI_OPENROUTER_MODEL, choices: [{ message: { role: "assistant", content: "done" } }], usage: { prompt_tokens: 12, completion_tokens: 4, cost: 0.000016 } }), { status: 200, headers: { "content-type": "application/json" } }) + return new Response(JSON.stringify({ id: "mock-provider-response", model: RSI_OPENROUTER_MODEL, openrouter_metadata: { endpoints: { available: [{ provider: "deepinfra", selected: true }] } }, choices: [{ message: { role: "assistant", content: "done" } }], usage: { prompt_tokens: 12, completion_tokens: 4, cost: 0.00000144 } }), { status: 200, headers: { "content-type": "application/json" } }) } }) brokerBaseUrl = broker.localBaseUrl lastCapability = broker.capability diff --git a/src/rsi/config.ts b/src/rsi/config.ts index befe9ce..dfee3dd 100644 --- a/src/rsi/config.ts +++ b/src/rsi/config.ts @@ -126,6 +126,7 @@ export function parseRsiArgs(argv: string[], cwd = process.cwd()): ParsedRsiArgs let workerHarnessMode: WorkerHarnessMode | undefined let workerHarnessPhase: "development" | "confirmation" = "development" let workerHarnessPairId: string | undefined + let workerHarnessEngineeringOnly = false let proofMinimumNetWins = DEFAULT_DEVELOPMENT_PROOF_RULE.minimumNetWins let proofRetentionTaskIds = [...DEFAULT_DEVELOPMENT_PROOF_RULE.retentionTaskIds] let proofRetentionWasConfigured = false @@ -298,6 +299,9 @@ export function parseRsiArgs(argv: string[], cwd = process.cwd()): ParsedRsiArgs index = next break } + case "--worker-harness-engineering-only": + workerHarnessEngineeringOnly = true + break case "--proof-min-net-wins": { const [value, next] = take(index, arg) proofMinimumNetWins = positiveInteger(value, arg) @@ -378,6 +382,8 @@ export function parseRsiArgs(argv: string[], cwd = process.cwd()): ParsedRsiArgs throw new Error(`unknown improve argument: ${arg}`) } } + if (workerHarnessEngineeringOnly && (!workerHarnessMode || !workerHarnessPairId)) throw new Error("--worker-harness-engineering-only requires an explicit paired campaign ID and arm") + if (workerHarnessEngineeringOnly && workerHarnessPhase !== "development") throw new Error("engineering-only worker harness runs support only the development phase") const resolvedRepo = path.resolve(repoRoot) const adaptiveTrajectories = maxTrajectories ?? population + 1 const adaptiveIterations = maxTotalIterations ?? maxIterations * adaptiveTrajectories @@ -428,6 +434,7 @@ export function parseRsiArgs(argv: string[], cwd = process.cwd()): ParsedRsiArgs ...(workerHarnessMode ? { workerHarnessMode } : {}), ...(workerHarnessMode ? { workerHarnessPhase } : {}), ...(workerHarnessPairId ? { workerHarnessPairId } : {}), + ...(workerHarnessPairId && workerHarnessEngineeringOnly ? { workerHarnessEngineeringOnly: true } : {}), }, } } catch (error) { diff --git a/src/rsi/controller.ts b/src/rsi/controller.ts index 2e4d8d6..2928bdd 100644 --- a/src/rsi/controller.ts +++ b/src/rsi/controller.ts @@ -77,12 +77,14 @@ export function createProofCampaignManifest(args: { plannedGenerations: number proofProtocol: NonNullable }): ProofCampaignManifest { + const engineeringOnly = args.config.workerHarnessEngineeringOnly === true + if (engineeringOnly && args.config.workerHarnessPhase === "confirmation") throw new Error("engineering-only proof runs cannot enter confirmation") if (args.provider !== "openshell-openrouter") throw new Error("restart-safe proof campaigns require supervisor-accounted OpenRouter inference") if (args.config.computePolicy === "adaptive-independent") throw new Error("restart-safe proof campaigns require every frozen generation to run; adaptive early-stop is not admitted") const benchmarkTasks = args.developmentManifest.tasks.filter((task) => task.split === "development") const confirmationTasks = args.developmentManifest.tasks.filter((task) => task.split === "confirmation") - if (!benchmarkTasks.length || !confirmationTasks.length) throw new Error("proof campaign requires nonempty development and confirmation benchmark splits") - if (RSI_CONFIRMATION_RETENTION_TASK_IDS.some((taskId) => !confirmationTasks.some((task) => task.id === taskId))) throw new Error("frozen benchmark manifest is missing a preregistered confirmation retention task") + if (!benchmarkTasks.length || (!engineeringOnly && !confirmationTasks.length)) throw new Error("proof campaign requires nonempty development tasks and, unless engineering-only, confirmation tasks") + if (!engineeringOnly && RSI_CONFIRMATION_RETENTION_TASK_IDS.some((taskId) => !confirmationTasks.some((task) => task.id === taskId))) throw new Error("frozen benchmark manifest is missing a preregistered confirmation retention task") const taskDigest = (tasks: typeof args.developmentManifest.tasks) => createHash("sha256").update(JSON.stringify(tasks)).digest("hex") const mutationCampaignId = proofBrokerCampaignId(args.campaignId, "mutation") const benchmarkCampaignId = proofBrokerCampaignId(args.campaignId, "benchmark") @@ -100,7 +102,7 @@ export function createProofCampaignManifest(args: { } if (jobCostCeiling(mutationLimits, mutationLimits.maxCalls, mutationLimits.maxOutputTokensPerCall) > mutationLimits.maxJobSpendUsd) throw new Error("OpenRouter mutation job spend cap is below its admitted token and price ceilings") for (const task of benchmarkTasks) if (jobCostCeiling(benchmarkLimits, task.ceilings.maxCalls, task.ceilings.maxOutputTokensPerCall) > benchmarkLimits.maxJobSpendUsd) throw new Error(`OpenRouter benchmark job spend cap is below the frozen token and price ceiling for ${task.id}`) - for (const task of confirmationTasks) if (jobCostCeiling(confirmationLimits, task.ceilings.maxCalls, task.ceilings.maxOutputTokensPerCall) > confirmationLimits.maxJobSpendUsd) throw new Error(`OpenRouter confirmation job spend cap is below the frozen token and price ceiling for ${task.id}`) + if (!engineeringOnly) for (const task of confirmationTasks) if (jobCostCeiling(confirmationLimits, task.ceilings.maxCalls, task.ceilings.maxOutputTokensPerCall) > confirmationLimits.maxJobSpendUsd) throw new Error(`OpenRouter confirmation job spend cap is below the frozen token and price ceiling for ${task.id}`) const candidateCounts = Array.from({ length: args.plannedGenerations }, (_, generation) => args.config.computePolicy === "adaptive-independent" && generation > 0 ? 1 : args.config.population) const schedule: ProofCampaignManifest["schedule"] = [] for (const armId of ["fixed-control", "self-hosted"]) for (let generation = 0; generation < args.plannedGenerations; generation++) { @@ -114,19 +116,19 @@ export function createProofCampaignManifest(args: { } } const finalGeneration = args.plannedGenerations - 1 - for (const armId of ["fixed-control", "self-hosted"]) { + if (!engineeringOnly) for (const armId of ["fixed-control", "self-hosted"]) { for (let generation = 0; generation < args.plannedGenerations; generation++) { for (let index = 0; index < candidateCounts[generation]!; index++) for (const task of confirmationTasks) schedule.push({ armId, generation, candidateId: proofCandidateSlot(generation, index), taskId: task.id, seed: String(task.seed), attemptKind: "confirmation", phase: "confirmation" }) } for (const task of confirmationTasks) schedule.push({ armId, generation: finalGeneration, candidateId: "root-control", taskId: task.id, seed: String(task.seed), attemptKind: "fixed-control", phase: "confirmation" }) } const developmentCallsPerProofRole = benchmarkTasks.reduce((sum, task) => sum + task.ceilings.maxCalls, 0) - const confirmationCalls = 4 * confirmationTasks.reduce((sum, task) => sum + task.ceilings.maxCalls, 0) // selected final and root-control, for both arms + const confirmationCalls = engineeringOnly ? 0 : 4 * confirmationTasks.reduce((sum, task) => sum + task.ceilings.maxCalls, 0) // selected final and root-control, for both arms const rootCalls = 2 * candidateCounts.reduce((sum, count) => sum + count * (mutationLimits.maxCalls + 3 * developmentCallsPerProofRole), 0) + confirmationCalls const developmentTokensPerProofRole = benchmarkTasks.reduce((sum, task) => sum + Math.min(benchmarkLimits.maxTotalTokens, task.ceilings.maxCalls * (benchmarkLimits.maxInputTokensPerCall + task.ceilings.maxOutputTokensPerCall)), 0) const confirmationTokensPerHarness = confirmationTasks.reduce((sum, task) => sum + Math.min(confirmationLimits.maxTotalTokens, task.ceilings.maxCalls * (confirmationLimits.maxInputTokensPerCall + task.ceilings.maxOutputTokensPerCall)), 0) const mutationTokens = Math.min(mutationLimits.maxTotalTokens, mutationLimits.maxCalls * (mutationLimits.maxInputTokensPerCall + mutationLimits.maxOutputTokensPerCall)) - const rootTokens = 2 * candidateCounts.reduce((sum, count) => sum + count * (mutationTokens + 3 * developmentTokensPerProofRole), 0) + 4 * confirmationTokensPerHarness + const rootTokens = 2 * candidateCounts.reduce((sum, count) => sum + count * (mutationTokens + 3 * developmentTokensPerProofRole), 0) + (engineeringOnly ? 0 : 4 * confirmationTokensPerHarness) if (!Number.isSafeInteger(rootCalls) || !Number.isSafeInteger(rootTokens)) throw new Error("proof campaign root call or token ceiling exceeds safe integer accounting") const spendAllocation = Number(process.env.HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD) if (!Number.isFinite(spendAllocation) || spendAllocation <= 0) throw new Error("HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD is required for a bounded proof campaign") @@ -135,26 +137,26 @@ export function createProofCampaignManifest(args: { const slotsPerArm = candidateCounts.reduce((sum, count) => sum + count, 0) const mutationReservationsUsd = 2 * slotsPerArm * mutationLimits.maxJobSpendUsd const developmentReservationsUsd = 2 * slotsPerArm * 3 * benchmarkTasks.length * benchmarkLimits.maxJobSpendUsd - const confirmationReservationsUsd = 4 * confirmationTasks.length * confirmationLimits.maxJobSpendUsd + const confirmationReservationsUsd = engineeringOnly ? 0 : 4 * confirmationTasks.length * confirmationLimits.maxJobSpendUsd const requiredBenchmarkUsd = developmentReservationsUsd + confirmationReservationsUsd if (openRouterBrokerLimitsFromEnvironment(openRouterAllocatedEnvironment(process.env, "mutation"), mutationCampaignId).maxCampaignSpendUsd < mutationReservationsUsd) throw new Error("mutation broker campaign allocation is below the sum of frozen mutation job reservations") - if (benchmarkLimits.maxCampaignSpendUsd < requiredBenchmarkUsd) throw new Error("benchmark broker campaign allocation is below development plus selected-final/root confirmation job reservations") + if (benchmarkLimits.maxCampaignSpendUsd < requiredBenchmarkUsd) throw new Error(engineeringOnly ? "benchmark broker campaign allocation is below development proof job reservations" : "benchmark broker campaign allocation is below development plus selected-final/root confirmation job reservations") if (totalBudgetUsd < mutationReservationsUsd + requiredBenchmarkUsd) throw new Error("proof campaign role allocations cannot fund every frozen provider job reservation") - const taskWallBoundMs = proofCampaignTaskWallBoundMs(args.config, args.plannedGenerations, benchmarkTasks, confirmationTasks) + const taskWallBoundMs = proofCampaignTaskWallBoundMs(args.config, args.plannedGenerations, benchmarkTasks, confirmationTasks, !engineeringOnly) const maxRuntimeMs = args.config.maxRuntimeMs ?? 60 * 60_000 if (maxRuntimeMs < taskWallBoundMs) throw new Error(`proof campaign runtime ceiling ${maxRuntimeMs}ms is below the summed frozen task-time ceiling ${taskWallBoundMs}ms`) return { - schemaVersion: 1, campaignId: args.campaignId, rootCommit: args.baseCommit, - benchmarkDigests: { manifest: args.developmentManifestDigest, development: taskDigest(benchmarkTasks), confirmation: taskDigest(confirmationTasks) }, + schemaVersion: 1, phaseMode: engineeringOnly ? "engineering-only" : "paired-confirmation", campaignId: args.campaignId, rootCommit: args.baseCommit, + benchmarkDigests: { manifest: args.developmentManifestDigest, development: taskDigest(benchmarkTasks), confirmation: taskDigest(engineeringOnly ? [] : confirmationTasks) }, armAssignment: { fixedControl: "fixed-control", selfHosted: "self-hosted" }, model: { provider: args.provider, id: args.model, settings: { sampling: args.proofProtocol.model.sampling, roles: args.config.roles ?? {}, maxIterations: args.config.maxIterations, maxTokens: args.config.maxTokens ?? null } }, - seed: args.config.seed, taskOrder: ["mutation", "regression", "typecheck", ...args.config.evalCommands.map((_, index) => `visible-${index}`), ...benchmarkTasks.map((task) => task.id), ...confirmationTasks.map((task) => task.id)], schedule, + seed: args.config.seed, taskOrder: ["mutation", "regression", "typecheck", ...args.config.evalCommands.map((_, index) => `visible-${index}`), ...benchmarkTasks.map((task) => task.id), ...(engineeringOnly ? [] : confirmationTasks.map((task) => task.id))], schedule, budgets: { - attempt: { maxCalls: Math.max(mutationLimits.maxCalls, benchmarkLimits.maxCalls, confirmationLimits.maxCalls), maxTokens: Math.max(mutationLimits.maxTotalTokens, benchmarkLimits.maxTotalTokens, confirmationLimits.maxTotalTokens), maxSpendUsd: Math.max(mutationLimits.maxJobSpendUsd, benchmarkLimits.maxJobSpendUsd, confirmationLimits.maxJobSpendUsd), maxDurationMs: Math.max(args.config.commandTimeoutMs, ...benchmarkTasks.map((task) => task.ceilings.timeoutMs), ...confirmationTasks.map((task) => task.ceilings.timeoutMs)) }, + attempt: { maxCalls: Math.max(mutationLimits.maxCalls, benchmarkLimits.maxCalls, ...(!engineeringOnly ? [confirmationLimits.maxCalls] : [])), maxTokens: Math.max(mutationLimits.maxTotalTokens, benchmarkLimits.maxTotalTokens, ...(!engineeringOnly ? [confirmationLimits.maxTotalTokens] : [])), maxSpendUsd: Math.max(mutationLimits.maxJobSpendUsd, benchmarkLimits.maxJobSpendUsd, ...(!engineeringOnly ? [confirmationLimits.maxJobSpendUsd] : [])), maxDurationMs: Math.max(args.config.commandTimeoutMs, ...benchmarkTasks.map((task) => task.ceilings.timeoutMs), ...(!engineeringOnly ? confirmationTasks.map((task) => task.ceilings.timeoutMs) : [])) }, root: { maxCalls: rootCalls, maxTokens: rootTokens, maxSpendUsd: Math.min(spendAllocation, totalBudgetUsd, mutationReservationsUsd + requiredBenchmarkUsd), maxRuntimeMs }, }, - acceptanceRule: { ...args.proofProtocol.rule, confirmationRetentionTaskIds: [...RSI_CONFIRMATION_RETENTION_TASK_IDS], minimumSelfHostedCampaigns: 2, requiredSuccessiveTransitions: 2, clusterConfidence: 0.9 }, runtime: { imageDigest: args.runtime.imageDigest, harnessCommit: args.runtime.harnessCommit, dependencyLockDigest: args.runtime.dependencyLockDigest }, - analysisVersion: args.proofProtocol.schemaVersion.toString(), settings: { configDigest: workerHarnessConfigDigest(args.config, { baseCommit: args.baseCommit, model: args.model, provider: args.provider, proofProtocolDigest: args.proofProtocol.protocolDigest }), candidateCounts, taskWallBoundMs, evaluationCells: ["regression", "typecheck", ...args.config.evalCommands.map((_, index) => `visible-${index}`)], developmentTaskIdentities: benchmarkTasks.map((task) => ({ id: task.id, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings })), confirmationTaskIdentities: confirmationTasks.map((task) => ({ id: task.id, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings })), inferenceCeilings: { maxInputTokensPerCall: benchmarkLimits.maxInputTokensPerCall, maxOutputTokensPerCall: benchmarkLimits.maxOutputTokensPerCall, maxCallsPerAttempt: benchmarkLimits.maxCalls, maxTotalTokens: benchmarkLimits.maxTotalTokens, maxCallSpendUsd: benchmarkLimits.maxCallSpendUsd, maxJobSpendUsd: benchmarkLimits.maxJobSpendUsd, maxInputPricePerMillionUsd: benchmarkLimits.maxInputPricePerMillionUsd, maxOutputPricePerMillionUsd: benchmarkLimits.maxOutputPricePerMillionUsd }, inferenceCampaignIds: { mutation: mutationCampaignId, benchmark: benchmarkCampaignId }, inferenceAllocationsUsd: { mutation: mutationLimits.maxCampaignSpendUsd, benchmark: benchmarkLimits.maxCampaignSpendUsd } }, + acceptanceRule: { ...args.proofProtocol.rule, confirmationRetentionTaskIds: engineeringOnly ? [] : [...RSI_CONFIRMATION_RETENTION_TASK_IDS], minimumSelfHostedCampaigns: 2, requiredSuccessiveTransitions: 2, clusterConfidence: 0.9 }, runtime: { imageDigest: args.runtime.imageDigest, harnessCommit: args.runtime.harnessCommit, dependencyLockDigest: args.runtime.dependencyLockDigest }, + analysisVersion: args.proofProtocol.schemaVersion.toString(), settings: { phaseMode: engineeringOnly ? "engineering-only" : "paired-confirmation", configDigest: workerHarnessConfigDigest(args.config, { baseCommit: args.baseCommit, model: args.model, provider: args.provider, proofProtocolDigest: args.proofProtocol.protocolDigest }), effectiveConfig: Object.fromEntries(Object.entries(args.config).filter(([key]) => !["workerHarnessMode", "workerHarnessPhase", "workerHarnessEngineeringOnly", "workerHarnessPairId", "workerHarnessRunId", "resumeRunId", "candidateIdNamespace", "workerHarnessParentCommit", "workerHarnessSnapshotTreeDigest"].includes(key))), candidateCounts, taskWallBoundMs, evaluationCells: ["regression", "typecheck", ...args.config.evalCommands.map((_, index) => `visible-${index}`)], developmentTaskIdentities: benchmarkTasks.map((task) => ({ id: task.id, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings })), confirmationTaskIdentities: engineeringOnly ? [] : confirmationTasks.map((task) => ({ id: task.id, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings })), inferenceCeilings: { maxInputTokensPerCall: benchmarkLimits.maxInputTokensPerCall, maxOutputTokensPerCall: benchmarkLimits.maxOutputTokensPerCall, maxCallsPerAttempt: benchmarkLimits.maxCalls, maxTotalTokens: benchmarkLimits.maxTotalTokens, maxCallSpendUsd: benchmarkLimits.maxCallSpendUsd, maxJobSpendUsd: benchmarkLimits.maxJobSpendUsd, maxInputPricePerMillionUsd: benchmarkLimits.maxInputPricePerMillionUsd, maxOutputPricePerMillionUsd: benchmarkLimits.maxOutputPricePerMillionUsd }, inferenceCampaignIds: { mutation: mutationCampaignId, benchmark: benchmarkCampaignId }, inferenceAllocationsUsd: { mutation: mutationLimits.maxCampaignSpendUsd, benchmark: benchmarkLimits.maxCampaignSpendUsd } }, } } @@ -186,6 +188,10 @@ export async function rebuildFinishedProofCampaignArchive(args: { const expectedDigest = proofCampaignManifestDigest(args.manifest) if (!campaign || !arm || arm.state !== "finished" || String(campaign.manifest_sha256).trim() !== expectedDigest) throw new Error("finished proof campaign snapshot does not match the frozen manifest and arm") if (arm.run_id !== args.run.runId || args.run.candidates.some((candidate) => !isTerminal(candidate))) throw new Error("finished proof campaign arm cannot rebuild an incomplete or foreign run archive") + if (args.manifest.phaseMode === "engineering-only") { + const events = args.snapshot.events as Array> | undefined + if (!events?.some((event) => event.arm_id === args.armId && event.event_type === "engineering-development-finished") || args.run.proofConfirmation || args.run.confirmationAttempts?.length) throw new Error("engineering-only recovery must have a terminal development event and no confirmation evidence") + } else await verifyFinishedConfirmationCheckpoint(args.run, args.config.archiveDir, args.snapshot, args.manifest, args.armId) args.run.proofCampaign = { manifest: args.manifest, manifestSha256: expectedDigest, campaignId: args.manifest.campaignId, armId: args.armId, epoch: Number(arm.epoch), state: "finished", ledgerSnapshot: args.snapshot } const selections = (args.snapshot.events as Array>).filter((event) => event.arm_id === args.armId && event.event_type === "generation-selected").sort((left, right) => Number(left.sequence) - Number(right.sequence)) const finalSelection = selections.at(-1)?.payload as { generation?: number; selectedCandidateIds?: string[] } | undefined @@ -206,6 +212,41 @@ export async function rebuildFinishedProofCampaignArchive(args: { return args.run } +async function verifyFinishedConfirmationCheckpoint(run: RsiRunRecord, archiveDir: string, snapshot: Record, manifest: ProofCampaignManifest, armId: string): Promise { + const events = snapshot.events as Array> | undefined + const attempts = snapshot.attempts as Array> | undefined + const selection = events?.filter((event) => event.arm_id === armId && event.event_type === "final-harness-selected") + const payload = selection?.length === 1 ? selection[0]!.payload as { generation?: number; candidateId?: string; commit?: string } : undefined + const saved = run.proofConfirmation + if (!payload || !saved || saved.selectedHarness.slotId !== payload.candidateId || saved.selectedHarness.generation !== payload.generation || saved.selectedHarness.commit !== payload.commit || saved.rootHarness.commit !== manifest.rootCommit || run.confirmationEvidence !== "complete") throw new Error("finished campaign recovery requires a matching checkpointed final-harness selection and confirmation result") + const cells = manifest.schedule.filter((cell) => cell.armId === armId && cell.phase === "confirmation" && (cell.candidateId === "root-control" || (cell.candidateId === payload.candidateId && cell.generation === payload.generation))) + const byCandidateKind = (candidateId: string, kind: string) => cells.filter((cell) => cell.candidateId === candidateId && cell.attemptKind === kind) + const selectedRows = byCandidateKind(String(payload.candidateId), payload.candidateId === "root-control" ? "fixed-control" : "confirmation") + const rootRows = byCandidateKind("root-control", "fixed-control") + const rowFor = (cell: typeof cells[number]) => attempts?.filter((row) => row.arm_id === armId && Number(row.generation) === cell.generation && row.candidate_id === cell.candidateId && row.task_id === cell.taskId && row.seed === cell.seed && row.attempt_kind === cell.attemptKind && row.status === "completed") ?? [] + const expectedSelectedRows = selectedRows.flatMap((cell) => rowFor(cell)) + const expectedRootRows = rootRows.flatMap((cell) => rowFor(cell)) + const orderedIds = (rows: Array>) => rows.map((row) => String(row.broker_job_id)).sort() + if (expectedSelectedRows.length !== selectedRows.length || expectedRootRows.length !== rootRows.length || JSON.stringify(orderedIds(expectedSelectedRows)) !== JSON.stringify([...saved.selectedHarness.attemptIds].sort()) || JSON.stringify(orderedIds(expectedRootRows)) !== JSON.stringify([...saved.rootHarness.attemptIds].sort())) throw new Error("finished campaign recovery checkpoint does not name the completed selected and root confirmation cells") + const artifacts = new Map((run.confirmationAttempts ?? []).map((attempt) => [attempt.attemptId, attempt])) + for (const row of [...expectedSelectedRows, ...expectedRootRows]) { + const attemptId = String(row.broker_job_id) + const attempt = artifacts.get(attemptId) + if (!attempt || attempt.validity !== "valid" || !attempt.proofQualified || attempt.resourceAccounting !== "supervisor-enforced" || !attempt.inferenceReceipt?.qualified || attempt.inferenceReceipt.endpointProvider !== "deepinfra" || attempt.inferenceReceipt.returnedProviders?.join(",") !== "deepinfra" || createHash("sha256").update(JSON.stringify(attempt)).digest("hex") !== String(row.result_sha256).trim()) throw new Error(`finished campaign recovery attempt ${attemptId} does not match its verified durable result`) + const storedRefs = row.artifact_refs as Record | undefined + if (!storedRefs || Object.entries(attempt.artifactRefs).some(([name, digest]) => storedRefs[name]?.sha256 !== digest)) throw new Error(`finished campaign recovery attempt ${attemptId} has different artifact references`) + const candidateId = String(row.candidate_id) + const kind = String(row.attempt_kind) + const outputRoot = path.join(archiveDir, "confirmation", manifest.campaignId, armId, candidateId, kind, "harness", attempt.harnessCommit) + for (const digest of Object.values(attempt.artifactRefs)) { + if (!/^[a-f0-9]{64}$/.test(digest)) throw new Error(`finished campaign recovery attempt ${attemptId} has a malformed artifact digest`) + const artifact = path.join(outputRoot, "artifacts", "sha256", digest.slice(0, 2), digest) + const stat = await fs.lstat(artifact) + if (!stat.isFile() || stat.isSymbolicLink() || createHash("sha256").update(await fs.readFile(artifact)).digest("hex") !== digest) throw new Error(`finished campaign recovery artifact ${digest} is missing or changed`) + } + } +} + /** Rebuild only the durable development-phase checkpoint after its DB commit. */ export async function rebuildEvaluatedProofCampaignArchive(args: { run: RsiRunRecord @@ -269,10 +310,18 @@ function combinedArchive(archive: RsiArchive, run: RsiRunRecord): RsiArchive { return { ...archive, candidates: [...archive.candidates.filter((candidate) => !currentIds.has(candidate.id)), ...run.candidates] } } +function validateDeterministicWorkerHarnessRunId(config: RsiConfig): void { + if (config.workerHarnessRunId) { + const suffix = config.workerHarnessMode === "fixed-control" ? "fixed" : config.workerHarnessMode === "self-hosted" ? "self" : undefined + if (!config.workerHarnessPairId || !suffix || config.workerHarnessRunId !== `${config.workerHarnessPairId}-${suffix}` || !/^rsi-proof-[a-f0-9]{20}-(fixed|self)$/.test(config.workerHarnessRunId)) throw new Error("deterministic worker harness run ID must match its paired campaign and arm") + } + } + function initialRun(config: RsiConfig, baseCommit: string, now: string): RsiRunRecord { const modelCandidateId = config.modelCandidateId ?? "model-base" + validateDeterministicWorkerHarnessRunId(config) return { - runId: `rsi-${now.replace(/[^0-9]/g, "").slice(0, 14)}-${randomUUID().slice(0, 8)}`, + runId: config.workerHarnessRunId ?? `rsi-${now.replace(/[^0-9]/g, "").slice(0, 14)}-${randomUUID().slice(0, 8)}`, startedAt: now, model: config.model, baseRef: config.baseRef, @@ -726,8 +775,11 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverr : undefined if (workerHarnessRuntime && !validWorkerHarnessRuntimeIdentity(workerHarnessRuntime)) throw new Error("worker harness runtime identity is malformed") const requestedPhase = config.workerHarnessPhase ?? "development" + if (config.workerHarnessEngineeringOnly && (!config.workerHarnessPairId || !config.workerHarnessMode || requestedPhase !== "development")) throw new Error("engineering-only paired runs require an explicit campaign, arm, and development phase") + validateDeterministicWorkerHarnessRunId(config) const resumed = config.resumeRunId ? requestedPhase === "confirmation" ? await findRun(config.archiveDir, config.resumeRunId) : await findActiveRun(config.archiveDir, config.resumeRunId) : undefined if (config.resumeRunId && !resumed) throw new Error(`no active RSI run found for --resume ${config.resumeRunId}`) + if (resumed && config.workerHarnessRunId && resumed.runId !== config.workerHarnessRunId) throw new Error("worker harness deterministic run identity differs from the resumed archive record") const baseCommit = resumed?.baseCommit ?? (await resolveBaseCommit(config)) const run = resumed ?? initialRun({ ...config, model: workerModel, roles }, baseCommit, now()) const stageCount = requestedPhase === "confirmation" ? 0 : config.computePolicy === "adaptive-independent" ? 2 : config.generations @@ -892,6 +944,7 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverr } if (config.workerHarnessPairId && !config.dryRun) { if (!config.workerHarnessMode || !workerHarnessRuntime || !fleetQueue) throw new Error("restart-safe paired worker-harness campaigns require an explicit arm, frozen runtime identity, and PostgreSQL fleet queue") + if (config.workerHarnessEngineeringOnly && (requestedPhase !== "development" || config.resumeRunId && resumed?.proofCampaign?.manifest.phaseMode !== "engineering-only")) throw new Error("engineering-only execution cannot resume or enter a confirmation campaign") const manifest = createProofCampaignManifest({ config: runConfig, campaignId: config.workerHarnessPairId, baseCommit, model: workerModel, provider: workerProvider, runtime: workerHarnessRuntime, developmentManifest: loadedProofBenchmark.manifest, developmentManifestDigest: loadedProofBenchmark.manifestDigest, plannedGenerations, proofProtocol }) campaignStore = fleetQueue.proofCampaignCoordinator() if (hooks.runDevelopmentBenchmark) throw new Error("paired proof campaigns cannot use an injected benchmark hook that bypasses durable campaign admission and settlement") @@ -920,13 +973,22 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverr if (requestedPhase === "confirmation") { if (!campaignStore || !campaignLease || campaignLease.phase !== "confirmation" || !config.workerHarnessPairId || config.dryRun) throw new Error("confirmation execution requires a resumed, frozen, supervised paired proof campaign") const confirmation = await runProofConfirmation({ run, config, manifest: run.proofCampaign!.manifest, benchmark: loadedProofBenchmark.manifest, fixturesRoot: proofFixturesRoot, store: campaignStore, lease: campaignLease, assertLease: assertCampaignLease, model: workerModel, provider: "openshell-openrouter" }) - run.confirmationAttempts = [...(run.confirmationAttempts ?? []), ...confirmation.attempts] + const attemptsById = new Map((run.confirmationAttempts ?? []).map((attempt) => [attempt.attemptId, attempt])) + for (const attempt of confirmation.attempts) attemptsById.set(attempt.attemptId, attempt) + run.confirmationAttempts = [...attemptsById.values()] run.proofConfirmation = { selectedHarness: { candidateId: confirmation.selectedHarnessCandidate, slotId: confirmation.selectedCandidateId, generation: confirmation.selectedGeneration, commit: confirmation.selectedCommit, attemptIds: confirmation.selectedAttempts.map((attempt) => attempt.attemptId) }, rootHarness: { commit: run.baseCommit, attemptIds: confirmation.rootAttempts.map((attempt) => attempt.attemptId) }, } run.confirmationEvidence = confirmation.attempts.length > 0 && confirmation.attempts.every((attempt) => attempt.validity === "valid" && attempt.proofQualified && attempt.inferenceReceipt?.qualified) ? "complete" : "incomplete" await assertCampaignLease() + // Persist the exact attempts and selected/root identities before the + // authoritative DB terminal transition. If the process dies after + // finishArm but before appendRun, recovery can rebuild from this + // checkpoint and the durable campaign snapshot without paid replay. + run.proofCampaign = { ...run.proofCampaign!, state: "running", ledgerSnapshot: await campaignStore.snapshot(campaignLease.campaignId) } + await checkpointRun(config.archiveDir, run) + await assertCampaignLease() await campaignStore.finishArm(campaignLease, [confirmation.selectedCandidateId]) run.proofCampaign = { ...run.proofCampaign!, state: "finished", ledgerSnapshot: await campaignStore.snapshot(campaignLease.campaignId) } run.finishedAt = now() @@ -1443,7 +1505,18 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverr if (campaignStore && campaignLease) { await assertCampaignLease() await campaignStore.finishDevelopmentArm(campaignLease) - run.proofCampaign = { ...run.proofCampaign!, state: "evaluated", ledgerSnapshot: await campaignStore.snapshot(campaignLease.campaignId) } + const snapshot = await campaignStore.snapshot(campaignLease.campaignId) + const engineeringOnly = run.proofCampaign?.manifest.phaseMode === "engineering-only" + run.proofCampaign = { ...run.proofCampaign!, state: engineeringOnly ? "finished" : "evaluated", ledgerSnapshot: snapshot } + if (engineeringOnly) { + run.finishedAt = now() + const reportPath = `${config.archiveDir}/${run.runId}.md` + await fs.mkdir(config.archiveDir, { recursive: true }) + await fs.writeFile(reportPath, formatRunReport(run, config), "utf8") + run.reports = [...new Set([...run.reports, reportPath])] + await appendRun(config.archiveDir, run) + return run + } } run.finishedAt = now() diff --git a/src/rsi/proof-campaign.ts b/src/rsi/proof-campaign.ts index 9f3eb75..547cf00 100644 --- a/src/rsi/proof-campaign.ts +++ b/src/rsi/proof-campaign.ts @@ -6,6 +6,8 @@ export type ProofCampaignAttemptKind = "mutation" | "development-proof" | "verif export interface ProofCampaignManifest { schemaVersion: 1 + /** Engineering-only roots terminate after development and never freeze confirmation task IDs. */ + phaseMode?: "paired-confirmation" | "engineering-only" campaignId: string rootCommit: string benchmarkDigests: { manifest: string; development: string; confirmation: string } @@ -125,6 +127,8 @@ export function validateProofCampaignManifest(manifest: ProofCampaignManifest): if (manifest.budgets.root.maxRuntimeMs < 1) throw new Error("proof campaign root runtime ceiling must be positive") if (!/^sha256:[a-f0-9]{64}$/.test(manifest.runtime.imageDigest) || !/^[a-f0-9]{40}$/.test(manifest.runtime.harnessCommit) || !/^[a-f0-9]{64}$/.test(manifest.runtime.dependencyLockDigest)) throw new Error("proof campaign runtime identity is malformed") if (!manifest.analysisVersion.trim()) throw new Error("proof campaign analysis version is required") + if (manifest.phaseMode !== undefined && !["paired-confirmation", "engineering-only"].includes(manifest.phaseMode)) throw new Error("proof campaign phase mode is invalid") + if (manifest.phaseMode === "engineering-only" && manifest.schedule.some((cell) => (cell.phase ?? "development") === "confirmation")) throw new Error("engineering-only proof campaign may not freeze confirmation cells") stableJson(manifest) } @@ -531,13 +535,15 @@ export class PostgresProofCampaign { async finishDevelopmentArm(lease: ProofCampaignArmLease): Promise { await transaction(this.pool, async (client) => { - const stateResult = await client.query(`SELECT campaign.manifest, campaign.manifest_sha256, arm.phase AS arm_phase, arm.epoch AS arm_epoch, arm.run_id + const stateResult = await client.query(`SELECT campaign.manifest, campaign.manifest_sha256, arm.phase AS arm_phase, arm.epoch AS arm_epoch, arm.state AS arm_state, arm.run_id FROM headlesscode_rsi_proof_campaigns campaign JOIN headlesscode_rsi_proof_campaign_arms arm USING(campaign_id) WHERE campaign.campaign_id=$1 AND arm.arm_id=$2 FOR UPDATE OF campaign,arm`, [lease.campaignId, lease.armId]) const state = stateResult.rows[0] if (!state || state.manifest_sha256.trim() !== lease.manifestSha256 || state.run_id !== lease.runId) throw new Error("proof campaign development completion has a stale campaign or arm identity") const priorFinish = await client.query("SELECT 1 FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND epoch=$3 AND event_type='development-phase-finished' AND payload->>'remainingPhase'='confirmation'", [lease.campaignId, lease.armId, lease.epoch]) if (state.arm_phase === "confirmation" && lease.phase === "development" && priorFinish.rowCount === 1) return + const priorEngineeringFinish = await client.query("SELECT 1 FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='engineering-development-finished'", [lease.campaignId, lease.armId]) + if (state.arm_state === "finished" && lease.phase === "development" && priorEngineeringFinish.rowCount === 1) return const campaign = await this.lockLease(client, lease) if (lease.phase === "confirmation" && campaign.arm_phase === "confirmation") { const finished = await client.query("SELECT 1 FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='development-phase-finished' AND payload->>'remainingPhase'='confirmation' LIMIT 1", [lease.campaignId, lease.armId]) @@ -556,9 +562,18 @@ export class PostgresProofCampaign { const result = await client.query("SELECT status FROM headlesscode_rsi_proof_campaign_attempts WHERE campaign_id=$1 AND arm_id=$2 AND generation=$3 AND candidate_id=$4 AND task_id=$5 AND seed=$6 AND attempt_kind=$7", [lease.campaignId, lease.armId, cell.generation, cell.candidateId, cell.taskId, cell.seed, cell.attemptKind]) if (result.rowCount !== 1 || !["completed", "skipped-after-hard-gate"].includes(String(result.rows[0].status))) throw new Error("cannot finish development phase with incomplete or uncertain development evidence") } - await client.query("UPDATE headlesscode_rsi_proof_campaign_arms SET state='evaluated',phase='confirmation',coordinator_id=NULL,lease_token_sha256=NULL,lease_expires_at=NULL,updated_at=clock_timestamp() WHERE campaign_id=$1 AND arm_id=$2 AND epoch=$3", [lease.campaignId, lease.armId, lease.epoch]) - await this.appendEvent(client, lease.campaignId, lease.armId, lease.epoch, "development-phase-finished", { remainingPhase: "confirmation" }) - await client.query("UPDATE headlesscode_rsi_proof_campaigns SET state='evaluated',updated_at=clock_timestamp() WHERE campaign_id=$1 AND state <> 'finished'", [lease.campaignId]) + if (manifest.phaseMode === "engineering-only") { + const prior = await client.query("SELECT 1 FROM headlesscode_rsi_proof_campaign_events WHERE campaign_id=$1 AND arm_id=$2 AND event_type='engineering-development-finished'", [lease.campaignId, lease.armId]) + if (prior.rowCount) return + await client.query("UPDATE headlesscode_rsi_proof_campaign_arms SET state='finished',coordinator_id=NULL,lease_token_sha256=NULL,lease_expires_at=NULL,updated_at=clock_timestamp() WHERE campaign_id=$1 AND arm_id=$2 AND epoch=$3", [lease.campaignId, lease.armId, lease.epoch]) + await this.appendEvent(client, lease.campaignId, lease.armId, lease.epoch, "engineering-development-finished", { confirmationFrozen: false }) + const arms = await client.query("SELECT count(*)::int AS total,count(*) FILTER (WHERE state='finished')::int AS finished FROM headlesscode_rsi_proof_campaign_arms WHERE campaign_id=$1", [lease.campaignId]) + if (Number(arms.rows[0]?.total) === 2 && Number(arms.rows[0]?.finished) === 2) await client.query("UPDATE headlesscode_rsi_proof_campaigns SET state='finished',updated_at=clock_timestamp() WHERE campaign_id=$1", [lease.campaignId]) + } else { + await client.query("UPDATE headlesscode_rsi_proof_campaign_arms SET state='evaluated',phase='confirmation',coordinator_id=NULL,lease_token_sha256=NULL,lease_expires_at=NULL,updated_at=clock_timestamp() WHERE campaign_id=$1 AND arm_id=$2 AND epoch=$3", [lease.campaignId, lease.armId, lease.epoch]) + await this.appendEvent(client, lease.campaignId, lease.armId, lease.epoch, "development-phase-finished", { remainingPhase: "confirmation" }) + await client.query("UPDATE headlesscode_rsi_proof_campaigns SET state='evaluated',updated_at=clock_timestamp() WHERE campaign_id=$1 AND state <> 'finished'", [lease.campaignId]) + } }) } diff --git a/src/rsi/proof-demonstration-manifest.ts b/src/rsi/proof-demonstration-manifest.ts index a1d06cf..5cbc407 100644 --- a/src/rsi/proof-demonstration-manifest.ts +++ b/src/rsi/proof-demonstration-manifest.ts @@ -25,6 +25,9 @@ export interface FrozenProofDemonstrationManifest { campaignIds: { engineering: string; campaigns: [string, string, string] } operatorCapUsd: number plan: ProofDemonstrationPlan + analysis: { bootstrapSeed: string; bootstrapReplicates: number; confirmationRetentionTaskIds: string[] } + /** Digest-bound development-only engineering pair that had to pass before this holdout manifest was frozen. */ + engineeringEvidence: { path: string; sha256: string; campaignId: string; rootCommit: string; manifestSha256: string; snapshotSha256: string } digestSha256: string } @@ -55,7 +58,10 @@ export function freezeProofDemonstrationManifest(args: Omit Number.isFinite(value) && value > 0) || args.priceCeilingsPerMillionUsd.input < args.endpointPrice.inputPricePerMillionUsd || args.priceCeilingsPerMillionUsd.output < args.endpointPrice.outputPricePerMillionUsd) throw new Error("configured price ceilings do not cover the observed endpoint rates") if (!/^[a-f0-9]{64}$/.test(args.execution.configSha256) || args.execution.configSha256 !== frozenProofConfigDigest(args.execution.configs) || !/^[a-f0-9]{64}$/.test(args.execution.analysisSourceSha256)) throw new Error("proof demonstration config or analysis-source digest is malformed") if (!args.execution.taskIdentities.length || args.execution.taskIdentities.some((task) => !/^[a-f0-9]{64}$/.test(task.verifierDigest) || !/^[a-f0-9]{64}$/.test(task.knownGoodDigest) || !/^[a-f0-9]{64}$/.test(task.knownBadDigest))) throw new Error("proof demonstration task identities are incomplete") - if (args.operatorCapUsd + 1e-9 < args.plan.totalRootSpendCeilingUsd) throw new Error(`operator cap $${args.operatorCapUsd.toFixed(6)} is below the frozen root reservation $${args.plan.totalRootSpendCeilingUsd.toFixed(6)}`) + const confirmationIds = new Set(args.execution.taskIdentities.filter((task) => task.split === "confirmation").map((task) => task.id)) + if (!args.analysis.bootstrapSeed.trim() || !Number.isSafeInteger(args.analysis.bootstrapReplicates) || args.analysis.bootstrapReplicates < 1 || !args.analysis.confirmationRetentionTaskIds.length || new Set(args.analysis.confirmationRetentionTaskIds).size !== args.analysis.confirmationRetentionTaskIds.length || args.analysis.confirmationRetentionTaskIds.some((id) => !confirmationIds.has(id))) throw new Error("proof demonstration analysis settings or confirmation retention set are invalid") + if (args.operatorCapUsd + 1e-9 < args.plan.campaignRootsSpendCeilingUsd) throw new Error(`operator cap $${args.operatorCapUsd.toFixed(6)} is below the three frozen campaign roots reservation $${args.plan.campaignRootsSpendCeilingUsd.toFixed(6)}`) + if (!args.engineeringEvidence.path || !/^[a-f0-9]{64}$/.test(args.engineeringEvidence.sha256) || !/^[A-Za-z0-9._:-]{1,160}$/.test(args.engineeringEvidence.campaignId) || args.engineeringEvidence.rootCommit !== args.rootCommit || !/^[a-f0-9]{64}$/.test(args.engineeringEvidence.manifestSha256) || !/^[a-f0-9]{64}$/.test(args.engineeringEvidence.snapshotSha256)) throw new Error("engineering evidence identity is malformed or belongs to another root") const body = { schemaVersion: 1 as const, ...args } return { ...body, digestSha256: frozenProofDemonstrationDigest(body) } } @@ -67,7 +73,10 @@ export function verifyFrozenProofDemonstrationManifest(value: unknown): FrozenPr const { digestSha256, ...body } = manifest if (frozenProofDemonstrationDigest(body) !== digestSha256) throw new Error("frozen proof demonstration manifest digest mismatch") if (manifest.endpointPrice.model !== manifest.model.requestedId || manifest.endpointPrice.provider !== RSI_OPENROUTER_ENDPOINT_PROVIDER || manifest.endpointPrice.endpointTag !== RSI_OPENROUTER_ENDPOINT_TAG || manifest.endpointPrice.quantization !== RSI_OPENROUTER_ENDPOINT_QUANTIZATION || !["max_tokens", "temperature", "tools", "tool_choice"].every((parameter) => manifest.endpointPrice.supportedParameters.includes(parameter))) throw new Error("frozen proof demonstration endpoint is not the pinned DeepInfra fp8 route with required parameters") - if (manifest.operatorCapUsd + 1e-9 < manifest.plan.totalRootSpendCeilingUsd) throw new Error("frozen proof demonstration operator cap is below root reservations") + if (manifest.operatorCapUsd + 1e-9 < manifest.plan.campaignRootsSpendCeilingUsd) throw new Error("frozen proof demonstration operator cap is below the three campaign root reservations") + if (!manifest.engineeringEvidence || !manifest.engineeringEvidence.path || !/^[a-f0-9]{64}$/.test(manifest.engineeringEvidence.sha256) || manifest.engineeringEvidence.rootCommit !== manifest.rootCommit || !/^[a-f0-9]{64}$/.test(manifest.engineeringEvidence.manifestSha256) || !/^[a-f0-9]{64}$/.test(manifest.engineeringEvidence.snapshotSha256)) throw new Error("frozen engineering evidence reference is missing or malformed") if (frozenProofConfigDigest(manifest.execution.configs) !== manifest.execution.configSha256) throw new Error("frozen proof demonstration config digest mismatch") + const confirmationIds = new Set(manifest.execution.taskIdentities.filter((task) => task.split === "confirmation").map((task) => task.id)) + if (!manifest.analysis.bootstrapSeed.trim() || !Number.isSafeInteger(manifest.analysis.bootstrapReplicates) || manifest.analysis.bootstrapReplicates < 1 || !manifest.analysis.confirmationRetentionTaskIds.length || new Set(manifest.analysis.confirmationRetentionTaskIds).size !== manifest.analysis.confirmationRetentionTaskIds.length || manifest.analysis.confirmationRetentionTaskIds.some((id) => !confirmationIds.has(id))) throw new Error("frozen proof demonstration analysis settings or confirmation retention set is invalid") return manifest } diff --git a/src/rsi/proof-demonstration-plan.ts b/src/rsi/proof-demonstration-plan.ts index 1dae229..d31158f 100644 --- a/src/rsi/proof-demonstration-plan.ts +++ b/src/rsi/proof-demonstration-plan.ts @@ -9,6 +9,8 @@ export interface ProofDemonstrationPlan { campaign: ProofDemonstrationArmPlan totalReservedSpendUsd: number totalRootSpendCeilingUsd: number + engineeringRootSpendCeilingUsd: number + campaignRootsSpendCeilingUsd: number totalPriceDerivedSpendCeilingUsd: number totalRootPriceDerivedSpendCeilingUsd: number totalCalls: number @@ -23,7 +25,7 @@ export interface ProofDemonstrationArmPlan { verificationJobs: number developmentProofCells: number confirmationScheduleCells: number - confirmationDisposition: "pre-registered-deferred" | "executed-after-development" + confirmationDisposition: "not-frozen-until-engineering-passes" | "executed-after-development" confirmationNotSelectedCells: number paidConfirmationCells: number rootControlCells: number @@ -92,9 +94,9 @@ export function planProofDemonstrationArm(args: { const candidateConfirmationCells = candidateSlotsBothArms * confirmation.length const paidConfirmationCells = args.phase === "campaign" ? 2 * confirmation.length : 0 const rootControlCells = args.phase === "campaign" ? 2 * confirmation.length : 0 - const deferredConfirmationReservationCells = args.phase === "engineering" ? 4 * confirmation.length : 0 + const deferredConfirmationReservationCells = 0 const confirmationNotSelectedCells = args.phase === "campaign" ? candidateConfirmationCells - paidConfirmationCells : 0 - const confirmationScheduleCells = candidateConfirmationCells + 2 * confirmation.length + const confirmationScheduleCells = args.phase === "campaign" ? candidateConfirmationCells + 2 * confirmation.length : 0 const mutationCalls = mutationJobs * args.mutationLimits.maxCalls const mutationTokensPerJob = Math.min(args.mutationLimits.maxTotalTokens, args.mutationLimits.maxCalls * (args.mutationLimits.maxInputTokensPerCall + args.mutationLimits.maxOutputTokensPerCall)) const mutationPriceMicros = assertPriceFeasible("mutation", args.mutationLimits.maxCalls, args.mutationLimits.maxOutputTokensPerCall, args.mutationLimits) @@ -116,11 +118,6 @@ export function planProofDemonstrationArm(args: { priceDerivedSpendMicros += proofRoles * devPriceMicrosPerRole rootSpendMicros = reservedSpendMicros rootPriceSpendMicros = priceDerivedSpendMicros - if (args.phase === "engineering") { - const deferredFinalAndRoot = 4 * confirmation.length - rootSpendMicros += deferredFinalAndRoot * microsCeil(args.benchmarkLimits.maxJobSpendUsd) - rootPriceSpendMicros += confirmation.reduce((sum, task) => sum + taskPrice(task), 0) * 4 - } if (args.phase === "campaign") { const finalAndRoot = 4 // selected final and root control, in both arms const confirmationCalls = confirmation.reduce((sum, task) => sum + taskCalls(task), 0) * finalAndRoot @@ -138,12 +135,12 @@ export function planProofDemonstrationArm(args: { return { phase: args.phase, generations: args.generations, population: args.population, mutationJobs, verificationJobs, developmentProofCells, confirmationScheduleCells, - confirmationDisposition: args.phase === "campaign" ? "executed-after-development" : "pre-registered-deferred", + confirmationDisposition: args.phase === "campaign" ? "executed-after-development" : "not-frozen-until-engineering-passes", confirmationNotSelectedCells, paidConfirmationCells, rootControlCells, deferredConfirmationReservationCells, maximumCalls, maximumTokens, reservedSpendUsd: fromMicros(reservedSpendMicros), rootSpendCeilingUsd: fromMicros(rootSpendMicros), priceDerivedSpendCeilingUsd: fromMicros(priceDerivedSpendMicros), rootPriceDerivedSpendCeilingUsd: fromMicros(rootPriceSpendMicros), taskWallBoundMs, - developmentTaskIds: development.map((task) => task.id), confirmationTaskIds: confirmation.map((task) => task.id), + developmentTaskIds: development.map((task) => task.id), confirmationTaskIds: args.phase === "campaign" ? confirmation.map((task) => task.id) : [], } } @@ -158,6 +155,8 @@ export function planProofDemonstration(args: { return { engineering, campaign, totalReservedSpendUsd: fromMicros(microsCeil(engineering.reservedSpendUsd) + 3 * microsCeil(campaign.reservedSpendUsd)), + engineeringRootSpendCeilingUsd: fromMicros(microsCeil(engineering.rootSpendCeilingUsd)), + campaignRootsSpendCeilingUsd: fromMicros(3 * microsCeil(campaign.rootSpendCeilingUsd)), totalRootSpendCeilingUsd: fromMicros(microsCeil(engineering.rootSpendCeilingUsd) + 3 * microsCeil(campaign.rootSpendCeilingUsd)), totalPriceDerivedSpendCeilingUsd: fromMicros(microsCeil(engineering.priceDerivedSpendCeilingUsd) + 3 * microsCeil(campaign.priceDerivedSpendCeilingUsd)), totalRootPriceDerivedSpendCeilingUsd: fromMicros(microsCeil(engineering.rootPriceDerivedSpendCeilingUsd) + 3 * microsCeil(campaign.rootPriceDerivedSpendCeilingUsd)), @@ -166,3 +165,17 @@ export function planProofDemonstration(args: { totalTaskWallMs: engineering.taskWallBoundMs + 3 * campaign.taskWallBoundMs, } } + +function canonical(value: unknown): unknown { + if (Array.isArray(value)) return value.map(canonical) + if (value && typeof value === "object") return Object.fromEntries(Object.entries(value as Record) + .filter(([, item]) => item !== undefined) + .sort(([a], [b]) => Buffer.compare(Buffer.from(a), Buffer.from(b))) + .map(([key, item]) => [key, canonical(item)])) + return value +} + +export function assertFrozenProofDemonstrationPlanMatches(frozen: ProofDemonstrationPlan, args: Parameters[0]): void { + const recomputed = planProofDemonstration(args) + if (JSON.stringify(canonical(frozen)) !== JSON.stringify(canonical(recomputed))) throw new Error("frozen proof budget plan differs from the recomputed schedule and resource ceilings") +} diff --git a/src/rsi/proof-demonstration.ts b/src/rsi/proof-demonstration.ts index bb6fccf..76663d9 100644 --- a/src/rsi/proof-demonstration.ts +++ b/src/rsi/proof-demonstration.ts @@ -1,8 +1,11 @@ import { createHash } from "node:crypto" +import * as fs from "node:fs/promises" +import * as path from "node:path" import type { BenchmarkAttempt } from "./benchmark/types.js" import type { RsiRunRecord } from "./types.js" import { proofCampaignManifestDigest } from "./proof-campaign.js" import { RSI_OPENROUTER_MODEL } from "./inference-broker.js" +import { frozenProofConfigDigest, verifyFrozenProofDemonstrationManifest, type FrozenProofDemonstrationManifest } from "./proof-demonstration-manifest.js" export type ProofDemonstrationVerdict = "demonstrated" | "inconclusive" | "not-demonstrated" export interface ProofDemonstrationReport { @@ -12,16 +15,27 @@ export interface ProofDemonstrationReport { verdictReasons: string[] campaigns: Array<{ campaignId: string - selfHosted: { runId: string; finalCommit: string; selectedCandidateId: string; adjacentAcceptedDepth: number; confirmation: Array>; root: Array>; netWinsVsRoot: number; retentionLosses: string[]; resources: Record; complete: boolean } - fixedControl: { runId: string; finalCommit: string; selectedCandidateId: string; confirmation: Array>; root: Array>; resources: Record; complete: boolean } + selfHosted: { runId: string; finalCommit: string; selectedCandidateId: string; lineage: Array<{ generation: number; candidateId: string; parentId: string; parentCommit: string | null; commit: string }>; development: Record; adjacentAcceptedDepth: number; confirmation: Array>; root: Array>; netWinsVsRoot: number; retentionLosses: string[]; resources: Record; complete: boolean } + fixedControl: { runId: string; finalCommit: string; selectedCandidateId: string; development: Record; confirmation: Array>; root: Array>; resources: Record; complete: boolean } }> pooled: { taskIds: string[]; campaignCount: number; selfHostedSuccesses: number; fixedControlSuccesses: number; difference: number; cluster90PercentInterval: [number, number] | null; bootstrapReplicates: number; bootstrapSeed: string } denominators: { campaigns: number; pairedConfirmationTasks: number; missingCells: string[]; excludedAttempts: string[] } limitations: string[] } -function sha256(value: string): string { return createHash("sha256").update(value).digest("hex") } +function sha256(value: string | Buffer): string { return createHash("sha256").update(value).digest("hex") } function sorted(items: T[]): T[] { return [...items].sort((a, b) => Buffer.compare(Buffer.from(a), Buffer.from(b))) } +function campaignConfig(value: unknown): Record { + if (!value || typeof value !== "object" || Array.isArray(value)) return {} + const config = { ...(value as Record) } + for (const key of ["workerHarnessMode", "workerHarnessPairId", "candidateIdNamespace", "workerHarnessParentCommit", "workerHarnessSnapshotTreeDigest"]) delete config[key] + return config +} +function campaignConfigWithoutSeed(value: unknown): string { + const config = campaignConfig(value) + delete config.seed + return frozenProofConfigDigest(config) +} function percentile(values: number[], fraction: number): number { if (!values.length) throw new Error("cannot calculate percentile from no bootstrap samples") return values[Math.min(values.length - 1, Math.max(0, Math.ceil(fraction * values.length) - 1))]! @@ -64,6 +78,16 @@ function chainDepth(run: RsiRunRecord): number { current = parent } } +function selectedLineage(run: RsiRunRecord): Array<{ generation: number; candidateId: string; parentId: string; parentCommit: string | null; commit: string }> { + const selectedId = run.proofConfirmation?.selectedHarness.candidateId + let current = selectedId ? run.candidates.find((candidate) => candidate.id === selectedId) : undefined + const lineage: Array<{ generation: number; candidateId: string; parentId: string; parentCommit: string | null; commit: string }> = [] + while (current && current.status === "accepted" && current.commits[0] && durablySelected(run, current)) { + lineage.push({ generation: current.generation, candidateId: current.id, parentId: current.parent ?? "baseline", parentCommit: current.parentCommit ?? null, commit: current.commits[0] }) + current = current.parent && current.parent !== "baseline" ? run.candidates.find((candidate) => candidate.id === current!.parent) : undefined + } + return lineage.reverse() +} function attemptSummary(attempt: BenchmarkAttempt): Record { return { @@ -97,13 +121,73 @@ function validateScheduledConfirmation(run: RsiRunRecord, missing: string[]): bo } return complete } +function developmentDenominators(run: RsiRunRecord): Record { + const manifest = run.proofCampaign?.manifest + const rows = (run.proofCampaign?.ledgerSnapshot?.attempts as Array> | undefined) ?? [] + const candidates = run.candidates.map((candidate) => ({ id: candidate.id, generation: candidate.generation, status: candidate.status, proofAccepted: candidate.developmentProof?.accepted ?? false, reason: candidate.developmentProof?.reason ?? candidate.failure ?? "no durable development decision" })) + const slotsByGeneration = (manifest?.settings.candidateCounts as number[] | undefined) ?? [] + const scheduledCells = manifest?.schedule.filter((cell) => cell.armId === run.proofCampaign?.armId && cell.phase === "development") ?? [] + const scheduledCandidateSlots = slotsByGeneration.reduce((sum, value) => sum + (Number.isSafeInteger(value) && value >= 0 ? value : 0), 0) + const completedCells = scheduledCells.filter((cell) => rows.filter((row) => row.arm_id === run.proofCampaign?.armId && Number(row.generation) === cell.generation && row.candidate_id === cell.candidateId && row.task_id === cell.taskId && row.seed === cell.seed && row.attempt_kind === cell.attemptKind && row.status === "completed").length === 1) + const cellStatus = scheduledCells.map((cell) => { + const matches = rows.filter((row) => row.arm_id === run.proofCampaign?.armId && Number(row.generation) === cell.generation && row.candidate_id === cell.candidateId && row.task_id === cell.taskId && row.seed === cell.seed && row.attempt_kind === cell.attemptKind) + return { generation: cell.generation, candidateId: cell.candidateId, taskId: cell.taskId, attemptKind: cell.attemptKind, status: matches.length === 1 ? matches[0]!.status : matches.length === 0 ? "missing" : "duplicate" } + }) + return { + scheduledCandidateSlots, + missingCandidateSlots: Math.max(0, scheduledCandidateSlots - candidates.length), + candidateCounts: { accepted: candidates.filter((candidate) => candidate.status === "accepted").length, rejected: candidates.filter((candidate) => candidate.status === "rejected").length, failed: candidates.filter((candidate) => candidate.status === "failed").length, incomplete: candidates.filter((candidate) => !["accepted", "rejected", "failed"].includes(candidate.status)).length + Math.max(0, scheduledCandidateSlots - candidates.length) }, + candidates, scheduledDevelopmentProofCells: scheduledCells.length, completedDevelopmentProofCells: completedCells.length, + otherDevelopmentCells: cellStatus.filter((cell) => cell.status !== "completed"), + } +} + +/** Verify confirmation attempt bytes against the immutable ledger references before offline analysis. */ +export async function verifyProofDemonstrationArtifactBytes(runs: RsiRunRecord[], archiveDir: string): Promise { + const root = await fs.realpath(archiveDir) + for (const run of runs) { + const campaign = run.proofCampaign + if (!campaign || !run.proofConfirmation) continue + const rows = campaign.ledgerSnapshot?.attempts as Array> | undefined + if (!rows) throw new Error(`campaign ${campaign.campaignId}/${campaign.armId} has no durable attempt rows`) + const ids = [...run.proofConfirmation.selectedHarness.attemptIds, ...run.proofConfirmation.rootHarness.attemptIds] + for (const attemptId of ids) { + const attemptMatches = (run.confirmationAttempts ?? []).filter((attempt) => attempt.attemptId === attemptId) + const rowMatches = rows.filter((row) => row.broker_job_id === attemptId && row.arm_id === campaign.armId && row.status === "completed") + if (attemptMatches.length !== 1 || rowMatches.length !== 1) throw new Error(`campaign ${campaign.campaignId}/${campaign.armId} attempt ${attemptId} is missing or duplicated`) + const attempt = attemptMatches[0]! + const row = rowMatches[0]! + if (String(row.result_sha256).trim() !== sha256(JSON.stringify(attempt))) throw new Error(`campaign ${campaign.campaignId}/${campaign.armId} attempt ${attemptId} result digest mismatch`) + const savedRefs = row.artifact_refs as Record | undefined + if (!savedRefs || Object.entries(attempt.artifactRefs).some(([name, digest]) => savedRefs[name]?.sha256 !== digest)) throw new Error(`campaign ${campaign.campaignId}/${campaign.armId} attempt ${attemptId} artifact references differ from the durable ledger`) + const candidateId = String(row.candidate_id) + const kind = String(row.attempt_kind) + const relativeRoot = path.join("confirmation", campaign.campaignId, campaign.armId, candidateId, kind, "harness", attempt.harnessCommit, "artifacts", "sha256") + for (const digest of Object.values(attempt.artifactRefs)) { + if (!/^[a-f0-9]{64}$/.test(digest)) throw new Error(`campaign ${campaign.campaignId} attempt ${attemptId} has malformed artifact digest`) + const relativeFile = path.join(relativeRoot, digest.slice(0, 2), digest) + const file = path.resolve(root, relativeFile) + if (file !== root && !file.startsWith(root + path.sep)) throw new Error("confirmation artifact path escapes the archive") + let current = root + for (const segment of relativeFile.split(path.sep)) { + current = path.join(current, segment) + const stat = await fs.lstat(current) + if (stat.isSymbolicLink()) throw new Error(`confirmation artifact path contains a symlink: ${attemptId}`) + } + const stat = await fs.lstat(file) + const bytes = await fs.readFile(file) + if (!stat.isFile() || sha256(bytes) !== digest) throw new Error(`confirmation artifact ${digest} is missing or changed`) + } + } + } +} type FrozenTaskIdentity = { id: string; seed: number; startDigest: string; verifierDigest: string; ceilings: BenchmarkAttempt["ceilings"] } function validateAttempt(run: RsiRunRecord, attempt: BenchmarkAttempt, expectedCommit: string, expectedTask: FrozenTaskIdentity | undefined, missing: string[], excluded: string[]): boolean { const expectedTaskId = expectedTask?.id ?? "unknown-task" const label = run.runId + "/" + expectedTaskId if (!expectedTask || attempt.taskId !== expectedTaskId || attempt.harnessCommit !== expectedCommit || attempt.split !== "confirmation" || attempt.seed !== expectedTask.seed || attempt.startDigest !== expectedTask.startDigest || attempt.evaluatorDigest !== expectedTask.verifierDigest || JSON.stringify(attempt.ceilings) !== JSON.stringify(expectedTask.ceilings)) { missing.push(label + ":identity"); return false } const expectedModel = run.proofCampaign?.manifest.model - if (!expectedModel || attempt.model.provider !== expectedModel.provider || attempt.model.id !== expectedModel.id || JSON.stringify(attempt.model.sampling) !== JSON.stringify(expectedModel.settings.sampling) || attempt.inferenceReceipt?.model !== expectedModel.id || attempt.inferenceReceipt.returnedModels.length !== 1 || attempt.inferenceReceipt.returnedModels[0] !== expectedModel.id) { missing.push(label + ":model-identity"); return false } + if (!expectedModel || attempt.model.provider !== expectedModel.provider || attempt.model.id !== expectedModel.id || JSON.stringify(attempt.model.sampling) !== JSON.stringify(expectedModel.settings.sampling) || attempt.inferenceReceipt?.model !== expectedModel.id || attempt.inferenceReceipt.returnedModels.length !== 1 || attempt.inferenceReceipt.returnedModels[0] !== expectedModel.id || attempt.inferenceReceipt.endpointProvider !== "deepinfra" || attempt.inferenceReceipt.returnedProviders.length !== 1 || attempt.inferenceReceipt.returnedProviders[0] !== "deepinfra") { missing.push(label + ":model-or-provider-identity"); return false } const limits = run.proofCampaign?.manifest.settings.inferenceCeilings as { maxInputTokensPerCall?: unknown; maxTotalTokens?: unknown; maxJobSpendUsd?: unknown } | undefined const maxTokens = Math.min(Number(limits?.maxTotalTokens), expectedTask.ceilings.maxCalls * (Number(limits?.maxInputTokensPerCall) + expectedTask.ceilings.maxOutputTokensPerCall)) if (!Number.isSafeInteger(maxTokens) || attempt.wallTimeMs > expectedTask.ceilings.timeoutMs || attempt.calls === null || attempt.calls > expectedTask.ceilings.maxCalls || attempt.tokens === null || attempt.tokens > maxTokens || attempt.inferenceReceipt?.costUsd === null || attempt.inferenceReceipt?.costUsd === undefined || attempt.inferenceReceipt.costUsd > Number(limits?.maxJobSpendUsd) || attempt.inferenceReceipt.calls !== attempt.calls || attempt.inferenceReceipt.totalTokens !== attempt.tokens) { missing.push(label + ":resource-ceiling"); return false } @@ -131,14 +215,22 @@ function matchAttempts(run: RsiRunRecord, attemptIds: string[], commit: string, } /** Recompute the preregistered verdict from immutable attempt artifacts and durable campaign snapshots. */ -export function analyzeProofDemonstration(runs: RsiRunRecord[], options: { expectedCampaignIds: string[]; retentionTaskIds?: string[]; bootstrapSeed: string; bootstrapReplicates?: number }): ProofDemonstrationReport { +export function analyzeProofDemonstration(runs: RsiRunRecord[], options: { frozenPlan: FrozenProofDemonstrationManifest; expectedCampaignIds?: string[]; retentionTaskIds?: string[]; bootstrapSeed?: string; bootstrapReplicates?: number }): ProofDemonstrationReport { const missing: string[] = [] const excluded: string[] = [] - const campaignIds = sorted([...new Set(options.expectedCampaignIds)]) - if (campaignIds.length !== 3 || options.expectedCampaignIds.length !== 3) throw new Error("analysis requires exactly three distinct preregistered campaign IDs") - const confirmationTasks = sorted([...new Set(runs.flatMap((run) => run.proofCampaign?.manifest.schedule.filter((cell) => cell.phase === "confirmation" && cell.candidateId === "root-control").map((cell) => cell.taskId) ?? []))]) - if (!confirmationTasks.length) throw new Error("confirmation task schedule is empty") - const frozenRetentionTaskIds = (runs.find((run) => run.proofCampaign)?.proofCampaign?.manifest.acceptanceRule.confirmationRetentionTaskIds as string[] | undefined) ?? [] + const frozenPlan = verifyFrozenProofDemonstrationManifest(options.frozenPlan) + const campaignIds = sorted([...frozenPlan.campaignIds.campaigns]) + if (new Set(campaignIds).size !== 3 || (options.expectedCampaignIds && JSON.stringify(sorted(options.expectedCampaignIds)) !== JSON.stringify(campaignIds))) throw new Error("analysis campaign IDs differ from the frozen top-level plan") + if ((options.bootstrapSeed && options.bootstrapSeed !== frozenPlan.analysis.bootstrapSeed) || (options.bootstrapReplicates !== undefined && options.bootstrapReplicates !== frozenPlan.analysis.bootstrapReplicates)) throw new Error("analysis bootstrap settings differ from the frozen top-level plan") + if (campaignConfigWithoutSeed(frozenPlan.execution.configs[1]) !== campaignConfigWithoutSeed(frozenPlan.execution.configs[2]) || campaignConfigWithoutSeed(frozenPlan.execution.configs[1]) !== campaignConfigWithoutSeed(frozenPlan.execution.configs[3])) throw new Error("frozen campaign configs differ in settings other than the seed") + const frozenConfirmationIds = sorted(frozenPlan.execution.taskIdentities.filter((task) => task.split === "confirmation").map((task) => task.id)) + if (!frozenConfirmationIds.length) throw new Error("frozen plan confirmation task set is empty") + const confirmationTasks = frozenConfirmationIds + for (const run of runs) { + const rootSchedule = run.proofCampaign?.manifest.schedule.filter((cell) => cell.armId === run.proofCampaign?.armId && cell.phase === "confirmation" && cell.candidateId === "root-control").map((cell) => cell.taskId) + if (rootSchedule && JSON.stringify(sorted(rootSchedule)) !== JSON.stringify(confirmationTasks)) missing.push(run.runId + ":confirmation-task-schedule-differs-from-frozen-plan") + } + const frozenRetentionTaskIds = frozenPlan.analysis.confirmationRetentionTaskIds if (!frozenRetentionTaskIds.length || frozenRetentionTaskIds.some((id) => !confirmationTasks.includes(id))) throw new Error("frozen confirmation retention set is missing or outside the confirmation split") if (options.retentionTaskIds && JSON.stringify(sorted(options.retentionTaskIds)) !== JSON.stringify(sorted(frozenRetentionTaskIds))) throw new Error("analysis retention set differs from the frozen confirmation rule") const campaigns: ProofDemonstrationReport["campaigns"] = [] @@ -150,18 +242,45 @@ export function analyzeProofDemonstration(runs: RsiRunRecord[], options: { expec let selfSuccesses = 0 let fixedSuccesses = 0 for (const campaignId of campaignIds) { + const campaignIndex = frozenPlan.campaignIds.campaigns.indexOf(campaignId) + const expectedSeed = frozenPlan.seeds.campaigns[campaignIndex] + const expectedConfig = campaignConfig(frozenPlan.execution.configs[campaignIndex + 1]) + const expectedConfigDigest = frozenProofConfigDigest(expectedConfig) const arms = runs.filter((run) => run.workerHarnessPairId === campaignId) - const selfRun = arms.find((run) => run.workerHarnessMode === "self-hosted") - const fixedRun = arms.find((run) => run.workerHarnessMode === "fixed-control") - if (!selfRun || !fixedRun) { missing.push(campaignId + ":missing-arm"); continue } + const selfRuns = arms.filter((run) => run.workerHarnessMode === "self-hosted") + const fixedRuns = arms.filter((run) => run.workerHarnessMode === "fixed-control") + if (selfRuns.length !== 1 || fixedRuns.length !== 1) { + missing.push(campaignId + (selfRuns.length > 1 || fixedRuns.length > 1 ? ":duplicate-arm-runs" : ":missing-arm")) + continue + } + const selfRun = selfRuns[0]! + const fixedRun = fixedRuns[0]! + for (const run of [selfRun, fixedRun]) { + if (!run.finishedAt || run.proofCampaign?.state !== "finished") missing.push(campaignId + ":arm-run-not-terminal") + } if (selfRun.proofCampaign?.manifestSha256 !== fixedRun.proofCampaign?.manifestSha256 || selfRun.workerHarnessConfigDigest !== fixedRun.workerHarnessConfigDigest || selfRun.workerHarnessSamplingSeed !== fixedRun.workerHarnessSamplingSeed) missing.push(campaignId + ":paired-manifest-or-seed-mismatch") for (const run of [selfRun, fixedRun]) { const frozen = run.proofCampaign?.manifest if (frozen) { + if (frozen.phaseMode !== "paired-confirmation") missing.push(campaignId + ":campaign-manifest-is-not-the-frozen-confirmation-phase") + if (frozen.campaignId !== campaignId || frozen.seed !== expectedSeed || run.workerHarnessSamplingSeed !== expectedSeed) missing.push(campaignId + ":campaign-id-or-seed-differs-from-frozen-plan") + if (frozen.rootCommit !== frozenPlan.rootCommit || frozen.benchmarkDigests.manifest !== frozenPlan.benchmarkManifestSha256 || run.proofCampaign?.manifestSha256 !== proofCampaignManifestDigest(frozen)) missing.push(campaignId + ":arm-manifest-root-or-benchmark-identity-mismatch") + if (frozenProofConfigDigest(campaignConfig(frozen.settings.effectiveConfig)) !== expectedConfigDigest) missing.push(campaignId + ":effective-config-differs-from-frozen-seeded-config") + if (JSON.stringify(frozen.runtime) !== JSON.stringify({ imageDigest: frozenPlan.runtime.imageDigest, harnessCommit: frozenPlan.runtime.harnessCommit, dependencyLockDigest: frozenPlan.runtime.dependencyLockDigest }) || frozenProofConfigDigest(run.workerHarnessRuntime) !== frozenProofConfigDigest(frozenPlan.runtime)) missing.push(campaignId + ":runtime-identity-differs-from-frozen-plan") + if (frozen.model.provider !== frozenPlan.model.provider || frozen.model.id !== frozenPlan.model.requestedId || frozenProofConfigDigest(frozen.model.settings.sampling) !== frozenProofConfigDigest(frozenPlan.model.sampling)) missing.push(campaignId + ":model-identity-differs-from-frozen-plan") + const armConfirmationTasks = frozen.settings.confirmationTaskIdentities as FrozenTaskIdentity[] | undefined + const expectedConfirmationTasks = frozenPlan.execution.taskIdentities.filter((task) => task.split === "confirmation").map(({ id, seed, startDigest, verifierDigest, ceilings }) => ({ id, seed, startDigest, verifierDigest, ceilings })) + if (!armConfirmationTasks || frozenProofConfigDigest(armConfirmationTasks) !== frozenProofConfigDigest(expectedConfirmationTasks)) missing.push(campaignId + ":confirmation-task-identities-differ-from-frozen-plan") benchmarkManifestDigests.add(frozen.benchmarkDigests.manifest) providerModels.add(frozen.model.provider + "/" + frozen.model.id) campaignSeeds.add(frozen.seed) const frozenSettings = Object.fromEntries(Object.entries(frozen.settings).filter(([key]) => key !== "configDigest" && key !== "inferenceCampaignIds")) + const protocolConfig = frozenSettings.effectiveConfig + if (protocolConfig && typeof protocolConfig === "object" && !Array.isArray(protocolConfig)) { + const normalized = { ...(protocolConfig as Record) } + delete normalized.seed + frozenSettings.effectiveConfig = normalized + } protocolIdentities.add(JSON.stringify({ rootCommit: frozen.rootCommit, benchmarkDigests: frozen.benchmarkDigests, provider: frozen.model.provider, model: frozen.model.id, modelSettings: frozen.model.settings, settings: frozenSettings, acceptanceRule: frozen.acceptanceRule, runtime: frozen.runtime, budgets: frozen.budgets, taskOrder: frozen.taskOrder, analysisVersion: frozen.analysisVersion })) for (const armId of [frozen.armAssignment.fixedControl, frozen.armAssignment.selfHosted]) { const rootTaskIds = sorted(frozen.schedule.filter((cell) => cell.armId === armId && cell.phase === "confirmation" && cell.candidateId === "root-control").map((cell) => cell.taskId)) @@ -224,8 +343,8 @@ export function analyzeProofDemonstration(runs: RsiRunRecord[], options: { expec }) campaigns.push({ campaignId, - selfHosted: { runId: selfRun.runId, finalCommit: selfMeta.selectedHarness.commit, selectedCandidateId: selfMeta.selectedHarness.candidateId, adjacentAcceptedDepth: depth, confirmation: [...selfFinal.values()].map(attemptSummary), root: [...selfRoot.values()].map(attemptSummary), netWinsVsRoot: wins - losses, retentionLosses, resources: sumResources([...selfFinal.values(), ...selfRoot.values()]), complete: selfComplete }, - fixedControl: { runId: fixedRun.runId, finalCommit: fixedMeta.selectedHarness.commit, selectedCandidateId: fixedMeta.selectedHarness.candidateId, confirmation: [...fixedFinal.values()].map(attemptSummary), root: [...fixedRoot.values()].map(attemptSummary), resources: sumResources([...fixedFinal.values(), ...fixedRoot.values()]), complete: fixedComplete }, + selfHosted: { runId: selfRun.runId, finalCommit: selfMeta.selectedHarness.commit, selectedCandidateId: selfMeta.selectedHarness.candidateId, lineage: selectedLineage(selfRun), development: developmentDenominators(selfRun), adjacentAcceptedDepth: depth, confirmation: [...selfFinal.values()].map(attemptSummary), root: [...selfRoot.values()].map(attemptSummary), netWinsVsRoot: wins - losses, retentionLosses, resources: sumResources([...selfFinal.values(), ...selfRoot.values()]), complete: selfComplete }, + fixedControl: { runId: fixedRun.runId, finalCommit: fixedMeta.selectedHarness.commit, selectedCandidateId: fixedMeta.selectedHarness.candidateId, development: developmentDenominators(fixedRun), confirmation: [...fixedFinal.values()].map(attemptSummary), root: [...fixedRoot.values()].map(attemptSummary), resources: sumResources([...fixedFinal.values(), ...fixedRoot.values()]), complete: fixedComplete }, }) } if (benchmarkManifestDigests.size !== 1) missing.push("paired:benchmark-manifest-digest-mismatch") @@ -234,7 +353,7 @@ export function analyzeProofDemonstration(runs: RsiRunRecord[], options: { expec if (protocolIdentities.size !== 1) missing.push("paired:frozen-protocol-runtime-budget-or-selection-mismatch") if (campaignSeeds.size !== 3) missing.push("paired:campaign-seeds-are-not-three-independent-values") const taskIds = sorted([...new Set(paired.map((entry) => entry.taskId))]) - const replicates = options.bootstrapReplicates ?? 10_000 + const replicates = frozenPlan.analysis.bootstrapReplicates if (!Number.isSafeInteger(replicates) || replicates < 1) throw new Error("bootstrap replicates must be a positive safe integer") let interval: [number, number] | null = null let difference = 0 @@ -244,7 +363,7 @@ export function analyzeProofDemonstration(runs: RsiRunRecord[], options: { expec return cluster.reduce((sum, entry) => sum + entry.self - entry.fixed, 0) / cluster.length }) difference = clusterMeans.reduce((sum, value) => sum + value, 0) / clusterMeans.length - const random = seededRandom(options.bootstrapSeed) + const random = seededRandom(frozenPlan.analysis.bootstrapSeed) const samples: number[] = [] for (let sample = 0; sample < replicates; sample++) { let total = 0 @@ -268,7 +387,7 @@ export function analyzeProofDemonstration(runs: RsiRunRecord[], options: { expec verdict: demonstrated ? "demonstrated" : allComplete ? "not-demonstrated" : "inconclusive", verdictReasons: demonstrated ? ["all frozen campaign, lineage, confirmation, retention, and task-cluster interval gates passed"] : reasons, campaigns, - pooled: { taskIds, campaignCount: campaignIds.length, selfHostedSuccesses: selfSuccesses, fixedControlSuccesses: fixedSuccesses, difference, cluster90PercentInterval: interval, bootstrapReplicates: replicates, bootstrapSeed: options.bootstrapSeed }, + pooled: { taskIds, campaignCount: campaignIds.length, selfHostedSuccesses: selfSuccesses, fixedControlSuccesses: fixedSuccesses, difference, cluster90PercentInterval: interval, bootstrapReplicates: replicates, bootstrapSeed: frozenPlan.analysis.bootstrapSeed }, denominators: { campaigns: campaignIds.length, pairedConfirmationTasks: taskIds.length, missingCells: sorted(missing), excludedAttempts: sorted(excluded) }, limitations: ["This analysis covers only the registered tasks, runtime, model endpoint, prompts, and budgets.", "A result does not establish model-weight improvement or broad coding ability.", "Input and output artifacts must remain available for independent digest and verifier replay."], } diff --git a/src/rsi/types.ts b/src/rsi/types.ts index 4531782..c14a086 100644 --- a/src/rsi/types.ts +++ b/src/rsi/types.ts @@ -245,7 +245,11 @@ export interface RsiConfig { benchmarkHarnessCommit?: string workerHarnessMode?: WorkerHarnessMode workerHarnessPhase?: WorkerHarnessPhase + /** Terminal, development-only paired run used to qualify the engineering chain before holdout freeze. */ + workerHarnessEngineeringOnly?: boolean workerHarnessPairId?: string + /** Deterministic supervisor-owned run ID for restart-safe paired campaign admission. */ + workerHarnessRunId?: string workerHarnessRuntime?: WorkerHarnessRuntimeIdentity workerHarnessParentCommit?: string workerHarnessSnapshotTreeDigest?: string From a01afb1040358a0fae1e2da94a5998b55b48c8df Mon Sep 17 00:00:00 2001 From: RSI smoke Date: Wed, 30 Sep 2026 17:31:02 -0600 Subject: [PATCH 3/6] Make benchmark dependency preflight idempotent --- src/rsi/__tests__/openshell.test.ts | 1 + src/rsi/openshell.ts | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/src/rsi/__tests__/openshell.test.ts b/src/rsi/__tests__/openshell.test.ts index cda4880..9080f87 100644 --- a/src/rsi/__tests__/openshell.test.ts +++ b/src/rsi/__tests__/openshell.test.ts @@ -131,6 +131,7 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker assert.equal(fs.existsSync(path.join(workspace, "package.json")), false, "candidate-controlled package metadata must not enter the mutation guest") if (mode === "self-hosted-missing-cli") fs.rmSync(path.join(workspace, "src", "cli.ts")) if (mode === "self-hosted-corrupt-cli") fs.writeFileSync(path.join(workspace, "src", "cli.ts"), "export const = ;\n") + if (mode === "self-hosted-corrupt-cli") fs.symlinkSync(path.join(trustedRuntimeRoot, "node_modules"), path.join(workspace, "node_modules")) if (mode === "dependency-escape") fs.symlinkSync("/tmp/candidate-owned-node-modules", path.join(workspace, "node_modules")) return { id: request.worktreeSpec.name, provider: "mock-openshell", address: workspace } }, diff --git a/src/rsi/openshell.ts b/src/rsi/openshell.ts index 6c58c0a..f0f9591 100644 --- a/src/rsi/openshell.ts +++ b/src/rsi/openshell.ts @@ -634,7 +634,7 @@ export async function runOpenShellMutation(candidate: CandidateRecord, config: R mutationTask(candidate, config), ].map(shellQuote).join(" ") const boundaryChecks = [ - "ln -s /opt/headlesscode/node_modules /workspace/node_modules", + "if test -L /workspace/node_modules; then :; else test ! -e /workspace/node_modules && ln -s /opt/headlesscode/node_modules /workspace/node_modules; fi", 'test "$(readlink /workspace/node_modules)" = /opt/headlesscode/node_modules', "test -r /workspace/src/cli.ts", "test -r /opt/headlesscode/tsconfig.json", From 276e17a1089c39c86824ec04e9ca56fafe2b7945 Mon Sep 17 00:00:00 2001 From: RSI smoke Date: Wed, 30 Sep 2026 17:35:15 -0600 Subject: [PATCH 4/6] Report the failed OpenShell preflight check --- src/rsi/__tests__/openshell.test.ts | 1 + src/rsi/openshell.ts | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/src/rsi/__tests__/openshell.test.ts b/src/rsi/__tests__/openshell.test.ts index 9080f87..17b9608 100644 --- a/src/rsi/__tests__/openshell.test.ts +++ b/src/rsi/__tests__/openshell.test.ts @@ -161,6 +161,7 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker if (mode === "self-hosted-missing-cli" || mode === "dependency-escape") { assert.equal(boundary.status, 79, "the actual shell preflight must stop on the missing CLI or escaped dependency link") assert.match(boundary.stderr, /RSI_BENCHMARK_PREFLIGHT_FAILURE/) + assert.match(boundary.stderr, mode === "dependency-escape" ? /RSI_BENCHMARK_PREFLIGHT_FAILURE:1/ : /RSI_BENCHMARK_PREFLIGHT_FAILURE:2/) return { exitCode: 79, output: boundary.stderr } } assert.equal(boundary.status, 0, "an existing but corrupted CLI passes only the readability/link preflight") diff --git a/src/rsi/openshell.ts b/src/rsi/openshell.ts index f0f9591..27a0c44 100644 --- a/src/rsi/openshell.ts +++ b/src/rsi/openshell.ts @@ -663,7 +663,7 @@ export async function runOpenShellMutation(candidate: CandidateRecord, config: R : "test -z \"$(find /workspace /opt/headlesscode/src -type f \\( -path '*/__tests__/*' -o -name '*.test.*' -o -name '*.spec.*' \\) -print -quit)\"", 'test -z "${HEADLESSCODE_OPENROUTER_API_KEY:-}${OPENROUTER_API_KEY:-}${OPENAI_API_KEY:-}${ANTHROPIC_API_KEY:-}"', ] - const boundaryGuard = `${boundaryChecks.map((check) => `(${check}) || { echo RSI_BENCHMARK_PREFLIGHT_FAILURE >&2; exit 79; }`).join(" && ")} && exec ${workerCommand}` + const boundaryGuard = `${boundaryChecks.map((check, index) => `(${check}) || { echo RSI_BENCHMARK_PREFLIGHT_FAILURE:${index} >&2; exit 79; }`).join(" && ")} && exec ${workerCommand}` const env = rsiMutationGuestEnvironment(config, candidate, inferenceBroker) const isolation = await runSnapshot({ config, From ea12df5ee67abb3d610dd26a68988c5dff8af25f Mon Sep 17 00:00:00 2001 From: RSI smoke Date: Wed, 30 Sep 2026 17:42:42 -0600 Subject: [PATCH 5/6] fix: select worker CLI by harness mode --- src/rsi/__tests__/openshell.test.ts | 10 +++++++--- src/rsi/openshell.ts | 2 +- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/src/rsi/__tests__/openshell.test.ts b/src/rsi/__tests__/openshell.test.ts index 17b9608..bb32bf3 100644 --- a/src/rsi/__tests__/openshell.test.ts +++ b/src/rsi/__tests__/openshell.test.ts @@ -50,7 +50,7 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker execFileSync("git", ["worktree", "add", "--quiet", "-b", `rsi/${mode}`, worktree, baseCommit], { cwd: repo }) const candidate: CandidateRecord = { id: `candidate-${mode}`, - generation: 0, + generation: ["self-hosted-missing-cli", "self-hosted-corrupt-cli", "dependency-escape"].includes(mode) ? 1 : 0, parent: "baseline", branch: `rsi/${mode}`, worktree, @@ -141,7 +141,11 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker assert.match(command, /rev-list --count HEAD/) assert.match(command, /\/opt\/headlesscode\/src\/rsi/) assert.match(command, /OPENROUTER_API_KEY/) - assert.match(command, /test -r \/workspace\/src\/cli\.ts/, "guest preflight must reject a missing self-hosted CLI") + if (["self-hosted-missing-cli", "self-hosted-corrupt-cli", "dependency-escape"].includes(mode)) { + assert.match(command, /test -r \/workspace\/src\/cli\.ts/, "self-hosted preflight checks the selected-parent CLI") + } else { + assert.match(command, /test -r \/opt\/headlesscode\/src\/cli\.ts/, "fixed-control preflight checks the pinned release CLI") + } assert.match(command, /readlink \/workspace\/node_modules/) assert.match(command, /test \! -e \/workspace\/__rsi_benchmark_verifier__/) assert.match(command, /test \! -e \/opt\/headlesscode\/scripts\/rsi-benchmark\/task-spec\.mjs/) @@ -173,7 +177,7 @@ async function run(mode: "source" | "escape" | "symlink" | "preflight" | "worker assert.match(command, /headlesscode-harness-commit/) return { exitCode: 79, output: "RSI_BENCHMARK_PREFLIGHT_FAILURE wrong harness image commit" } } - assert.equal(command.split("/opt/headlesscode/src/cli.ts").length - 1, 1, "the benchmark coding worker command must execute exactly once") + assert.equal(command.split(" && exec ").length - 1, 1, "the benchmark coding worker command must execute exactly once after preflight") if (mode === "worker-failure") return { exitCode: 1, output: "model process failed" } if (mode === "source") { fs.writeFileSync(path.join(guestPath, "src", "engine", "loop.ts"), "export const value = 2\n") diff --git a/src/rsi/openshell.ts b/src/rsi/openshell.ts index 27a0c44..9fc8856 100644 --- a/src/rsi/openshell.ts +++ b/src/rsi/openshell.ts @@ -636,7 +636,7 @@ export async function runOpenShellMutation(candidate: CandidateRecord, config: R const boundaryChecks = [ "if test -L /workspace/node_modules; then :; else test ! -e /workspace/node_modules && ln -s /opt/headlesscode/node_modules /workspace/node_modules; fi", 'test "$(readlink /workspace/node_modules)" = /opt/headlesscode/node_modules', - "test -r /workspace/src/cli.ts", + effectiveMode === "self-hosted" ? "test -r /workspace/src/cli.ts" : "test -r /opt/headlesscode/src/cli.ts", "test -r /opt/headlesscode/tsconfig.json", "test -r /opt/headlesscode/package-lock.json", 'test "$(git -C /workspace rev-list --count HEAD)" = 1', From 3800b39e747e9e33a6270f52680d47016c697880 Mon Sep 17 00:00:00 2001 From: RSI smoke Date: Wed, 30 Sep 2026 18:06:36 -0600 Subject: [PATCH 6/6] test: align proof fixtures with frozen provider rules --- src/rsi/__tests__/postgres-queue.test.ts | 2 +- src/rsi/__tests__/proof-campaign.test.ts | 36 ++++++++++--------- .../__tests__/openshell-integration.test.ts | 5 ++- 3 files changed, 25 insertions(+), 18 deletions(-) diff --git a/src/rsi/__tests__/postgres-queue.test.ts b/src/rsi/__tests__/postgres-queue.test.ts index ae5b679..b8724c0 100644 --- a/src/rsi/__tests__/postgres-queue.test.ts +++ b/src/rsi/__tests__/postgres-queue.test.ts @@ -81,7 +81,7 @@ test("PostgreSQL migrations upgrade an earlier artifact and lease schema", { ski assert.ok(artifactColumns.rows.some((row) => row.column_name === "byte_length")) assert.equal(artifactColumns.rows.some((row) => row.column_name === "content"), false, "empty legacy NOT NULL PostgreSQL blob column is removed") assert.ok(jobColumns.rows.some((row) => row.column_name === "lease_duration_ms")) - assert.equal((await scopedPool.query("SELECT version FROM headlesscode_rsi_schema_migrations ORDER BY version")).rowCount, 7) + assert.equal((await scopedPool.query("SELECT version FROM headlesscode_rsi_schema_migrations ORDER BY version")).rowCount, 8) const campaignColumns = await scopedPool.query("SELECT column_name FROM information_schema.columns WHERE table_schema=$1 AND table_name='headlesscode_rsi_inference_campaigns'", [schema]) assert.ok(campaignColumns.rows.some((row) => row.column_name === "blocked"), "migration 004 adds the campaign block state used by admission and settlement") const inferenceJobColumns = await scopedPool.query("SELECT column_name FROM information_schema.columns WHERE table_schema=$1 AND table_name='headlesscode_rsi_inference_jobs'", [schema]) diff --git a/src/rsi/__tests__/proof-campaign.test.ts b/src/rsi/__tests__/proof-campaign.test.ts index 53e884d..5786e41 100644 --- a/src/rsi/__tests__/proof-campaign.test.ts +++ b/src/rsi/__tests__/proof-campaign.test.ts @@ -11,7 +11,7 @@ import { fleetJobMatches, fleetPayloadDigest, PostgresRsiJobQueue, type FleetJob import { PostgresProofCampaign, proofCampaignManifestDigest, type ProofCampaignManifest } from "../proof-campaign.js" import { openRouterAllocatedEnvironment, openRouterBrokerLimitsFromEnvironment, RSI_OPENROUTER_MODEL, type OpenRouterBrokerGrant, type OpenRouterBrokerLimits } from "../inference-broker.js" import { readArchive, writeArchive } from "../archive.js" -import { createProofCampaignManifest, proofBrokerCampaignId, reconcileProofCampaignJob, rebuildEvaluatedProofCampaignArchive, rebuildFinishedProofCampaignArchive } from "../controller.js" +import { createProofCampaignManifest, proofBrokerCampaignId, reconcileProofCampaignJob, rebuildEvaluatedProofCampaignArchive, rebuildFinishedProofCampaignArchive, RSI_CONFIRMATION_RETENTION_TASK_IDS } from "../controller.js" import { loadBenchmark, runBenchmark, type BenchmarkVerifier } from "../benchmark/runner.js" import type { BenchmarkAgent, BenchmarkAttempt, BenchmarkImageIdentity, BenchmarkTask } from "../benchmark/types.js" import type { CandidateRecord, RsiConfig, RsiRunRecord } from "../types.js" @@ -141,8 +141,8 @@ test("controller freezes paired broker allocations as separate campaigns under o knownBadPath: "bad.patch", knownBadCommit: "d".repeat(40), knownBadDigest: "d".repeat(64), seed: 17, ceilings: { timeoutMs: 5_000, maxCalls: 2, maxOutputTokensPerCall: 128, maxPatchBytes: 16_384 }, } - const confirmationTask = { ...task, id: "confirmation-task", family: "confirmation-integration", split: "confirmation" as const, seed: 23 } - const developmentManifest = { schemaVersion: 1 as const, runnerVersion: "test", createdFromCommit: "e".repeat(40), tasks: [task, confirmationTask] } + const confirmationTasks = RSI_CONFIRMATION_RETENTION_TASK_IDS.map((id, index) => ({ ...task, id, family: `confirmation-integration-${index}`, split: "confirmation" as const, seed: 23 + index })) + const developmentManifest = { schemaVersion: 1 as const, runnerVersion: "test", createdFromCommit: "e".repeat(40), tasks: [task, ...confirmationTasks] } const protocol = { schemaVersion: 1 as const, manifestDigest: "1".repeat(64), taskSetDigest: "2".repeat(64), taskIds: [task.id], taskIdentities: [{ id: task.id, family: task.family, seed: task.seed, startDigest: task.startDigest, verifierDigest: task.verifierDigest, ceilings: task.ceilings }], @@ -164,13 +164,13 @@ test("controller freezes paired broker allocations as separate campaigns under o assert.equal(frozen.benchmarkDigests.manifest, "4".repeat(64), "the full loaded manifest digest, including runner/version metadata, is frozen") assert.deepEqual(frozen.settings.inferenceCampaignIds, { mutation: `${campaignId}:mutation`, benchmark: `${campaignId}:benchmark` }) assert.deepEqual(frozen.settings.inferenceAllocationsUsd, { mutation: 0.04, benchmark: 0.28 }) - assert.equal(frozen.schedule.filter((cell) => cell.armId === "fixed-control").length, 14) - assert.equal(frozen.schedule.filter((cell) => cell.armId === "self-hosted").length, 14) - assert.equal(frozen.schedule.filter((cell) => cell.armId === "fixed-control" && cell.phase === "confirmation").length, 5, "all generation candidate slots and root-control confirmation cells are frozen") - assert.equal(frozen.budgets.root.maxCalls, 40, "root calls sum mutation calls, all three development proof roles, and selected-final/root confirmation grants for both arms") - assert.equal(frozen.budgets.root.maxTokens, 645_120, "root tokens sum effective task-specific grants rather than the base per-call defaults") - assert.equal(frozen.budgets.root.maxSpendUsd, 0.2, "root spend is the sum of every frozen provider-job reservation") - assert.equal(frozen.settings.taskWallBoundMs, 140_000, "the runtime plan sums mutation, required evaluation, development proof, and selected-final/root confirmation deadlines") + assert.equal(frozen.schedule.filter((cell) => cell.armId === "fixed-control").length, 24) + assert.equal(frozen.schedule.filter((cell) => cell.armId === "self-hosted").length, 24) + assert.equal(frozen.schedule.filter((cell) => cell.armId === "fixed-control" && cell.phase === "confirmation").length, 12, "all generation candidate slots and root-control confirmation cells are frozen") + assert.equal(frozen.budgets.root.maxCalls, 64, "root calls sum mutation calls, all three development proof roles, and selected-final/root confirmation grants for both arms") + assert.equal(frozen.budgets.root.maxTokens, 1_032_192, "root tokens sum effective task-specific grants rather than the base per-call defaults") + assert.equal(frozen.budgets.root.maxSpendUsd, 0.32, "root spend is the sum of every frozen provider-job reservation") + assert.equal(frozen.settings.taskWallBoundMs, 200_000, "the runtime plan sums mutation, required evaluation, development proof, and selected-final/root confirmation deadlines") const engineering = createProofCampaignManifest({ config: { ...config, workerHarnessEngineeringOnly: true }, campaignId: `${campaignId}-engineering`, baseCommit: "f".repeat(40), model: RSI_OPENROUTER_MODEL, provider: "openshell-openrouter", runtime: { imageName: "rsi:test", imageDigest: `sha256:${"9".repeat(64)}`, harnessCommit: "8".repeat(40), dependencyLockDigest: "7".repeat(64) }, @@ -186,18 +186,21 @@ test("controller freezes paired broker allocations as separate campaigns under o process.env.HEADLESSCODE_RSI_OPENROUTER_MUTATION_BUDGET_USD = "0.04" process.env.HEADLESSCODE_RSI_OPENROUTER_BENCHMARK_BUDGET_USD = "0.15" assert.throws(() => createProofCampaignManifest({ config, campaignId: `${campaignId}-underfunded`, baseCommit: "f".repeat(40), model: RSI_OPENROUTER_MODEL, provider: "openshell-openrouter", runtime: { imageName: "rsi:test", imageDigest: `sha256:${"9".repeat(64)}`, harnessCommit: "8".repeat(40), dependencyLockDigest: "7".repeat(64) }, developmentManifest, developmentManifestDigest: "4".repeat(64), plannedGenerations: 2, proofProtocol: protocol }), /below development plus selected-final\/root confirmation job reservations/) + process.env.HEADLESSCODE_RSI_OPENROUTER_MAX_CAMPAIGN_SPEND_USD = "0.32" + process.env.HEADLESSCODE_RSI_OPENROUTER_MUTATION_BUDGET_USD = "0.04" + process.env.HEADLESSCODE_RSI_OPENROUTER_BENCHMARK_BUDGET_USD = "0.28" const fixed = await campaign.acquire(frozen, "fixed-control", "run-fixed", "coordinator-fixed") const selfHosted = await campaign.acquire(frozen, "self-hosted", "run-self", "coordinator-self") assert.equal(fixed.manifestSha256, selfHosted.manifestSha256) await assert.rejects(() => campaign.acquire({ ...frozen, benchmarkDigests: { ...frozen.benchmarkDigests, manifest: "5".repeat(64) } }, "fixed-control", "run-fixed", "coordinator-tampered"), /manifest or root budget changed/) const mutationCell = frozen.schedule.find((cell) => cell.armId === fixed.armId && cell.attemptKind === "mutation")! const benchmarkCell = frozen.schedule.find((cell) => cell.armId === fixed.armId && cell.attemptKind === "development-proof")! - await campaign.admit(fixed, { attemptKey: "role-mutation-attempt", kind: "mutation", identity: { generation: mutationCell.generation, candidateId: mutationCell.candidateId, taskId: mutationCell.taskId, seed: mutationCell.seed, attemptKind: mutationCell.attemptKind }, requestSha256: "a".repeat(64), calls: 1, tokens: 10, spendUsd: 0.1 }) - await campaign.admit(fixed, { attemptKey: "role-benchmark-attempt", kind: "development-proof", identity: { generation: benchmarkCell.generation, candidateId: benchmarkCell.candidateId, taskId: benchmarkCell.taskId, seed: benchmarkCell.seed, attemptKind: benchmarkCell.attemptKind }, sourceCandidateId: "g0-candidate-0", requestSha256: "b".repeat(64), calls: 1, tokens: 10, spendUsd: 0.1 }) + await campaign.admit(fixed, { attemptKey: "role-mutation-attempt", kind: "mutation", identity: { generation: mutationCell.generation, candidateId: mutationCell.candidateId, taskId: mutationCell.taskId, seed: mutationCell.seed, attemptKind: mutationCell.attemptKind }, requestSha256: "a".repeat(64), calls: 1, tokens: 10, spendUsd: 0.01 }) + await campaign.admit(fixed, { attemptKey: "role-benchmark-attempt", kind: "development-proof", identity: { generation: benchmarkCell.generation, candidateId: benchmarkCell.candidateId, taskId: benchmarkCell.taskId, seed: benchmarkCell.seed, attemptKind: benchmarkCell.attemptKind }, sourceCandidateId: "g0-candidate-0", requestSha256: "b".repeat(64), calls: 1, tokens: 10, spendUsd: 0.01 }) const proofSnapshot = await campaign.snapshot(campaignId) - assert.equal(Number((proofSnapshot.campaign as Record).reserved_spend_microusd), 200_000, "both allocations reserve against the same root spend ledger") + assert.equal(Number((proofSnapshot.campaign as Record).reserved_spend_microusd), 20_000, "both allocations reserve against the same root spend ledger") const limits = ["mutation", "benchmark"].map((role) => openRouterBrokerLimitsFromEnvironment(openRouterAllocatedEnvironment(process.env, role as "mutation" | "benchmark"), proofBrokerCampaignId(campaignId, role as "mutation" | "benchmark"), { maxCalls: 1, maxDurationMs: 5_000 })) - assert.deepEqual(limits.map((limit) => limit.maxCampaignSpendUsd), [0.4, 0.2]) + assert.deepEqual(limits.map((limit) => limit.maxCampaignSpendUsd), [0.04, 0.28]) await queue.registerWorker({ workerId: "paired-broker-worker", gatewayId: "proof-gateway", hostname: "proof-host", maxActive: 1, resourceClasses: ["REMOTE_API"] }) for (const [index, role] of (["mutation", "benchmark"] as const).entries()) { const artifact = await queue.putArtifact(Buffer.from(`paired ${role} broker request`)) @@ -212,7 +215,7 @@ test("controller freezes paired broker allocations as separate campaigns under o } const brokerCampaigns = (await _pool.query("SELECT campaign_id,max_spend_microusd FROM headlesscode_rsi_inference_campaigns WHERE campaign_id=ANY($1::text[]) ORDER BY campaign_id", [[proofBrokerCampaignId(campaignId, "benchmark"), proofBrokerCampaignId(campaignId, "mutation")]])).rows assert.equal(brokerCampaigns.length, 2) - assert.deepEqual(brokerCampaigns.map((row) => [row.campaign_id, Number(row.max_spend_microusd)]), [[`${campaignId}:benchmark`, 200_000], [`${campaignId}:mutation`, 400_000]]) + assert.deepEqual(brokerCampaigns.map((row) => [row.campaign_id, Number(row.max_spend_microusd)]), [[`${campaignId}:benchmark`, 280_000], [`${campaignId}:mutation`, 40_000]]) const snapshot = await campaign.snapshot(campaignId) assert.deepEqual((snapshot.arms as Array>).map((arm) => [arm.arm_id, arm.run_id]).sort(), [ ["fixed-control", "run-fixed"], ["self-hosted", "run-self"], @@ -1043,9 +1046,10 @@ test("FileArtifactStore PostgreSQL proof-bound mutation imports only matching du const grant: OpenRouterBrokerGrant = { jobId: job.jobId, leaseToken: jobLease.token, capability: randomUUID(), model: RSI_OPENROUTER_MODEL, limits } await queue.register(grant) await queue.reserve(grant, { callId: "mock-call-000001", requestSha256: "2".repeat(64), reservedUsd: 0.01, reservedTokens: 100, startedAt: new Date().toISOString() }) - await queue.settle(grant, { callId: "mock-call-000001", state: "complete", qualified: true, returnedModel: RSI_OPENROUTER_MODEL, promptTokens: 10, completionTokens: 20, costUsd: 0.001, responseSha256: "3".repeat(64), completedAt: new Date().toISOString() }) + await queue.settle(grant, { callId: "mock-call-000001", state: "complete", qualified: true, returnedModel: RSI_OPENROUTER_MODEL, returnedProvider: "deepinfra", promptTokens: 10, completionTokens: 20, costUsd: 0.001, responseSha256: "3".repeat(64), completedAt: new Date().toISOString() }) const receipt = await queue.receipt(grant) assert.equal(receipt.qualified, true) + assert.deepEqual(receipt.returnedProviders, ["deepinfra"]) const patch = await queue.putArtifact(Buffer.from("verified candidate patch bytes")) const result = { patchSha256: patch.sha256, patchBytes: patch.byteLength, inferenceReceipt: receipt } assert.equal((await queue.complete(jobLease, result)).accepted, true) diff --git a/src/rsi/benchmark/__tests__/openshell-integration.test.ts b/src/rsi/benchmark/__tests__/openshell-integration.test.ts index bc6b5fc..a5627ff 100644 --- a/src/rsi/benchmark/__tests__/openshell-integration.test.ts +++ b/src/rsi/benchmark/__tests__/openshell-integration.test.ts @@ -113,7 +113,10 @@ try { async runHarness(_handle, command) { if (phase === "agent") { events.push("agent-command") - assert.equal(command.split("/opt/headlesscode/src/cli.ts").length - 1, 1, "coding worker command should execute once") + const workerCommand = command.split(" && exec ")[1] ?? "" + assert.equal(command.split(" && exec ").length - 1, 1, "coding worker command should execute once after preflight") + assert.equal(workerCommand.split("/opt/headlesscode/src/cli.ts").length - 1, 1, "fixed-control must execute the pinned release CLI exactly once") + assert.equal(workerCommand.includes("/workspace/src/cli.ts"), false, "fixed-control must not execute the candidate CLI") assert.equal(guestEnvironment.HEADLESSCODE_RSI_INFERENCE_CAPABILITY, lastCapability) assert.equal(guestEnvironment.HEADLESSCODE_OPENROUTER_API_KEY, undefined, "raw provider credential must stay on the supervisor") assert.equal(guestEnvironment.OPENROUTER_API_KEY, undefined)