headlesscode 1.2.2 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,10 +1,13 @@
1
- import { randomUUID } from "node:crypto"
1
+ import { createHash, randomUUID } from "node:crypto"
2
+ import { execFileSync } from "node:child_process"
2
3
  import * as fs from "node:fs/promises"
4
+ import * as path from "node:path"
5
+ import { setTimeout as delay } from "node:timers/promises"
3
6
  import { appendRun, checkpointRun, findActiveRun, readArchive } from "./archive.js"
4
7
  import { parseRsiArgs, rsiHelp } from "./config.js"
5
- import { generateCurriculumProposals, writeCurriculumProposals } from "./curriculum.js"
6
- import { evaluateCandidate, runCommand } from "./evaluator.js"
7
- import { computeFitness, compareFitness } from "./fitness.js"
8
+ import { generateCurriculumProposals, validateCurriculumProposals, writeCurriculumProposals } from "./curriculum.js"
9
+ import { evaluateCandidate } from "./evaluator.js"
10
+ import { computeFitness, compareFitness, comparePairedEvaluation } from "./fitness.js"
8
11
  import { baseModelCandidate, createCombination } from "./models.js"
9
12
  import { runMutation } from "./mutation.js"
10
13
  import { formatRunReport } from "./reports.js"
@@ -12,7 +15,12 @@ import { protectedPathViolations } from "./sandbox.js"
12
15
  import { resolveRoles } from "./roles.js"
13
16
  import { newExperimentJob, paretoFront, selectParentChoices, transitionJob } from "./selection.js"
14
17
  import { captureCandidateTrajectory, exportTrajectoryDatasets } from "./trajectory.js"
15
- import type { CandidateRecord, ExperimentJob, RsiArchive, RsiConfig, RsiHooks, RsiRunRecord, TrajectoryRecord } from "./types.js"
18
+ import { decideAdaptiveContinuation } from "./adaptive.js"
19
+ import { createOpenShellCommandRunner } from "./openshell.js"
20
+ import { createAdversarialEvaluationSnapshotBundle, createEvaluationSnapshotBundle, createMutationSnapshotBundle, applyFleetMutationPatch } from "./openshell.js"
21
+ import { ADVERSARIAL_COMMAND, ADVERSARIAL_PROMPT_VERSION, ADVERSARIAL_TESTS_PATH, buildAdversarialPrompt, parseAdversarialReview, requestAdversarialReview, sha256 } from "./adversarial.js"
22
+ import { fleetJobMatches, PostgresRsiJobQueue, fleetQueuePolicy, requiredFleetJobLeaseMs, type FleetJob, type FleetJobIdentity, type FleetJobPayload } from "./postgres-queue.js"
23
+ import type { AdversarialReviewRecord, CandidateRecord, ExperimentJob, RsiArchive, RsiConfig, RsiHooks, RsiRunRecord, TrajectoryRecord, TrialResult } from "./types.js"
16
24
  import {
17
25
  candidateCommits,
18
26
  changedFiles,
@@ -59,11 +67,298 @@ function initialRun(config: RsiConfig, baseCommit: string, now: string): RsiRunR
59
67
  jobs: [],
60
68
  trajectoryRefs: [],
61
69
  curriculumTasks: [],
70
+ adversarialReviews: [],
62
71
  candidates: [],
63
72
  reports: [],
64
73
  }
65
74
  }
66
75
 
76
+ async function waitForFleetJob(queue: PostgresRsiJobQueue, jobId: string, timeoutMs: number): Promise<FleetJob> {
77
+ const deadline = Date.now() + timeoutMs
78
+ while (Date.now() < deadline) {
79
+ const job = await queue.getJob(jobId)
80
+ if (!job) throw new Error(`fleet job ${jobId} disappeared`)
81
+ if (job.status === "completed" || job.status === "failed" || job.status === "cancelled") return job
82
+ await delay(500)
83
+ }
84
+ await queue.cancel(jobId, "coordinator wait timeout")
85
+ throw new Error(`fleet job ${jobId} exceeded coordinator timeout`)
86
+ }
87
+
88
+ function fleetPayload(config: RsiConfig, candidate: CandidateRecord, snapshot: { content: Buffer; snapshotCommit: string }): FleetJobPayload {
89
+ return {
90
+ schemaVersion: 1,
91
+ snapshotSha256: "", // replaced with the verified artifact digest before enqueue
92
+ snapshotBytes: snapshot.content.byteLength,
93
+ snapshotCommit: snapshot.snapshotCommit,
94
+ task: config.mutationTask,
95
+ model: candidate.model,
96
+ maxIterations: config.maxIterations,
97
+ timeoutMs: config.commandTimeoutMs,
98
+ protectedPaths: config.protectedPaths,
99
+ candidate: {
100
+ id: candidate.id, generation: candidate.generation, parent: candidate.parent,
101
+ mutationKind: candidate.mutationKind, hypothesis: candidate.hypothesis,
102
+ modelCandidateId: candidate.modelCandidateId, computePolicy: config.computePolicy,
103
+ },
104
+ regressionCommand: config.regressionCommand,
105
+ evalCommands: config.evalCommands,
106
+ }
107
+ }
108
+
109
+ function fleetEvaluationResult(value: unknown): { regression: TrialResult; visible: TrialResult[] } {
110
+ if (!value || typeof value !== "object") throw new Error("fleet evaluation result is malformed")
111
+ const result = value as {
112
+ evaluation?: { regression?: unknown; visible?: unknown[] }
113
+ outputArtifact?: { sha256?: unknown; bytes?: unknown }
114
+ }
115
+ const output = result.outputArtifact
116
+ if (!output || typeof output.sha256 !== "string" || !/^[a-f0-9]{64}$/.test(output.sha256) || !Number.isSafeInteger(output.bytes) || Number(output.bytes) < 1) {
117
+ throw new Error("fleet evaluation output artifact reference is missing or malformed")
118
+ }
119
+ const compactTrial = (trial: unknown): TrialResult => {
120
+ if (!trial || typeof trial !== "object") throw new Error("fleet evaluation trial is malformed")
121
+ const item = trial as TrialResult
122
+ if (typeof item.ok !== "boolean" || (item.exitCode !== null && !Number.isSafeInteger(item.exitCode)) || !Number.isFinite(item.durationMs) || item.durationMs < 0 || typeof item.command !== "string" || item.command.length > 512 || typeof item.stdout !== "string" || item.stdout.length > 2048 || typeof item.stderr !== "string" || item.stderr.length > 2048 || item.outputArtifactSha256 !== output.sha256 || item.outputArtifactBytes !== output.bytes) {
123
+ throw new Error("fleet evaluation trial exceeds bounded result fields or lacks its output artifact reference")
124
+ }
125
+ return item
126
+ }
127
+ const evaluation = result.evaluation
128
+ if (!evaluation || !Array.isArray(evaluation.visible)) throw new Error("fleet evaluation trial list is malformed")
129
+ return { regression: compactTrial(evaluation.regression), visible: evaluation.visible.map(compactTrial) }
130
+ }
131
+
132
+ async function validateCurriculumOnFleet(
133
+ tasks: RsiRunRecord["curriculumTasks"],
134
+ config: RsiConfig,
135
+ run: RsiRunRecord,
136
+ queue: PostgresRsiJobQueue,
137
+ now: () => string,
138
+ ): Promise<NonNullable<RsiRunRecord["curriculumTasks"]>> {
139
+ if (!tasks?.length) return []
140
+ return validateCurriculumProposals(tasks, run.baseCommit, async (task, attempt, replayKey) => {
141
+ const replayCandidate: CandidateRecord = {
142
+ id: `${task.fixtureId}-replay-${attempt}`,
143
+ generation: 0,
144
+ parent: run.baseCommit,
145
+ branch: `curriculum-${task.fixtureId}`,
146
+ worktree: config.repoRoot,
147
+ baseCommit: run.baseCommit,
148
+ status: "evaluating",
149
+ model: config.model,
150
+ mutation: task.task,
151
+ createdAt: now(),
152
+ updatedAt: now(),
153
+ commits: [],
154
+ changedFiles: [],
155
+ protectedPathViolations: [],
156
+ }
157
+ const idempotencyKey = `${run.runId}:${replayKey}`
158
+ const snapshot = createEvaluationSnapshotBundle(config.repoRoot, run.baseCommit, config)
159
+ const artifact = await queue.putArtifact(snapshot.content)
160
+ const payload = fleetPayload(config, replayCandidate, snapshot)
161
+ payload.snapshotSha256 = artifact.sha256
162
+ payload.task = `Replay registered curriculum fixture ${task.fixtureId}`
163
+ payload.regressionCommand = "true"
164
+ payload.evalCommands = [task.groundTruthCommand]
165
+ const expected = {
166
+ idempotencyKey,
167
+ runId: run.runId,
168
+ candidateId: replayCandidate.id,
169
+ jobKind: "evaluation" as const,
170
+ resourceClass: "CPU" as const,
171
+ concurrencyKey: `curriculum:${task.fixtureId}`,
172
+ artifactSha256: artifact.sha256,
173
+ payload,
174
+ }
175
+ let job = await queue.getJobByIdempotencyKey(idempotencyKey)
176
+ if (!job) {
177
+ job = await queue.enqueue({
178
+ idempotencyKey,
179
+ runId: run.runId,
180
+ candidateId: replayCandidate.id,
181
+ jobKind: "evaluation",
182
+ resourceClass: "CPU",
183
+ concurrencyKey: expected.concurrencyKey,
184
+ artifactSha256: artifact.sha256,
185
+ payload,
186
+ })
187
+ }
188
+ if (!fleetJobMatches(job, expected)) return { ok: false, exitCode: null, failure: "idempotent curriculum replay key refers to a stale or mismatched fleet job payload" }
189
+ const completed = await waitForFleetJob(queue, job.jobId, requiredFleetJobLeaseMs("evaluation", config.commandTimeoutMs, 1) + 60_000)
190
+ if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") return { ok: false, exitCode: null, failure: completed.error ?? `OpenShell curriculum replay ended ${completed.status}` }
191
+ try {
192
+ const evaluation = fleetEvaluationResult(completed.result)
193
+ const trial = evaluation.visible[0]
194
+ if (!trial) return { ok: false, exitCode: null, failure: "OpenShell curriculum replay returned no fixture trial" }
195
+ return {
196
+ ok: trial.ok,
197
+ exitCode: trial.exitCode,
198
+ outputSha256: trial.outputArtifactSha256,
199
+ ...(!trial.ok ? { failure: `fixture command exited ${trial.exitCode ?? "unknown"}` } : {}),
200
+ }
201
+ } catch (error) {
202
+ return { ok: false, exitCode: null, failure: error instanceof Error ? error.message : String(error) }
203
+ }
204
+ }, now())
205
+ }
206
+
207
+ async function runAdversarialStage(
208
+ candidate: CandidateRecord,
209
+ evaluation: { regression: TrialResult; visible: TrialResult[] },
210
+ config: RsiConfig,
211
+ run: RsiRunRecord,
212
+ queue: PostgresRsiJobQueue,
213
+ hooks: RsiHooks,
214
+ now: () => string,
215
+ ): Promise<AdversarialReviewRecord> {
216
+ const role = config.roles?.adversary
217
+ if (!role) throw new Error("adversary role is not configured")
218
+ const base: AdversarialReviewRecord = {
219
+ candidateId: candidate.id, role: "adversary", provider: role.provider, model: role.model,
220
+ promptVersion: ADVERSARIAL_PROMPT_VERSION, promptSha256: "", promptArtifactSha256: "",
221
+ createdAt: now(), status: "error", findings: [], testResults: [], penaltyPoints: 0,
222
+ }
223
+ let internalJob: ExperimentJob | undefined
224
+ try {
225
+ const workerRole = config.roles?.worker
226
+ const workerModel = workerRole?.model ?? candidate.model
227
+ if (!role.model?.trim() || role.model === workerModel) throw new Error("adversary role must use a different model from the configured worker role")
228
+ const patch = execFileSync("git", ["diff", "--binary", "--no-ext-diff", "--no-renames", candidate.baseCommit + "...HEAD"], { cwd: candidate.worktree, encoding: "buffer", maxBuffer: 4 * 1024 * 1024 }).toString("utf8")
229
+ if (!patch || Buffer.byteLength(patch) > 512 * 1024) throw new Error("candidate diff is empty or exceeds the 512 KiB reviewer input limit")
230
+ const prompt = buildAdversarialPrompt(candidate, patch, evaluation)
231
+ const promptBytes = Buffer.from(prompt)
232
+ if (promptBytes.byteLength > 600 * 1024) throw new Error("adversarial prompt exceeds the 600 KiB limit")
233
+ const promptArtifact = await queue.putArtifact(promptBytes)
234
+ base.promptSha256 = sha256(promptBytes)
235
+ base.promptArtifactSha256 = promptArtifact.sha256
236
+ const raw = await (hooks.runAdversary ? hooks.runAdversary({ role, prompt }) : requestAdversarialReview(role, prompt))
237
+ const resultBytes = Buffer.from(raw)
238
+ const resultArtifact = await queue.putArtifact(resultBytes)
239
+ base.resultSha256 = sha256(resultBytes)
240
+ base.resultArtifactSha256 = resultArtifact.sha256
241
+ const parsed = parseAdversarialReview(raw, candidate.changedFiles)
242
+ base.summary = parsed.summary
243
+ base.findings = parsed.findings
244
+ base.penaltyPoints = Math.min(10, parsed.findings.reduce((points, finding) => points + (finding.severity === "major" ? 5 : 1), 0))
245
+ const tests = Buffer.from(JSON.stringify({ schemaVersion: 1, tests: parsed.tests }))
246
+ const testsArtifact = await queue.putArtifact(tests)
247
+ base.testArtifactSha256 = testsArtifact.sha256
248
+ const candidateCommit = execFileSync("git", ["rev-parse", "HEAD"], { cwd: candidate.worktree, encoding: "utf8" }).trim()
249
+ const snapshot = createAdversarialEvaluationSnapshotBundle(candidate.worktree, candidateCommit, config, tests)
250
+ const snapshotArtifact = await queue.putArtifact(snapshot.content)
251
+ base.snapshotArtifactSha256 = snapshotArtifact.sha256
252
+ internalJob = newExperimentJob("adversarial", { class: "CPU", units: 1, concurrencyKey: "adversarial:" + candidate.id }, now(), candidate.id)
253
+ run.jobs ??= []
254
+ run.jobs.push(internalJob)
255
+ const candidateId = candidate.id + "-adversarial"
256
+ const payload = fleetPayload(config, { ...candidate, id: candidateId, baseCommit: snapshot.snapshotCommit }, snapshot)
257
+ payload.snapshotSha256 = snapshotArtifact.sha256
258
+ payload.task = "Run supervisor-generated adversarial checks for " + candidate.id
259
+ payload.regressionCommand = "true"
260
+ payload.evalCommands = [ADVERSARIAL_COMMAND]
261
+ const idempotencyKey = run.runId + ":" + candidate.id + ":adversarial:" + testsArtifact.sha256
262
+ const expected = {
263
+ idempotencyKey, runId: run.runId, candidateId, jobKind: "evaluation" as const,
264
+ resourceClass: "CPU" as const, concurrencyKey: "adversarial:" + candidate.id,
265
+ artifactSha256: snapshotArtifact.sha256, payload,
266
+ }
267
+ let job = await queue.getJobByIdempotencyKey(idempotencyKey)
268
+ if (!job) job = await queue.enqueue({ ...expected, payload })
269
+ if (!fleetJobMatches(job, expected)) throw new Error("existing adversarial evaluation idempotency key has mismatched test artifact or execution payload")
270
+ const completed = await waitForFleetJob(queue, job.jobId, requiredFleetJobLeaseMs("evaluation", config.commandTimeoutMs, 1) + 60_000)
271
+ if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") throw new Error(completed.error ?? "OpenShell adversarial tests ended " + completed.status)
272
+ const result = fleetEvaluationResult(completed.result)
273
+ const trial = result.visible[0]
274
+ if (!trial) throw new Error("OpenShell adversarial job returned no test result")
275
+ const passedIds = new Set(trial.stdout.split("\n").filter((line) => line.startsWith("PASS ")).map((line) => line.slice(5).trim()))
276
+ base.testResults = parsed.tests.map((test) => ({ id: test.id, passed: trial.ok && passedIds.has(test.id), outputArtifactSha256: trial.outputArtifactSha256, ...(!trial.ok || !passedIds.has(test.id) ? { failure: trial.stdout.slice(0, 1000) || "OpenShell adversarial test failed with exit " + (trial.exitCode ?? "unknown") } : {}) }))
277
+ base.jobId = job.jobId
278
+ const allTestsPassed = trial.ok && base.testResults.length === parsed.tests.length && base.testResults.every((entry) => entry.passed)
279
+ base.status = allTestsPassed ? "passed" : "failed"
280
+ internalJob.status = allTestsPassed ? "completed" : "failed"
281
+ internalJob.updatedAt = now()
282
+ if (!allTestsPassed) internalJob.error = "adversarial test failure (" + (trial.exitCode ?? "unknown") + ")"
283
+ return base
284
+ } catch (error) {
285
+ const message = error instanceof Error ? error.message : String(error)
286
+ base.error = message.slice(0, 2000)
287
+ base.status = "error"
288
+ if (internalJob) {
289
+ internalJob.status = "failed"
290
+ internalJob.updatedAt = now()
291
+ internalJob.error = base.error
292
+ }
293
+ try {
294
+ const failureArtifact = await queue.putArtifact(Buffer.from(base.error))
295
+ base.errorSha256 = sha256(base.error)
296
+ base.errorArtifactSha256 = failureArtifact.sha256
297
+ } catch { /* preserve the primary failure if artifact storage itself is unavailable */ }
298
+ return base
299
+ }
300
+ }
301
+
302
+ async function evaluateBaselineLocal(config: RsiConfig, baseCommit: string, runner: RsiHooks["runCommand"]): Promise<NonNullable<RsiRunRecord["baselineEvaluation"]>> {
303
+ const name = `rsi-baseline-${randomUUID().slice(0, 12)}`
304
+ const baselineRoot = path.join(config.repoRoot, ".worktrees", name)
305
+ execFileSync("git", ["worktree", "add", "--detach", baselineRoot, baseCommit], { cwd: config.repoRoot, stdio: "pipe" })
306
+ try {
307
+ return await evaluateCandidate(config, baselineRoot, baseCommit, runner)
308
+ } finally {
309
+ execFileSync("git", ["worktree", "remove", "--force", baselineRoot], { cwd: config.repoRoot, stdio: "pipe" })
310
+ }
311
+ }
312
+
313
+ async function evaluateBaselineFleet(
314
+ config: RsiConfig,
315
+ baseCommit: string,
316
+ queue: PostgresRsiJobQueue,
317
+ runId: string,
318
+ ): Promise<NonNullable<RsiRunRecord["baselineEvaluation"]>> {
319
+ const idempotencyKey = `${runId}:baseline:evaluation`
320
+ const snapshot = createEvaluationSnapshotBundle(config.repoRoot, baseCommit, config)
321
+ const artifact = await queue.putArtifact(snapshot.content)
322
+ const baseline: CandidateRecord = {
323
+ id: "baseline",
324
+ generation: 0,
325
+ parent: "baseline",
326
+ branch: "baseline",
327
+ worktree: config.repoRoot,
328
+ baseCommit,
329
+ status: "evaluating",
330
+ model: config.model,
331
+ mutation: "Baseline visible evaluation",
332
+ createdAt: new Date().toISOString(),
333
+ updatedAt: new Date().toISOString(),
334
+ commits: [],
335
+ changedFiles: [],
336
+ protectedPathViolations: [],
337
+ }
338
+ const payload = fleetPayload(config, baseline, snapshot)
339
+ payload.snapshotSha256 = artifact.sha256
340
+ const expected: FleetJobIdentity = {
341
+ idempotencyKey, runId, candidateId: "baseline", jobKind: "evaluation",
342
+ resourceClass: "CPU", concurrencyKey: "eval:baseline", artifactSha256: artifact.sha256, payload,
343
+ }
344
+ const existing = await queue.getJobByIdempotencyKey(idempotencyKey)
345
+ const job = existing ?? await queue.enqueue(expected)
346
+ if (!fleetJobMatches(job, expected)) throw new Error("existing baseline evaluation idempotency key has mismatched execution identity or payload")
347
+
348
+ const completed = await waitForFleetJob(queue, job.jobId, requiredFleetJobLeaseMs("evaluation", config.commandTimeoutMs, config.evalCommands.length) + 60_000)
349
+ if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") throw new Error(completed.error ?? `fleet baseline evaluation ended ${completed.status}`)
350
+ const evaluation = fleetEvaluationResult(completed.result)
351
+ const trials = [evaluation.regression, ...evaluation.visible]
352
+ let resultIndex = 0
353
+ const baselineRoot = path.join(config.repoRoot, ".worktrees", `rsi-baseline-${randomUUID().slice(0, 12)}`)
354
+ execFileSync("git", ["worktree", "add", "--detach", baselineRoot, baseCommit], { cwd: config.repoRoot, stdio: "pipe" })
355
+ try {
356
+ return await evaluateCandidate(config, baselineRoot, baseCommit, async () => trials[resultIndex++] as TrialResult)
357
+ } finally {
358
+ execFileSync("git", ["worktree", "remove", "--force", baselineRoot], { cwd: config.repoRoot, stdio: "pipe" })
359
+ }
360
+ }
361
+
67
362
  function recoverInterruptedCandidates(run: RsiRunRecord, now: string): void {
68
363
  for (const candidate of run.candidates) {
69
364
  if (candidate.status === "mutating" || candidate.status === "evaluating") {
@@ -74,6 +369,24 @@ function recoverInterruptedCandidates(run: RsiRunRecord, now: string): void {
74
369
  }
75
370
  }
76
371
 
372
+ async function ensureCandidateWorktree(config: RsiConfig, candidate: CandidateRecord): Promise<{ created: boolean; commits: number }> {
373
+ let created = false
374
+ try {
375
+ await fs.access(candidate.worktree)
376
+ } catch (error) {
377
+ if ((error as NodeJS.ErrnoException).code !== "ENOENT") throw error
378
+ await createCandidateWorktree(config, candidate)
379
+ created = true
380
+ }
381
+ const top = execFileSync("git", ["rev-parse", "--show-toplevel"], { cwd: candidate.worktree, encoding: "utf8" }).trim()
382
+ if (path.resolve(top) !== path.resolve(candidate.worktree)) throw new Error("candidate worktree path resolved outside its recorded location")
383
+ const commits = Number(execFileSync("git", ["rev-list", "--count", `${candidate.baseCommit}..HEAD`], { cwd: candidate.worktree, encoding: "utf8" }).trim())
384
+ if (!Number.isSafeInteger(commits) || commits < 0 || commits > 1) throw new Error("candidate worktree has an unexpected commit count while resuming")
385
+ const status = execFileSync("git", ["status", "--porcelain=v1", "-z", "--untracked-files=all"], { cwd: candidate.worktree, encoding: "buffer" })
386
+ if (status.byteLength > 0) throw new Error("candidate worktree has uncommitted state; refusing unsafe fleet resume")
387
+ return { created, commits }
388
+ }
389
+
77
390
  async function captureCandidate(
78
391
  candidate: CandidateRecord,
79
392
  config: RsiConfig,
@@ -93,7 +406,14 @@ async function captureCandidate(
93
406
  }
94
407
  }
95
408
 
96
- export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<RsiRunRecord> {
409
+ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverride?: PostgresRsiJobQueue): Promise<RsiRunRecord> {
410
+ if (config.computePolicy === "adaptive-independent") {
411
+ const trajectoryLimit = config.maxTrajectories ?? config.population + 1
412
+ const iterationLimit = config.maxTotalIterations ?? trajectoryLimit * config.maxIterations
413
+ if (config.population < 2 || config.generations !== 1 || trajectoryLimit < config.population || trajectoryLimit > config.population + 1 || iterationLimit < config.population * config.maxIterations || iterationLimit > trajectoryLimit * config.maxIterations) {
414
+ throw new Error("invalid adaptive-independent budgets: require population >= 2, exactly one follow-up generation, and trajectory/iteration limits for only the initial cohort plus at most one candidate")
415
+ }
416
+ }
97
417
  const now = hooks.now ?? (() => new Date().toISOString())
98
418
  const archive = await readArchive(config.archiveDir)
99
419
  const roles = resolveRoles(config.roles, process.env, config.model)
@@ -102,7 +422,8 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
102
422
  if (config.resumeRunId && !resumed) throw new Error(`no active RSI run found for --resume ${config.resumeRunId}`)
103
423
  const baseCommit = resumed?.baseCommit ?? (await resolveBaseCommit(config))
104
424
  const run = resumed ?? initialRun({ ...config, model: workerModel, roles }, baseCommit, now())
105
- run.generations = config.generations
425
+ const stageCount = config.computePolicy === "adaptive-independent" ? 2 : config.generations
426
+ run.generations = stageCount
106
427
  run.model = workerModel
107
428
  run.parentSelectionPolicy = config.parentSelectionPolicy ?? run.parentSelectionPolicy ?? "champion-specialist-novelty"
108
429
  run.modelCandidates ??= [baseModelCandidate(config.model, now(), config.modelCandidateId ?? "model-base")]
@@ -113,18 +434,55 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
113
434
  run.jobs ??= []
114
435
  run.trajectoryRefs ??= []
115
436
  run.curriculumTasks ??= []
437
+ run.adversarialReviews ??= []
116
438
  run.parentChoices ??= {}
117
- if (resumed) recoverInterruptedCandidates(run, now())
118
439
  const runConfig: RsiConfig = { ...config, model: workerModel, roles, seed: `${config.seed}-${run.runId.slice(-8)}` }
119
440
  const trajectories: TrajectoryRecord[] = []
120
441
 
121
- if (!config.dryRun && !run.baseline) {
122
- run.baseline = await (hooks.runCommand ?? runCommand)("npm test", config.repoRoot, config.commandTimeoutMs)
442
+ if (!config.dryRun && config.hiddenEvalCommands.length > 0) {
443
+ throw new Error("RSI hidden evaluations are disabled until their evaluator assets can remain unavailable to candidate processes")
444
+ }
445
+ const fleetQueue = !config.dryRun && (queueOverride !== undefined || !hooks.runMutation)
446
+ ? queueOverride ?? PostgresRsiJobQueue.fromEnvironment(fleetQueuePolicy(process.env.HEADLESSCODE_RSI_QUEUE_NAME?.trim() || "rsi-default", config.maxConcurrent, process.env), process.env)
447
+ : undefined
448
+ if (!config.dryRun && config.roles?.adversary && !fleetQueue) throw new Error("RSI adversarial break tests require the OpenShell fleet evaluation path")
449
+ const commandRunner = hooks.runCommand ?? (fleetQueue ? undefined : createOpenShellCommandRunner(runConfig))
450
+ try {
451
+ if (fleetQueue) {
452
+ if (fleetQueue.artifactBackend !== "s3") throw new Error("RSI production runs require a shared S3-compatible artifact store")
453
+ await fleetQueue.migrate()
454
+ if (await fleetQueue.activeWorkerCount() < 1) throw new Error("no live OpenShell RSI worker is registered for the configured queue")
455
+ }
456
+ if (resumed && !fleetQueue) recoverInterruptedCandidates(run, now())
457
+ if (!config.dryRun && !run.baselineEvaluation) {
458
+ run.baselineEvaluation = fleetQueue
459
+ ? await evaluateBaselineFleet(runConfig, baseCommit, fleetQueue, run.runId)
460
+ : await evaluateBaselineLocal(runConfig, baseCommit, commandRunner)
461
+ run.baselineFitness = computeFitness(run.baselineEvaluation)
462
+ run.baseline = run.baselineEvaluation.regression
123
463
  log(hooks, `[rsi] baseline: ${run.baseline.ok ? "pass" : "fail"}`)
124
464
  await checkpointRun(config.archiveDir, run)
125
465
  }
126
466
 
127
- for (let generation = 0; generation < config.generations; generation++) {
467
+ for (let generation = 0; generation < stageCount; generation++) {
468
+ if (config.computePolicy === "adaptive-independent" && generation > 0) {
469
+ let previousDecision = run.adaptiveSearch?.find((entry) => entry.generation === generation - 1)
470
+ if (!previousDecision) {
471
+ const priorCandidates = run.candidates.filter((candidate) => candidate.generation === generation - 1)
472
+ previousDecision = decideAdaptiveContinuation({
473
+ generation: generation - 1,
474
+ generationCandidates: priorCandidates,
475
+ allCandidates: run.candidates,
476
+ config,
477
+ elapsedMs: Math.max(0, Date.now() - Date.parse(run.startedAt)),
478
+ decidedAt: now(),
479
+ })
480
+ run.adaptiveSearch ??= []
481
+ run.adaptiveSearch.push(previousDecision)
482
+ await checkpointRun(config.archiveDir, run)
483
+ }
484
+ if (previousDecision.decision !== "continue") break
485
+ }
128
486
  let candidates = run.candidates.filter((candidate) => candidate.generation === generation)
129
487
  if (candidates.length === 0) {
130
488
  const merged = combinedArchive(archive, run)
@@ -134,7 +492,8 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
134
492
  const fallbackBase = run.selected
135
493
  ? run.candidates.find((candidate) => candidate.id === run.selected)?.commits[0] ?? run.baseCommit
136
494
  : run.baseCommit
137
- candidates = Array.from({ length: config.population }, (_, index) => {
495
+ const candidateCount = config.computePolicy === "adaptive-independent" && generation > 0 ? 1 : config.population
496
+ candidates = Array.from({ length: candidateCount }, (_, index) => {
138
497
  const parentChoice = choices[index]
139
498
  const candidate = newCandidate(runConfig, generation, index, parentChoice?.baseCommit ?? fallbackBase, now(), parentChoice)
140
499
  if (parentChoice) run.parentChoices![candidate.id] = parentChoice.reason
@@ -148,6 +507,67 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
148
507
  }
149
508
  if (config.dryRun) continue
150
509
 
510
+ const mutationJobs = new Map<string, FleetJob>()
511
+ const evaluationJobs = new Map<string, FleetJob>()
512
+ const managedWorktrees = new Set<string>()
513
+ if (fleetQueue) {
514
+ try {
515
+ for (const candidate of candidates) {
516
+ if (isTerminal(candidate)) continue
517
+ const worktree = await ensureCandidateWorktree(runConfig, candidate)
518
+ managedWorktrees.add(candidate.id)
519
+ const key = `${run.runId}:${candidate.id}:mutation`
520
+ const snapshot = createMutationSnapshotBundle(candidate.worktree, candidate.baseCommit, runConfig)
521
+ const artifact = await fleetQueue.putArtifact(snapshot.content)
522
+ const payload = fleetPayload(runConfig, candidate, snapshot)
523
+ payload.snapshotSha256 = artifact.sha256
524
+ const expected: FleetJobIdentity = {
525
+ idempotencyKey: key, runId: run.runId, candidateId: candidate.id,
526
+ jobKind: "mutation", resourceClass: "LOCAL_GPU", concurrencyKey: `ollama:${candidate.model}`,
527
+ artifactSha256: artifact.sha256, payload,
528
+ }
529
+ const existing = await fleetQueue.getJobByIdempotencyKey(key)
530
+ if (existing && !fleetJobMatches(existing, expected)) throw new Error("existing idempotent mutation job has mismatched execution identity, snapshot, or task payload")
531
+ if (worktree.commits > 0) {
532
+ const completed = existing
533
+ if (completed?.candidateId !== candidate.id || completed.jobKind !== "mutation" || completed.status !== "completed" || !completed.result || typeof completed.result !== "object") {
534
+ throw new Error("candidate worktree contains an applied mutation commit without its completed idempotent fleet job")
535
+ }
536
+ const result = completed.result as { patchSha256?: string; patchBytes?: number }
537
+ if (!result.patchSha256 || !Number.isSafeInteger(result.patchBytes) || Number(result.patchBytes) < 1) throw new Error("completed fleet mutation has no valid patch artifact reference for resume")
538
+ const patch = await fleetQueue.getArtifact(result.patchSha256)
539
+ const appliedPatch = execFileSync("git", ["diff", "--binary", "--no-ext-diff", "--no-renames", `${candidate.baseCommit}...HEAD`], { cwd: candidate.worktree, encoding: "buffer" })
540
+ if (patch.byteLength !== result.patchBytes || appliedPatch.byteLength !== result.patchBytes || createHash("sha256").update(appliedPatch).digest("hex") !== result.patchSha256) {
541
+ throw new Error("candidate worktree commit does not match the completed fleet mutation patch")
542
+ }
543
+ let recorded = run.jobs.find((entry) => entry.candidateId === candidate.id && entry.kind === "mutation")
544
+ if (!recorded) {
545
+ recorded = newExperimentJob("mutation", { class: "LOCAL_GPU", units: 1, concurrencyKey: "ollama" }, now(), candidate.id)
546
+ run.jobs.push(recorded)
547
+ }
548
+ if (recorded.status !== "completed") replaceJob(run, transitionJob(recorded, "completed", now()))
549
+ candidate.status = "evaluating"
550
+ candidate.changedFiles = await changedFiles(candidate.worktree, candidate.baseCommit)
551
+ candidate.commits = await candidateCommits(candidate.worktree, candidate.baseCommit)
552
+ candidate.protectedPathViolations = protectedPathViolations(candidate.changedFiles, config.protectedPaths)
553
+ candidate.updatedAt = now()
554
+ await checkpointRun(config.archiveDir, run)
555
+ continue
556
+ }
557
+ const job = existing ?? await fleetQueue.enqueue(expected)
558
+ if (!fleetJobMatches(job, expected)) throw new Error("enqueued mutation job does not match its expected execution identity or payload")
559
+ mutationJobs.set(candidate.id, job)
560
+ }
561
+ } catch (error) {
562
+ for (const candidate of candidates.filter((entry) => managedWorktrees.has(entry.id))) {
563
+ await removeCandidateWorktree(runConfig, candidate).catch(() => undefined)
564
+ }
565
+ throw error
566
+ }
567
+ }
568
+
569
+ // Resolve every mutation first. Evaluation submission is a separate
570
+ // generation-wide phase so CPU workers can claim the full ready set.
151
571
  for (const candidate of candidates) {
152
572
  if (isTerminal(candidate)) continue
153
573
  let job = run.jobs.find((entry) => entry.candidateId === candidate.id && entry.kind === "mutation")
@@ -155,37 +575,115 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
155
575
  job = newExperimentJob("mutation", { class: "LOCAL_GPU", units: 1, concurrencyKey: "ollama" }, now(), candidate.id)
156
576
  run.jobs.push(job)
157
577
  }
158
- candidate.status = "mutating"
578
+ try {
579
+ const worktree = fleetQueue ? await ensureCandidateWorktree(runConfig, candidate) : { created: false, commits: 0 }
580
+ managedWorktrees.add(candidate.id)
581
+ if (worktree.commits === 0) {
582
+ candidate.status = "mutating"
583
+ candidate.updatedAt = now()
584
+ if (job.status !== "running") job = transitionJob(job, "running", now())
585
+ replaceJob(run, job)
586
+ await checkpointRun(config.archiveDir, run)
587
+ if (fleetQueue) {
588
+ const submitted = mutationJobs.get(candidate.id)
589
+ if (!submitted) throw new Error("candidate mutation was not admitted to the RSI fleet queue")
590
+ const completed = await waitForFleetJob(fleetQueue, submitted.jobId, config.commandTimeoutMs * 4 + 180_000)
591
+ if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") throw new Error(completed.error ?? `fleet mutation ended ${completed.status}`)
592
+ const result = completed.result as { patchSha256?: string; patchBytes?: number }
593
+ if (!result.patchSha256 || !Number.isSafeInteger(result.patchBytes) || Number(result.patchBytes) < 1) throw new Error("fleet mutation result is missing its patch artifact reference")
594
+ const patch = await fleetQueue.getArtifact(result.patchSha256)
595
+ if (patch.byteLength !== result.patchBytes) throw new Error("fleet mutation patch length did not match result metadata")
596
+ applyFleetMutationPatch(patch, candidate, runConfig)
597
+ } else {
598
+ await createCandidateWorktree(runConfig, candidate)
599
+ managedWorktrees.add(candidate.id)
600
+ const mutation = await (hooks.runMutation ?? runMutation)(candidate, runConfig)
601
+ if (!mutation.ok) throw new Error(mutation.error ?? "RSI mutation failed")
602
+ }
603
+ job = transitionJob(job, "completed", now())
604
+ replaceJob(run, job)
605
+ }
606
+ candidate.status = "evaluating"
607
+ candidate.changedFiles = await changedFiles(candidate.worktree, candidate.baseCommit)
608
+ candidate.commits = await candidateCommits(candidate.worktree, candidate.baseCommit)
609
+ candidate.protectedPathViolations = protectedPathViolations(candidate.changedFiles, config.protectedPaths)
610
+ candidate.updatedAt = now()
611
+ await checkpointRun(config.archiveDir, run)
612
+ } catch (error) {
613
+ candidate.status = "failed"
614
+ candidate.failure = error instanceof Error ? error.message : String(error)
159
615
  candidate.updatedAt = now()
160
- job = transitionJob(job, "running", now())
616
+ job = transitionJob(job, "failed", now(), candidate.failure)
161
617
  replaceJob(run, job)
618
+ await captureCandidate(candidate, runConfig, run, trajectories, now(), hooks)
619
+ log(hooks, `[rsi] ${candidate.id}: mutation failed: ${candidate.failure}`)
620
+ if (!config.keepWorktrees && managedWorktrees.has(candidate.id)) await removeCandidateWorktree(runConfig, candidate).catch(() => undefined)
162
621
  await checkpointRun(config.archiveDir, run)
163
- let worktreeCreated = false
164
- try {
165
- await createCandidateWorktree(runConfig, candidate)
166
- worktreeCreated = true
167
- const mutation = await (hooks.runMutation ?? runMutation)(candidate, runConfig)
168
- if (!mutation.ok) {
622
+ }
623
+ await checkpointRun(config.archiveDir, run)
624
+ }
625
+
626
+ if (fleetQueue) {
627
+ for (const candidate of candidates.filter((entry) => entry.status === "evaluating")) {
628
+ try {
629
+ const key = `${run.runId}:${candidate.id}:evaluation`
630
+ const head = execFileSync("git", ["rev-parse", "HEAD"], { cwd: candidate.worktree, encoding: "utf8" }).trim()
631
+ const snapshot = createEvaluationSnapshotBundle(candidate.worktree, head, runConfig)
632
+ const artifact = await fleetQueue.putArtifact(snapshot.content)
633
+ const payload = fleetPayload(runConfig, candidate, snapshot)
634
+ payload.snapshotSha256 = artifact.sha256
635
+ const expected: FleetJobIdentity = {
636
+ idempotencyKey: key, runId: run.runId, candidateId: candidate.id,
637
+ jobKind: "evaluation", resourceClass: "CPU", concurrencyKey: `eval:${candidate.id}`,
638
+ artifactSha256: artifact.sha256, payload,
639
+ }
640
+ const existing = await fleetQueue.getJobByIdempotencyKey(key)
641
+ if (existing && !fleetJobMatches(existing, expected)) throw new Error("existing idempotent evaluation job has mismatched execution identity, snapshot, or evaluation configuration")
642
+ const job = existing ?? await fleetQueue.enqueue(expected)
643
+ if (!fleetJobMatches(job, expected)) throw new Error("enqueued evaluation job does not match its expected execution identity or payload")
644
+ evaluationJobs.set(candidate.id, job)
645
+ } catch (error) {
169
646
  candidate.status = "failed"
170
- candidate.failure = mutation.error
171
- candidate.result = mutation.result
647
+ candidate.failure = error instanceof Error ? error.message : String(error)
172
648
  candidate.updatedAt = now()
173
- job = transitionJob(job, "failed", now(), mutation.error)
174
- replaceJob(run, job)
175
- await captureCandidate(candidate, runConfig, run, trajectories, now(), hooks)
176
- log(hooks, `[rsi] ${candidate.id}: mutation failed${mutation.error ? `: ${mutation.error}` : ""}`)
177
- continue
649
+ log(hooks, `[rsi] ${candidate.id}: evaluation admission failed: ${candidate.failure}`)
650
+ if (!config.keepWorktrees && managedWorktrees.has(candidate.id)) await removeCandidateWorktree(runConfig, candidate).catch(() => undefined)
651
+ await checkpointRun(config.archiveDir, run)
652
+ }
653
+ }
654
+ }
655
+
656
+ for (const candidate of candidates.filter((entry) => entry.status === "evaluating")) {
657
+ let job = run.jobs.find((entry) => entry.candidateId === candidate.id && entry.kind === "mutation")!
658
+ try {
659
+ let evaluation
660
+ if (fleetQueue) {
661
+ const submitted = evaluationJobs.get(candidate.id) ?? await fleetQueue.getJobByIdempotencyKey(`${run.runId}:${candidate.id}:evaluation`)
662
+ if (!submitted) throw new Error("candidate evaluation was not admitted to the RSI fleet queue")
663
+ const completed = await waitForFleetJob(fleetQueue, submitted.jobId, requiredFleetJobLeaseMs("evaluation", config.commandTimeoutMs, config.evalCommands.length) + 60_000)
664
+ if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") throw new Error(completed.error ?? `fleet evaluation ended ${completed.status}`)
665
+ const visible = fleetEvaluationResult(completed.result)
666
+ const trials = [visible.regression, ...visible.visible]
667
+ let resultIndex = 0
668
+ evaluation = await evaluateCandidate(runConfig, candidate.worktree, candidate.baseCommit, async () => trials[resultIndex++] as Awaited<ReturnType<NonNullable<typeof hooks.runCommand>>>)
669
+ } else {
670
+ evaluation = await evaluateCandidate(runConfig, candidate.worktree, candidate.baseCommit, commandRunner)
671
+ }
672
+ if (runConfig.roles?.adversary) {
673
+ if (!fleetQueue) throw new Error("adversarial break tests require an OpenShell fleet worker")
674
+ const review = await runAdversarialStage(candidate, evaluation, runConfig, run, fleetQueue, hooks, now)
675
+ run.adversarialReviews ??= []
676
+ run.adversarialReviews.push(review)
677
+ evaluation.adversarialPass = review.status === "passed" && review.testResults.length > 0 && review.testResults.every((entry) => entry.passed)
678
+ evaluation.adversarialPenalty = review.penaltyPoints
679
+ await checkpointRun(config.archiveDir, run)
178
680
  }
179
- candidate.status = "evaluating"
180
- candidate.changedFiles = await changedFiles(candidate.worktree, candidate.baseCommit)
181
- candidate.commits = await candidateCommits(candidate.worktree, candidate.baseCommit)
182
- candidate.protectedPathViolations = protectedPathViolations(candidate.changedFiles, config.protectedPaths)
183
- const evaluation = await evaluateCandidate(runConfig, candidate.worktree, candidate.baseCommit, hooks.runCommand ?? runCommand)
184
681
  candidate.changedFiles = evaluation.changedFiles
185
682
  candidate.commits = candidate.commits.length > 0 ? candidate.commits : await candidateCommits(candidate.worktree, candidate.baseCommit)
186
683
  candidate.result = evaluation.regression
187
684
  candidate.fitness = computeFitness({ ...evaluation, protectedPathViolation: candidate.protectedPathViolations.length > 0 })
188
- candidate.status = candidate.fitness.score > 0 ? "accepted" : "rejected"
685
+ if (run.baselineEvaluation) candidate.pairedComparison = comparePairedEvaluation(run.baselineEvaluation, evaluation)
686
+ candidate.status = candidate.fitness.score > 0 && candidate.pairedComparison?.improved === true ? "accepted" : "rejected"
189
687
  candidate.updatedAt = now()
190
688
  const combination = createCombination(candidate.id, candidate.modelCandidateId ?? "model-base", now())
191
689
  combination.status = candidate.status === "accepted" ? "accepted" : "rejected"
@@ -202,14 +700,11 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
202
700
  candidate.updatedAt = now()
203
701
  job = transitionJob(job, "failed", now(), candidate.failure)
204
702
  replaceJob(run, job)
205
- log(hooks, `[rsi] ${candidate.id}: lifecycle failed: ${candidate.failure}`)
703
+ await captureCandidate(candidate, runConfig, run, trajectories, now(), hooks)
704
+ log(hooks, `[rsi] ${candidate.id}: evaluation failed: ${candidate.failure}`)
206
705
  } finally {
207
- if (worktreeCreated && !config.keepWorktrees) {
208
- try {
209
- await removeCandidateWorktree(runConfig, candidate)
210
- } catch (error) {
211
- log(hooks, `[rsi] ${candidate.id}: worktree cleanup failed: ${String(error)}`)
212
- }
706
+ if (!config.keepWorktrees && managedWorktrees.has(candidate.id)) {
707
+ await removeCandidateWorktree(runConfig, candidate).catch((error) => log(hooks, `[rsi] ${candidate.id}: worktree cleanup failed: ${String(error)}`))
213
708
  }
214
709
  await checkpointRun(config.archiveDir, run)
215
710
  }
@@ -220,12 +715,33 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
220
715
  const ranked = [...accepted].sort((a, b) => compareFitness(b.fitness, a.fitness))
221
716
  run.selectedCandidates = front.map((candidate) => candidate.id)
222
717
  if (ranked[0]) run.selected = ranked[0].id
718
+ if (config.computePolicy === "adaptive-independent") {
719
+ run.adaptiveSearch ??= []
720
+ const decision = decideAdaptiveContinuation({
721
+ generation,
722
+ generationCandidates: candidates,
723
+ allCandidates: run.candidates,
724
+ config,
725
+ elapsedMs: Math.max(0, Date.now() - Date.parse(run.startedAt)),
726
+ decidedAt: now(),
727
+ })
728
+ const existingDecision = run.adaptiveSearch.findIndex((entry) => entry.generation === generation)
729
+ if (existingDecision >= 0) run.adaptiveSearch[existingDecision] = decision
730
+ else run.adaptiveSearch.push(decision)
731
+ log(hooks, `[rsi] adaptive search: ${decision.decision} after generation ${generation} (${decision.reason})`)
732
+ }
223
733
  await checkpointRun(config.archiveDir, run)
224
734
  }
225
735
 
226
736
  const merged = combinedArchive(archive, run)
227
- const curriculum = generateCurriculumProposals(merged, now())
228
- run.curriculumTasks = curriculum
737
+ let curriculum = generateCurriculumProposals(merged, now())
738
+ if (fleetQueue && !config.dryRun && curriculum.length > 0) {
739
+ curriculum = await validateCurriculumOnFleet(curriculum, runConfig, run, fleetQueue, now)
740
+ run.curriculumTasks = curriculum
741
+ await checkpointRun(config.archiveDir, run)
742
+ } else {
743
+ run.curriculumTasks = curriculum
744
+ }
229
745
  if (!config.dryRun && curriculum.length > 0) {
230
746
  await writeCurriculumProposals(curriculum, config.curriculumDir ?? `${config.archiveDir}/curriculum`)
231
747
  }
@@ -245,6 +761,9 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
245
761
  log(hooks, report.trimEnd())
246
762
  }
247
763
  return run
764
+ } finally {
765
+ if (fleetQueue) await fleetQueue.close()
766
+ }
248
767
  }
249
768
 
250
769
  export async function improveMain(argv: string[]): Promise<number> {