headlesscode 1.2.2 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -16
- package/package.json +7 -3
- package/src/cli.ts +47 -18
- package/src/cloud/openshell-preflight.ts +2 -2
- package/src/cloud/openshell-provider.ts +84 -24
- package/src/llm/ollama.ts +25 -19
- package/src/project-store.ts +4 -1
- package/src/rsi/adaptive.ts +49 -0
- package/src/rsi/adversarial.ts +106 -0
- package/src/rsi/archive.ts +6 -5
- package/src/rsi/artifact-store.ts +158 -0
- package/src/rsi/config.ts +50 -2
- package/src/rsi/controller.ts +561 -42
- package/src/rsi/curriculum.ts +135 -16
- package/src/rsi/evaluator.ts +6 -37
- package/src/rsi/fitness.ts +39 -4
- package/src/rsi/index.ts +1 -0
- package/src/rsi/migrations/001_postgres_fleet_queue.sql +65 -0
- package/src/rsi/migrations/002_external_artifacts_and_job_leases.sql +39 -0
- package/src/rsi/migrations/003_model_training_jobs.sql +6 -0
- package/src/rsi/model-training.ts +256 -0
- package/src/rsi/mutation.ts +1 -77
- package/src/rsi/openshell.ts +639 -0
- package/src/rsi/postgres-queue.ts +424 -0
- package/src/rsi/promote-curriculum.ts +21 -0
- package/src/rsi/reports.ts +29 -2
- package/src/rsi/roles.ts +13 -3
- package/src/rsi/selection.ts +7 -1
- package/src/rsi/training-data.ts +103 -0
- package/src/rsi/trajectory.ts +1 -1
- package/src/rsi/types.ts +115 -2
- package/src/rsi/worker.ts +264 -0
- package/src/rsi/workspace.ts +14 -3
package/src/rsi/controller.ts
CHANGED
|
@@ -1,10 +1,13 @@
|
|
|
1
|
-
import { randomUUID } from "node:crypto"
|
|
1
|
+
import { createHash, randomUUID } from "node:crypto"
|
|
2
|
+
import { execFileSync } from "node:child_process"
|
|
2
3
|
import * as fs from "node:fs/promises"
|
|
4
|
+
import * as path from "node:path"
|
|
5
|
+
import { setTimeout as delay } from "node:timers/promises"
|
|
3
6
|
import { appendRun, checkpointRun, findActiveRun, readArchive } from "./archive.js"
|
|
4
7
|
import { parseRsiArgs, rsiHelp } from "./config.js"
|
|
5
|
-
import { generateCurriculumProposals, writeCurriculumProposals } from "./curriculum.js"
|
|
6
|
-
import { evaluateCandidate
|
|
7
|
-
import { computeFitness, compareFitness } from "./fitness.js"
|
|
8
|
+
import { generateCurriculumProposals, validateCurriculumProposals, writeCurriculumProposals } from "./curriculum.js"
|
|
9
|
+
import { evaluateCandidate } from "./evaluator.js"
|
|
10
|
+
import { computeFitness, compareFitness, comparePairedEvaluation } from "./fitness.js"
|
|
8
11
|
import { baseModelCandidate, createCombination } from "./models.js"
|
|
9
12
|
import { runMutation } from "./mutation.js"
|
|
10
13
|
import { formatRunReport } from "./reports.js"
|
|
@@ -12,7 +15,12 @@ import { protectedPathViolations } from "./sandbox.js"
|
|
|
12
15
|
import { resolveRoles } from "./roles.js"
|
|
13
16
|
import { newExperimentJob, paretoFront, selectParentChoices, transitionJob } from "./selection.js"
|
|
14
17
|
import { captureCandidateTrajectory, exportTrajectoryDatasets } from "./trajectory.js"
|
|
15
|
-
import
|
|
18
|
+
import { decideAdaptiveContinuation } from "./adaptive.js"
|
|
19
|
+
import { createOpenShellCommandRunner } from "./openshell.js"
|
|
20
|
+
import { createAdversarialEvaluationSnapshotBundle, createEvaluationSnapshotBundle, createMutationSnapshotBundle, applyFleetMutationPatch } from "./openshell.js"
|
|
21
|
+
import { ADVERSARIAL_COMMAND, ADVERSARIAL_PROMPT_VERSION, ADVERSARIAL_TESTS_PATH, buildAdversarialPrompt, parseAdversarialReview, requestAdversarialReview, sha256 } from "./adversarial.js"
|
|
22
|
+
import { fleetJobMatches, PostgresRsiJobQueue, fleetQueuePolicy, requiredFleetJobLeaseMs, type FleetJob, type FleetJobIdentity, type FleetJobPayload } from "./postgres-queue.js"
|
|
23
|
+
import type { AdversarialReviewRecord, CandidateRecord, ExperimentJob, RsiArchive, RsiConfig, RsiHooks, RsiRunRecord, TrajectoryRecord, TrialResult } from "./types.js"
|
|
16
24
|
import {
|
|
17
25
|
candidateCommits,
|
|
18
26
|
changedFiles,
|
|
@@ -59,11 +67,298 @@ function initialRun(config: RsiConfig, baseCommit: string, now: string): RsiRunR
|
|
|
59
67
|
jobs: [],
|
|
60
68
|
trajectoryRefs: [],
|
|
61
69
|
curriculumTasks: [],
|
|
70
|
+
adversarialReviews: [],
|
|
62
71
|
candidates: [],
|
|
63
72
|
reports: [],
|
|
64
73
|
}
|
|
65
74
|
}
|
|
66
75
|
|
|
76
|
+
async function waitForFleetJob(queue: PostgresRsiJobQueue, jobId: string, timeoutMs: number): Promise<FleetJob> {
|
|
77
|
+
const deadline = Date.now() + timeoutMs
|
|
78
|
+
while (Date.now() < deadline) {
|
|
79
|
+
const job = await queue.getJob(jobId)
|
|
80
|
+
if (!job) throw new Error(`fleet job ${jobId} disappeared`)
|
|
81
|
+
if (job.status === "completed" || job.status === "failed" || job.status === "cancelled") return job
|
|
82
|
+
await delay(500)
|
|
83
|
+
}
|
|
84
|
+
await queue.cancel(jobId, "coordinator wait timeout")
|
|
85
|
+
throw new Error(`fleet job ${jobId} exceeded coordinator timeout`)
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function fleetPayload(config: RsiConfig, candidate: CandidateRecord, snapshot: { content: Buffer; snapshotCommit: string }): FleetJobPayload {
|
|
89
|
+
return {
|
|
90
|
+
schemaVersion: 1,
|
|
91
|
+
snapshotSha256: "", // replaced with the verified artifact digest before enqueue
|
|
92
|
+
snapshotBytes: snapshot.content.byteLength,
|
|
93
|
+
snapshotCommit: snapshot.snapshotCommit,
|
|
94
|
+
task: config.mutationTask,
|
|
95
|
+
model: candidate.model,
|
|
96
|
+
maxIterations: config.maxIterations,
|
|
97
|
+
timeoutMs: config.commandTimeoutMs,
|
|
98
|
+
protectedPaths: config.protectedPaths,
|
|
99
|
+
candidate: {
|
|
100
|
+
id: candidate.id, generation: candidate.generation, parent: candidate.parent,
|
|
101
|
+
mutationKind: candidate.mutationKind, hypothesis: candidate.hypothesis,
|
|
102
|
+
modelCandidateId: candidate.modelCandidateId, computePolicy: config.computePolicy,
|
|
103
|
+
},
|
|
104
|
+
regressionCommand: config.regressionCommand,
|
|
105
|
+
evalCommands: config.evalCommands,
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function fleetEvaluationResult(value: unknown): { regression: TrialResult; visible: TrialResult[] } {
|
|
110
|
+
if (!value || typeof value !== "object") throw new Error("fleet evaluation result is malformed")
|
|
111
|
+
const result = value as {
|
|
112
|
+
evaluation?: { regression?: unknown; visible?: unknown[] }
|
|
113
|
+
outputArtifact?: { sha256?: unknown; bytes?: unknown }
|
|
114
|
+
}
|
|
115
|
+
const output = result.outputArtifact
|
|
116
|
+
if (!output || typeof output.sha256 !== "string" || !/^[a-f0-9]{64}$/.test(output.sha256) || !Number.isSafeInteger(output.bytes) || Number(output.bytes) < 1) {
|
|
117
|
+
throw new Error("fleet evaluation output artifact reference is missing or malformed")
|
|
118
|
+
}
|
|
119
|
+
const compactTrial = (trial: unknown): TrialResult => {
|
|
120
|
+
if (!trial || typeof trial !== "object") throw new Error("fleet evaluation trial is malformed")
|
|
121
|
+
const item = trial as TrialResult
|
|
122
|
+
if (typeof item.ok !== "boolean" || (item.exitCode !== null && !Number.isSafeInteger(item.exitCode)) || !Number.isFinite(item.durationMs) || item.durationMs < 0 || typeof item.command !== "string" || item.command.length > 512 || typeof item.stdout !== "string" || item.stdout.length > 2048 || typeof item.stderr !== "string" || item.stderr.length > 2048 || item.outputArtifactSha256 !== output.sha256 || item.outputArtifactBytes !== output.bytes) {
|
|
123
|
+
throw new Error("fleet evaluation trial exceeds bounded result fields or lacks its output artifact reference")
|
|
124
|
+
}
|
|
125
|
+
return item
|
|
126
|
+
}
|
|
127
|
+
const evaluation = result.evaluation
|
|
128
|
+
if (!evaluation || !Array.isArray(evaluation.visible)) throw new Error("fleet evaluation trial list is malformed")
|
|
129
|
+
return { regression: compactTrial(evaluation.regression), visible: evaluation.visible.map(compactTrial) }
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
async function validateCurriculumOnFleet(
|
|
133
|
+
tasks: RsiRunRecord["curriculumTasks"],
|
|
134
|
+
config: RsiConfig,
|
|
135
|
+
run: RsiRunRecord,
|
|
136
|
+
queue: PostgresRsiJobQueue,
|
|
137
|
+
now: () => string,
|
|
138
|
+
): Promise<NonNullable<RsiRunRecord["curriculumTasks"]>> {
|
|
139
|
+
if (!tasks?.length) return []
|
|
140
|
+
return validateCurriculumProposals(tasks, run.baseCommit, async (task, attempt, replayKey) => {
|
|
141
|
+
const replayCandidate: CandidateRecord = {
|
|
142
|
+
id: `${task.fixtureId}-replay-${attempt}`,
|
|
143
|
+
generation: 0,
|
|
144
|
+
parent: run.baseCommit,
|
|
145
|
+
branch: `curriculum-${task.fixtureId}`,
|
|
146
|
+
worktree: config.repoRoot,
|
|
147
|
+
baseCommit: run.baseCommit,
|
|
148
|
+
status: "evaluating",
|
|
149
|
+
model: config.model,
|
|
150
|
+
mutation: task.task,
|
|
151
|
+
createdAt: now(),
|
|
152
|
+
updatedAt: now(),
|
|
153
|
+
commits: [],
|
|
154
|
+
changedFiles: [],
|
|
155
|
+
protectedPathViolations: [],
|
|
156
|
+
}
|
|
157
|
+
const idempotencyKey = `${run.runId}:${replayKey}`
|
|
158
|
+
const snapshot = createEvaluationSnapshotBundle(config.repoRoot, run.baseCommit, config)
|
|
159
|
+
const artifact = await queue.putArtifact(snapshot.content)
|
|
160
|
+
const payload = fleetPayload(config, replayCandidate, snapshot)
|
|
161
|
+
payload.snapshotSha256 = artifact.sha256
|
|
162
|
+
payload.task = `Replay registered curriculum fixture ${task.fixtureId}`
|
|
163
|
+
payload.regressionCommand = "true"
|
|
164
|
+
payload.evalCommands = [task.groundTruthCommand]
|
|
165
|
+
const expected = {
|
|
166
|
+
idempotencyKey,
|
|
167
|
+
runId: run.runId,
|
|
168
|
+
candidateId: replayCandidate.id,
|
|
169
|
+
jobKind: "evaluation" as const,
|
|
170
|
+
resourceClass: "CPU" as const,
|
|
171
|
+
concurrencyKey: `curriculum:${task.fixtureId}`,
|
|
172
|
+
artifactSha256: artifact.sha256,
|
|
173
|
+
payload,
|
|
174
|
+
}
|
|
175
|
+
let job = await queue.getJobByIdempotencyKey(idempotencyKey)
|
|
176
|
+
if (!job) {
|
|
177
|
+
job = await queue.enqueue({
|
|
178
|
+
idempotencyKey,
|
|
179
|
+
runId: run.runId,
|
|
180
|
+
candidateId: replayCandidate.id,
|
|
181
|
+
jobKind: "evaluation",
|
|
182
|
+
resourceClass: "CPU",
|
|
183
|
+
concurrencyKey: expected.concurrencyKey,
|
|
184
|
+
artifactSha256: artifact.sha256,
|
|
185
|
+
payload,
|
|
186
|
+
})
|
|
187
|
+
}
|
|
188
|
+
if (!fleetJobMatches(job, expected)) return { ok: false, exitCode: null, failure: "idempotent curriculum replay key refers to a stale or mismatched fleet job payload" }
|
|
189
|
+
const completed = await waitForFleetJob(queue, job.jobId, requiredFleetJobLeaseMs("evaluation", config.commandTimeoutMs, 1) + 60_000)
|
|
190
|
+
if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") return { ok: false, exitCode: null, failure: completed.error ?? `OpenShell curriculum replay ended ${completed.status}` }
|
|
191
|
+
try {
|
|
192
|
+
const evaluation = fleetEvaluationResult(completed.result)
|
|
193
|
+
const trial = evaluation.visible[0]
|
|
194
|
+
if (!trial) return { ok: false, exitCode: null, failure: "OpenShell curriculum replay returned no fixture trial" }
|
|
195
|
+
return {
|
|
196
|
+
ok: trial.ok,
|
|
197
|
+
exitCode: trial.exitCode,
|
|
198
|
+
outputSha256: trial.outputArtifactSha256,
|
|
199
|
+
...(!trial.ok ? { failure: `fixture command exited ${trial.exitCode ?? "unknown"}` } : {}),
|
|
200
|
+
}
|
|
201
|
+
} catch (error) {
|
|
202
|
+
return { ok: false, exitCode: null, failure: error instanceof Error ? error.message : String(error) }
|
|
203
|
+
}
|
|
204
|
+
}, now())
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
async function runAdversarialStage(
|
|
208
|
+
candidate: CandidateRecord,
|
|
209
|
+
evaluation: { regression: TrialResult; visible: TrialResult[] },
|
|
210
|
+
config: RsiConfig,
|
|
211
|
+
run: RsiRunRecord,
|
|
212
|
+
queue: PostgresRsiJobQueue,
|
|
213
|
+
hooks: RsiHooks,
|
|
214
|
+
now: () => string,
|
|
215
|
+
): Promise<AdversarialReviewRecord> {
|
|
216
|
+
const role = config.roles?.adversary
|
|
217
|
+
if (!role) throw new Error("adversary role is not configured")
|
|
218
|
+
const base: AdversarialReviewRecord = {
|
|
219
|
+
candidateId: candidate.id, role: "adversary", provider: role.provider, model: role.model,
|
|
220
|
+
promptVersion: ADVERSARIAL_PROMPT_VERSION, promptSha256: "", promptArtifactSha256: "",
|
|
221
|
+
createdAt: now(), status: "error", findings: [], testResults: [], penaltyPoints: 0,
|
|
222
|
+
}
|
|
223
|
+
let internalJob: ExperimentJob | undefined
|
|
224
|
+
try {
|
|
225
|
+
const workerRole = config.roles?.worker
|
|
226
|
+
const workerModel = workerRole?.model ?? candidate.model
|
|
227
|
+
if (!role.model?.trim() || role.model === workerModel) throw new Error("adversary role must use a different model from the configured worker role")
|
|
228
|
+
const patch = execFileSync("git", ["diff", "--binary", "--no-ext-diff", "--no-renames", candidate.baseCommit + "...HEAD"], { cwd: candidate.worktree, encoding: "buffer", maxBuffer: 4 * 1024 * 1024 }).toString("utf8")
|
|
229
|
+
if (!patch || Buffer.byteLength(patch) > 512 * 1024) throw new Error("candidate diff is empty or exceeds the 512 KiB reviewer input limit")
|
|
230
|
+
const prompt = buildAdversarialPrompt(candidate, patch, evaluation)
|
|
231
|
+
const promptBytes = Buffer.from(prompt)
|
|
232
|
+
if (promptBytes.byteLength > 600 * 1024) throw new Error("adversarial prompt exceeds the 600 KiB limit")
|
|
233
|
+
const promptArtifact = await queue.putArtifact(promptBytes)
|
|
234
|
+
base.promptSha256 = sha256(promptBytes)
|
|
235
|
+
base.promptArtifactSha256 = promptArtifact.sha256
|
|
236
|
+
const raw = await (hooks.runAdversary ? hooks.runAdversary({ role, prompt }) : requestAdversarialReview(role, prompt))
|
|
237
|
+
const resultBytes = Buffer.from(raw)
|
|
238
|
+
const resultArtifact = await queue.putArtifact(resultBytes)
|
|
239
|
+
base.resultSha256 = sha256(resultBytes)
|
|
240
|
+
base.resultArtifactSha256 = resultArtifact.sha256
|
|
241
|
+
const parsed = parseAdversarialReview(raw, candidate.changedFiles)
|
|
242
|
+
base.summary = parsed.summary
|
|
243
|
+
base.findings = parsed.findings
|
|
244
|
+
base.penaltyPoints = Math.min(10, parsed.findings.reduce((points, finding) => points + (finding.severity === "major" ? 5 : 1), 0))
|
|
245
|
+
const tests = Buffer.from(JSON.stringify({ schemaVersion: 1, tests: parsed.tests }))
|
|
246
|
+
const testsArtifact = await queue.putArtifact(tests)
|
|
247
|
+
base.testArtifactSha256 = testsArtifact.sha256
|
|
248
|
+
const candidateCommit = execFileSync("git", ["rev-parse", "HEAD"], { cwd: candidate.worktree, encoding: "utf8" }).trim()
|
|
249
|
+
const snapshot = createAdversarialEvaluationSnapshotBundle(candidate.worktree, candidateCommit, config, tests)
|
|
250
|
+
const snapshotArtifact = await queue.putArtifact(snapshot.content)
|
|
251
|
+
base.snapshotArtifactSha256 = snapshotArtifact.sha256
|
|
252
|
+
internalJob = newExperimentJob("adversarial", { class: "CPU", units: 1, concurrencyKey: "adversarial:" + candidate.id }, now(), candidate.id)
|
|
253
|
+
run.jobs ??= []
|
|
254
|
+
run.jobs.push(internalJob)
|
|
255
|
+
const candidateId = candidate.id + "-adversarial"
|
|
256
|
+
const payload = fleetPayload(config, { ...candidate, id: candidateId, baseCommit: snapshot.snapshotCommit }, snapshot)
|
|
257
|
+
payload.snapshotSha256 = snapshotArtifact.sha256
|
|
258
|
+
payload.task = "Run supervisor-generated adversarial checks for " + candidate.id
|
|
259
|
+
payload.regressionCommand = "true"
|
|
260
|
+
payload.evalCommands = [ADVERSARIAL_COMMAND]
|
|
261
|
+
const idempotencyKey = run.runId + ":" + candidate.id + ":adversarial:" + testsArtifact.sha256
|
|
262
|
+
const expected = {
|
|
263
|
+
idempotencyKey, runId: run.runId, candidateId, jobKind: "evaluation" as const,
|
|
264
|
+
resourceClass: "CPU" as const, concurrencyKey: "adversarial:" + candidate.id,
|
|
265
|
+
artifactSha256: snapshotArtifact.sha256, payload,
|
|
266
|
+
}
|
|
267
|
+
let job = await queue.getJobByIdempotencyKey(idempotencyKey)
|
|
268
|
+
if (!job) job = await queue.enqueue({ ...expected, payload })
|
|
269
|
+
if (!fleetJobMatches(job, expected)) throw new Error("existing adversarial evaluation idempotency key has mismatched test artifact or execution payload")
|
|
270
|
+
const completed = await waitForFleetJob(queue, job.jobId, requiredFleetJobLeaseMs("evaluation", config.commandTimeoutMs, 1) + 60_000)
|
|
271
|
+
if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") throw new Error(completed.error ?? "OpenShell adversarial tests ended " + completed.status)
|
|
272
|
+
const result = fleetEvaluationResult(completed.result)
|
|
273
|
+
const trial = result.visible[0]
|
|
274
|
+
if (!trial) throw new Error("OpenShell adversarial job returned no test result")
|
|
275
|
+
const passedIds = new Set(trial.stdout.split("\n").filter((line) => line.startsWith("PASS ")).map((line) => line.slice(5).trim()))
|
|
276
|
+
base.testResults = parsed.tests.map((test) => ({ id: test.id, passed: trial.ok && passedIds.has(test.id), outputArtifactSha256: trial.outputArtifactSha256, ...(!trial.ok || !passedIds.has(test.id) ? { failure: trial.stdout.slice(0, 1000) || "OpenShell adversarial test failed with exit " + (trial.exitCode ?? "unknown") } : {}) }))
|
|
277
|
+
base.jobId = job.jobId
|
|
278
|
+
const allTestsPassed = trial.ok && base.testResults.length === parsed.tests.length && base.testResults.every((entry) => entry.passed)
|
|
279
|
+
base.status = allTestsPassed ? "passed" : "failed"
|
|
280
|
+
internalJob.status = allTestsPassed ? "completed" : "failed"
|
|
281
|
+
internalJob.updatedAt = now()
|
|
282
|
+
if (!allTestsPassed) internalJob.error = "adversarial test failure (" + (trial.exitCode ?? "unknown") + ")"
|
|
283
|
+
return base
|
|
284
|
+
} catch (error) {
|
|
285
|
+
const message = error instanceof Error ? error.message : String(error)
|
|
286
|
+
base.error = message.slice(0, 2000)
|
|
287
|
+
base.status = "error"
|
|
288
|
+
if (internalJob) {
|
|
289
|
+
internalJob.status = "failed"
|
|
290
|
+
internalJob.updatedAt = now()
|
|
291
|
+
internalJob.error = base.error
|
|
292
|
+
}
|
|
293
|
+
try {
|
|
294
|
+
const failureArtifact = await queue.putArtifact(Buffer.from(base.error))
|
|
295
|
+
base.errorSha256 = sha256(base.error)
|
|
296
|
+
base.errorArtifactSha256 = failureArtifact.sha256
|
|
297
|
+
} catch { /* preserve the primary failure if artifact storage itself is unavailable */ }
|
|
298
|
+
return base
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
async function evaluateBaselineLocal(config: RsiConfig, baseCommit: string, runner: RsiHooks["runCommand"]): Promise<NonNullable<RsiRunRecord["baselineEvaluation"]>> {
|
|
303
|
+
const name = `rsi-baseline-${randomUUID().slice(0, 12)}`
|
|
304
|
+
const baselineRoot = path.join(config.repoRoot, ".worktrees", name)
|
|
305
|
+
execFileSync("git", ["worktree", "add", "--detach", baselineRoot, baseCommit], { cwd: config.repoRoot, stdio: "pipe" })
|
|
306
|
+
try {
|
|
307
|
+
return await evaluateCandidate(config, baselineRoot, baseCommit, runner)
|
|
308
|
+
} finally {
|
|
309
|
+
execFileSync("git", ["worktree", "remove", "--force", baselineRoot], { cwd: config.repoRoot, stdio: "pipe" })
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
async function evaluateBaselineFleet(
|
|
314
|
+
config: RsiConfig,
|
|
315
|
+
baseCommit: string,
|
|
316
|
+
queue: PostgresRsiJobQueue,
|
|
317
|
+
runId: string,
|
|
318
|
+
): Promise<NonNullable<RsiRunRecord["baselineEvaluation"]>> {
|
|
319
|
+
const idempotencyKey = `${runId}:baseline:evaluation`
|
|
320
|
+
const snapshot = createEvaluationSnapshotBundle(config.repoRoot, baseCommit, config)
|
|
321
|
+
const artifact = await queue.putArtifact(snapshot.content)
|
|
322
|
+
const baseline: CandidateRecord = {
|
|
323
|
+
id: "baseline",
|
|
324
|
+
generation: 0,
|
|
325
|
+
parent: "baseline",
|
|
326
|
+
branch: "baseline",
|
|
327
|
+
worktree: config.repoRoot,
|
|
328
|
+
baseCommit,
|
|
329
|
+
status: "evaluating",
|
|
330
|
+
model: config.model,
|
|
331
|
+
mutation: "Baseline visible evaluation",
|
|
332
|
+
createdAt: new Date().toISOString(),
|
|
333
|
+
updatedAt: new Date().toISOString(),
|
|
334
|
+
commits: [],
|
|
335
|
+
changedFiles: [],
|
|
336
|
+
protectedPathViolations: [],
|
|
337
|
+
}
|
|
338
|
+
const payload = fleetPayload(config, baseline, snapshot)
|
|
339
|
+
payload.snapshotSha256 = artifact.sha256
|
|
340
|
+
const expected: FleetJobIdentity = {
|
|
341
|
+
idempotencyKey, runId, candidateId: "baseline", jobKind: "evaluation",
|
|
342
|
+
resourceClass: "CPU", concurrencyKey: "eval:baseline", artifactSha256: artifact.sha256, payload,
|
|
343
|
+
}
|
|
344
|
+
const existing = await queue.getJobByIdempotencyKey(idempotencyKey)
|
|
345
|
+
const job = existing ?? await queue.enqueue(expected)
|
|
346
|
+
if (!fleetJobMatches(job, expected)) throw new Error("existing baseline evaluation idempotency key has mismatched execution identity or payload")
|
|
347
|
+
|
|
348
|
+
const completed = await waitForFleetJob(queue, job.jobId, requiredFleetJobLeaseMs("evaluation", config.commandTimeoutMs, config.evalCommands.length) + 60_000)
|
|
349
|
+
if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") throw new Error(completed.error ?? `fleet baseline evaluation ended ${completed.status}`)
|
|
350
|
+
const evaluation = fleetEvaluationResult(completed.result)
|
|
351
|
+
const trials = [evaluation.regression, ...evaluation.visible]
|
|
352
|
+
let resultIndex = 0
|
|
353
|
+
const baselineRoot = path.join(config.repoRoot, ".worktrees", `rsi-baseline-${randomUUID().slice(0, 12)}`)
|
|
354
|
+
execFileSync("git", ["worktree", "add", "--detach", baselineRoot, baseCommit], { cwd: config.repoRoot, stdio: "pipe" })
|
|
355
|
+
try {
|
|
356
|
+
return await evaluateCandidate(config, baselineRoot, baseCommit, async () => trials[resultIndex++] as TrialResult)
|
|
357
|
+
} finally {
|
|
358
|
+
execFileSync("git", ["worktree", "remove", "--force", baselineRoot], { cwd: config.repoRoot, stdio: "pipe" })
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
|
|
67
362
|
function recoverInterruptedCandidates(run: RsiRunRecord, now: string): void {
|
|
68
363
|
for (const candidate of run.candidates) {
|
|
69
364
|
if (candidate.status === "mutating" || candidate.status === "evaluating") {
|
|
@@ -74,6 +369,24 @@ function recoverInterruptedCandidates(run: RsiRunRecord, now: string): void {
|
|
|
74
369
|
}
|
|
75
370
|
}
|
|
76
371
|
|
|
372
|
+
async function ensureCandidateWorktree(config: RsiConfig, candidate: CandidateRecord): Promise<{ created: boolean; commits: number }> {
|
|
373
|
+
let created = false
|
|
374
|
+
try {
|
|
375
|
+
await fs.access(candidate.worktree)
|
|
376
|
+
} catch (error) {
|
|
377
|
+
if ((error as NodeJS.ErrnoException).code !== "ENOENT") throw error
|
|
378
|
+
await createCandidateWorktree(config, candidate)
|
|
379
|
+
created = true
|
|
380
|
+
}
|
|
381
|
+
const top = execFileSync("git", ["rev-parse", "--show-toplevel"], { cwd: candidate.worktree, encoding: "utf8" }).trim()
|
|
382
|
+
if (path.resolve(top) !== path.resolve(candidate.worktree)) throw new Error("candidate worktree path resolved outside its recorded location")
|
|
383
|
+
const commits = Number(execFileSync("git", ["rev-list", "--count", `${candidate.baseCommit}..HEAD`], { cwd: candidate.worktree, encoding: "utf8" }).trim())
|
|
384
|
+
if (!Number.isSafeInteger(commits) || commits < 0 || commits > 1) throw new Error("candidate worktree has an unexpected commit count while resuming")
|
|
385
|
+
const status = execFileSync("git", ["status", "--porcelain=v1", "-z", "--untracked-files=all"], { cwd: candidate.worktree, encoding: "buffer" })
|
|
386
|
+
if (status.byteLength > 0) throw new Error("candidate worktree has uncommitted state; refusing unsafe fleet resume")
|
|
387
|
+
return { created, commits }
|
|
388
|
+
}
|
|
389
|
+
|
|
77
390
|
async function captureCandidate(
|
|
78
391
|
candidate: CandidateRecord,
|
|
79
392
|
config: RsiConfig,
|
|
@@ -93,7 +406,14 @@ async function captureCandidate(
|
|
|
93
406
|
}
|
|
94
407
|
}
|
|
95
408
|
|
|
96
|
-
export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<RsiRunRecord> {
|
|
409
|
+
export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}, queueOverride?: PostgresRsiJobQueue): Promise<RsiRunRecord> {
|
|
410
|
+
if (config.computePolicy === "adaptive-independent") {
|
|
411
|
+
const trajectoryLimit = config.maxTrajectories ?? config.population + 1
|
|
412
|
+
const iterationLimit = config.maxTotalIterations ?? trajectoryLimit * config.maxIterations
|
|
413
|
+
if (config.population < 2 || config.generations !== 1 || trajectoryLimit < config.population || trajectoryLimit > config.population + 1 || iterationLimit < config.population * config.maxIterations || iterationLimit > trajectoryLimit * config.maxIterations) {
|
|
414
|
+
throw new Error("invalid adaptive-independent budgets: require population >= 2, exactly one follow-up generation, and trajectory/iteration limits for only the initial cohort plus at most one candidate")
|
|
415
|
+
}
|
|
416
|
+
}
|
|
97
417
|
const now = hooks.now ?? (() => new Date().toISOString())
|
|
98
418
|
const archive = await readArchive(config.archiveDir)
|
|
99
419
|
const roles = resolveRoles(config.roles, process.env, config.model)
|
|
@@ -102,7 +422,8 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
|
|
|
102
422
|
if (config.resumeRunId && !resumed) throw new Error(`no active RSI run found for --resume ${config.resumeRunId}`)
|
|
103
423
|
const baseCommit = resumed?.baseCommit ?? (await resolveBaseCommit(config))
|
|
104
424
|
const run = resumed ?? initialRun({ ...config, model: workerModel, roles }, baseCommit, now())
|
|
105
|
-
|
|
425
|
+
const stageCount = config.computePolicy === "adaptive-independent" ? 2 : config.generations
|
|
426
|
+
run.generations = stageCount
|
|
106
427
|
run.model = workerModel
|
|
107
428
|
run.parentSelectionPolicy = config.parentSelectionPolicy ?? run.parentSelectionPolicy ?? "champion-specialist-novelty"
|
|
108
429
|
run.modelCandidates ??= [baseModelCandidate(config.model, now(), config.modelCandidateId ?? "model-base")]
|
|
@@ -113,18 +434,55 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
|
|
|
113
434
|
run.jobs ??= []
|
|
114
435
|
run.trajectoryRefs ??= []
|
|
115
436
|
run.curriculumTasks ??= []
|
|
437
|
+
run.adversarialReviews ??= []
|
|
116
438
|
run.parentChoices ??= {}
|
|
117
|
-
if (resumed) recoverInterruptedCandidates(run, now())
|
|
118
439
|
const runConfig: RsiConfig = { ...config, model: workerModel, roles, seed: `${config.seed}-${run.runId.slice(-8)}` }
|
|
119
440
|
const trajectories: TrajectoryRecord[] = []
|
|
120
441
|
|
|
121
|
-
if (!config.dryRun &&
|
|
122
|
-
|
|
442
|
+
if (!config.dryRun && config.hiddenEvalCommands.length > 0) {
|
|
443
|
+
throw new Error("RSI hidden evaluations are disabled until their evaluator assets can remain unavailable to candidate processes")
|
|
444
|
+
}
|
|
445
|
+
const fleetQueue = !config.dryRun && (queueOverride !== undefined || !hooks.runMutation)
|
|
446
|
+
? queueOverride ?? PostgresRsiJobQueue.fromEnvironment(fleetQueuePolicy(process.env.HEADLESSCODE_RSI_QUEUE_NAME?.trim() || "rsi-default", config.maxConcurrent, process.env), process.env)
|
|
447
|
+
: undefined
|
|
448
|
+
if (!config.dryRun && config.roles?.adversary && !fleetQueue) throw new Error("RSI adversarial break tests require the OpenShell fleet evaluation path")
|
|
449
|
+
const commandRunner = hooks.runCommand ?? (fleetQueue ? undefined : createOpenShellCommandRunner(runConfig))
|
|
450
|
+
try {
|
|
451
|
+
if (fleetQueue) {
|
|
452
|
+
if (fleetQueue.artifactBackend !== "s3") throw new Error("RSI production runs require a shared S3-compatible artifact store")
|
|
453
|
+
await fleetQueue.migrate()
|
|
454
|
+
if (await fleetQueue.activeWorkerCount() < 1) throw new Error("no live OpenShell RSI worker is registered for the configured queue")
|
|
455
|
+
}
|
|
456
|
+
if (resumed && !fleetQueue) recoverInterruptedCandidates(run, now())
|
|
457
|
+
if (!config.dryRun && !run.baselineEvaluation) {
|
|
458
|
+
run.baselineEvaluation = fleetQueue
|
|
459
|
+
? await evaluateBaselineFleet(runConfig, baseCommit, fleetQueue, run.runId)
|
|
460
|
+
: await evaluateBaselineLocal(runConfig, baseCommit, commandRunner)
|
|
461
|
+
run.baselineFitness = computeFitness(run.baselineEvaluation)
|
|
462
|
+
run.baseline = run.baselineEvaluation.regression
|
|
123
463
|
log(hooks, `[rsi] baseline: ${run.baseline.ok ? "pass" : "fail"}`)
|
|
124
464
|
await checkpointRun(config.archiveDir, run)
|
|
125
465
|
}
|
|
126
466
|
|
|
127
|
-
for (let generation = 0; generation <
|
|
467
|
+
for (let generation = 0; generation < stageCount; generation++) {
|
|
468
|
+
if (config.computePolicy === "adaptive-independent" && generation > 0) {
|
|
469
|
+
let previousDecision = run.adaptiveSearch?.find((entry) => entry.generation === generation - 1)
|
|
470
|
+
if (!previousDecision) {
|
|
471
|
+
const priorCandidates = run.candidates.filter((candidate) => candidate.generation === generation - 1)
|
|
472
|
+
previousDecision = decideAdaptiveContinuation({
|
|
473
|
+
generation: generation - 1,
|
|
474
|
+
generationCandidates: priorCandidates,
|
|
475
|
+
allCandidates: run.candidates,
|
|
476
|
+
config,
|
|
477
|
+
elapsedMs: Math.max(0, Date.now() - Date.parse(run.startedAt)),
|
|
478
|
+
decidedAt: now(),
|
|
479
|
+
})
|
|
480
|
+
run.adaptiveSearch ??= []
|
|
481
|
+
run.adaptiveSearch.push(previousDecision)
|
|
482
|
+
await checkpointRun(config.archiveDir, run)
|
|
483
|
+
}
|
|
484
|
+
if (previousDecision.decision !== "continue") break
|
|
485
|
+
}
|
|
128
486
|
let candidates = run.candidates.filter((candidate) => candidate.generation === generation)
|
|
129
487
|
if (candidates.length === 0) {
|
|
130
488
|
const merged = combinedArchive(archive, run)
|
|
@@ -134,7 +492,8 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
|
|
|
134
492
|
const fallbackBase = run.selected
|
|
135
493
|
? run.candidates.find((candidate) => candidate.id === run.selected)?.commits[0] ?? run.baseCommit
|
|
136
494
|
: run.baseCommit
|
|
137
|
-
|
|
495
|
+
const candidateCount = config.computePolicy === "adaptive-independent" && generation > 0 ? 1 : config.population
|
|
496
|
+
candidates = Array.from({ length: candidateCount }, (_, index) => {
|
|
138
497
|
const parentChoice = choices[index]
|
|
139
498
|
const candidate = newCandidate(runConfig, generation, index, parentChoice?.baseCommit ?? fallbackBase, now(), parentChoice)
|
|
140
499
|
if (parentChoice) run.parentChoices![candidate.id] = parentChoice.reason
|
|
@@ -148,6 +507,67 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
|
|
|
148
507
|
}
|
|
149
508
|
if (config.dryRun) continue
|
|
150
509
|
|
|
510
|
+
const mutationJobs = new Map<string, FleetJob>()
|
|
511
|
+
const evaluationJobs = new Map<string, FleetJob>()
|
|
512
|
+
const managedWorktrees = new Set<string>()
|
|
513
|
+
if (fleetQueue) {
|
|
514
|
+
try {
|
|
515
|
+
for (const candidate of candidates) {
|
|
516
|
+
if (isTerminal(candidate)) continue
|
|
517
|
+
const worktree = await ensureCandidateWorktree(runConfig, candidate)
|
|
518
|
+
managedWorktrees.add(candidate.id)
|
|
519
|
+
const key = `${run.runId}:${candidate.id}:mutation`
|
|
520
|
+
const snapshot = createMutationSnapshotBundle(candidate.worktree, candidate.baseCommit, runConfig)
|
|
521
|
+
const artifact = await fleetQueue.putArtifact(snapshot.content)
|
|
522
|
+
const payload = fleetPayload(runConfig, candidate, snapshot)
|
|
523
|
+
payload.snapshotSha256 = artifact.sha256
|
|
524
|
+
const expected: FleetJobIdentity = {
|
|
525
|
+
idempotencyKey: key, runId: run.runId, candidateId: candidate.id,
|
|
526
|
+
jobKind: "mutation", resourceClass: "LOCAL_GPU", concurrencyKey: `ollama:${candidate.model}`,
|
|
527
|
+
artifactSha256: artifact.sha256, payload,
|
|
528
|
+
}
|
|
529
|
+
const existing = await fleetQueue.getJobByIdempotencyKey(key)
|
|
530
|
+
if (existing && !fleetJobMatches(existing, expected)) throw new Error("existing idempotent mutation job has mismatched execution identity, snapshot, or task payload")
|
|
531
|
+
if (worktree.commits > 0) {
|
|
532
|
+
const completed = existing
|
|
533
|
+
if (completed?.candidateId !== candidate.id || completed.jobKind !== "mutation" || completed.status !== "completed" || !completed.result || typeof completed.result !== "object") {
|
|
534
|
+
throw new Error("candidate worktree contains an applied mutation commit without its completed idempotent fleet job")
|
|
535
|
+
}
|
|
536
|
+
const result = completed.result as { patchSha256?: string; patchBytes?: number }
|
|
537
|
+
if (!result.patchSha256 || !Number.isSafeInteger(result.patchBytes) || Number(result.patchBytes) < 1) throw new Error("completed fleet mutation has no valid patch artifact reference for resume")
|
|
538
|
+
const patch = await fleetQueue.getArtifact(result.patchSha256)
|
|
539
|
+
const appliedPatch = execFileSync("git", ["diff", "--binary", "--no-ext-diff", "--no-renames", `${candidate.baseCommit}...HEAD`], { cwd: candidate.worktree, encoding: "buffer" })
|
|
540
|
+
if (patch.byteLength !== result.patchBytes || appliedPatch.byteLength !== result.patchBytes || createHash("sha256").update(appliedPatch).digest("hex") !== result.patchSha256) {
|
|
541
|
+
throw new Error("candidate worktree commit does not match the completed fleet mutation patch")
|
|
542
|
+
}
|
|
543
|
+
let recorded = run.jobs.find((entry) => entry.candidateId === candidate.id && entry.kind === "mutation")
|
|
544
|
+
if (!recorded) {
|
|
545
|
+
recorded = newExperimentJob("mutation", { class: "LOCAL_GPU", units: 1, concurrencyKey: "ollama" }, now(), candidate.id)
|
|
546
|
+
run.jobs.push(recorded)
|
|
547
|
+
}
|
|
548
|
+
if (recorded.status !== "completed") replaceJob(run, transitionJob(recorded, "completed", now()))
|
|
549
|
+
candidate.status = "evaluating"
|
|
550
|
+
candidate.changedFiles = await changedFiles(candidate.worktree, candidate.baseCommit)
|
|
551
|
+
candidate.commits = await candidateCommits(candidate.worktree, candidate.baseCommit)
|
|
552
|
+
candidate.protectedPathViolations = protectedPathViolations(candidate.changedFiles, config.protectedPaths)
|
|
553
|
+
candidate.updatedAt = now()
|
|
554
|
+
await checkpointRun(config.archiveDir, run)
|
|
555
|
+
continue
|
|
556
|
+
}
|
|
557
|
+
const job = existing ?? await fleetQueue.enqueue(expected)
|
|
558
|
+
if (!fleetJobMatches(job, expected)) throw new Error("enqueued mutation job does not match its expected execution identity or payload")
|
|
559
|
+
mutationJobs.set(candidate.id, job)
|
|
560
|
+
}
|
|
561
|
+
} catch (error) {
|
|
562
|
+
for (const candidate of candidates.filter((entry) => managedWorktrees.has(entry.id))) {
|
|
563
|
+
await removeCandidateWorktree(runConfig, candidate).catch(() => undefined)
|
|
564
|
+
}
|
|
565
|
+
throw error
|
|
566
|
+
}
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
// Resolve every mutation first. Evaluation submission is a separate
|
|
570
|
+
// generation-wide phase so CPU workers can claim the full ready set.
|
|
151
571
|
for (const candidate of candidates) {
|
|
152
572
|
if (isTerminal(candidate)) continue
|
|
153
573
|
let job = run.jobs.find((entry) => entry.candidateId === candidate.id && entry.kind === "mutation")
|
|
@@ -155,37 +575,115 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
|
|
|
155
575
|
job = newExperimentJob("mutation", { class: "LOCAL_GPU", units: 1, concurrencyKey: "ollama" }, now(), candidate.id)
|
|
156
576
|
run.jobs.push(job)
|
|
157
577
|
}
|
|
158
|
-
|
|
578
|
+
try {
|
|
579
|
+
const worktree = fleetQueue ? await ensureCandidateWorktree(runConfig, candidate) : { created: false, commits: 0 }
|
|
580
|
+
managedWorktrees.add(candidate.id)
|
|
581
|
+
if (worktree.commits === 0) {
|
|
582
|
+
candidate.status = "mutating"
|
|
583
|
+
candidate.updatedAt = now()
|
|
584
|
+
if (job.status !== "running") job = transitionJob(job, "running", now())
|
|
585
|
+
replaceJob(run, job)
|
|
586
|
+
await checkpointRun(config.archiveDir, run)
|
|
587
|
+
if (fleetQueue) {
|
|
588
|
+
const submitted = mutationJobs.get(candidate.id)
|
|
589
|
+
if (!submitted) throw new Error("candidate mutation was not admitted to the RSI fleet queue")
|
|
590
|
+
const completed = await waitForFleetJob(fleetQueue, submitted.jobId, config.commandTimeoutMs * 4 + 180_000)
|
|
591
|
+
if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") throw new Error(completed.error ?? `fleet mutation ended ${completed.status}`)
|
|
592
|
+
const result = completed.result as { patchSha256?: string; patchBytes?: number }
|
|
593
|
+
if (!result.patchSha256 || !Number.isSafeInteger(result.patchBytes) || Number(result.patchBytes) < 1) throw new Error("fleet mutation result is missing its patch artifact reference")
|
|
594
|
+
const patch = await fleetQueue.getArtifact(result.patchSha256)
|
|
595
|
+
if (patch.byteLength !== result.patchBytes) throw new Error("fleet mutation patch length did not match result metadata")
|
|
596
|
+
applyFleetMutationPatch(patch, candidate, runConfig)
|
|
597
|
+
} else {
|
|
598
|
+
await createCandidateWorktree(runConfig, candidate)
|
|
599
|
+
managedWorktrees.add(candidate.id)
|
|
600
|
+
const mutation = await (hooks.runMutation ?? runMutation)(candidate, runConfig)
|
|
601
|
+
if (!mutation.ok) throw new Error(mutation.error ?? "RSI mutation failed")
|
|
602
|
+
}
|
|
603
|
+
job = transitionJob(job, "completed", now())
|
|
604
|
+
replaceJob(run, job)
|
|
605
|
+
}
|
|
606
|
+
candidate.status = "evaluating"
|
|
607
|
+
candidate.changedFiles = await changedFiles(candidate.worktree, candidate.baseCommit)
|
|
608
|
+
candidate.commits = await candidateCommits(candidate.worktree, candidate.baseCommit)
|
|
609
|
+
candidate.protectedPathViolations = protectedPathViolations(candidate.changedFiles, config.protectedPaths)
|
|
610
|
+
candidate.updatedAt = now()
|
|
611
|
+
await checkpointRun(config.archiveDir, run)
|
|
612
|
+
} catch (error) {
|
|
613
|
+
candidate.status = "failed"
|
|
614
|
+
candidate.failure = error instanceof Error ? error.message : String(error)
|
|
159
615
|
candidate.updatedAt = now()
|
|
160
|
-
job = transitionJob(job, "
|
|
616
|
+
job = transitionJob(job, "failed", now(), candidate.failure)
|
|
161
617
|
replaceJob(run, job)
|
|
618
|
+
await captureCandidate(candidate, runConfig, run, trajectories, now(), hooks)
|
|
619
|
+
log(hooks, `[rsi] ${candidate.id}: mutation failed: ${candidate.failure}`)
|
|
620
|
+
if (!config.keepWorktrees && managedWorktrees.has(candidate.id)) await removeCandidateWorktree(runConfig, candidate).catch(() => undefined)
|
|
162
621
|
await checkpointRun(config.archiveDir, run)
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
622
|
+
}
|
|
623
|
+
await checkpointRun(config.archiveDir, run)
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
if (fleetQueue) {
|
|
627
|
+
for (const candidate of candidates.filter((entry) => entry.status === "evaluating")) {
|
|
628
|
+
try {
|
|
629
|
+
const key = `${run.runId}:${candidate.id}:evaluation`
|
|
630
|
+
const head = execFileSync("git", ["rev-parse", "HEAD"], { cwd: candidate.worktree, encoding: "utf8" }).trim()
|
|
631
|
+
const snapshot = createEvaluationSnapshotBundle(candidate.worktree, head, runConfig)
|
|
632
|
+
const artifact = await fleetQueue.putArtifact(snapshot.content)
|
|
633
|
+
const payload = fleetPayload(runConfig, candidate, snapshot)
|
|
634
|
+
payload.snapshotSha256 = artifact.sha256
|
|
635
|
+
const expected: FleetJobIdentity = {
|
|
636
|
+
idempotencyKey: key, runId: run.runId, candidateId: candidate.id,
|
|
637
|
+
jobKind: "evaluation", resourceClass: "CPU", concurrencyKey: `eval:${candidate.id}`,
|
|
638
|
+
artifactSha256: artifact.sha256, payload,
|
|
639
|
+
}
|
|
640
|
+
const existing = await fleetQueue.getJobByIdempotencyKey(key)
|
|
641
|
+
if (existing && !fleetJobMatches(existing, expected)) throw new Error("existing idempotent evaluation job has mismatched execution identity, snapshot, or evaluation configuration")
|
|
642
|
+
const job = existing ?? await fleetQueue.enqueue(expected)
|
|
643
|
+
if (!fleetJobMatches(job, expected)) throw new Error("enqueued evaluation job does not match its expected execution identity or payload")
|
|
644
|
+
evaluationJobs.set(candidate.id, job)
|
|
645
|
+
} catch (error) {
|
|
169
646
|
candidate.status = "failed"
|
|
170
|
-
candidate.failure =
|
|
171
|
-
candidate.result = mutation.result
|
|
647
|
+
candidate.failure = error instanceof Error ? error.message : String(error)
|
|
172
648
|
candidate.updatedAt = now()
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
await
|
|
176
|
-
|
|
177
|
-
|
|
649
|
+
log(hooks, `[rsi] ${candidate.id}: evaluation admission failed: ${candidate.failure}`)
|
|
650
|
+
if (!config.keepWorktrees && managedWorktrees.has(candidate.id)) await removeCandidateWorktree(runConfig, candidate).catch(() => undefined)
|
|
651
|
+
await checkpointRun(config.archiveDir, run)
|
|
652
|
+
}
|
|
653
|
+
}
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
for (const candidate of candidates.filter((entry) => entry.status === "evaluating")) {
|
|
657
|
+
let job = run.jobs.find((entry) => entry.candidateId === candidate.id && entry.kind === "mutation")!
|
|
658
|
+
try {
|
|
659
|
+
let evaluation
|
|
660
|
+
if (fleetQueue) {
|
|
661
|
+
const submitted = evaluationJobs.get(candidate.id) ?? await fleetQueue.getJobByIdempotencyKey(`${run.runId}:${candidate.id}:evaluation`)
|
|
662
|
+
if (!submitted) throw new Error("candidate evaluation was not admitted to the RSI fleet queue")
|
|
663
|
+
const completed = await waitForFleetJob(fleetQueue, submitted.jobId, requiredFleetJobLeaseMs("evaluation", config.commandTimeoutMs, config.evalCommands.length) + 60_000)
|
|
664
|
+
if (completed.status !== "completed" || !completed.result || typeof completed.result !== "object") throw new Error(completed.error ?? `fleet evaluation ended ${completed.status}`)
|
|
665
|
+
const visible = fleetEvaluationResult(completed.result)
|
|
666
|
+
const trials = [visible.regression, ...visible.visible]
|
|
667
|
+
let resultIndex = 0
|
|
668
|
+
evaluation = await evaluateCandidate(runConfig, candidate.worktree, candidate.baseCommit, async () => trials[resultIndex++] as Awaited<ReturnType<NonNullable<typeof hooks.runCommand>>>)
|
|
669
|
+
} else {
|
|
670
|
+
evaluation = await evaluateCandidate(runConfig, candidate.worktree, candidate.baseCommit, commandRunner)
|
|
671
|
+
}
|
|
672
|
+
if (runConfig.roles?.adversary) {
|
|
673
|
+
if (!fleetQueue) throw new Error("adversarial break tests require an OpenShell fleet worker")
|
|
674
|
+
const review = await runAdversarialStage(candidate, evaluation, runConfig, run, fleetQueue, hooks, now)
|
|
675
|
+
run.adversarialReviews ??= []
|
|
676
|
+
run.adversarialReviews.push(review)
|
|
677
|
+
evaluation.adversarialPass = review.status === "passed" && review.testResults.length > 0 && review.testResults.every((entry) => entry.passed)
|
|
678
|
+
evaluation.adversarialPenalty = review.penaltyPoints
|
|
679
|
+
await checkpointRun(config.archiveDir, run)
|
|
178
680
|
}
|
|
179
|
-
candidate.status = "evaluating"
|
|
180
|
-
candidate.changedFiles = await changedFiles(candidate.worktree, candidate.baseCommit)
|
|
181
|
-
candidate.commits = await candidateCommits(candidate.worktree, candidate.baseCommit)
|
|
182
|
-
candidate.protectedPathViolations = protectedPathViolations(candidate.changedFiles, config.protectedPaths)
|
|
183
|
-
const evaluation = await evaluateCandidate(runConfig, candidate.worktree, candidate.baseCommit, hooks.runCommand ?? runCommand)
|
|
184
681
|
candidate.changedFiles = evaluation.changedFiles
|
|
185
682
|
candidate.commits = candidate.commits.length > 0 ? candidate.commits : await candidateCommits(candidate.worktree, candidate.baseCommit)
|
|
186
683
|
candidate.result = evaluation.regression
|
|
187
684
|
candidate.fitness = computeFitness({ ...evaluation, protectedPathViolation: candidate.protectedPathViolations.length > 0 })
|
|
188
|
-
candidate.
|
|
685
|
+
if (run.baselineEvaluation) candidate.pairedComparison = comparePairedEvaluation(run.baselineEvaluation, evaluation)
|
|
686
|
+
candidate.status = candidate.fitness.score > 0 && candidate.pairedComparison?.improved === true ? "accepted" : "rejected"
|
|
189
687
|
candidate.updatedAt = now()
|
|
190
688
|
const combination = createCombination(candidate.id, candidate.modelCandidateId ?? "model-base", now())
|
|
191
689
|
combination.status = candidate.status === "accepted" ? "accepted" : "rejected"
|
|
@@ -202,14 +700,11 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
|
|
|
202
700
|
candidate.updatedAt = now()
|
|
203
701
|
job = transitionJob(job, "failed", now(), candidate.failure)
|
|
204
702
|
replaceJob(run, job)
|
|
205
|
-
|
|
703
|
+
await captureCandidate(candidate, runConfig, run, trajectories, now(), hooks)
|
|
704
|
+
log(hooks, `[rsi] ${candidate.id}: evaluation failed: ${candidate.failure}`)
|
|
206
705
|
} finally {
|
|
207
|
-
if (
|
|
208
|
-
|
|
209
|
-
await removeCandidateWorktree(runConfig, candidate)
|
|
210
|
-
} catch (error) {
|
|
211
|
-
log(hooks, `[rsi] ${candidate.id}: worktree cleanup failed: ${String(error)}`)
|
|
212
|
-
}
|
|
706
|
+
if (!config.keepWorktrees && managedWorktrees.has(candidate.id)) {
|
|
707
|
+
await removeCandidateWorktree(runConfig, candidate).catch((error) => log(hooks, `[rsi] ${candidate.id}: worktree cleanup failed: ${String(error)}`))
|
|
213
708
|
}
|
|
214
709
|
await checkpointRun(config.archiveDir, run)
|
|
215
710
|
}
|
|
@@ -220,12 +715,33 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
|
|
|
220
715
|
const ranked = [...accepted].sort((a, b) => compareFitness(b.fitness, a.fitness))
|
|
221
716
|
run.selectedCandidates = front.map((candidate) => candidate.id)
|
|
222
717
|
if (ranked[0]) run.selected = ranked[0].id
|
|
718
|
+
if (config.computePolicy === "adaptive-independent") {
|
|
719
|
+
run.adaptiveSearch ??= []
|
|
720
|
+
const decision = decideAdaptiveContinuation({
|
|
721
|
+
generation,
|
|
722
|
+
generationCandidates: candidates,
|
|
723
|
+
allCandidates: run.candidates,
|
|
724
|
+
config,
|
|
725
|
+
elapsedMs: Math.max(0, Date.now() - Date.parse(run.startedAt)),
|
|
726
|
+
decidedAt: now(),
|
|
727
|
+
})
|
|
728
|
+
const existingDecision = run.adaptiveSearch.findIndex((entry) => entry.generation === generation)
|
|
729
|
+
if (existingDecision >= 0) run.adaptiveSearch[existingDecision] = decision
|
|
730
|
+
else run.adaptiveSearch.push(decision)
|
|
731
|
+
log(hooks, `[rsi] adaptive search: ${decision.decision} after generation ${generation} (${decision.reason})`)
|
|
732
|
+
}
|
|
223
733
|
await checkpointRun(config.archiveDir, run)
|
|
224
734
|
}
|
|
225
735
|
|
|
226
736
|
const merged = combinedArchive(archive, run)
|
|
227
|
-
|
|
228
|
-
|
|
737
|
+
let curriculum = generateCurriculumProposals(merged, now())
|
|
738
|
+
if (fleetQueue && !config.dryRun && curriculum.length > 0) {
|
|
739
|
+
curriculum = await validateCurriculumOnFleet(curriculum, runConfig, run, fleetQueue, now)
|
|
740
|
+
run.curriculumTasks = curriculum
|
|
741
|
+
await checkpointRun(config.archiveDir, run)
|
|
742
|
+
} else {
|
|
743
|
+
run.curriculumTasks = curriculum
|
|
744
|
+
}
|
|
229
745
|
if (!config.dryRun && curriculum.length > 0) {
|
|
230
746
|
await writeCurriculumProposals(curriculum, config.curriculumDir ?? `${config.archiveDir}/curriculum`)
|
|
231
747
|
}
|
|
@@ -245,6 +761,9 @@ export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<R
|
|
|
245
761
|
log(hooks, report.trimEnd())
|
|
246
762
|
}
|
|
247
763
|
return run
|
|
764
|
+
} finally {
|
|
765
|
+
if (fleetQueue) await fleetQueue.close()
|
|
766
|
+
}
|
|
248
767
|
}
|
|
249
768
|
|
|
250
769
|
export async function improveMain(argv: string[]): Promise<number> {
|