headlesscode 1.2.2 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -16
- package/package.json +7 -3
- package/src/cli.ts +47 -18
- package/src/cloud/openshell-preflight.ts +2 -2
- package/src/cloud/openshell-provider.ts +84 -24
- package/src/llm/ollama.ts +25 -19
- package/src/project-store.ts +4 -1
- package/src/rsi/adaptive.ts +49 -0
- package/src/rsi/adversarial.ts +106 -0
- package/src/rsi/archive.ts +6 -5
- package/src/rsi/artifact-store.ts +158 -0
- package/src/rsi/config.ts +50 -2
- package/src/rsi/controller.ts +561 -42
- package/src/rsi/curriculum.ts +135 -16
- package/src/rsi/evaluator.ts +6 -37
- package/src/rsi/fitness.ts +39 -4
- package/src/rsi/index.ts +1 -0
- package/src/rsi/migrations/001_postgres_fleet_queue.sql +65 -0
- package/src/rsi/migrations/002_external_artifacts_and_job_leases.sql +39 -0
- package/src/rsi/migrations/003_model_training_jobs.sql +6 -0
- package/src/rsi/model-training.ts +256 -0
- package/src/rsi/mutation.ts +1 -77
- package/src/rsi/openshell.ts +639 -0
- package/src/rsi/postgres-queue.ts +424 -0
- package/src/rsi/promote-curriculum.ts +21 -0
- package/src/rsi/reports.ts +29 -2
- package/src/rsi/roles.ts +13 -3
- package/src/rsi/selection.ts +7 -1
- package/src/rsi/training-data.ts +103 -0
- package/src/rsi/trajectory.ts +1 -1
- package/src/rsi/types.ts +115 -2
- package/src/rsi/worker.ts +264 -0
- package/src/rsi/workspace.ts +14 -3
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
import { createHash } from "node:crypto"
|
|
2
|
+
import { execFileSync } from "node:child_process"
|
|
3
|
+
import { appendRun, checkpointRun } from "./archive.js"
|
|
4
|
+
import { createModelJobSnapshotBundle } from "./openshell.js"
|
|
5
|
+
import { fleetJobMatches, requiredFleetJobLeaseMs, type FleetJob, type FleetJobIdentity, type FleetJobPayload, type PostgresRsiJobQueue } from "./postgres-queue.js"
|
|
6
|
+
import { buildVerifiedTrainingDataset } from "./training-data.js"
|
|
7
|
+
import type { CandidateRecord, ModelCandidate, RsiConfig, RsiRunRecord } from "./types.js"
|
|
8
|
+
|
|
9
|
+
const SEED = 1337
|
|
10
|
+
const CELL_IDS = ["generalization:negative number", "generalization:positive numeric string", "generalization:zero string"]
|
|
11
|
+
|
|
12
|
+
function digest(value: Buffer | string): string {
|
|
13
|
+
return createHash("sha256").update(value).digest("hex")
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
function modelCandidateId(value: string): string {
|
|
17
|
+
if (!/^[A-Za-z0-9._-]{1,80}$/.test(value)) throw new Error("model candidate id must contain only letters, numbers, dot, underscore, or dash")
|
|
18
|
+
return value
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
function modelJobPayload(config: RsiConfig, candidate: CandidateRecord, snapshot: { content: Buffer; snapshotCommit: string }, artifactSha256: string, task: NonNullable<FleetJobPayload["modelTask"]>): FleetJobPayload {
|
|
22
|
+
return {
|
|
23
|
+
schemaVersion: 1,
|
|
24
|
+
snapshotSha256: artifactSha256,
|
|
25
|
+
snapshotBytes: snapshot.content.byteLength,
|
|
26
|
+
snapshotCommit: snapshot.snapshotCommit,
|
|
27
|
+
task: "Run the fixed offline RSI model training/evaluation program.",
|
|
28
|
+
model: candidate.model,
|
|
29
|
+
maxIterations: 1,
|
|
30
|
+
timeoutMs: config.commandTimeoutMs,
|
|
31
|
+
protectedPaths: config.protectedPaths,
|
|
32
|
+
candidate: { id: candidate.id, generation: 0, parent: "base", modelCandidateId: task.modelCandidateId, computePolicy: "single" },
|
|
33
|
+
regressionCommand: "true",
|
|
34
|
+
evalCommands: [],
|
|
35
|
+
modelTask: task,
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
async function waitForModelJob(queue: PostgresRsiJobQueue, jobId: string, timeoutMs: number): Promise<FleetJob> {
|
|
40
|
+
const deadline = Date.now() + timeoutMs
|
|
41
|
+
while (Date.now() < deadline) {
|
|
42
|
+
const job = await queue.getJob(jobId)
|
|
43
|
+
if (!job) throw new Error(`model job ${jobId} disappeared`)
|
|
44
|
+
if (["completed", "failed", "cancelled"].includes(job.status)) return job
|
|
45
|
+
await new Promise((resolve) => setTimeout(resolve, 500))
|
|
46
|
+
}
|
|
47
|
+
// Keep a live lease and capacity reserved until the OpenShell worker confirms
|
|
48
|
+
// teardown or the lease expires. The coordinator does not clear it on timeout.
|
|
49
|
+
throw new Error(`model job ${jobId} exceeded coordinator wait timeout`)
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function assertModelJob(job: FleetJob, expected: FleetJobIdentity): void {
|
|
53
|
+
if (!fleetJobMatches(job, expected)) {
|
|
54
|
+
throw new Error("existing model job idempotency key refers to a different payload or artifact")
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function readJsonArtifact(queue: PostgresRsiJobQueue, sha256: string, bytes: number): Promise<Record<string, unknown>> {
|
|
59
|
+
return queue.getArtifact(sha256).then((content) => {
|
|
60
|
+
if (content.byteLength !== bytes) throw new Error("model result artifact length differs from its recorded metadata")
|
|
61
|
+
return JSON.parse(content.toString("utf8")) as Record<string, unknown>
|
|
62
|
+
})
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function completedModelResult(job: FleetJob): Record<string, unknown> {
|
|
66
|
+
if (job.status !== "completed" || !job.result || typeof job.result !== "object") throw new Error(job.error ?? `model job ended ${job.status}`)
|
|
67
|
+
const result = job.result as { modelResult?: unknown }
|
|
68
|
+
if (!result.modelResult || typeof result.modelResult !== "object") throw new Error("completed model job has no bounded result record")
|
|
69
|
+
return result.modelResult as Record<string, unknown>
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function validatePairedResult(result: Record<string, unknown>, modelKind: "base" | "adapter", dataset: ReturnType<typeof buildVerifiedTrainingDataset>, baseModelId: string): { exactMatchRate: number; outputSha256: string; resultSha256: string; baseModelSha256: string; adapterSha256?: string } {
|
|
73
|
+
if (result.status !== "completed" || result.modelKind !== modelKind || result.seed !== SEED || result.cellSetSha256 !== digest(Buffer.from([...dataset.evaluationCellIds].sort().join("\n")))) {
|
|
74
|
+
throw new Error(`${modelKind} evaluation did not complete the fixed seed and cell set`)
|
|
75
|
+
}
|
|
76
|
+
if (result.baseModelId !== baseModelId) throw new Error(`${modelKind} evaluation used a different configured base model identity`)
|
|
77
|
+
if (!Array.isArray(result.cellSet) || JSON.stringify([...result.cellSet].sort()) !== JSON.stringify([...dataset.evaluationCellIds].sort())) throw new Error(`${modelKind} evaluation cell set is incomplete or changed`)
|
|
78
|
+
if (!Array.isArray(result.results) || result.results.length !== dataset.evaluationCellIds.length) throw new Error(`${modelKind} evaluation omitted one or more paired cells`)
|
|
79
|
+
const ids = result.results.map((item) => item && typeof item === "object" ? (item as { cellId?: unknown }).cellId : undefined)
|
|
80
|
+
if (JSON.stringify([...ids].sort()) !== JSON.stringify([...dataset.evaluationCellIds].sort())) throw new Error(`${modelKind} evaluation result IDs differ from the registered harness cells`)
|
|
81
|
+
const metric = result.exactMatchRate
|
|
82
|
+
if (typeof metric !== "number" || !Number.isFinite(metric) || metric < 0 || metric > 1) throw new Error(`${modelKind} evaluation metric is invalid`)
|
|
83
|
+
const outputs = result.results.map((item) => (item as { outputSha256?: unknown }).outputSha256)
|
|
84
|
+
if (outputs.some((value) => typeof value !== "string" || !/^[0-9a-f]{64}$/.test(value))) throw new Error(`${modelKind} evaluation output digest is invalid`)
|
|
85
|
+
const baseModelSha256 = String(result.baseModelSha256 ?? "")
|
|
86
|
+
if (!/^[0-9a-f]{64}$/.test(baseModelSha256)) throw new Error(`${modelKind} evaluation omitted base model provenance`)
|
|
87
|
+
const adapterSha256 = modelKind === "adapter" ? String(result.adapterSha256 ?? "") : undefined
|
|
88
|
+
if (modelKind === "adapter" && !/^[0-9a-f]{64}$/.test(adapterSha256!)) throw new Error("adapter evaluation omitted adapter digest provenance")
|
|
89
|
+
return { exactMatchRate: metric, outputSha256: digest(JSON.stringify(outputs)), resultSha256: String(result.resultSha256 ?? ""), baseModelSha256, ...(adapterSha256 ? { adapterSha256 } : {}) }
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** Enqueue bounded OpenShell QLoRA training and paired base/adapter model evaluation. */
|
|
93
|
+
export async function runBoundedModelCandidate(config: RsiConfig, run: RsiRunRecord, queue: PostgresRsiJobQueue, id: string, baseModelId = process.env.HEADLESSCODE_RSI_BASE_MODEL_ID?.trim() ?? ""): Promise<ModelCandidate> {
|
|
94
|
+
if (config.dryRun) throw new Error("model training does not support dry-run mode")
|
|
95
|
+
if (queue.artifactBackend !== "s3") throw new Error("model training requires the shared S3-compatible RSI artifact store")
|
|
96
|
+
const configuredBasePath = process.env.HEADLESSCODE_RSI_BASE_MODEL_PATH?.trim() ?? ""
|
|
97
|
+
if (/\.gguf$/i.test(configuredBasePath) || /\.gguf$/i.test(baseModelId)) {
|
|
98
|
+
throw new Error("GGUF base models are not supported by the current QLoRA runner: Transformers GGUF loading cannot be combined with bitsandbytes training, and the paired evaluator requires a Hugging Face checkpoint. No model job was enqueued.")
|
|
99
|
+
}
|
|
100
|
+
if (!configuredBasePath) throw new Error("HEADLESSCODE_RSI_BASE_MODEL_PATH is required on the coordinator and model workers; no model job was enqueued")
|
|
101
|
+
if (await queue.activeWorkerCount("TRAINING_GPU") < 1) throw new Error("no fresh OpenShell worker advertises TRAINING_GPU; model training was not enqueued")
|
|
102
|
+
if (await queue.activeWorkerCount("CPU") < 1) throw new Error("no fresh OpenShell worker advertises CPU; paired model evaluation was not enqueued")
|
|
103
|
+
const candidateId = modelCandidateId(id)
|
|
104
|
+
if (!/^[A-Za-z0-9._:@/-]{1,180}$/.test(baseModelId)) throw new Error("HEADLESSCODE_RSI_BASE_MODEL_ID must identify the operator-provided local Hugging Face checkpoint")
|
|
105
|
+
const existing = run.modelCandidates?.find((entry) => entry.id === candidateId)
|
|
106
|
+
if (existing?.eligibleForSelection) throw new Error(`model candidate ${candidateId} is already eligible and cannot be retrained in place`)
|
|
107
|
+
const model: ModelCandidate = existing ?? {
|
|
108
|
+
id: candidateId,
|
|
109
|
+
parentModel: baseModelId,
|
|
110
|
+
trainingMethod: "qlora",
|
|
111
|
+
datasetVersion: undefined,
|
|
112
|
+
trainingConfig: { method: "qlora", seed: SEED, parameters: { maxSteps: 8, rank: 8, alpha: 16, dropout: 0.05, quantization: "nf4-double" } },
|
|
113
|
+
status: "prepared",
|
|
114
|
+
createdAt: new Date().toISOString(),
|
|
115
|
+
eligibleForSelection: false,
|
|
116
|
+
provenance: {},
|
|
117
|
+
}
|
|
118
|
+
run.modelCandidates ??= []
|
|
119
|
+
if (!existing) run.modelCandidates.push(model)
|
|
120
|
+
const candidate: CandidateRecord = {
|
|
121
|
+
id: `model-${candidateId}`,
|
|
122
|
+
generation: 0,
|
|
123
|
+
parent: "base",
|
|
124
|
+
branch: `rsi-model-${candidateId}`,
|
|
125
|
+
worktree: config.repoRoot,
|
|
126
|
+
baseCommit: run.baseCommit,
|
|
127
|
+
status: "evaluating",
|
|
128
|
+
model: config.model,
|
|
129
|
+
mutation: "Run bounded QLoRA model candidate",
|
|
130
|
+
createdAt: new Date().toISOString(),
|
|
131
|
+
updatedAt: new Date().toISOString(),
|
|
132
|
+
commits: [],
|
|
133
|
+
changedFiles: [],
|
|
134
|
+
protectedPathViolations: [],
|
|
135
|
+
modelCandidateId: candidateId,
|
|
136
|
+
}
|
|
137
|
+
try {
|
|
138
|
+
const dataset = buildVerifiedTrainingDataset(config.repoRoot)
|
|
139
|
+
model.datasetVersion = dataset.version
|
|
140
|
+
model.provenance = { datasetSha256: dataset.datasetSha256, manifestSha256: dataset.manifestSha256, evaluationCellsSha256: dataset.evaluationCellsSha256, seed: String(SEED), maxSteps: "8", image: "headlesscode-openshell-rsi-training:local", baseModelId }
|
|
141
|
+
await checkpointRun(config.archiveDir, run)
|
|
142
|
+
const trainingSnapshot = createModelJobSnapshotBundle(config.repoRoot, run.baseCommit, config, { dataset: dataset.dataset, manifest: dataset.manifest, cells: dataset.evaluationCells })
|
|
143
|
+
const trainingArtifact = await queue.putArtifact(trainingSnapshot.content)
|
|
144
|
+
const trainingTask: NonNullable<FleetJobPayload["modelTask"]> = {
|
|
145
|
+
kind: "qlora-train", datasetVersion: dataset.version, datasetSha256: dataset.datasetSha256,
|
|
146
|
+
manifestSha256: dataset.manifestSha256, baseModelId, baseModelPath: "/workspace/.rsi-base-model", seed: SEED,
|
|
147
|
+
maxSteps: 8, modelCandidateId: candidateId,
|
|
148
|
+
cellSetSha256: digest(Buffer.from([...dataset.evaluationCellIds].sort().join("\n"))), cellIds: [...dataset.evaluationCellIds].sort(),
|
|
149
|
+
}
|
|
150
|
+
const trainingExpected = {
|
|
151
|
+
idempotencyKey: `${run.runId}:${candidateId}:qlora-train`, runId: run.runId, candidateId: `model-${candidateId}`,
|
|
152
|
+
jobKind: "training" as const, resourceClass: "TRAINING_GPU" as const, concurrencyKey: `training:${config.model}`,
|
|
153
|
+
artifactSha256: trainingArtifact.sha256,
|
|
154
|
+
payload: modelJobPayload(config, candidate, trainingSnapshot, trainingArtifact.sha256, trainingTask),
|
|
155
|
+
}
|
|
156
|
+
let trainingJob = await queue.getJobByIdempotencyKey(trainingExpected.idempotencyKey)
|
|
157
|
+
if (!trainingJob) trainingJob = await queue.enqueue(trainingExpected)
|
|
158
|
+
assertModelJob(trainingJob, trainingExpected)
|
|
159
|
+
const trainingCompleted = await waitForModelJob(queue, trainingJob.jobId, requiredFleetJobLeaseMs("training", config.commandTimeoutMs) + 60_000)
|
|
160
|
+
const trainResult = completedModelResult(trainingCompleted)
|
|
161
|
+
const status = trainResult.status
|
|
162
|
+
if (status !== "completed") {
|
|
163
|
+
model.status = status === "partial" ? "partial" : "failed"
|
|
164
|
+
model.trainingResult = { status: model.status, datasetVersion: dataset.version, datasetSha256: dataset.datasetSha256, baseModelSha256: String(trainResult.baseModelSha256 ?? ""), stepsCompleted: Number(trainResult.stepsCompleted ?? 0), maxSteps: 8, error: String(trainResult.error ?? "training did not complete") }
|
|
165
|
+
model.eligibleForSelection = false
|
|
166
|
+
model.provenance.trainingJobId = trainingJob.jobId
|
|
167
|
+
await checkpointRun(config.archiveDir, run)
|
|
168
|
+
if (run.finishedAt) await appendRun(config.archiveDir, run)
|
|
169
|
+
return model
|
|
170
|
+
}
|
|
171
|
+
const resultArtifactSha = String(trainResult.outputArtifactSha256 ?? "")
|
|
172
|
+
const resultArtifactBytes = Number(trainResult.outputArtifactBytes)
|
|
173
|
+
if (!/^[0-9a-f]{64}$/.test(resultArtifactSha) || !Number.isSafeInteger(resultArtifactBytes) || resultArtifactBytes < 1) throw new Error("training result artifact reference is invalid")
|
|
174
|
+
const fullTrainingResult = await readJsonArtifact(queue, resultArtifactSha, resultArtifactBytes)
|
|
175
|
+
const baseModelSha256 = String(fullTrainingResult.baseModelSha256 ?? "")
|
|
176
|
+
if (fullTrainingResult.status !== "completed" || fullTrainingResult.datasetSha256 !== dataset.datasetSha256 || fullTrainingResult.manifestSha256 !== dataset.manifestSha256 || fullTrainingResult.baseModelId !== baseModelId || fullTrainingResult.stepsCompleted !== 8 || !/^[0-9a-f]{64}$/.test(baseModelSha256)) throw new Error("training artifact does not prove complete execution against the verified dataset and base model")
|
|
177
|
+
const adapterSha = String(trainResult.adapterArtifactSha256 ?? "")
|
|
178
|
+
const adapterBytes = Number(trainResult.adapterArtifactBytes)
|
|
179
|
+
if (!/^[0-9a-f]{64}$/.test(adapterSha) || !Number.isSafeInteger(adapterBytes) || adapterBytes < 1) throw new Error("trained adapter artifact reference is invalid")
|
|
180
|
+
const adapterArchive = await queue.getArtifact(adapterSha)
|
|
181
|
+
if (adapterArchive.byteLength !== adapterBytes || digest(adapterArchive) !== adapterSha) throw new Error("trained adapter artifact failed content digest verification")
|
|
182
|
+
model.status = "trained"
|
|
183
|
+
model.artifactHash = adapterSha
|
|
184
|
+
model.artifactPath = `sha256:${adapterSha}`
|
|
185
|
+
model.trainingResult = { status: "completed", datasetVersion: dataset.version, datasetSha256: dataset.datasetSha256, baseModelSha256, artifactSha256: adapterSha, artifactBytes: adapterBytes, stepsCompleted: 8, maxSteps: 8 }
|
|
186
|
+
model.provenance.trainingJobId = trainingJob.jobId
|
|
187
|
+
await checkpointRun(config.archiveDir, run)
|
|
188
|
+
|
|
189
|
+
const pair = [] as Array<{ kind: "base" | "adapter"; job: FleetJob; result: Record<string, unknown> }>
|
|
190
|
+
for (const kind of ["base", "adapter"] as const) {
|
|
191
|
+
const snapshot = createModelJobSnapshotBundle(config.repoRoot, run.baseCommit, config, { dataset: dataset.dataset, manifest: dataset.manifest, cells: dataset.evaluationCells, ...(kind === "adapter" ? { adapterArchive } : {}) })
|
|
192
|
+
const artifact = await queue.putArtifact(snapshot.content)
|
|
193
|
+
const modelTask: NonNullable<FleetJobPayload["modelTask"]> = {
|
|
194
|
+
kind: "model-evaluation", datasetVersion: dataset.version, datasetSha256: dataset.datasetSha256,
|
|
195
|
+
manifestSha256: dataset.manifestSha256, baseModelId, baseModelPath: "/workspace/.rsi-base-model", seed: SEED,
|
|
196
|
+
modelCandidateId: candidateId, adapterPath: kind === "adapter" ? "/workspace/.headlesscode-rsi-model/adapter.tar" : undefined,
|
|
197
|
+
modelKind: kind, cellSetSha256: trainingTask.cellSetSha256, cellIds: trainingTask.cellIds,
|
|
198
|
+
}
|
|
199
|
+
const expected = {
|
|
200
|
+
idempotencyKey: `${run.runId}:${candidateId}:paired-${kind}`, runId: run.runId, candidateId: `model-${candidateId}`,
|
|
201
|
+
jobKind: "model-evaluation" as const, resourceClass: "CPU" as const, concurrencyKey: `model-eval:${candidateId}:${kind}`,
|
|
202
|
+
artifactSha256: artifact.sha256, payload: modelJobPayload(config, candidate, snapshot, artifact.sha256, modelTask),
|
|
203
|
+
}
|
|
204
|
+
let job = await queue.getJobByIdempotencyKey(expected.idempotencyKey)
|
|
205
|
+
if (!job) job = await queue.enqueue(expected)
|
|
206
|
+
assertModelJob(job, expected)
|
|
207
|
+
pair.push({ kind, job, result: {} })
|
|
208
|
+
}
|
|
209
|
+
// Both cells are admitted before waiting, allowing independent CPU workers to run them concurrently.
|
|
210
|
+
for (const entry of pair) {
|
|
211
|
+
const completed = await waitForModelJob(queue, entry.job.jobId, requiredFleetJobLeaseMs("model-evaluation", config.commandTimeoutMs) + 60_000)
|
|
212
|
+
const bounded = completedModelResult(completed)
|
|
213
|
+
const resultSha = String(bounded.outputArtifactSha256 ?? "")
|
|
214
|
+
const resultBytes = Number(bounded.outputArtifactBytes)
|
|
215
|
+
if (!/^[0-9a-f]{64}$/.test(resultSha) || !Number.isSafeInteger(resultBytes) || resultBytes < 1) throw new Error(`${entry.kind} paired result has no verified artifact reference`)
|
|
216
|
+
const full = await readJsonArtifact(queue, resultSha, resultBytes)
|
|
217
|
+
entry.result = { ...full, resultSha256: resultSha }
|
|
218
|
+
}
|
|
219
|
+
const baseEval = validatePairedResult(pair[0]!.result, "base", dataset, baseModelId)
|
|
220
|
+
const adapterEval = validatePairedResult(pair[1]!.result, "adapter", dataset, baseModelId)
|
|
221
|
+
if (baseEval.baseModelSha256 !== baseModelSha256 || adapterEval.baseModelSha256 !== baseModelSha256) throw new Error("paired evaluations did not use the exact checkpoint used for training")
|
|
222
|
+
if (adapterEval.adapterSha256 !== fullTrainingResult.adapterSha256) throw new Error("paired adapter evaluation digest differs from the trained adapter artifact")
|
|
223
|
+
model.pairedEvaluation = {
|
|
224
|
+
cellSetSha256: trainingTask.cellSetSha256,
|
|
225
|
+
baseResultSha256: baseEval.resultSha256 || baseEval.outputSha256,
|
|
226
|
+
adapterResultSha256: adapterEval.resultSha256 || adapterEval.outputSha256,
|
|
227
|
+
baseMetric: baseEval.exactMatchRate,
|
|
228
|
+
adapterMetric: adapterEval.exactMatchRate,
|
|
229
|
+
completedCellCount: dataset.evaluationCellIds.length,
|
|
230
|
+
expectedCellCount: dataset.evaluationCellIds.length,
|
|
231
|
+
seed: SEED,
|
|
232
|
+
}
|
|
233
|
+
model.provenance.baseEvaluationJobId = pair[0]!.job.jobId
|
|
234
|
+
model.provenance.adapterEvaluationJobId = pair[1]!.job.jobId
|
|
235
|
+
model.provenance.baseOutputSha256 = baseEval.outputSha256
|
|
236
|
+
model.provenance.adapterOutputSha256 = adapterEval.outputSha256
|
|
237
|
+
model.provenance.adapterTreeSha256 = adapterEval.adapterSha256!
|
|
238
|
+
model.status = adapterEval.exactMatchRate > baseEval.exactMatchRate ? "accepted" : "rejected"
|
|
239
|
+
model.eligibleForSelection = model.status === "accepted"
|
|
240
|
+
await checkpointRun(config.archiveDir, run)
|
|
241
|
+
if (run.finishedAt) await appendRun(config.archiveDir, run)
|
|
242
|
+
return model
|
|
243
|
+
} catch (error) {
|
|
244
|
+
model.status = model.status === "trained" ? "partial" : "failed"
|
|
245
|
+
model.eligibleForSelection = false
|
|
246
|
+
model.provenance = { ...(model.provenance ?? {}), failure: error instanceof Error ? error.message : String(error) }
|
|
247
|
+
model.trainingResult ??= { status: model.status, datasetVersion: model.datasetVersion ?? "unavailable", datasetSha256: model.provenance.datasetSha256 ?? "", baseModelSha256: "", stepsCompleted: 0, maxSteps: 8, error: model.provenance.failure }
|
|
248
|
+
await checkpointRun(config.archiveDir, run)
|
|
249
|
+
if (run.finishedAt) await appendRun(config.archiveDir, run)
|
|
250
|
+
return model
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
export function currentModelBaseCommit(repoRoot: string, baseRef: string): string {
|
|
255
|
+
return execFileSync("git", ["rev-parse", "--verify", `${baseRef}^{commit}`], { cwd: repoRoot, encoding: "utf8" }).trim()
|
|
256
|
+
}
|
package/src/rsi/mutation.ts
CHANGED
|
@@ -1,77 +1 @@
|
|
|
1
|
-
|
|
2
|
-
import * as path from "node:path"
|
|
3
|
-
import { promisify } from "node:util"
|
|
4
|
-
import type { MutationRunner, RsiConfig, TrialResult, CandidateRecord } from "./types.js"
|
|
5
|
-
import { DEFAULT_RSI_MODEL } from "./config.js"
|
|
6
|
-
|
|
7
|
-
const exec = promisify(execCallback)
|
|
8
|
-
|
|
9
|
-
export function mutationPrompt(candidate: CandidateRecord, config: RsiConfig): string {
|
|
10
|
-
const hypothesis = candidate.hypothesis
|
|
11
|
-
return [
|
|
12
|
-
"You are the bounded RSI worker for headlesscode.",
|
|
13
|
-
`Improve this candidate checkout for generation ${candidate.generation}.`,
|
|
14
|
-
`Mutation class: ${candidate.mutationKind ?? "corrective"}. Compute policy: ${config.computePolicy ?? "single"}.`,
|
|
15
|
-
`Objective: ${config.mutationTask}`,
|
|
16
|
-
hypothesis
|
|
17
|
-
? `Hypothesis: ${hypothesis.statement}\nExpected effect: ${hypothesis.expectedEffect}\nPotential downside: ${hypothesis.potentialDownside}${hypothesis.evidence ? `\nEvidence: ${hypothesis.evidence}` : ""}`
|
|
18
|
-
: "",
|
|
19
|
-
"You may change agent implementation, but you must not modify tests, evaluator/scoring/archive code, hidden evaluation commands, git metadata, or protected paths.",
|
|
20
|
-
"Inspect the existing repository first. Make a small, test-backed change. Run the relevant existing tests. Commit the candidate change before completing.",
|
|
21
|
-
].join("\n\n")
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
export const runMutation: MutationRunner = async (candidate, config) => {
|
|
25
|
-
const started = Date.now()
|
|
26
|
-
const workerRole = config.roles?.worker
|
|
27
|
-
const workerModel = workerRole?.model ?? config.model
|
|
28
|
-
const workerProvider = workerRole?.provider ?? "ollama"
|
|
29
|
-
const env: NodeJS.ProcessEnv = {
|
|
30
|
-
...process.env,
|
|
31
|
-
HEADLESSCODE_OPENROUTER_API_KEY: process.env.HEADLESSCODE_OPENROUTER_API_KEY ?? "rsi-local-placeholder",
|
|
32
|
-
HEADLESSCODE_CODE_MODE_BACKEND: workerProvider === "ollama" ? "ollama" : "openrouter",
|
|
33
|
-
HEADLESSCODE_LOCAL_BACKEND_MODES: "code",
|
|
34
|
-
HEADLESSCODE_CODE_MODE_MODEL: workerModel || DEFAULT_RSI_MODEL,
|
|
35
|
-
HEADLESSCODE_OLLAMA_THINK: "0",
|
|
36
|
-
HEADLESSCODE_CAPTURE_TRANSCRIPT_DIR: path.join(candidate.worktree, ".headlesscode", "rsi-transcripts"),
|
|
37
|
-
}
|
|
38
|
-
const command = [
|
|
39
|
-
"npx tsx src/cli.ts",
|
|
40
|
-
"--mode code",
|
|
41
|
-
`--max-iterations ${config.maxIterations}`,
|
|
42
|
-
"--no-checkpoints",
|
|
43
|
-
`--task ${JSON.stringify(mutationPrompt(candidate, config))}`,
|
|
44
|
-
].join(" ")
|
|
45
|
-
try {
|
|
46
|
-
const result = await exec(command, {
|
|
47
|
-
cwd: candidate.worktree,
|
|
48
|
-
env,
|
|
49
|
-
timeout: config.commandTimeoutMs,
|
|
50
|
-
maxBuffer: 16 * 1024 * 1024,
|
|
51
|
-
})
|
|
52
|
-
const trial: TrialResult = {
|
|
53
|
-
ok: true,
|
|
54
|
-
command,
|
|
55
|
-
exitCode: 0,
|
|
56
|
-
durationMs: Date.now() - started,
|
|
57
|
-
stdout: result.stdout,
|
|
58
|
-
stderr: result.stderr,
|
|
59
|
-
}
|
|
60
|
-
return { ok: true, result: trial }
|
|
61
|
-
} catch (error) {
|
|
62
|
-
const failure = error as { code?: number; killed?: boolean; stdout?: string; stderr?: string }
|
|
63
|
-
return {
|
|
64
|
-
ok: false,
|
|
65
|
-
result: {
|
|
66
|
-
ok: false,
|
|
67
|
-
command,
|
|
68
|
-
exitCode: typeof failure.code === "number" ? failure.code : null,
|
|
69
|
-
durationMs: Date.now() - started,
|
|
70
|
-
stdout: failure.stdout ?? "",
|
|
71
|
-
stderr: failure.stderr ?? String(error),
|
|
72
|
-
timedOut: failure.killed === true,
|
|
73
|
-
},
|
|
74
|
-
error: error instanceof Error ? error.message : String(error),
|
|
75
|
-
}
|
|
76
|
-
}
|
|
77
|
-
}
|
|
1
|
+
export { runOpenShellMutation as runMutation } from "./openshell.js"
|