headlesscode 1.0.3 → 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -56
- package/package.json +1 -1
- package/src/cli.ts +115 -4
- package/src/engine/claims.ts +503 -0
- package/src/engine/events.ts +32 -0
- package/src/engine/lazy-tools.ts +30 -9
- package/src/engine/loop.ts +473 -11
- package/src/engine/prompt.ts +5 -1
- package/src/engine/types.ts +36 -0
- package/src/rsi/archive.ts +129 -0
- package/src/rsi/config.ts +312 -0
- package/src/rsi/controller.ts +268 -0
- package/src/rsi/curriculum.ts +68 -0
- package/src/rsi/evaluator.ts +106 -0
- package/src/rsi/fitness.ts +64 -0
- package/src/rsi/index.ts +16 -0
- package/src/rsi/models.ts +89 -0
- package/src/rsi/mutation.ts +77 -0
- package/src/rsi/reports.ts +47 -0
- package/src/rsi/roles.ts +37 -0
- package/src/rsi/sandbox.ts +10 -0
- package/src/rsi/search.ts +32 -0
- package/src/rsi/selection.ts +132 -0
- package/src/rsi/trajectory.ts +143 -0
- package/src/rsi/types.ts +317 -0
- package/src/rsi/workspace.ts +96 -0
- package/src/tools/executor.ts +79 -3
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
import { randomUUID } from "node:crypto"
|
|
2
|
+
import * as fs from "node:fs/promises"
|
|
3
|
+
import { appendRun, checkpointRun, findActiveRun, readArchive } from "./archive.js"
|
|
4
|
+
import { parseRsiArgs, rsiHelp } from "./config.js"
|
|
5
|
+
import { generateCurriculumProposals, writeCurriculumProposals } from "./curriculum.js"
|
|
6
|
+
import { evaluateCandidate, runCommand } from "./evaluator.js"
|
|
7
|
+
import { computeFitness, compareFitness } from "./fitness.js"
|
|
8
|
+
import { baseModelCandidate, createCombination } from "./models.js"
|
|
9
|
+
import { runMutation } from "./mutation.js"
|
|
10
|
+
import { formatRunReport } from "./reports.js"
|
|
11
|
+
import { protectedPathViolations } from "./sandbox.js"
|
|
12
|
+
import { resolveRoles } from "./roles.js"
|
|
13
|
+
import { newExperimentJob, paretoFront, selectParentChoices, transitionJob } from "./selection.js"
|
|
14
|
+
import { captureCandidateTrajectory, exportTrajectoryDatasets } from "./trajectory.js"
|
|
15
|
+
import type { CandidateRecord, ExperimentJob, RsiArchive, RsiConfig, RsiHooks, RsiRunRecord, TrajectoryRecord } from "./types.js"
|
|
16
|
+
import {
|
|
17
|
+
candidateCommits,
|
|
18
|
+
changedFiles,
|
|
19
|
+
createCandidateWorktree,
|
|
20
|
+
newCandidate,
|
|
21
|
+
removeCandidateWorktree,
|
|
22
|
+
resolveBaseCommit,
|
|
23
|
+
} from "./workspace.js"
|
|
24
|
+
|
|
25
|
+
function log(hooks: RsiHooks, line: string): void {
|
|
26
|
+
(hooks.log ?? ((message) => process.stdout.write(`${message}\n`)))(line)
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
function isTerminal(candidate: CandidateRecord): boolean {
|
|
30
|
+
return candidate.status === "accepted" || candidate.status === "rejected" || candidate.status === "failed"
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function replaceJob(run: RsiRunRecord, job: ExperimentJob): void {
|
|
34
|
+
run.jobs ??= []
|
|
35
|
+
const index = run.jobs.findIndex((entry) => entry.id === job.id)
|
|
36
|
+
if (index >= 0) run.jobs[index] = job
|
|
37
|
+
else run.jobs.push(job)
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function combinedArchive(archive: RsiArchive, run: RsiRunRecord): RsiArchive {
|
|
41
|
+
const currentIds = new Set(run.candidates.map((candidate) => candidate.id))
|
|
42
|
+
return { ...archive, candidates: [...archive.candidates.filter((candidate) => !currentIds.has(candidate.id)), ...run.candidates] }
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function initialRun(config: RsiConfig, baseCommit: string, now: string): RsiRunRecord {
|
|
46
|
+
const modelCandidateId = config.modelCandidateId ?? "model-base"
|
|
47
|
+
return {
|
|
48
|
+
runId: `rsi-${now.replace(/[^0-9]/g, "").slice(0, 14)}-${randomUUID().slice(0, 8)}`,
|
|
49
|
+
startedAt: now,
|
|
50
|
+
model: config.model,
|
|
51
|
+
baseRef: config.baseRef,
|
|
52
|
+
baseCommit,
|
|
53
|
+
generations: config.generations,
|
|
54
|
+
parentSelectionPolicy: config.parentSelectionPolicy ?? "champion-specialist-novelty",
|
|
55
|
+
selectedCandidates: [],
|
|
56
|
+
parentChoices: {},
|
|
57
|
+
modelCandidates: [baseModelCandidate(config.model, now, modelCandidateId)],
|
|
58
|
+
combinations: [],
|
|
59
|
+
jobs: [],
|
|
60
|
+
trajectoryRefs: [],
|
|
61
|
+
curriculumTasks: [],
|
|
62
|
+
candidates: [],
|
|
63
|
+
reports: [],
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
function recoverInterruptedCandidates(run: RsiRunRecord, now: string): void {
|
|
68
|
+
for (const candidate of run.candidates) {
|
|
69
|
+
if (candidate.status === "mutating" || candidate.status === "evaluating") {
|
|
70
|
+
candidate.status = "failed"
|
|
71
|
+
candidate.failure = "run interrupted before candidate reached a terminal state"
|
|
72
|
+
candidate.updatedAt = now
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
async function captureCandidate(
|
|
78
|
+
candidate: CandidateRecord,
|
|
79
|
+
config: RsiConfig,
|
|
80
|
+
run: RsiRunRecord,
|
|
81
|
+
trajectories: TrajectoryRecord[],
|
|
82
|
+
now: string,
|
|
83
|
+
hooks: RsiHooks,
|
|
84
|
+
): Promise<void> {
|
|
85
|
+
try {
|
|
86
|
+
const captured = await captureCandidateTrajectory(candidate, config, now)
|
|
87
|
+
candidate.trajectory = captured.summary
|
|
88
|
+
trajectories.push(captured.record)
|
|
89
|
+
run.trajectoryRefs ??= []
|
|
90
|
+
run.trajectoryRefs.push(captured.summary.path)
|
|
91
|
+
} catch (error) {
|
|
92
|
+
log(hooks, `[rsi] ${candidate.id}: trajectory capture failed: ${String(error)}`)
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export async function runRsi(config: RsiConfig, hooks: RsiHooks = {}): Promise<RsiRunRecord> {
|
|
97
|
+
const now = hooks.now ?? (() => new Date().toISOString())
|
|
98
|
+
const archive = await readArchive(config.archiveDir)
|
|
99
|
+
const roles = resolveRoles(config.roles, process.env, config.model)
|
|
100
|
+
const workerModel = roles.worker?.model ?? config.model
|
|
101
|
+
const resumed = config.resumeRunId ? await findActiveRun(config.archiveDir, config.resumeRunId) : undefined
|
|
102
|
+
if (config.resumeRunId && !resumed) throw new Error(`no active RSI run found for --resume ${config.resumeRunId}`)
|
|
103
|
+
const baseCommit = resumed?.baseCommit ?? (await resolveBaseCommit(config))
|
|
104
|
+
const run = resumed ?? initialRun({ ...config, model: workerModel, roles }, baseCommit, now())
|
|
105
|
+
run.generations = config.generations
|
|
106
|
+
run.model = workerModel
|
|
107
|
+
run.parentSelectionPolicy = config.parentSelectionPolicy ?? run.parentSelectionPolicy ?? "champion-specialist-novelty"
|
|
108
|
+
run.modelCandidates ??= [baseModelCandidate(config.model, now(), config.modelCandidateId ?? "model-base")]
|
|
109
|
+
if (config.modelCandidates?.length) {
|
|
110
|
+
run.modelCandidates = [...run.modelCandidates, ...config.modelCandidates.filter((model) => !run.modelCandidates!.some((entry) => entry.id === model.id))]
|
|
111
|
+
}
|
|
112
|
+
run.combinations ??= []
|
|
113
|
+
run.jobs ??= []
|
|
114
|
+
run.trajectoryRefs ??= []
|
|
115
|
+
run.curriculumTasks ??= []
|
|
116
|
+
run.parentChoices ??= {}
|
|
117
|
+
if (resumed) recoverInterruptedCandidates(run, now())
|
|
118
|
+
const runConfig: RsiConfig = { ...config, model: workerModel, roles, seed: `${config.seed}-${run.runId.slice(-8)}` }
|
|
119
|
+
const trajectories: TrajectoryRecord[] = []
|
|
120
|
+
|
|
121
|
+
if (!config.dryRun && !run.baseline) {
|
|
122
|
+
run.baseline = await (hooks.runCommand ?? runCommand)("npm test", config.repoRoot, config.commandTimeoutMs)
|
|
123
|
+
log(hooks, `[rsi] baseline: ${run.baseline.ok ? "pass" : "fail"}`)
|
|
124
|
+
await checkpointRun(config.archiveDir, run)
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
for (let generation = 0; generation < config.generations; generation++) {
|
|
128
|
+
let candidates = run.candidates.filter((candidate) => candidate.generation === generation)
|
|
129
|
+
if (candidates.length === 0) {
|
|
130
|
+
const merged = combinedArchive(archive, run)
|
|
131
|
+
const choices = generation === 0
|
|
132
|
+
? []
|
|
133
|
+
: selectParentChoices(merged, run.parentSelectionPolicy ?? "champion-specialist-novelty", config.population)
|
|
134
|
+
const fallbackBase = run.selected
|
|
135
|
+
? run.candidates.find((candidate) => candidate.id === run.selected)?.commits[0] ?? run.baseCommit
|
|
136
|
+
: run.baseCommit
|
|
137
|
+
candidates = Array.from({ length: config.population }, (_, index) => {
|
|
138
|
+
const parentChoice = choices[index]
|
|
139
|
+
const candidate = newCandidate(runConfig, generation, index, parentChoice?.baseCommit ?? fallbackBase, now(), parentChoice)
|
|
140
|
+
if (parentChoice) run.parentChoices![candidate.id] = parentChoice.reason
|
|
141
|
+
return candidate
|
|
142
|
+
})
|
|
143
|
+
run.candidates.push(...candidates)
|
|
144
|
+
log(hooks, `[rsi] generation ${generation}: planning ${candidates.length} candidate(s) with policy ${run.parentSelectionPolicy}`)
|
|
145
|
+
if (!config.dryRun) await checkpointRun(config.archiveDir, run)
|
|
146
|
+
} else {
|
|
147
|
+
log(hooks, `[rsi] generation ${generation}: resuming ${candidates.length} persisted candidate(s)`)
|
|
148
|
+
}
|
|
149
|
+
if (config.dryRun) continue
|
|
150
|
+
|
|
151
|
+
for (const candidate of candidates) {
|
|
152
|
+
if (isTerminal(candidate)) continue
|
|
153
|
+
let job = run.jobs.find((entry) => entry.candidateId === candidate.id && entry.kind === "mutation")
|
|
154
|
+
if (!job) {
|
|
155
|
+
job = newExperimentJob("mutation", { class: "LOCAL_GPU", units: 1, concurrencyKey: "ollama" }, now(), candidate.id)
|
|
156
|
+
run.jobs.push(job)
|
|
157
|
+
}
|
|
158
|
+
candidate.status = "mutating"
|
|
159
|
+
candidate.updatedAt = now()
|
|
160
|
+
job = transitionJob(job, "running", now())
|
|
161
|
+
replaceJob(run, job)
|
|
162
|
+
await checkpointRun(config.archiveDir, run)
|
|
163
|
+
let worktreeCreated = false
|
|
164
|
+
try {
|
|
165
|
+
await createCandidateWorktree(runConfig, candidate)
|
|
166
|
+
worktreeCreated = true
|
|
167
|
+
const mutation = await (hooks.runMutation ?? runMutation)(candidate, runConfig)
|
|
168
|
+
if (!mutation.ok) {
|
|
169
|
+
candidate.status = "failed"
|
|
170
|
+
candidate.failure = mutation.error
|
|
171
|
+
candidate.result = mutation.result
|
|
172
|
+
candidate.updatedAt = now()
|
|
173
|
+
job = transitionJob(job, "failed", now(), mutation.error)
|
|
174
|
+
replaceJob(run, job)
|
|
175
|
+
await captureCandidate(candidate, runConfig, run, trajectories, now(), hooks)
|
|
176
|
+
log(hooks, `[rsi] ${candidate.id}: mutation failed${mutation.error ? `: ${mutation.error}` : ""}`)
|
|
177
|
+
continue
|
|
178
|
+
}
|
|
179
|
+
candidate.status = "evaluating"
|
|
180
|
+
candidate.changedFiles = await changedFiles(candidate.worktree, candidate.baseCommit)
|
|
181
|
+
candidate.commits = await candidateCommits(candidate.worktree, candidate.baseCommit)
|
|
182
|
+
candidate.protectedPathViolations = protectedPathViolations(candidate.changedFiles, config.protectedPaths)
|
|
183
|
+
const evaluation = await evaluateCandidate(runConfig, candidate.worktree, candidate.baseCommit, hooks.runCommand ?? runCommand)
|
|
184
|
+
candidate.changedFiles = evaluation.changedFiles
|
|
185
|
+
candidate.commits = candidate.commits.length > 0 ? candidate.commits : await candidateCommits(candidate.worktree, candidate.baseCommit)
|
|
186
|
+
candidate.result = evaluation.regression
|
|
187
|
+
candidate.fitness = computeFitness({ ...evaluation, protectedPathViolation: candidate.protectedPathViolations.length > 0 })
|
|
188
|
+
candidate.status = candidate.fitness.score > 0 ? "accepted" : "rejected"
|
|
189
|
+
candidate.updatedAt = now()
|
|
190
|
+
const combination = createCombination(candidate.id, candidate.modelCandidateId ?? "model-base", now())
|
|
191
|
+
combination.status = candidate.status === "accepted" ? "accepted" : "rejected"
|
|
192
|
+
combination.fitness = candidate.fitness
|
|
193
|
+
combination.metrics = candidate.fitness.metrics
|
|
194
|
+
run.combinations.push(combination)
|
|
195
|
+
job = transitionJob(job, "completed", now())
|
|
196
|
+
replaceJob(run, job)
|
|
197
|
+
await captureCandidate(candidate, runConfig, run, trajectories, now(), hooks)
|
|
198
|
+
log(hooks, `[rsi] ${candidate.id}: ${candidate.status}, score=${candidate.fitness.score}`)
|
|
199
|
+
} catch (error) {
|
|
200
|
+
candidate.status = "failed"
|
|
201
|
+
candidate.failure = error instanceof Error ? error.message : String(error)
|
|
202
|
+
candidate.updatedAt = now()
|
|
203
|
+
job = transitionJob(job, "failed", now(), candidate.failure)
|
|
204
|
+
replaceJob(run, job)
|
|
205
|
+
log(hooks, `[rsi] ${candidate.id}: lifecycle failed: ${candidate.failure}`)
|
|
206
|
+
} finally {
|
|
207
|
+
if (worktreeCreated && !config.keepWorktrees) {
|
|
208
|
+
try {
|
|
209
|
+
await removeCandidateWorktree(runConfig, candidate)
|
|
210
|
+
} catch (error) {
|
|
211
|
+
log(hooks, `[rsi] ${candidate.id}: worktree cleanup failed: ${String(error)}`)
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
await checkpointRun(config.archiveDir, run)
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
const accepted = candidates.filter((candidate) => candidate.status === "accepted")
|
|
219
|
+
const front = paretoFront(accepted)
|
|
220
|
+
const ranked = [...accepted].sort((a, b) => compareFitness(b.fitness, a.fitness))
|
|
221
|
+
run.selectedCandidates = front.map((candidate) => candidate.id)
|
|
222
|
+
if (ranked[0]) run.selected = ranked[0].id
|
|
223
|
+
await checkpointRun(config.archiveDir, run)
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
const merged = combinedArchive(archive, run)
|
|
227
|
+
const curriculum = generateCurriculumProposals(merged, now())
|
|
228
|
+
run.curriculumTasks = curriculum
|
|
229
|
+
if (!config.dryRun && curriculum.length > 0) {
|
|
230
|
+
await writeCurriculumProposals(curriculum, config.curriculumDir ?? `${config.archiveDir}/curriculum`)
|
|
231
|
+
}
|
|
232
|
+
if (!config.dryRun && trajectories.length > 0) {
|
|
233
|
+
await exportTrajectoryDatasets(trajectories, `${config.trajectoryDir ?? `${config.archiveDir}/trajectories`}/datasets`)
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
run.finishedAt = now()
|
|
237
|
+
const report = formatRunReport(run, config)
|
|
238
|
+
const reportPath = `${config.archiveDir}/${run.runId}.md`
|
|
239
|
+
if (!config.dryRun) {
|
|
240
|
+
await fs.mkdir(config.archiveDir, { recursive: true })
|
|
241
|
+
await fs.writeFile(reportPath, report, "utf8")
|
|
242
|
+
run.reports = [...new Set([...run.reports, reportPath])]
|
|
243
|
+
await appendRun(config.archiveDir, run)
|
|
244
|
+
} else {
|
|
245
|
+
log(hooks, report.trimEnd())
|
|
246
|
+
}
|
|
247
|
+
return run
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
export async function improveMain(argv: string[]): Promise<number> {
|
|
251
|
+
const parsed = parseRsiArgs(argv)
|
|
252
|
+
if (parsed.help) {
|
|
253
|
+
process.stdout.write(rsiHelp())
|
|
254
|
+
return 0
|
|
255
|
+
}
|
|
256
|
+
if (parsed.error || !parsed.config) {
|
|
257
|
+
process.stderr.write(`headlesscode improve: ${parsed.error ?? "invalid configuration"}\n`)
|
|
258
|
+
return 2
|
|
259
|
+
}
|
|
260
|
+
try {
|
|
261
|
+
const run = await runRsi(parsed.config)
|
|
262
|
+
process.stdout.write(`[rsi] complete: ${run.runId}; selected=${run.selected ?? "none"}\n`)
|
|
263
|
+
return run.selected || parsed.config.dryRun ? 0 : 1
|
|
264
|
+
} catch (error) {
|
|
265
|
+
process.stderr.write(`headlesscode improve: ${error instanceof Error ? error.message : String(error)}\n`)
|
|
266
|
+
return 1
|
|
267
|
+
}
|
|
268
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import * as fs from "node:fs/promises"
|
|
2
|
+
import * as path from "node:path"
|
|
3
|
+
import type { CandidateRecord, CurriculumTask, RsiArchive } from "./types.js"
|
|
4
|
+
|
|
5
|
+
const FAILURE_TASKS: Record<string, { capability: string; difficulty: CurriculumTask["difficulty"]; template: string }> = {
|
|
6
|
+
"iteration-cap": {
|
|
7
|
+
capability: "completion-discipline",
|
|
8
|
+
difficulty: 3,
|
|
9
|
+
template: "Complete a bounded implementation task and prove it with a final verification command before the iteration budget expires.",
|
|
10
|
+
},
|
|
11
|
+
"regression-failure": {
|
|
12
|
+
capability: "regression-recovery",
|
|
13
|
+
difficulty: 3,
|
|
14
|
+
template: "Repair a deliberately failing change while preserving the existing regression suite and report the verified result.",
|
|
15
|
+
},
|
|
16
|
+
"timeout": {
|
|
17
|
+
capability: "tool-efficiency",
|
|
18
|
+
difficulty: 4,
|
|
19
|
+
template: "Solve a repository task under a strict command-time budget without repeating unproductive exploration.",
|
|
20
|
+
},
|
|
21
|
+
"evaluation-failure": {
|
|
22
|
+
capability: "generalization",
|
|
23
|
+
difficulty: 4,
|
|
24
|
+
template: "Make a focused repository improvement that passes the visible suite and an adjacent executable behavior check.",
|
|
25
|
+
},
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function candidateFailure(candidate: CandidateRecord): string | undefined {
|
|
29
|
+
if (candidate.failure?.includes("Max iterations")) return "iteration-cap"
|
|
30
|
+
if (candidate.result?.timedOut) return "timeout"
|
|
31
|
+
if (candidate.result && !candidate.result.ok) return "evaluation-failure"
|
|
32
|
+
return candidate.fitness?.hardGates.regressionPass === false ? "regression-failure" : undefined
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export function generateCurriculumProposals(archive: RsiArchive, now = new Date().toISOString()): CurriculumTask[] {
|
|
36
|
+
const grouped = new Map<string, CandidateRecord[]>()
|
|
37
|
+
for (const candidate of archive.candidates) {
|
|
38
|
+
const failure = candidateFailure(candidate)
|
|
39
|
+
if (!failure) continue
|
|
40
|
+
const entries = grouped.get(failure) ?? []
|
|
41
|
+
entries.push(candidate)
|
|
42
|
+
grouped.set(failure, entries)
|
|
43
|
+
}
|
|
44
|
+
return [...grouped.entries()].flatMap(([failureClass, candidates]) => {
|
|
45
|
+
const definition = FAILURE_TASKS[failureClass] ?? FAILURE_TASKS["evaluation-failure"]
|
|
46
|
+
return [{
|
|
47
|
+
id: `curriculum-${failureClass}`,
|
|
48
|
+
task: definition.template,
|
|
49
|
+
difficulty: definition.difficulty,
|
|
50
|
+
capability: definition.capability,
|
|
51
|
+
groundTruthCommand: "npm test",
|
|
52
|
+
provenance: { sourceCandidateIds: candidates.map((candidate) => candidate.id), failureClass, generatedAt: now },
|
|
53
|
+
validated: false,
|
|
54
|
+
}]
|
|
55
|
+
})
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export function validateCurriculumTask(task: CurriculumTask): boolean {
|
|
59
|
+
return task.task.trim() !== "" && task.groundTruthCommand.trim() !== "" && task.provenance.sourceCandidateIds.length > 0
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export async function writeCurriculumProposals(tasks: CurriculumTask[], outputDir: string): Promise<string> {
|
|
63
|
+
await fs.mkdir(outputDir, { recursive: true })
|
|
64
|
+
const validated = tasks.map((task) => ({ ...task, validated: validateCurriculumTask(task) }))
|
|
65
|
+
const outputPath = path.join(outputDir, "proposals.json")
|
|
66
|
+
await fs.writeFile(outputPath, `${JSON.stringify(validated, null, 2)}\n`, "utf8")
|
|
67
|
+
return outputPath
|
|
68
|
+
}
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import { exec as execCallback } from "node:child_process"
|
|
2
|
+
import { promisify } from "node:util"
|
|
3
|
+
import type { CommandRunner, EvaluationSummary, RsiConfig, TrialResult } from "./types.js"
|
|
4
|
+
import { changedFiles, candidateCommits, gitOutput } from "./workspace.js"
|
|
5
|
+
import { protectedPathViolations } from "./sandbox.js"
|
|
6
|
+
|
|
7
|
+
const exec = promisify(execCallback)
|
|
8
|
+
|
|
9
|
+
export const runCommand: CommandRunner = async (command, cwd, timeoutMs, env) => {
|
|
10
|
+
const started = Date.now()
|
|
11
|
+
try {
|
|
12
|
+
const result = await exec(command, { cwd, env, timeout: timeoutMs, maxBuffer: 8 * 1024 * 1024 })
|
|
13
|
+
return { ok: true, command, exitCode: 0, durationMs: Date.now() - started, stdout: result.stdout, stderr: result.stderr }
|
|
14
|
+
} catch (error) {
|
|
15
|
+
const failure = error as { code?: number | string; killed?: boolean; stdout?: string; stderr?: string }
|
|
16
|
+
return {
|
|
17
|
+
ok: false,
|
|
18
|
+
command,
|
|
19
|
+
exitCode: typeof failure.code === "number" ? failure.code : null,
|
|
20
|
+
durationMs: Date.now() - started,
|
|
21
|
+
stdout: failure.stdout ?? "",
|
|
22
|
+
stderr: failure.stderr ?? String(error),
|
|
23
|
+
timedOut: failure.killed === true,
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export async function evaluateCandidate(
|
|
29
|
+
config: RsiConfig,
|
|
30
|
+
candidateRoot: string,
|
|
31
|
+
baseCommit: string,
|
|
32
|
+
runner: CommandRunner = runCommand,
|
|
33
|
+
): Promise<EvaluationSummary> {
|
|
34
|
+
const files = await changedFiles(candidateRoot, baseCommit)
|
|
35
|
+
const commits = await candidateCommits(candidateRoot, baseCommit)
|
|
36
|
+
const complexity = await complexityMetrics(candidateRoot, baseCommit, files)
|
|
37
|
+
const violations = protectedPathViolations(files, config.protectedPaths)
|
|
38
|
+
// Refuse to execute a candidate that touched evaluator/scoring inputs. This
|
|
39
|
+
// check happens before any candidate-controlled test command runs.
|
|
40
|
+
if (violations.length > 0) {
|
|
41
|
+
const blocked: TrialResult = {
|
|
42
|
+
ok: false,
|
|
43
|
+
command: "protected-path-check",
|
|
44
|
+
exitCode: 1,
|
|
45
|
+
durationMs: 0,
|
|
46
|
+
stdout: "",
|
|
47
|
+
stderr: `protected paths changed: ${violations.join(", ")}`,
|
|
48
|
+
}
|
|
49
|
+
return {
|
|
50
|
+
regression: blocked,
|
|
51
|
+
visible: [],
|
|
52
|
+
hidden: [],
|
|
53
|
+
completed: false,
|
|
54
|
+
crashed: false,
|
|
55
|
+
protectedPathViolation: true,
|
|
56
|
+
changedFiles: files,
|
|
57
|
+
committed: commits.length > 0,
|
|
58
|
+
complexity,
|
|
59
|
+
failureClassification: "protected-path-violation",
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
const regression = await runner("npm test", candidateRoot, config.commandTimeoutMs)
|
|
63
|
+
const visible: TrialResult[] = []
|
|
64
|
+
for (const command of config.evalCommands) {
|
|
65
|
+
visible.push(await runner(command, candidateRoot, config.commandTimeoutMs))
|
|
66
|
+
}
|
|
67
|
+
const hidden: TrialResult[] = []
|
|
68
|
+
for (const command of config.hiddenEvalCommands) {
|
|
69
|
+
// Hidden commands belong to the supervisor checkout. They receive the
|
|
70
|
+
// candidate path explicitly so a candidate cannot replace the evaluator
|
|
71
|
+
// script or package metadata in the process that scores it.
|
|
72
|
+
hidden.push(
|
|
73
|
+
await runner(command, config.repoRoot, config.commandTimeoutMs, {
|
|
74
|
+
...process.env,
|
|
75
|
+
HEADLESSCODE_RSI_CANDIDATE_ROOT: candidateRoot,
|
|
76
|
+
}),
|
|
77
|
+
)
|
|
78
|
+
}
|
|
79
|
+
return {
|
|
80
|
+
regression,
|
|
81
|
+
visible,
|
|
82
|
+
hidden,
|
|
83
|
+
completed: regression.ok && visible.every((trial) => trial.ok) && hidden.every((trial) => trial.ok),
|
|
84
|
+
crashed: [regression, ...visible, ...hidden].some((trial) => trial.exitCode === null && !trial.timedOut),
|
|
85
|
+
protectedPathViolation: false,
|
|
86
|
+
changedFiles: files,
|
|
87
|
+
committed: commits.length > 0,
|
|
88
|
+
complexity,
|
|
89
|
+
failureClassification: [regression, ...visible, ...hidden].some((trial) => !trial.ok) ? "evaluation-failure" : undefined,
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
async function complexityMetrics(candidateRoot: string, baseCommit: string, files: string[]): Promise<NonNullable<EvaluationSummary["complexity"]>> {
|
|
94
|
+
try {
|
|
95
|
+
const numstat = await gitOutput(candidateRoot, ["diff", "--numstat", `${baseCommit}...HEAD`])
|
|
96
|
+
const diffLines = numstat.split("\n").reduce((total, line) => {
|
|
97
|
+
const [added, deleted] = line.split(/\s+/)
|
|
98
|
+
const a = Number(added)
|
|
99
|
+
const d = Number(deleted)
|
|
100
|
+
return total + (Number.isFinite(a) ? a : 0) + (Number.isFinite(d) ? d : 0)
|
|
101
|
+
}, 0)
|
|
102
|
+
return { diffLines, changedFiles: files.length, newDependencies: 0, additionalModelCalls: 0, runtimeOverheadMs: 0 }
|
|
103
|
+
} catch {
|
|
104
|
+
return { diffLines: 0, changedFiles: files.length, newDependencies: 0, additionalModelCalls: 0, runtimeOverheadMs: 0 }
|
|
105
|
+
}
|
|
106
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type { EvaluationSummary, Fitness, HardGates } from "./types.js"
|
|
2
|
+
|
|
3
|
+
function rate(passed: number, total: number): number {
|
|
4
|
+
return total === 0 ? 1 : passed / total
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
export function hardGatesFor(summary: EvaluationSummary): HardGates {
|
|
8
|
+
return {
|
|
9
|
+
regressionPass: summary.regression.ok,
|
|
10
|
+
visibleEvalPass: summary.visible.every((trial) => trial.ok),
|
|
11
|
+
hiddenEvalPass: summary.hidden.every((trial) => trial.ok),
|
|
12
|
+
noProtectedPathViolation: !summary.protectedPathViolation,
|
|
13
|
+
completed: summary.completed,
|
|
14
|
+
noCrash: !summary.crashed,
|
|
15
|
+
committed: summary.committed !== false,
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export function computeFitness(summary: EvaluationSummary): Fitness {
|
|
20
|
+
const hardGates = hardGatesFor(summary)
|
|
21
|
+
const visibleRate = rate(summary.visible.filter((trial) => trial.ok).length, summary.visible.length)
|
|
22
|
+
const hiddenRate = rate(summary.hidden.filter((trial) => trial.ok).length, summary.hidden.length)
|
|
23
|
+
const regression = hardGates.regressionPass ? 1 : 0
|
|
24
|
+
const efficiency = Math.max(0, 1 - Math.min(1, summary.visible.reduce((sum, trial) => sum + trial.durationMs, 0) / 600_000))
|
|
25
|
+
const recovery = summary.visible.length === 0 ? 0 : visibleRate
|
|
26
|
+
const complexity = summary.complexity ?? {
|
|
27
|
+
diffLines: 0,
|
|
28
|
+
changedFiles: summary.changedFiles.length,
|
|
29
|
+
newDependencies: 0,
|
|
30
|
+
additionalModelCalls: 0,
|
|
31
|
+
runtimeOverheadMs: 0,
|
|
32
|
+
}
|
|
33
|
+
const latency = [summary.regression, ...summary.visible, ...summary.hidden].reduce((sum, trial) => sum + trial.durationMs, 0)
|
|
34
|
+
const complexityPenalty = Math.min(1, complexity.diffLines / 20_000 + complexity.newDependencies / 10 + complexity.additionalModelCalls / 20)
|
|
35
|
+
const metrics = {
|
|
36
|
+
correctness: regression,
|
|
37
|
+
reliability: hardGates.noCrash ? 1 : 0,
|
|
38
|
+
generalization: visibleRate,
|
|
39
|
+
hidden: hiddenRate,
|
|
40
|
+
efficiency,
|
|
41
|
+
latency: latency,
|
|
42
|
+
tokenUse: 0,
|
|
43
|
+
recovery,
|
|
44
|
+
fabricationRate: summary.completed ? 0 : 1,
|
|
45
|
+
complexityPenalty,
|
|
46
|
+
}
|
|
47
|
+
const components = { regression, visible: visibleRate, hidden: hiddenRate, efficiency, recovery }
|
|
48
|
+
const score = Math.round(
|
|
49
|
+
(regression * 45 + visibleRate * 25 + hiddenRate * 20 + efficiency * 5 + recovery * 5 - complexityPenalty * 5) * 100,
|
|
50
|
+
) / 100
|
|
51
|
+
const failedGate = Object.entries(hardGates).find(([, passed]) => !passed)?.[0]
|
|
52
|
+
return {
|
|
53
|
+
score: failedGate ? 0 : score,
|
|
54
|
+
components,
|
|
55
|
+
metrics,
|
|
56
|
+
complexity,
|
|
57
|
+
hardGates,
|
|
58
|
+
reason: failedGate ? `hard gate failed: ${failedGate}` : "all hard gates passed",
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function compareFitness(left: Fitness | undefined, right: Fitness | undefined): number {
|
|
63
|
+
return (left?.score ?? 0) - (right?.score ?? 0)
|
|
64
|
+
}
|
package/src/rsi/index.ts
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export * from "./archive.js"
|
|
2
|
+
export * from "./config.js"
|
|
3
|
+
export * from "./controller.js"
|
|
4
|
+
export * from "./evaluator.js"
|
|
5
|
+
export * from "./fitness.js"
|
|
6
|
+
export * from "./mutation.js"
|
|
7
|
+
export * from "./models.js"
|
|
8
|
+
export * from "./reports.js"
|
|
9
|
+
export * from "./sandbox.js"
|
|
10
|
+
export * from "./selection.js"
|
|
11
|
+
export * from "./types.js"
|
|
12
|
+
export * from "./workspace.js"
|
|
13
|
+
export * from "./trajectory.js"
|
|
14
|
+
export * from "./curriculum.js"
|
|
15
|
+
export * from "./roles.js"
|
|
16
|
+
export * from "./search.js"
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
import { createHash } from "node:crypto"
|
|
2
|
+
import { exec as execCallback } from "node:child_process"
|
|
3
|
+
import * as fs from "node:fs/promises"
|
|
4
|
+
import { promisify } from "node:util"
|
|
5
|
+
import type {
|
|
6
|
+
HarnessModelCombination,
|
|
7
|
+
ModelCandidate,
|
|
8
|
+
TrainingConfig,
|
|
9
|
+
TrialResult,
|
|
10
|
+
} from "./types.js"
|
|
11
|
+
|
|
12
|
+
const exec = promisify(execCallback)
|
|
13
|
+
|
|
14
|
+
export function baseModelCandidate(model: string, now: string, id = "model-base"): ModelCandidate {
|
|
15
|
+
return {
|
|
16
|
+
id,
|
|
17
|
+
parentModel: model,
|
|
18
|
+
trainingMethod: "none",
|
|
19
|
+
trainingConfig: { method: "none" },
|
|
20
|
+
status: "base",
|
|
21
|
+
createdAt: now,
|
|
22
|
+
provenance: { source: "configured-worker-model" },
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export function modelCombinationId(harnessCandidateId: string, modelCandidateId: string): string {
|
|
27
|
+
return `${harnessCandidateId}::${modelCandidateId}`
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export function createCombination(
|
|
31
|
+
harnessCandidateId: string,
|
|
32
|
+
modelCandidateId: string,
|
|
33
|
+
now: string,
|
|
34
|
+
): HarnessModelCombination {
|
|
35
|
+
return {
|
|
36
|
+
id: modelCombinationId(harnessCandidateId, modelCandidateId),
|
|
37
|
+
harnessCandidateId,
|
|
38
|
+
modelCandidateId,
|
|
39
|
+
status: "planned",
|
|
40
|
+
createdAt: now,
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export function factorialCombinations(harnessIds: string[], modelIds: string[], now: string): HarnessModelCombination[] {
|
|
45
|
+
return harnessIds.flatMap((harnessId) => modelIds.map((modelId) => createCombination(harnessId, modelId, now)))
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export interface TrainingBackend {
|
|
49
|
+
prepareDataset(datasetVersion: string, outputDir: string): Promise<void>
|
|
50
|
+
train(config: TrainingConfig, outputDir: string): Promise<TrialResult>
|
|
51
|
+
inspectArtifact(artifactPath: string): Promise<{ hash: string; bytes: number }>
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export class ExternalTrainingBackend implements TrainingBackend {
|
|
55
|
+
constructor(private readonly timeoutMs = 60 * 60_000) {}
|
|
56
|
+
|
|
57
|
+
async prepareDataset(datasetVersion: string, outputDir: string): Promise<void> {
|
|
58
|
+
await fs.mkdir(outputDir, { recursive: true })
|
|
59
|
+
await fs.writeFile(`${outputDir}/dataset-version.txt`, `${datasetVersion}\n`, "utf8")
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
async train(config: TrainingConfig, outputDir: string): Promise<TrialResult> {
|
|
63
|
+
const command = config.command
|
|
64
|
+
if (!command) {
|
|
65
|
+
return { ok: false, command: "training-command", exitCode: 2, durationMs: 0, stdout: "", stderr: "no training command configured" }
|
|
66
|
+
}
|
|
67
|
+
const started = Date.now()
|
|
68
|
+
try {
|
|
69
|
+
const result = await exec(command, { cwd: outputDir, timeout: this.timeoutMs, maxBuffer: 16 * 1024 * 1024 })
|
|
70
|
+
return { ok: true, command, exitCode: 0, durationMs: Date.now() - started, stdout: result.stdout, stderr: result.stderr }
|
|
71
|
+
} catch (error) {
|
|
72
|
+
const failure = error as { code?: number; killed?: boolean; stdout?: string; stderr?: string }
|
|
73
|
+
return {
|
|
74
|
+
ok: false,
|
|
75
|
+
command,
|
|
76
|
+
exitCode: typeof failure.code === "number" ? failure.code : null,
|
|
77
|
+
durationMs: Date.now() - started,
|
|
78
|
+
stdout: failure.stdout ?? "",
|
|
79
|
+
stderr: failure.stderr ?? String(error),
|
|
80
|
+
timedOut: failure.killed === true,
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
async inspectArtifact(artifactPath: string): Promise<{ hash: string; bytes: number }> {
|
|
86
|
+
const bytes = await fs.readFile(artifactPath)
|
|
87
|
+
return { hash: createHash("sha256").update(bytes).digest("hex"), bytes: bytes.byteLength }
|
|
88
|
+
}
|
|
89
|
+
}
|