headlesscode 1.0.3 → 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,77 @@
1
+ import { exec as execCallback } from "node:child_process"
2
+ import * as path from "node:path"
3
+ import { promisify } from "node:util"
4
+ import type { MutationRunner, RsiConfig, TrialResult, CandidateRecord } from "./types.js"
5
+ import { DEFAULT_RSI_MODEL } from "./config.js"
6
+
7
+ const exec = promisify(execCallback)
8
+
9
+ export function mutationPrompt(candidate: CandidateRecord, config: RsiConfig): string {
10
+ const hypothesis = candidate.hypothesis
11
+ return [
12
+ "You are the bounded RSI worker for headlesscode.",
13
+ `Improve this candidate checkout for generation ${candidate.generation}.`,
14
+ `Mutation class: ${candidate.mutationKind ?? "corrective"}. Compute policy: ${config.computePolicy ?? "single"}.`,
15
+ `Objective: ${config.mutationTask}`,
16
+ hypothesis
17
+ ? `Hypothesis: ${hypothesis.statement}\nExpected effect: ${hypothesis.expectedEffect}\nPotential downside: ${hypothesis.potentialDownside}${hypothesis.evidence ? `\nEvidence: ${hypothesis.evidence}` : ""}`
18
+ : "",
19
+ "You may change agent implementation, but you must not modify tests, evaluator/scoring/archive code, hidden evaluation commands, git metadata, or protected paths.",
20
+ "Inspect the existing repository first. Make a small, test-backed change. Run the relevant existing tests. Commit the candidate change before completing.",
21
+ ].join("\n\n")
22
+ }
23
+
24
+ export const runMutation: MutationRunner = async (candidate, config) => {
25
+ const started = Date.now()
26
+ const workerRole = config.roles?.worker
27
+ const workerModel = workerRole?.model ?? config.model
28
+ const workerProvider = workerRole?.provider ?? "ollama"
29
+ const env: NodeJS.ProcessEnv = {
30
+ ...process.env,
31
+ HEADLESSCODE_OPENROUTER_API_KEY: process.env.HEADLESSCODE_OPENROUTER_API_KEY ?? "rsi-local-placeholder",
32
+ HEADLESSCODE_CODE_MODE_BACKEND: workerProvider === "ollama" ? "ollama" : "openrouter",
33
+ HEADLESSCODE_LOCAL_BACKEND_MODES: "code",
34
+ HEADLESSCODE_CODE_MODE_MODEL: workerModel || DEFAULT_RSI_MODEL,
35
+ HEADLESSCODE_OLLAMA_THINK: "0",
36
+ HEADLESSCODE_CAPTURE_TRANSCRIPT_DIR: path.join(candidate.worktree, ".headlesscode", "rsi-transcripts"),
37
+ }
38
+ const command = [
39
+ "npx tsx src/cli.ts",
40
+ "--mode code",
41
+ `--max-iterations ${config.maxIterations}`,
42
+ "--no-checkpoints",
43
+ `--task ${JSON.stringify(mutationPrompt(candidate, config))}`,
44
+ ].join(" ")
45
+ try {
46
+ const result = await exec(command, {
47
+ cwd: candidate.worktree,
48
+ env,
49
+ timeout: config.commandTimeoutMs,
50
+ maxBuffer: 16 * 1024 * 1024,
51
+ })
52
+ const trial: TrialResult = {
53
+ ok: true,
54
+ command,
55
+ exitCode: 0,
56
+ durationMs: Date.now() - started,
57
+ stdout: result.stdout,
58
+ stderr: result.stderr,
59
+ }
60
+ return { ok: true, result: trial }
61
+ } catch (error) {
62
+ const failure = error as { code?: number; killed?: boolean; stdout?: string; stderr?: string }
63
+ return {
64
+ ok: false,
65
+ result: {
66
+ ok: false,
67
+ command,
68
+ exitCode: typeof failure.code === "number" ? failure.code : null,
69
+ durationMs: Date.now() - started,
70
+ stdout: failure.stdout ?? "",
71
+ stderr: failure.stderr ?? String(error),
72
+ timedOut: failure.killed === true,
73
+ },
74
+ error: error instanceof Error ? error.message : String(error),
75
+ }
76
+ }
77
+ }
@@ -0,0 +1,47 @@
1
+ import type { CandidateRecord, Fitness, RsiConfig, RsiRunRecord } from "./types.js"
2
+
3
+ export function formatCandidate(candidate: CandidateRecord): string {
4
+ const fitness = candidate.fitness
5
+ const parent = candidate.parentSelection ? `, parent=${candidate.parent} (${candidate.parentSelection.strategy})` : ""
6
+ const mutation = candidate.mutationKind ? `, mutation=${candidate.mutationKind}` : ""
7
+ const metrics = fitness ? `, metrics=${JSON.stringify(fitness.metrics)}` : ""
8
+ return `${candidate.id}: ${candidate.status}${parent}${mutation}${fitness ? `, score=${fitness.score}, ${fitness.reason}` : ""}${metrics}`
9
+ }
10
+
11
+ export function formatRunReport(run: RsiRunRecord, config: RsiConfig): string {
12
+ const lines = [
13
+ `# RSI Generation Report: ${run.runId}`,
14
+ "",
15
+ `- Model: ${run.model}`,
16
+ `- Base commit: ${run.baseCommit}`,
17
+ `- Generations: ${run.generations}`,
18
+ `- Started: ${run.startedAt}`,
19
+ `- Finished: ${run.finishedAt ?? "in progress"}`,
20
+ `- Baseline: ${run.baseline ? (run.baseline.ok ? "pass" : "fail") : "not run (dry-run)"}`,
21
+ `- Selected candidate: ${run.selected ?? "none"}`,
22
+ `- Pareto candidates: ${run.selectedCandidates?.join(", ") || "none"}`,
23
+ `- Parent policy: ${run.parentSelectionPolicy ?? config.parentSelectionPolicy ?? "champion-specialist-novelty"}`,
24
+ `- Model candidates: ${run.modelCandidates?.map((model) => `${model.id} (${model.status})`).join(", ") || "none"}`,
25
+ `- Experiment jobs: ${run.jobs?.length ?? 0}`,
26
+ `- Trajectories: ${run.trajectoryRefs?.length ?? 0}`,
27
+ "",
28
+ "## Candidates",
29
+ "",
30
+ ...run.candidates.map((candidate) => `- ${formatCandidate(candidate)}${candidate.failure ? `; failure=${candidate.failure}` : ""}`),
31
+ "",
32
+ "## Guardrails",
33
+ "",
34
+ `- Visible evaluations: ${config.evalCommands.join("; ") || "none"}`,
35
+ `- Hidden evaluations: ${config.hiddenEvalCommands.length || 0}`,
36
+ `- Compute policy: ${config.computePolicy ?? "single"}`,
37
+ `- Protected paths: ${config.protectedPaths.join(", ")}`,
38
+ "",
39
+ "The archive is the durable machine-readable record for this run.",
40
+ ]
41
+ return `${lines.join("\n")}\n`
42
+ }
43
+
44
+ export function fitnessSummary(fitness: Fitness | undefined): string {
45
+ if (!fitness) return "not evaluated"
46
+ return `${fitness.score} (${fitness.reason})`
47
+ }
@@ -0,0 +1,37 @@
1
+ import { DEFAULT_RSI_MODEL } from "./config.js"
2
+ import type { ModelRole, RsiRoleConfig, RoleModelConfig } from "./types.js"
3
+
4
+ const ROLE_ENV_NAMES: ModelRole[] = [
5
+ "worker",
6
+ "mutation-architect",
7
+ "failure-analyst",
8
+ "critic",
9
+ "adversary",
10
+ "reviewer",
11
+ "curriculum-designer",
12
+ "training-data-curator",
13
+ ]
14
+
15
+ function envKey(role: ModelRole): string {
16
+ return `HEADLESSCODE_RSI_ROLE_${role.toUpperCase().replace(/-/g, "_")}_MODEL`
17
+ }
18
+
19
+ export function defaultRoles(workerModel = DEFAULT_RSI_MODEL): RsiRoleConfig {
20
+ return { worker: { provider: "ollama", model: workerModel } }
21
+ }
22
+
23
+ export function resolveRoles(
24
+ configured: RsiRoleConfig | undefined,
25
+ env: NodeJS.ProcessEnv = process.env,
26
+ workerModel = DEFAULT_RSI_MODEL,
27
+ ): RsiRoleConfig {
28
+ const roles: RsiRoleConfig = { ...defaultRoles(workerModel), ...configured }
29
+ for (const role of ROLE_ENV_NAMES) {
30
+ const model = env[envKey(role)]
31
+ if (model) {
32
+ const existing = roles[role]
33
+ roles[role] = { ...(existing ?? { provider: "ollama" }), model } as RoleModelConfig
34
+ }
35
+ }
36
+ return roles
37
+ }
@@ -0,0 +1,10 @@
1
+ import { findMatchingPattern } from "../permissions/protected-files.js"
2
+
3
+ export function protectedPathViolations(changedFiles: string[], patterns: string[]): string[] {
4
+ return changedFiles.filter((file) => findMatchingPattern(file, patterns) !== null)
5
+ }
6
+
7
+ export function protectedPathReport(changedFiles: string[], patterns: string[]): string {
8
+ const violations = protectedPathViolations(changedFiles, patterns)
9
+ return violations.length === 0 ? "none" : violations.join(", ")
10
+ }
@@ -0,0 +1,32 @@
1
+ import type { ComputePolicy } from "./types.js"
2
+
3
+ export interface DifficultySignals {
4
+ taskLength: number
5
+ priorFailures: number
6
+ uncertainty: number
7
+ repeatedFailure: boolean
8
+ }
9
+
10
+ export interface SearchPlan {
11
+ policy: ComputePolicy
12
+ attempts: number
13
+ criticAfterFailure: boolean
14
+ reason: string
15
+ }
16
+
17
+ export function chooseComputePolicy(signals: DifficultySignals): SearchPlan {
18
+ if (signals.repeatedFailure || signals.priorFailures >= 2) {
19
+ return { policy: "critic-retry", attempts: 2, criticAfterFailure: true, reason: "repeated failure warrants diagnosis before another attempt" }
20
+ }
21
+ if (signals.taskLength >= 1200 || signals.uncertainty >= 0.75) {
22
+ return { policy: "planner-executors", attempts: 3, criticAfterFailure: false, reason: "long or uncertain task needs a plan plus independent implementations" }
23
+ }
24
+ if (signals.taskLength >= 500 || signals.uncertainty >= 0.4) {
25
+ return { policy: "independent", attempts: 2, criticAfterFailure: false, reason: "moderate difficulty merits independent trajectories" }
26
+ }
27
+ return { policy: "single", attempts: 1, criticAfterFailure: false, reason: "easy task does not justify extra inference" }
28
+ }
29
+
30
+ export function boundedSearchBudget(plan: SearchPlan, baseIterations: number): number {
31
+ return Math.max(1, Math.floor(baseIterations * plan.attempts))
32
+ }
@@ -0,0 +1,132 @@
1
+ import { randomUUID } from "node:crypto"
2
+ import type {
3
+ CandidateRecord,
4
+ MetricVector,
5
+ ParentSelectionPolicy,
6
+ ParentSelectionReason,
7
+ RsiArchive,
8
+ ExperimentJob,
9
+ ExperimentJobKind,
10
+ ExperimentJobStatus,
11
+ ResourceRequirements,
12
+ } from "./types.js"
13
+
14
+ const SPECIALIST_METRICS: Array<keyof MetricVector> = ["correctness", "hidden", "efficiency", "recovery", "reliability"]
15
+
16
+ function metrics(candidate: CandidateRecord): MetricVector {
17
+ return (
18
+ candidate.fitness?.metrics ?? {
19
+ correctness: candidate.fitness?.components.regression ?? 0,
20
+ reliability: candidate.fitness?.components.recovery ?? 0,
21
+ generalization: candidate.fitness?.components.visible ?? 0,
22
+ hidden: candidate.fitness?.components.hidden ?? 0,
23
+ efficiency: candidate.fitness?.components.efficiency ?? 0,
24
+ latency: 0,
25
+ tokenUse: 0,
26
+ recovery: candidate.fitness?.components.recovery ?? 0,
27
+ fabricationRate: 1,
28
+ complexityPenalty: 0,
29
+ }
30
+ )
31
+ }
32
+
33
+ function eligible(candidates: CandidateRecord[]): CandidateRecord[] {
34
+ return candidates.filter((candidate) => candidate.fitness && candidate.fitness.hardGates.committed !== false && candidate.commits[0])
35
+ }
36
+
37
+ export interface ParentChoice {
38
+ candidateId: string
39
+ baseCommit: string
40
+ reason: ParentSelectionReason
41
+ }
42
+
43
+ export function paretoDominates(left: MetricVector, right: MetricVector): boolean {
44
+ let strictlyBetter = false
45
+ for (const key of Object.keys(left) as Array<keyof MetricVector>) {
46
+ const leftValue = key === "fabricationRate" || key === "complexityPenalty" || key === "latency" || key === "tokenUse" ? -left[key] : left[key]
47
+ const rightValue = key === "fabricationRate" || key === "complexityPenalty" || key === "latency" || key === "tokenUse" ? -right[key] : right[key]
48
+ if (leftValue < rightValue) return false
49
+ if (leftValue > rightValue) strictlyBetter = true
50
+ }
51
+ return strictlyBetter
52
+ }
53
+
54
+ export function paretoFront(candidates: CandidateRecord[]): CandidateRecord[] {
55
+ const pool = eligible(candidates)
56
+ return pool.filter((candidate) => !pool.some((other) => other !== candidate && paretoDominates(metrics(other), metrics(candidate))))
57
+ }
58
+
59
+ function byScore(a: CandidateRecord, b: CandidateRecord): number {
60
+ return (b.fitness?.score ?? 0) - (a.fitness?.score ?? 0)
61
+ }
62
+
63
+ function choice(candidate: CandidateRecord, strategy: ParentSelectionReason["strategy"], reason: string): ParentChoice {
64
+ return {
65
+ candidateId: candidate.id,
66
+ baseCommit: candidate.commits[0] ?? candidate.baseCommit,
67
+ reason: { strategy, reason, metrics: metrics(candidate) },
68
+ }
69
+ }
70
+
71
+ export function selectParentChoices(
72
+ archive: RsiArchive,
73
+ policy: ParentSelectionPolicy,
74
+ count: number,
75
+ ): ParentChoice[] {
76
+ const pool = eligible(archive.candidates).sort(byScore)
77
+ if (pool.length === 0 || count <= 0) return []
78
+ if (policy === "all-eligible") {
79
+ return Array.from({ length: count }, (_, index) => choice(pool[index % pool.length], "archive", "eligible archived candidate"))
80
+ }
81
+ if (policy === "pareto-front") {
82
+ const front = paretoFront(pool).sort(byScore)
83
+ return Array.from({ length: count }, (_, index) => choice(front[index % front.length], "pareto", "candidate retained on the multi-objective Pareto front"))
84
+ }
85
+
86
+ const selected: ParentChoice[] = []
87
+ const champion = pool[0]
88
+ selected.push(choice(champion, "champion", "highest hard-gated scalar fitness"))
89
+ for (const metric of SPECIALIST_METRICS) {
90
+ if (selected.length >= count) break
91
+ const specialist = [...pool].sort((a, b) => metrics(b)[metric] - metrics(a)[metric])[0]
92
+ if (specialist && !selected.some((entry) => entry.candidateId === specialist.id)) {
93
+ selected.push(choice(specialist, "specialist", `best archived candidate on ${metric}`))
94
+ }
95
+ }
96
+ for (const candidate of pool) {
97
+ if (selected.length >= count) break
98
+ if (!selected.some((entry) => entry.candidateId === candidate.id)) {
99
+ selected.push(choice(candidate, "novelty", "next eligible candidate preserves population diversity"))
100
+ }
101
+ }
102
+ return Array.from({ length: count }, (_, index) => selected[index % selected.length])
103
+ }
104
+
105
+ export function newExperimentJob(
106
+ kind: ExperimentJobKind,
107
+ resource: ResourceRequirements,
108
+ now: string,
109
+ candidateId?: string,
110
+ ): ExperimentJob {
111
+ return {
112
+ id: `job-${kind}-${randomUUID().slice(0, 8)}`,
113
+ kind,
114
+ status: "queued",
115
+ resource,
116
+ candidateId,
117
+ createdAt: now,
118
+ updatedAt: now,
119
+ attempts: 0,
120
+ }
121
+ }
122
+
123
+ export function transitionJob(job: ExperimentJob, status: ExperimentJobStatus, now: string, error?: string): ExperimentJob {
124
+ if (job.status === "completed" || job.status === "cancelled") return job
125
+ return {
126
+ ...job,
127
+ status,
128
+ updatedAt: now,
129
+ attempts: status === "running" ? job.attempts + 1 : job.attempts,
130
+ ...(error ? { error } : {}),
131
+ }
132
+ }
@@ -0,0 +1,143 @@
1
+ import { createHash } from "node:crypto"
2
+ import * as fs from "node:fs/promises"
3
+ import * as path from "node:path"
4
+ import type { CandidateRecord, RsiConfig, TrajectoryMessage, TrajectoryRecord, TrajectorySummary } from "./types.js"
5
+
6
+ interface CapturedCall {
7
+ messages?: TrajectoryMessage[]
8
+ response?: { message?: { tool_calls?: unknown[] } }
9
+ error?: string
10
+ provider?: string
11
+ model?: string
12
+ }
13
+
14
+ function failureClass(candidate: CandidateRecord): string | undefined {
15
+ if (candidate.failure) return candidate.failure.includes("Max iterations") ? "iteration-cap" : "mutation-failure"
16
+ if (candidate.result && !candidate.result.ok) {
17
+ const text = `${candidate.result.stderr}\n${candidate.result.stdout}`.toLowerCase()
18
+ if (candidate.result.timedOut) return "timeout"
19
+ if (text.includes("test") || text.includes("typecheck")) return "regression-failure"
20
+ return "evaluation-failure"
21
+ }
22
+ return undefined
23
+ }
24
+
25
+ async function readCapturedCalls(candidate: CandidateRecord): Promise<CapturedCall[]> {
26
+ const directory = path.join(candidate.worktree, ".headlesscode", "rsi-transcripts")
27
+ let names: string[]
28
+ try {
29
+ names = (await fs.readdir(directory)).filter((name) => name.endsWith(".jsonl")).sort()
30
+ } catch {
31
+ return []
32
+ }
33
+ const calls: CapturedCall[] = []
34
+ for (const name of names) {
35
+ const lines = (await fs.readFile(path.join(directory, name), "utf8")).split("\n").filter(Boolean)
36
+ for (const line of lines) {
37
+ try {
38
+ calls.push(JSON.parse(line) as CapturedCall)
39
+ } catch {
40
+ // A torn JSONL line is marked incomplete by the absence of a call.
41
+ }
42
+ }
43
+ }
44
+ return calls
45
+ }
46
+
47
+ export function trajectoryOutcome(candidate: CandidateRecord): TrajectoryRecord["outcome"] {
48
+ if (candidate.status === "accepted" && candidate.fitness?.hardGates.completed) return "success"
49
+ if (candidate.status === "failed" || candidate.status === "rejected") return "failure"
50
+ return "incomplete"
51
+ }
52
+
53
+ export async function captureCandidateTrajectory(
54
+ candidate: CandidateRecord,
55
+ config: RsiConfig,
56
+ now: string,
57
+ ): Promise<{ record: TrajectoryRecord; summary: TrajectorySummary }> {
58
+ const calls = await readCapturedCalls(candidate)
59
+ const messages = calls.flatMap((call) => call.messages ?? [])
60
+ const toolCalls = calls.reduce((count, call) => count + (call.response?.message?.tool_calls?.length ?? 0), 0)
61
+ const outcome = trajectoryOutcome(candidate)
62
+ const trusted = outcome === "success" && candidate.fitness?.hardGates.noProtectedPathViolation === true
63
+ const record: TrajectoryRecord = {
64
+ id: `trajectory-${candidate.id}`,
65
+ task: candidate.mutation,
66
+ environment: { repoRoot: config.repoRoot, baseCommit: candidate.baseCommit, generation: candidate.generation },
67
+ model: { id: candidate.model, candidateId: candidate.modelCandidateId },
68
+ harnessVersion: candidate.parentCommit ?? candidate.baseCommit,
69
+ promptConfig: {
70
+ mutationKind: candidate.mutationKind,
71
+ hypothesis: candidate.hypothesis,
72
+ parentSelection: candidate.parentSelection,
73
+ },
74
+ messages,
75
+ toolCalls,
76
+ outcome,
77
+ verification: {
78
+ verified: trusted,
79
+ regressionPass: candidate.fitness?.hardGates.regressionPass === true,
80
+ hiddenPass: candidate.fitness?.hardGates.hiddenEvalPass === true,
81
+ },
82
+ failureClassification: failureClass(candidate),
83
+ fitnessImpact: candidate.fitness?.score,
84
+ provenance: { source: "headlesscode-rsi", capturedAt: now, trusted },
85
+ }
86
+ const outputDir = config.trajectoryDir ?? path.join(config.archiveDir, "trajectories")
87
+ await fs.mkdir(outputDir, { recursive: true })
88
+ const outputPath = path.join(outputDir, `${candidate.id}.json`)
89
+ await fs.writeFile(outputPath, `${JSON.stringify(record, null, 2)}\n`, "utf8")
90
+ return {
91
+ record,
92
+ summary: {
93
+ path: outputPath,
94
+ messageCount: messages.length,
95
+ toolCalls,
96
+ trusted,
97
+ outcome,
98
+ failureClassification: record.failureClassification,
99
+ },
100
+ }
101
+ }
102
+
103
+ export async function readTrajectoryRecord(filePath: string): Promise<TrajectoryRecord | undefined> {
104
+ try {
105
+ return JSON.parse(await fs.readFile(filePath, "utf8")) as TrajectoryRecord
106
+ } catch {
107
+ return undefined
108
+ }
109
+ }
110
+
111
+ function jsonl(records: unknown[]): string {
112
+ return records.map((record) => JSON.stringify(record)).join("\n") + (records.length ? "\n" : "")
113
+ }
114
+
115
+ export async function exportTrajectoryDatasets(records: TrajectoryRecord[], outputDir: string): Promise<{
116
+ sft: number
117
+ preferences: number
118
+ failures: number
119
+ manifestPath: string
120
+ }> {
121
+ await fs.mkdir(outputDir, { recursive: true })
122
+ const trusted = records.filter((record) => record.provenance.trusted && record.outcome === "success")
123
+ const failures = records.filter((record) => record.outcome === "failure")
124
+ const preferences: Array<{ task: string; chosen: TrajectoryRecord; rejected: TrajectoryRecord }> = []
125
+ for (const task of new Set(records.map((record) => record.task))) {
126
+ const successful = records.find((record) => record.task === task && record.provenance.trusted && record.outcome === "success")
127
+ const failed = records.find((record) => record.task === task && record.outcome === "failure")
128
+ if (successful && failed) preferences.push({ task, chosen: successful, rejected: failed })
129
+ }
130
+ await fs.writeFile(path.join(outputDir, "sft.jsonl"), jsonl(trusted), "utf8")
131
+ await fs.writeFile(path.join(outputDir, "preferences.jsonl"), jsonl(preferences), "utf8")
132
+ await fs.writeFile(path.join(outputDir, "failures.jsonl"), jsonl(failures), "utf8")
133
+ const manifest = {
134
+ schemaVersion: 1,
135
+ createdAt: new Date().toISOString(),
136
+ counts: { sft: trusted.length, preferences: preferences.length, failures: failures.length },
137
+ trustedOnly: true,
138
+ sha256: createHash("sha256").update(JSON.stringify(records)).digest("hex"),
139
+ }
140
+ const manifestPath = path.join(outputDir, "manifest.json")
141
+ await fs.writeFile(manifestPath, `${JSON.stringify(manifest, null, 2)}\n`, "utf8")
142
+ return { ...manifest.counts, manifestPath }
143
+ }