headlesscode 1.2.1 → 1.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,103 @@
1
+ import { createHash } from "node:crypto"
2
+ import type { MonitorFeatureName, MonitorFeatureVector, PilotBoundary, PilotToolCall, PilotToolResult } from "./types.js"
3
+ import { MONITOR_FEATURE_ORDER } from "./types.js"
4
+
5
+ export function stableJson(value: unknown): string {
6
+ if (value === null || typeof value !== "object") return JSON.stringify(value)
7
+ if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`
8
+ const record = value as Record<string, unknown>
9
+ return `{${Object.keys(record)
10
+ .sort()
11
+ .map((key) => `${JSON.stringify(key)}:${stableJson(record[key])}`)
12
+ .join(",")}}`
13
+ }
14
+
15
+ export function sha256(value: unknown): string {
16
+ return createHash("sha256").update(typeof value === "string" ? value : stableJson(value)).digest("hex")
17
+ }
18
+
19
+ function isReadOnlyCall(call: PilotToolCall): boolean {
20
+ return call.name === "read_file" || call.name === "list_files" || call.name === "execute_command"
21
+ }
22
+
23
+ function isWriteCall(call: PilotToolCall): boolean {
24
+ return call.name === "apply_diff"
25
+ }
26
+
27
+ function callSignature(call: PilotToolCall): string {
28
+ return `${call.name}:${stableJson(call.args)}`
29
+ }
30
+
31
+ function resultSignature(result: PilotToolResult): string {
32
+ return `${result.name}:${result.isError}:${result.resultHash}`
33
+ }
34
+
35
+ function boundedFraction(numerator: number, denominator: number): number {
36
+ return denominator > 0 ? numerator / denominator : 0
37
+ }
38
+
39
+ /** Extract only current and recent action/result state. Oracle fields never enter here. */
40
+ export function extractMonitorFeatures(boundary: PilotBoundary): MonitorFeatureVector {
41
+ const turns = [...boundary.recentTurns].slice(-4)
42
+ const calls = turns.flatMap((turn) => turn.calls)
43
+ const results = turns.flatMap((turn) => turn.results)
44
+ const lastCall = boundary.calls.at(-1)
45
+ let identicalCallStreak = boundary.deterministicGuardrails.identicalCallStreak
46
+ if (lastCall !== undefined) {
47
+ identicalCallStreak = 0
48
+ for (const call of [...calls].reverse()) {
49
+ if (callSignature(call) !== callSignature(lastCall)) break
50
+ identicalCallStreak++
51
+ }
52
+ }
53
+ let sameToolErrorStreak = boundary.deterministicGuardrails.sameToolErrorStreak
54
+ const lastError = [...results].reverse().find((result) => result.isError)
55
+ if (lastError !== undefined) {
56
+ sameToolErrorStreak = 0
57
+ for (const result of [...results].reverse()) {
58
+ if (!result.isError || result.name !== lastError.name) break
59
+ sameToolErrorStreak++
60
+ }
61
+ }
62
+ const distinctCalls = new Set(calls.map(callSignature)).size
63
+ const errorCount = results.filter((result) => result.isError).length
64
+ const repeatedResults = results.length - new Set(results.map(resultSignature)).size
65
+ let turnsSinceSuccessfulVirtualWrite = 0
66
+ for (const turn of [...turns].reverse()) {
67
+ if (turn.successfulVirtualWrite) break
68
+ turnsSinceSuccessfulVirtualWrite++
69
+ }
70
+ const values: Record<MonitorFeatureName, number> = {
71
+ identicalCallStreak,
72
+ sameToolErrorStreak,
73
+ readOnlyStreak: boundary.deterministicGuardrails.readOnlyStreak,
74
+ uniqueCallFraction4: boundedFraction(distinctCalls, calls.length),
75
+ errorFraction4: boundedFraction(errorCount, results.length),
76
+ turnsSinceSuccessfulVirtualWrite,
77
+ repeatedResultFraction4: boundedFraction(repeatedResults, results.length),
78
+ iterationBudgetFraction: boundary.maxIterations > 0 ? boundary.iteration / boundary.maxIterations : 1,
79
+ }
80
+ return values
81
+ }
82
+
83
+ export function featureVector(features: MonitorFeatureVector): number[] {
84
+ return MONITOR_FEATURE_ORDER.map((name) => features[name])
85
+ }
86
+
87
+ export function featureHash(features: MonitorFeatureVector): string {
88
+ return sha256(featureVector(features))
89
+ }
90
+
91
+ export function assertFiniteFeatures(features: MonitorFeatureVector): void {
92
+ for (const name of MONITOR_FEATURE_ORDER) {
93
+ if (!Number.isFinite(features[name])) throw new Error(`non-finite feature: ${name}`)
94
+ }
95
+ }
96
+
97
+ export function currentTurnIsReadOnly(boundary: PilotBoundary): boolean {
98
+ return boundary.calls.length > 0 && boundary.calls.every(isReadOnlyCall)
99
+ }
100
+
101
+ export function currentTurnHasWrite(boundary: PilotBoundary): boolean {
102
+ return boundary.calls.some(isWriteCall) && boundary.results.some((result) => result.name === "apply_diff" && !result.isError)
103
+ }
@@ -0,0 +1,4 @@
1
+ export * from "./types.js"
2
+ export * from "./features.js"
3
+ export * from "./predictor.js"
4
+ export * from "./controller.js"
@@ -0,0 +1,246 @@
1
+ import type {
2
+ Calibration,
3
+ FrozenPredictorArtifact,
4
+ MonitorFeatureName,
5
+ MonitorFeatureVector,
6
+ PilotBoundary,
7
+ Prediction,
8
+ Standardization,
9
+ } from "./types.js"
10
+ import { MONITOR_FEATURE_ORDER } from "./types.js"
11
+ import { assertFiniteFeatures, extractMonitorFeatures, featureHash, sha256 } from "./features.js"
12
+
13
+ export interface LabeledWindow {
14
+ readonly features: MonitorFeatureVector
15
+ readonly label: 0 | 1
16
+ readonly trajectoryId: string
17
+ readonly boundarySequence?: number
18
+ }
19
+
20
+ export interface CalibrationPoint {
21
+ readonly score: number
22
+ readonly label: 0 | 1
23
+ }
24
+
25
+ export interface FitOptions {
26
+ readonly modelDigest: string
27
+ readonly templatesHash: string
28
+ readonly l2?: 1
29
+ }
30
+
31
+ function sigmoid(value: number): number {
32
+ if (value >= 0) {
33
+ const z = Math.exp(-value)
34
+ return 1 / (1 + z)
35
+ }
36
+ const z = Math.exp(value)
37
+ return z / (1 + z)
38
+ }
39
+
40
+ function emptyFeatureRecord(value: number): Record<MonitorFeatureName, number> {
41
+ return Object.fromEntries(MONITOR_FEATURE_ORDER.map((name) => [name, value])) as Record<MonitorFeatureName, number>
42
+ }
43
+
44
+ function standardize(rows: readonly LabeledWindow[]): Standardization {
45
+ const mean = emptyFeatureRecord(0)
46
+ const scale = emptyFeatureRecord(1)
47
+ for (const row of rows) {
48
+ for (const name of MONITOR_FEATURE_ORDER) mean[name] += row.features[name]
49
+ }
50
+ for (const name of MONITOR_FEATURE_ORDER) mean[name] /= Math.max(1, rows.length)
51
+ for (const row of rows) {
52
+ for (const name of MONITOR_FEATURE_ORDER) {
53
+ const delta = row.features[name] - mean[name]
54
+ scale[name] += delta * delta
55
+ }
56
+ }
57
+ for (const name of MONITOR_FEATURE_ORDER) scale[name] = Math.sqrt((scale[name] - 1) / Math.max(1, rows.length)) || 1
58
+ return { mean, scale }
59
+ }
60
+
61
+ function standardizedVector(features: MonitorFeatureVector, normalization: Standardization): number[] {
62
+ assertFiniteFeatures(features)
63
+ return MONITOR_FEATURE_ORDER.map((name) => (features[name] - normalization.mean[name]) / normalization.scale[name])
64
+ }
65
+
66
+ function fitLogistic(rows: readonly LabeledWindow[], normalization: Standardization): {
67
+ coefficients: Record<MonitorFeatureName, number>
68
+ intercept: number
69
+ } {
70
+ const coefficients = emptyFeatureRecord(0)
71
+ let intercept = 0
72
+ // Fixed optimizer settings are part of the frozen baseline, not tunable data.
73
+ const learningRate = 0.08
74
+ for (let step = 0; step < 2_000; step++) {
75
+ const gradient = emptyFeatureRecord(0)
76
+ let interceptGradient = 0
77
+ for (const row of rows) {
78
+ const x = standardizedVector(row.features, normalization)
79
+ let logit = intercept
80
+ for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) logit += coefficients[MONITOR_FEATURE_ORDER[i]] * x[i]
81
+ const error = sigmoid(logit) - row.label
82
+ interceptGradient += error
83
+ for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) gradient[MONITOR_FEATURE_ORDER[i]] += error * x[i]
84
+ }
85
+ const denominator = Math.max(1, rows.length)
86
+ intercept -= learningRate * (interceptGradient / denominator)
87
+ for (const name of MONITOR_FEATURE_ORDER) {
88
+ // Fixed L2 coefficient 1.0; intercept is not regularized.
89
+ coefficients[name] -= learningRate * (gradient[name] / denominator + coefficients[name] / denominator)
90
+ }
91
+ }
92
+ return { coefficients, intercept }
93
+ }
94
+
95
+ function rawScore(artifact: Omit<FrozenPredictorArtifact, "artifactHash">, features: MonitorFeatureVector): number {
96
+ const x = standardizedVector(features, artifact.standardization)
97
+ let logit = artifact.intercept
98
+ for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) logit += artifact.coefficients[MONITOR_FEATURE_ORDER[i]] * x[i]
99
+ return sigmoid(logit)
100
+ }
101
+
102
+ function fitCalibration(points: readonly CalibrationPoint[]): Calibration {
103
+ let intercept = 0
104
+ let slope = 1
105
+ for (let step = 0; step < 1_000; step++) {
106
+ let gi = 0
107
+ let gs = 0
108
+ for (const point of points) {
109
+ const logit = Math.log(Math.min(1 - 1e-6, Math.max(1e-6, point.score)) / Math.max(1e-6, 1 - point.score))
110
+ const probability = sigmoid(intercept + slope * logit)
111
+ const error = probability - point.label
112
+ gi += error
113
+ gs += error * logit
114
+ }
115
+ const denominator = Math.max(1, points.length)
116
+ intercept -= 0.05 * gi / denominator
117
+ slope -= 0.05 * gs / denominator
118
+ }
119
+ return { intercept, slope }
120
+ }
121
+
122
+ function calibratedScore(artifact: Omit<FrozenPredictorArtifact, "artifactHash">, features: MonitorFeatureVector): number {
123
+ const score = rawScore(artifact, features)
124
+ const logit = Math.log(Math.min(1 - 1e-6, Math.max(1e-6, score)) / Math.max(1e-6, 1 - score))
125
+ return sigmoid(artifact.calibration.intercept + artifact.calibration.slope * logit)
126
+ }
127
+
128
+ export function chooseThresholdFromScores(rows: readonly CalibrationPoint[]): number | undefined {
129
+ for (let step = 10; step <= 19; step++) {
130
+ const threshold = step / 20
131
+ const selected = rows.filter((row) => row.score >= threshold)
132
+ const truePositive = selected.filter((row) => row.label === 1).length
133
+ const falsePositive = selected.filter((row) => row.label === 0).length
134
+ const positives = rows.filter((row) => row.label === 1).length
135
+ const negatives = rows.filter((row) => row.label === 0).length
136
+ const precision = selected.length === 0 ? 1 : truePositive / selected.length
137
+ const falsePositiveRate = negatives === 0 ? 0 : falsePositive / negatives
138
+ if (precision >= 0.8 && falsePositiveRate <= 0.1 && positives >= 10) return threshold
139
+ }
140
+ return undefined
141
+ }
142
+
143
+ export function fitFrozenPredictor(
144
+ rows: readonly LabeledWindow[],
145
+ calibrationRows: readonly LabeledWindow[],
146
+ options: FitOptions,
147
+ ): FrozenPredictorArtifact {
148
+ if (rows.length === 0) throw new Error("cannot fit predictor without training rows")
149
+ const standardization = standardize(rows)
150
+ const model = fitLogistic(rows, standardization)
151
+ const provisional: Omit<FrozenPredictorArtifact, "artifactHash"> = {
152
+ schemaVersion: 1,
153
+ featureOrder: MONITOR_FEATURE_ORDER,
154
+ coefficients: model.coefficients,
155
+ intercept: model.intercept,
156
+ standardization,
157
+ calibration: fitCalibration(
158
+ calibrationRows.map((row) => ({
159
+ score: rawScore({
160
+ schemaVersion: 1,
161
+ featureOrder: MONITOR_FEATURE_ORDER,
162
+ coefficients: model.coefficients,
163
+ intercept: model.intercept,
164
+ standardization,
165
+ calibration: { intercept: 0, slope: 1 },
166
+ threshold: 0.5,
167
+ l2: 1,
168
+ modelDigest: options.modelDigest,
169
+ templatesHash: options.templatesHash,
170
+ }, row.features),
171
+ label: row.label,
172
+ })),
173
+ ),
174
+ threshold: 0.5,
175
+ l2: 1,
176
+ modelDigest: options.modelDigest,
177
+ templatesHash: options.templatesHash,
178
+ }
179
+ const calibrationScores = calibrationRows.map((row) => ({
180
+ score: calibratedScore(provisional, row.features),
181
+ label: row.label,
182
+ }))
183
+ const threshold = chooseThresholdFromScores(calibrationScores)
184
+ if (threshold === undefined) throw new Error("STOP: no qualifying calibration threshold")
185
+ const frozen = { ...provisional, threshold }
186
+ return { ...frozen, artifactHash: sha256(frozen) }
187
+ }
188
+
189
+ export function predict(artifact: FrozenPredictorArtifact, boundary: PilotBoundary): Prediction {
190
+ const started = process.hrtime.bigint()
191
+ try {
192
+ return predictFromFeatures(artifact, extractMonitorFeatures(boundary))
193
+ } catch (error) {
194
+ const elapsedNs = Number(process.hrtime.bigint() - started)
195
+ const features = emptyFeatureRecord(Number.NaN)
196
+ return {
197
+ probability: 0,
198
+ threshold: artifact.threshold,
199
+ features,
200
+ featureHash: sha256(features),
201
+ predictorHash: artifact.artifactHash,
202
+ latencyNs: elapsedNs,
203
+ abstentionReason: error instanceof Error && error.message.includes("finite")
204
+ ? "non-finite"
205
+ : error instanceof Error && error.message.includes("unknown model")
206
+ ? "unknown-model"
207
+ : "malformed",
208
+ }
209
+ }
210
+ }
211
+
212
+ export function predictFromFeatures(artifact: FrozenPredictorArtifact, features: MonitorFeatureVector): Prediction {
213
+ const started = process.hrtime.bigint()
214
+ try {
215
+ assertFiniteFeatures(features)
216
+ if (artifact.modelDigest.trim() === "") throw new Error("unknown model")
217
+ const provisional = artifact as Omit<FrozenPredictorArtifact, "artifactHash">
218
+ const probability = calibratedScore(provisional, features)
219
+ return {
220
+ probability,
221
+ threshold: artifact.threshold,
222
+ features,
223
+ featureHash: featureHash(features),
224
+ predictorHash: artifact.artifactHash,
225
+ latencyNs: Number(process.hrtime.bigint() - started),
226
+ }
227
+ } catch (error) {
228
+ return {
229
+ probability: 0,
230
+ threshold: artifact.threshold,
231
+ features,
232
+ featureHash: sha256(features),
233
+ predictorHash: artifact.artifactHash,
234
+ latencyNs: Number(process.hrtime.bigint() - started),
235
+ abstentionReason: error instanceof Error && error.message.includes("finite")
236
+ ? "non-finite"
237
+ : error instanceof Error && error.message.includes("unknown model")
238
+ ? "unknown-model"
239
+ : "malformed",
240
+ }
241
+ }
242
+ }
243
+
244
+ export function predictedIntervention(artifact: FrozenPredictorArtifact, prediction: Prediction): boolean {
245
+ return prediction.abstentionReason === undefined && prediction.probability >= artifact.threshold
246
+ }
@@ -0,0 +1,192 @@
1
+ import type { IterationInterventionTemplateId } from "../engine/types.js"
2
+
3
+ export { type IterationInterventionTemplateId }
4
+
5
+ export const MONITOR_FEATURE_ORDER = [
6
+ "identicalCallStreak",
7
+ "sameToolErrorStreak",
8
+ "readOnlyStreak",
9
+ "uniqueCallFraction4",
10
+ "errorFraction4",
11
+ "turnsSinceSuccessfulVirtualWrite",
12
+ "repeatedResultFraction4",
13
+ "iterationBudgetFraction",
14
+ ] as const
15
+
16
+ export type MonitorFeatureName = (typeof MONITOR_FEATURE_ORDER)[number]
17
+ export type MonitorFeatureVector = Record<MonitorFeatureName, number>
18
+
19
+ export type PilotArm = "B" | "M" | "R" | "G"
20
+ export type FixtureFamily = "wrong-key-path" | "revision-mismatch" | "failed-validation" | "legitimate-investigation"
21
+ export type FixtureSplit = "train" | "calibration" | "test"
22
+
23
+ export interface PilotFileSet {
24
+ readonly [relativePath: string]: string
25
+ }
26
+
27
+ export interface PilotFixture {
28
+ readonly id: string
29
+ readonly family: FixtureFamily
30
+ readonly split: FixtureSplit
31
+ readonly task: string
32
+ readonly initialFiles: PilotFileSet
33
+ readonly targetFiles: PilotFileSet
34
+ readonly unrelatedPaths: readonly string[]
35
+ readonly validator: (files: PilotFileSet) => { ok: boolean; message: string; milestone: number }
36
+ readonly oracle: (files: PilotFileSet) => { success: boolean; reason: string; milestone: number }
37
+ }
38
+
39
+ export interface PilotToolCall {
40
+ readonly id: string
41
+ readonly name: string
42
+ readonly args: Readonly<Record<string, unknown>>
43
+ }
44
+
45
+ export interface PilotToolResult {
46
+ readonly callId: string
47
+ readonly name: string
48
+ readonly isError: boolean
49
+ readonly content: string
50
+ readonly normalizedError?: string
51
+ readonly resultHash: string
52
+ }
53
+
54
+ export interface PilotTurnSummary {
55
+ readonly iteration: number
56
+ readonly calls: readonly PilotToolCall[]
57
+ readonly results: readonly PilotToolResult[]
58
+ readonly readOnly: boolean
59
+ readonly successfulVirtualWrite: boolean
60
+ readonly milestone: number
61
+ }
62
+
63
+ /** Complete supervisor-owned observation. It contains no oracle labels. */
64
+ export interface PilotBoundary {
65
+ readonly schemaVersion: 1
66
+ readonly experimentId: string
67
+ readonly runId: string
68
+ readonly attemptId: string
69
+ readonly epoch: number
70
+ readonly sequence: number
71
+ readonly iteration: number
72
+ readonly maxIterations: number
73
+ readonly calls: readonly PilotToolCall[]
74
+ readonly results: readonly PilotToolResult[]
75
+ readonly recentTurns: readonly PilotTurnSummary[]
76
+ readonly virtualRevisionHashes: Readonly<Record<string, string>>
77
+ readonly inputTokens: number
78
+ readonly outputTokens: number
79
+ readonly reasoningAvailable: boolean
80
+ readonly truncationFlags: readonly string[]
81
+ readonly deterministicGuardrails: Readonly<{
82
+ consecutiveMistakes: number
83
+ identicalCallStreak: number
84
+ readOnlyStreak: number
85
+ sameToolErrorStreak: number
86
+ repeatedToolFailureStreak: number
87
+ }>
88
+ }
89
+
90
+ export interface Standardization {
91
+ readonly mean: Record<MonitorFeatureName, number>
92
+ readonly scale: Record<MonitorFeatureName, number>
93
+ }
94
+
95
+ export interface Calibration {
96
+ readonly intercept: number
97
+ readonly slope: number
98
+ }
99
+
100
+ export interface FrozenPredictorArtifact {
101
+ readonly schemaVersion: 1
102
+ readonly featureOrder: readonly MonitorFeatureName[]
103
+ readonly coefficients: Record<MonitorFeatureName, number>
104
+ readonly intercept: number
105
+ readonly standardization: Standardization
106
+ readonly calibration: Calibration
107
+ readonly threshold: number
108
+ readonly l2: 1
109
+ readonly modelDigest: string
110
+ readonly templatesHash: string
111
+ readonly artifactHash: string
112
+ }
113
+
114
+ export interface Prediction {
115
+ readonly probability: number
116
+ readonly threshold: number
117
+ readonly features: MonitorFeatureVector
118
+ readonly featureHash: string
119
+ readonly predictorHash: string
120
+ readonly latencyNs: number
121
+ readonly abstentionReason?: "missing" | "non-finite" | "unknown-model" | "malformed"
122
+ }
123
+
124
+ export interface InterventionDecision {
125
+ readonly interventionId: string
126
+ readonly templateId: IterationInterventionTemplateId
127
+ readonly reason: "threshold" | "random-schedule"
128
+ }
129
+
130
+ export interface PilotInterventionRecord extends InterventionDecision {
131
+ readonly boundarySequence: number
132
+ readonly delivered: boolean
133
+ }
134
+
135
+ export interface PilotBudget {
136
+ readonly maxCalls: 16
137
+ readonly maxTokens: 20_000
138
+ readonly maxWallMs: number
139
+ readonly globalMaxTokens: number
140
+ readonly globalMaxWallMs: number
141
+ }
142
+
143
+ export interface PilotManifest {
144
+ readonly schemaVersion: 1
145
+ readonly experimentId: string
146
+ readonly createdAt: string
147
+ readonly modelDigest: string
148
+ readonly modelSettings: Readonly<Record<string, unknown>>
149
+ readonly fixtureHash: string
150
+ readonly splitHash: string
151
+ readonly templatesHash: string
152
+ readonly featureOrder: readonly MonitorFeatureName[]
153
+ readonly predictorHash?: string
154
+ readonly threshold?: number
155
+ readonly armAssignments: readonly PilotArm[]
156
+ readonly seeds: readonly number[]
157
+ readonly budget: PilotBudget
158
+ readonly randomProbability?: number
159
+ readonly randomBoundaryDistribution?: readonly Readonly<{ boundarySequence: number; weight: number }>[]
160
+ }
161
+
162
+ export interface PilotOutcome {
163
+ readonly experimentId: string
164
+ readonly runId: string
165
+ readonly attemptId: string
166
+ readonly epoch: number
167
+ readonly fixtureId: string
168
+ readonly split: FixtureSplit
169
+ readonly arm: PilotArm
170
+ readonly valid: boolean
171
+ readonly success: boolean
172
+ readonly finalArtifactHash: string
173
+ readonly oracleHash: string
174
+ readonly verifierHash: string
175
+ readonly milestoneCount: number
176
+ readonly redundantCalls: number
177
+ readonly invalidRetries: number
178
+ readonly forbiddenEffectAttempts: number
179
+ readonly experimentalInterventionAttempted: boolean
180
+ readonly correctionDelivered: boolean
181
+ readonly deterministicInterventionCount: number
182
+ readonly tokens: number
183
+ readonly calls: number
184
+ readonly elapsedMs: number
185
+ readonly stopReason: string
186
+ readonly predictionRecords: readonly Prediction[]
187
+ readonly interventions: readonly PilotInterventionRecord[]
188
+ readonly scheduledIntervention: boolean
189
+ readonly scheduledBoundarySequence?: number
190
+ /** Labels are retained only in the outcome analysis, never passed to prediction. */
191
+ readonly detectionLabels: readonly (0 | 1)[]
192
+ }
@@ -68,6 +68,7 @@ import {
68
68
  TRIVIAL_DRIFT_AHEAD,
69
69
  } from "./git-sync.js"
70
70
  import { resolveModelForMode } from "../config/mode-models.js"
71
+ import { openshellPreflight } from "../cloud/openshell-preflight.js"
71
72
  import { runPreflight, runLocalPreflight, type PreflightResult, type LocalPreflightResult } from "../llm/preflight.js"
72
73
  import { resolvePerModeEnv } from "../cli.js"
73
74
 
@@ -110,6 +111,7 @@ Options:
110
111
  --poll-interval-ms <n> Watcher poll interval (default: 5000)
111
112
  --memory-dir <path> Phase 3 memory dir for workers (passed as HEADLESSCODE_MEMORY_DIR,
112
113
  which run-worker.sh forwards as --memory-dir to each worker CLI)
114
+ --execution-provider <local|openshell> Worker runtime (default: local)
113
115
  --max-concurrent-sessions <n> Phase 6 GLOBAL cap on concurrent sessions across
114
116
  processes (default: $HEADLESSCODE_MAX_CONCURRENT_SESSIONS or 3).
115
117
  When the cap is already reached this run ABORTS with a clear
@@ -195,6 +197,7 @@ Environment:
195
197
  interface OrchestrateOptions {
196
198
  repo: string
197
199
  issues: number[]
200
+ executionProvider: "local" | "openshell"
198
201
  issuesJson?: string
199
202
  /**
200
203
  * Fix 3: file a REAL GitHub issue for every synthetic --issues-json entry
@@ -301,6 +304,7 @@ export function parseOrchestrateArgs(argv: string[]): { options: OrchestrateOpti
301
304
  const options: OrchestrateOptions = {
302
305
  repo: "",
303
306
  issues: [],
307
+ executionProvider: process.env.HEADLESSCODE_EXECUTION_PROVIDER === "openshell" ? "openshell" : "local",
304
308
  mode: process.env.ORCHESTRATOR_MODE ?? "code",
305
309
  reviewMode: "deepseek-reviewer",
306
310
  review: true,
@@ -341,6 +345,12 @@ export function parseOrchestrateArgs(argv: string[]): { options: OrchestrateOpti
341
345
  }
342
346
 
343
347
  switch (flag) {
348
+ case "--execution-provider": {
349
+ const v = next()
350
+ if (v !== "local" && v !== "openshell") return { options, error: "--execution-provider must be local or openshell" }
351
+ options.executionProvider = v
352
+ break
353
+ }
344
354
  case "--repo": {
345
355
  const v = next()
346
356
  if (v === undefined) {
@@ -1614,6 +1624,7 @@ export interface BuildSpawnEnvOptions {
1614
1624
  repo: string
1615
1625
  /** Harness mode for the workers (e.g. "code"). */
1616
1626
  mode: string
1627
+ executionProvider?: "local" | "openshell"
1617
1628
  /** An explicit --model flag value, if the caller passed one — always wins (resolveModelForMode rule 1). */
1618
1629
  explicitModel?: string
1619
1630
  memoryDir?: string
@@ -1649,6 +1660,7 @@ export function buildSpawnEnv(options: BuildSpawnEnvOptions): { env: NodeJS.Proc
1649
1660
  ...(options.env ?? process.env),
1650
1661
  TARGET_REPO: options.repo,
1651
1662
  ORCHESTRATOR_MODE: options.mode,
1663
+ HEADLESSCODE_EXECUTION_PROVIDER: options.executionProvider ?? process.env.HEADLESSCODE_EXECUTION_PROVIDER ?? "local",
1652
1664
  // The resolved worker model must reach the spawner (and the workers it
1653
1665
  // launches) — same pattern as HEADLESSCODE_PROJECT below.
1654
1666
  ...(workerModel ? { OPENROUTER_MODEL: workerModel } : {}),
@@ -2545,6 +2557,14 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
2545
2557
  process.stderr.write(`headlesscode orchestrate: not a git repo: ${repo}\n`)
2546
2558
  return 2
2547
2559
  }
2560
+ if (!options.dryRun && options.executionProvider === "openshell") {
2561
+ const issue = openshellPreflight()
2562
+ if (issue) {
2563
+ process.stderr.write(`headlesscode orchestrate: OpenShell preflight failed: ${issue}\n`)
2564
+ return 2
2565
+ }
2566
+ }
2567
+ process.env.HEADLESSCODE_EXECUTION_PROVIDER = options.executionProvider
2548
2568
 
2549
2569
  let issues: SplitIssue[]
2550
2570
  try {
@@ -2852,6 +2872,7 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
2852
2872
  const { env: spawnEnv, workerModel } = buildSpawnEnv({
2853
2873
  repo,
2854
2874
  mode: options.mode,
2875
+ executionProvider: options.executionProvider,
2855
2876
  explicitModel: options.model,
2856
2877
  memoryDir: options.memoryDir,
2857
2878
  maxIterations: options.maxIterations,
@@ -3032,6 +3053,7 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
3032
3053
  cwd: repo,
3033
3054
  env: {
3034
3055
  ...process.env,
3056
+ HEADLESSCODE_EXECUTION_PROVIDER: options.executionProvider,
3035
3057
  HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
3036
3058
  ...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
3037
3059
  },
@@ -3221,6 +3243,7 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
3221
3243
  cwd: repo,
3222
3244
  env: {
3223
3245
  ...process.env,
3246
+ HEADLESSCODE_EXECUTION_PROVIDER: options.executionProvider,
3224
3247
  HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
3225
3248
  ...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
3226
3249
  },
@@ -3376,6 +3399,7 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
3376
3399
  cwd: repo,
3377
3400
  env: {
3378
3401
  ...process.env,
3402
+ HEADLESSCODE_EXECUTION_PROVIDER: options.executionProvider,
3379
3403
  HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
3380
3404
  ...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
3381
3405
  },
@@ -36,6 +36,7 @@ import { getNativeTools } from "../vendor/zoo-code/src/core/prompts/tools/native
36
36
  import { addCustomInstructions } from "../vendor/zoo-code/src/core/prompts/sections/custom-instructions.js"
37
37
  import type { ChatTool, LlmClient, SessionResult } from "../engine/types.js"
38
38
  import type { SessionBudget } from "../budget/budget.js"
39
+ import { runOpenShellSubsession } from "../cloud/openshell-subsession.js"
39
40
 
40
41
  /** Default location of the reviewer checklist, relative to the harness repo. */
41
42
  export const DEFAULT_REVIEW_PROMPT_PATH = "shared/prompts/review-mode-prompt.md"
@@ -210,6 +211,16 @@ export async function runReview(options: ReviewOptions): Promise<ReviewResult> {
210
211
  maxIterations = 200,
211
212
  budget,
212
213
  } = options
214
+ if (process.env.HEADLESSCODE_EXECUTION_PROVIDER === "openshell" && !llmClient) {
215
+ try {
216
+ return await runOpenShellSubsession<ReviewResult>("review", workspaceRoot, {
217
+ workspaceRoot, mode, model, reviewPromptPath, taskText, issues, maxIterations, budget,
218
+ })
219
+ } catch (error) {
220
+ const message = error instanceof Error ? error.message : String(error)
221
+ return { findings: [`[review session error] ${message}`], verdict: "error", summary: `Review session failed: ${message}` }
222
+ }
223
+ }
213
224
 
214
225
  // 2026-08-27: verified live — a review session couldn't find the `curlee`
215
226
  // compiler at all ("curlee runtime is not available in this environment"),