headlesscode 1.2.1 → 1.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,119 @@
1
+ /** Run one task in an OpenShell sandbox, preserving its Git worktree afterwards. */
2
+ import * as fs from "node:fs"
3
+ import * as path from "node:path"
4
+ import { spawnSync } from "node:child_process"
5
+ import { writeTaskFiles } from "../orchestrator/cli.js"
6
+ import type { WorktreeSpec } from "../orchestrator/split.js"
7
+ import { runOpenShellWorktreeCommand } from "./openshell-worker.js"
8
+ import { OPENSHELL_HARNESS_ROOT } from "./openshell-provider.js"
9
+
10
+ const USAGE = `headlesscode openshell-session --repo <path> --task <text> [options]
11
+
12
+ Options:
13
+ --repo <path> Git repository root (required)
14
+ --task <text> Task prompt (required)
15
+ --mode <slug> HeadlessCode mode (default: code)
16
+ --model <id> Optional model override
17
+ --name <name> Worktree name (default: openshell-<timestamp>)
18
+ --timeout-ms <n> Maximum task runtime (default: 1800000)
19
+ --image <image> OpenShell sandbox image
20
+ --policy <path> Reviewed OpenShell policy YAML (required)
21
+ --provider <name> OpenShell credential provider (repeatable; default: headlesscode-openrouter)
22
+ --cpu <quantity> CPU limit (default: 2)
23
+ --memory <quantity> Memory limit (default: 4Gi)
24
+ `
25
+
26
+ export async function openShellSessionMain(argv: string[]): Promise<number> {
27
+ const parsed = parseArgs(argv)
28
+ if (parsed.help) {
29
+ process.stdout.write(USAGE)
30
+ return 0
31
+ }
32
+ if (parsed.error) {
33
+ process.stderr.write(`headlesscode openshell-session: ${parsed.error}\n\n${USAGE}`)
34
+ return 2
35
+ }
36
+ const repo = path.resolve(parsed.repo!)
37
+ if (!fs.existsSync(path.join(repo, ".git"))) {
38
+ process.stderr.write(`headlesscode openshell-session: ${repo} is not a Git repository root\n`)
39
+ return 2
40
+ }
41
+ const policy = parsed.policy ?? process.env.HEADLESSCODE_OPENSHELL_POLICY
42
+ if (!policy) {
43
+ process.stderr.write("headlesscode openshell-session: pass --policy or set HEADLESSCODE_OPENSHELL_POLICY to a reviewed policy\n")
44
+ return 2
45
+ }
46
+ const providers = parsed.providerNames ?? (process.env.HEADLESSCODE_OPENSHELL_PROVIDERS ?? "headlesscode-openrouter").split(",").map((value) => value.trim()).filter(Boolean)
47
+ if (providers.includes("headlesscode-openrouter") && !process.env.HEADLESSCODE_OPENROUTER_API_KEY) {
48
+ process.stderr.write("headlesscode openshell-session: export HEADLESSCODE_OPENROUTER_API_KEY so OpenShell can store it in the headlesscode-openrouter provider\n")
49
+ return 2
50
+ }
51
+ const name = parsed.name ?? `openshell-${Date.now()}`
52
+ if (!/^[A-Za-z0-9._-]+$/.test(name) || name === "." || name === "..") {
53
+ process.stderr.write("headlesscode openshell-session: --name may contain only letters, numbers, dot, underscore, and hyphen\n")
54
+ return 2
55
+ }
56
+ const taskFile = `${name}.md`
57
+ const spec: WorktreeSpec = { name, issues: [0], taskFile }
58
+ const issue = { number: 0, title: "User task", body: parsed.task }
59
+ writeTaskFiles(repo, [spec], [issue])
60
+ const worktree = path.join(repo, ".worktrees", name)
61
+ const branch = `headlesscode/${name}-${Date.now()}`
62
+ const added = spawnSync("git", ["worktree", "add", "-b", branch, worktree, "HEAD"], { cwd: repo, encoding: "utf8" })
63
+ if (added.status !== 0) {
64
+ process.stderr.write(`headlesscode openshell-session: worktree creation failed: ${added.stderr || added.stdout}\n`)
65
+ return 1
66
+ }
67
+ fs.copyFileSync(path.join(repo, "plans", "parallel-tasks", taskFile), path.join(worktree, "ORCHESTRATOR_TASK.md"))
68
+ try {
69
+ const cliArgs = [
70
+ "--task-file", path.join(worktree, "ORCHESTRATOR_TASK.md"),
71
+ "--workspace", worktree,
72
+ "--mode", parsed.mode ?? "code",
73
+ ...(parsed.model ? ["--model", parsed.model] : []),
74
+ ]
75
+ const command = `cd ${quote(OPENSHELL_HARNESS_ROOT)} && timeout --signal=TERM --kill-after=10s ${Math.ceil((parsed.timeoutMs ?? 1_800_000) / 1000)}s ./node_modules/.bin/tsx src/cli.ts ${cliArgs.map(quote).join(" ")}`
76
+ const result = await runOpenShellWorktreeCommand(repo, worktree, command, {
77
+ ...(parsed.model ? { OPENROUTER_MODEL: parsed.model } : {}),
78
+ ORCHESTRATOR_MODE: parsed.mode ?? "code",
79
+ }, { image: parsed.image, policy, providers, cpu: parsed.cpu, memory: parsed.memory })
80
+ process.stdout.write(`OpenShell task finished: ${worktree}\nExit code: ${result.exitCode}\n${result.output}\n`)
81
+ return result.exitCode === 0 ? 0 : 1
82
+ } catch (error) {
83
+ process.stderr.write(`headlesscode openshell-session: ${error instanceof Error ? error.message : String(error)}\n`)
84
+ return 1
85
+ }
86
+ }
87
+
88
+ interface Args { repo?: string; task?: string; mode?: string; model?: string; name?: string; policy?: string; image?: string; providerNames?: string[]; cpu?: string; memory?: string; timeoutMs?: number; help?: boolean; error?: string }
89
+
90
+ function parseArgs(argv: string[]): Args {
91
+ const out: Args = {}
92
+ for (let i = 0; i < argv.length; i++) {
93
+ const arg = argv[i]
94
+ if (arg === "--help" || arg === "-h") out.help = true
95
+ else if (["--repo", "--task", "--mode", "--model", "--name", "--policy", "--image", "--timeout-ms", "--provider", "--cpu", "--memory"].includes(arg)) {
96
+ const value = argv[++i]
97
+ if (!value || value.startsWith("--")) return { error: `missing value for ${arg}` }
98
+ if (arg === "--repo") out.repo = value
99
+ else if (arg === "--task") out.task = value
100
+ else if (arg === "--mode") out.mode = value
101
+ else if (arg === "--model") out.model = value
102
+ else if (arg === "--name") out.name = value
103
+ else if (arg === "--policy") out.policy = value
104
+ else if (arg === "--image") out.image = value
105
+ else if (arg === "--provider") (out.providerNames ??= []).push(value)
106
+ else if (arg === "--cpu") out.cpu = value
107
+ else if (arg === "--memory") out.memory = value
108
+ else {
109
+ const timeoutMs = Number(value)
110
+ if (!Number.isSafeInteger(timeoutMs) || timeoutMs < 1000) return { error: "--timeout-ms must be an integer of at least 1000" }
111
+ out.timeoutMs = timeoutMs
112
+ }
113
+ } else return { error: `unknown option ${arg}` }
114
+ }
115
+ if (!out.help && (!out.repo || !out.task)) return { error: "--repo and --task are required" }
116
+ return out
117
+ }
118
+
119
+ function quote(value: string): string { return `'${value.replace(/'/g, `'\\''`)}'` }
@@ -0,0 +1,63 @@
1
+ import * as fs from "node:fs"
2
+ import * as os from "node:os"
3
+ import * as path from "node:path"
4
+ import { mainRepositoryForWorktree, runOpenShellWorktreeCommand } from "./openshell-worker.js"
5
+ import { OPENSHELL_HARNESS_ROOT } from "./openshell-provider.js"
6
+
7
+ const SESSION_FILES_ROOT = "/workspace/.headlesscode-session"
8
+
9
+ export async function runOpenShellSubsession<T>(
10
+ role: "review" | "qa",
11
+ workspaceRoot: string,
12
+ payload: Record<string, unknown>,
13
+ ): Promise<T> {
14
+ const workspace = path.resolve(workspaceRoot)
15
+ const repo = mainRepositoryForWorktree(workspace)
16
+ const sessionPayload = remapWorkspacePaths(payload, workspace)
17
+ let stagedPromptDir: string | undefined
18
+ if (role === "review" && typeof sessionPayload.reviewPromptPath === "string") {
19
+ // Copy only the selected prompt file into the disposable session clone.
20
+ stagedPromptDir = fs.mkdtempSync(path.join(os.tmpdir(), "headlesscode-openshell-review-"))
21
+ try {
22
+ fs.chmodSync(stagedPromptDir, 0o755)
23
+ const stagedPrompt = path.join(stagedPromptDir, "review-prompt.md")
24
+ fs.copyFileSync(String(sessionPayload.reviewPromptPath), stagedPrompt)
25
+ fs.chmodSync(stagedPrompt, 0o644)
26
+ sessionPayload.reviewPromptPath = `${SESSION_FILES_ROOT}/review-prompt.md`
27
+ } catch (error) {
28
+ fs.rmSync(stagedPromptDir, { recursive: true, force: true })
29
+ throw error
30
+ }
31
+ }
32
+ const encoded = Buffer.from(JSON.stringify(sessionPayload), "utf8").toString("base64url")
33
+ const quote = (value: string): string => `'${value.replaceAll("'", "'\\''")}'`
34
+ const command = `cd ${quote(OPENSHELL_HARNESS_ROOT)} && ./node_modules/.bin/tsx src/cli.ts openshell-${role}-session ${quote(encoded)}`
35
+ let cleanedUp = false
36
+ try {
37
+ const result = await runOpenShellWorktreeCommand(repo, workspace, command, {}, stagedPromptDir ? {
38
+ readOnlyMounts: [{ source: stagedPromptDir, target: SESSION_FILES_ROOT }],
39
+ importResults: false,
40
+ } : { importResults: false })
41
+ cleanedUp = true
42
+ const line = result.output.split(/\r?\n/).reverse().find((item) => item.startsWith("HEADLESSCODE_RESULT_JSON:"))
43
+ if (!line) throw new Error(`OpenShell ${role} session returned no result JSON (exit ${result.exitCode}): ${result.output}`)
44
+ return JSON.parse(line.slice("HEADLESSCODE_RESULT_JSON:".length)) as T
45
+ } finally {
46
+ // If sandbox teardown failed, retain the mounted file for recovery. A
47
+ // missing marker means validation failed before sandbox creation began.
48
+ if (stagedPromptDir && (cleanedUp || !fs.existsSync(path.join(workspace, ".harness.openshell-sandbox")))) {
49
+ fs.rmSync(stagedPromptDir, { recursive: true, force: true })
50
+ }
51
+ }
52
+ }
53
+
54
+ function remapWorkspacePaths(value: unknown, workspace: string): any {
55
+ if (typeof value === "string") {
56
+ if (value === workspace) return "/workspace"
57
+ if (value.startsWith(`${workspace}${path.sep}`)) return `/workspace${value.slice(workspace.length).split(path.sep).join("/")}`
58
+ return value
59
+ }
60
+ if (Array.isArray(value)) return value.map((item) => remapWorkspacePaths(item, workspace))
61
+ if (value && typeof value === "object") return Object.fromEntries(Object.entries(value).map(([key, item]) => [key, remapWorkspacePaths(item, workspace)]))
62
+ return value
63
+ }
@@ -0,0 +1,124 @@
1
+ import * as fs from "node:fs"
2
+ import * as path from "node:path"
3
+ import { execFileSync } from "node:child_process"
4
+
5
+ import { OPENSHELL_HARNESS_ROOT, OpenShellSessionProvider } from "./openshell-provider.js"
6
+ import type { OpenShellSessionProviderOptions } from "./openshell-provider.js"
7
+ import type { CloudSessionRequest, CommandResult } from "./provider.js"
8
+
9
+ const FORWARDED_ENV = [
10
+ "OPENROUTER_MODEL", "OPENROUTER_BASE_URL", "HEADLESSCODE_PROJECT", "HEADLESSCODE_MEMORY_DIR",
11
+ "HEADLESSCODE_MAX_COST_USD", "HEADLESSCODE_MAX_DURATION_MS", "HEADLESSCODE_MAX_ITERATIONS",
12
+ "HEADLESSCODE_WINDOW_SIZE", "HEADLESSCODE_DECISION_TIMEOUT_MS", "HEADLESSCODE_PRICING_JSON",
13
+ "HEADLESSCODE_CODE_MODE_BACKEND", "HEADLESSCODE_LOCAL_BACKEND_MODES", "HEADLESSCODE_CODE_MODE_MODEL",
14
+ ]
15
+
16
+ function forwardedEnvironment(env: Record<string, string>): Record<string, string> {
17
+ const forwarded: Record<string, string> = {}
18
+ for (const [key, value] of Object.entries({ ...process.env, ...env })) {
19
+ if (value !== undefined && (FORWARDED_ENV.includes(key) || key.startsWith("HEADLESSCODE_OLLAMA_") || key.startsWith("HEADLESSCODE_CODE_MODE_MODEL__"))) {
20
+ forwarded[key] = value
21
+ }
22
+ }
23
+ return forwarded
24
+ }
25
+
26
+ function hostGitIdentity(repo: string): Record<string, string> {
27
+ const identity = (variable: "GIT_AUTHOR_IDENT" | "GIT_COMMITTER_IDENT"): { name: string; email: string } | undefined => {
28
+ const result = execFileSync("git", ["var", variable], { cwd: repo, encoding: "utf8" }).trim()
29
+ const match = /^(.*) <([^<>]+)> \d+ [+-]\d{4}$/.exec(result)
30
+ return match ? { name: match[1], email: match[2] } : undefined
31
+ }
32
+ try {
33
+ const author = identity("GIT_AUTHOR_IDENT")
34
+ const committer = identity("GIT_COMMITTER_IDENT")
35
+ if (!author || !committer) return {}
36
+ return {
37
+ GIT_AUTHOR_NAME: author.name,
38
+ GIT_AUTHOR_EMAIL: author.email,
39
+ GIT_COMMITTER_NAME: committer.name,
40
+ GIT_COMMITTER_EMAIL: committer.email,
41
+ }
42
+ } catch {
43
+ // The host can still inspect changes, but Git commits inside the image
44
+ // require a configured host identity or explicit GIT_* identity env.
45
+ return {}
46
+ }
47
+ }
48
+
49
+ /** Run a command inside an OpenShell sandbox for an existing host worktree. */
50
+ export async function runOpenShellWorktreeCommand(
51
+ repo: string,
52
+ worktree: string,
53
+ command: string,
54
+ env: Record<string, string> = {},
55
+ providerOptions: OpenShellSessionProviderOptions = {},
56
+ ): Promise<CommandResult> {
57
+ const policy = providerOptions.policy ?? process.env.HEADLESSCODE_OPENSHELL_POLICY
58
+ if (!policy || !fs.existsSync(policy)) throw new Error("set HEADLESSCODE_OPENSHELL_POLICY to an existing base policy YAML")
59
+ const providers = providerOptions.providers ?? (process.env.HEADLESSCODE_OPENSHELL_PROVIDERS ?? "headlesscode-openrouter").split(",").map((value) => value.trim()).filter(Boolean)
60
+ if (providers.includes("headlesscode-openrouter") && !process.env.HEADLESSCODE_OPENROUTER_API_KEY) throw new Error("set HEADLESSCODE_OPENROUTER_API_KEY so OpenShell can provide the imported credential profile")
61
+ const worktreePath = path.resolve(worktree)
62
+ const repoPath = path.resolve(repo)
63
+ const name = path.basename(worktreePath)
64
+ const memoryDir = env.HEADLESSCODE_MEMORY_DIR ?? process.env.HEADLESSCODE_MEMORY_DIR
65
+ const memoryPath = memoryDir ? path.resolve(memoryDir) : undefined
66
+ if (memoryPath) fs.mkdirSync(memoryPath, { recursive: true })
67
+ const provider = new OpenShellSessionProvider({ ...providerOptions, policy, providers })
68
+ const request: CloudSessionRequest = {
69
+ repo: repoPath,
70
+ issue: { number: 0, title: `HeadlessCode worker ${name}` },
71
+ worktreeSpec: { name, issues: [], taskFile: "ORCHESTRATOR_TASK.md" },
72
+ env: {
73
+ ...forwardedEnvironment(env),
74
+ ...hostGitIdentity(repoPath),
75
+ ...env,
76
+ },
77
+ }
78
+ let handle
79
+ try {
80
+ handle = await provider.spawnExistingWorktreeSession(request, worktreePath)
81
+ await provider.waitReady(handle)
82
+ return await provider.runHarness(handle, command)
83
+ } finally {
84
+ if (handle) await provider.teardown(handle)
85
+ }
86
+ }
87
+
88
+ export function mainRepositoryForWorktree(worktree: string): string {
89
+ const root = path.resolve(worktree)
90
+ const commonDir = execFileSync("git", ["rev-parse", "--git-common-dir"], { cwd: root, encoding: "utf8" }).trim()
91
+ return path.dirname(path.resolve(root, commonDir))
92
+ }
93
+
94
+ /** Internal CLI command used by scripts/run-worker.sh. */
95
+ export async function openShellWorkerMain(argv: string[]): Promise<number> {
96
+ const separator = argv.indexOf("--")
97
+ const flags = separator < 0 ? argv : argv.slice(0, separator)
98
+ const workerArgs = separator < 0 ? [] : argv.slice(separator + 1)
99
+ let repo = ""
100
+ let worktree = ""
101
+ for (let i = 0; i < flags.length; i++) {
102
+ if (flags[i] === "--repo") repo = flags[++i] ?? ""
103
+ else if (flags[i] === "--worktree") worktree = flags[++i] ?? ""
104
+ else {
105
+ process.stderr.write(`headlesscode openshell-worker: unknown argument ${flags[i]}\n`)
106
+ return 2
107
+ }
108
+ }
109
+ if (!repo || !worktree || workerArgs.length === 0) {
110
+ process.stderr.write("Usage: headlesscode openshell-worker --repo <repo> --worktree <path> -- <worker CLI args...>\n")
111
+ return 2
112
+ }
113
+ const quote = (value: string): string => `'${value.replaceAll("'", "'\\''")}'`
114
+ const cli = process.env.HEADLESSCODE_SANDBOX_CLI ?? "node_modules/.bin/tsx src/cli.ts"
115
+ const command = `cd ${quote(OPENSHELL_HARNESS_ROOT)} && ${cli} ${workerArgs.map(quote).join(" ")}`
116
+ try {
117
+ const result = await runOpenShellWorktreeCommand(repo, worktree, command)
118
+ process.stdout.write(result.output)
119
+ return result.exitCode
120
+ } catch (error) {
121
+ process.stderr.write(`headlesscode openshell-worker: ${error instanceof Error ? error.message : String(error)}\n`)
122
+ return 1
123
+ }
124
+ }
@@ -37,6 +37,9 @@
37
37
  * → `.harness.inject-message` marker; see loop.ts's checkInjectedMessage.
38
38
  * The text has been appended to the session's live history as a plain
39
39
  * user-role message and the next LLM call sees it)
40
+ * iteration_intervention_queued / applied / abstained (optional typed
41
+ * iteration-boundary monitor; fixed-template requests only, with abstains
42
+ * recording malformed, duplicate, timed-out, or late requests)
40
43
  *
41
44
  * `condensed` is the one structural event that was originally only a log line
42
45
  * (`[condense] oldest turns condensed…` in src/engine/condense.ts) — the
@@ -98,6 +98,10 @@ import type {
98
98
  SessionResult,
99
99
  SessionBudgetUsage,
100
100
  SessionCompletionVerification,
101
+ IterationBoundaryHook,
102
+ IterationBoundarySnapshot,
103
+ IterationBoundaryIntervention,
104
+ IterationBoundaryToolRecord,
101
105
  ToolContext,
102
106
  ToolResult,
103
107
  } from "./types.js"
@@ -146,6 +150,14 @@ const PAUSED_FILENAME = ".harness.paused"
146
150
  */
147
151
  export const INJECT_MESSAGE_FILENAME = ".harness.inject-message"
148
152
 
153
+ /** Fixed, repository-owned monitor corrections. Monitor output never supplies text. */
154
+ const ITERATION_INTERVENTION_TEMPLATES: Record<string, string> = {
155
+ "specific-correction":
156
+ "Check the latest tool result. Re-read the affected configuration if necessary, identify the failed precondition, and change the next action before retrying.",
157
+ "generic-review": "Pause and review your approach before continuing.",
158
+ }
159
+ const DEFAULT_ITERATION_BOUNDARY_TIMEOUT_MS = 250
160
+
149
161
  /**
150
162
  * 2026-08-08: raised from 50 — real rounds against non-trivial issues
151
163
  * routinely needed MORE than 50 (exploration alone regularly ate 30-45 of
@@ -1356,6 +1368,10 @@ export interface HeadlessSessionConfig {
1356
1368
  * pattern as the live-usage-snapshot tests).
1357
1369
  */
1358
1370
  eventHook?: (eventType: string, fields: Record<string, unknown>) => void
1371
+ /** Optional awaited monitor invoked between complete iterations. */
1372
+ iterationBoundaryHook?: IterationBoundaryHook
1373
+ /** Fail-closed timeout for the optional iteration-boundary monitor. */
1374
+ iterationBoundaryTimeoutMs?: number
1359
1375
  /**
1360
1376
  * OPT-IN local exploration phase (default OFF — see
1361
1377
  * plans/local-explore-phase-experiment.md): when truthy, HeadlessSession runs a
@@ -1509,6 +1525,8 @@ export interface ResolvedSessionConfig {
1509
1525
  pausePollIntervalMs: number
1510
1526
  /** Live event hook (see HeadlessSessionConfig.eventHook). */
1511
1527
  eventHook?: (eventType: string, fields: Record<string, unknown>) => void
1528
+ iterationBoundaryHook?: IterationBoundaryHook
1529
+ iterationBoundaryTimeoutMs: number
1512
1530
  /**
1513
1531
  * Opt-in local exploration phase (see HeadlessSessionConfig.localExplore).
1514
1532
  * undefined/false = OFF (zero behavior change); options object = enabled
@@ -1718,6 +1736,8 @@ export class HeadlessSession {
1718
1736
  * when HEADLESSCODE_LAZY_TOOL_CATALOG is off (the default).
1719
1737
  */
1720
1738
  private lazyToolsByName: Map<string, ChatTool> = new Map()
1739
+ /** Intervention ids already applied during this run. */
1740
+ private readonly appliedIterationInterventionIds = new Set<string>()
1721
1741
 
1722
1742
  constructor(config: HeadlessSessionConfig) {
1723
1743
  if (!config.llmClient) {
@@ -1798,6 +1818,8 @@ export class HeadlessSession {
1798
1818
  maxPauseMs: config.maxPauseMs ?? envMaxPauseMs() ?? DEFAULT_MAX_PAUSE_MS,
1799
1819
  pausePollIntervalMs: config.pausePollIntervalMs ?? DEFAULT_PAUSE_POLL_INTERVAL_MS,
1800
1820
  eventHook: config.eventHook,
1821
+ iterationBoundaryHook: config.iterationBoundaryHook,
1822
+ iterationBoundaryTimeoutMs: config.iterationBoundaryTimeoutMs ?? DEFAULT_ITERATION_BOUNDARY_TIMEOUT_MS,
1801
1823
  localExplore: config.localExplore,
1802
1824
  }
1803
1825
 
@@ -2280,6 +2302,97 @@ export class HeadlessSession {
2280
2302
  }
2281
2303
  }
2282
2304
 
2305
+ /**
2306
+ * Observe one complete iteration and, if requested, append exactly one
2307
+ * repository-owned correction before the next model request. Runtime
2308
+ * validation is deliberate: callers may be JavaScript or otherwise bypass
2309
+ * TypeScript, and malformed monitor output must abstain rather than gain
2310
+ * authority over the session.
2311
+ */
2312
+ private async runIterationBoundaryHook(
2313
+ snapshot: IterationBoundarySnapshot,
2314
+ messages: ChatMessage[],
2315
+ ): Promise<void> {
2316
+ const hook = this.config.iterationBoundaryHook
2317
+ if (hook === undefined || this.sessionEnded) {
2318
+ return
2319
+ }
2320
+ let timedOut = false
2321
+ let timer: ReturnType<typeof setTimeout> | undefined
2322
+ try {
2323
+ const result = await Promise.race([
2324
+ Promise.resolve().then(() => hook(snapshot)),
2325
+ new Promise<undefined>((resolve) => {
2326
+ timer = setTimeout(() => {
2327
+ timedOut = true
2328
+ resolve(undefined)
2329
+ }, this.config.iterationBoundaryTimeoutMs)
2330
+ }),
2331
+ ])
2332
+ if (timedOut || result === undefined) {
2333
+ await this.emitEvent("iteration_intervention_abstained", () =>
2334
+ this.eventFeed.emit("iteration_intervention_abstained", {
2335
+ iteration: snapshot.iteration,
2336
+ reason: timedOut ? "timeout" : "missing",
2337
+ }), { iteration: snapshot.iteration })
2338
+ return
2339
+ }
2340
+ const candidate = result as Partial<IterationBoundaryIntervention> | null
2341
+ if (
2342
+ candidate === null ||
2343
+ typeof candidate !== "object" ||
2344
+ typeof candidate.id !== "string" ||
2345
+ candidate.id.trim() === "" ||
2346
+ typeof candidate.templateId !== "string" ||
2347
+ !Object.prototype.hasOwnProperty.call(ITERATION_INTERVENTION_TEMPLATES, candidate.templateId)
2348
+ ) {
2349
+ await this.emitEvent("iteration_intervention_abstained", () =>
2350
+ this.eventFeed.emit("iteration_intervention_abstained", {
2351
+ iteration: snapshot.iteration,
2352
+ reason: "malformed",
2353
+ }), { iteration: snapshot.iteration })
2354
+ return
2355
+ }
2356
+ const id = candidate.id.trim()
2357
+ const templateId = candidate.templateId as string
2358
+ await this.emitEvent("iteration_intervention_queued", () =>
2359
+ this.eventFeed.emit("iteration_intervention_queued", {
2360
+ iteration: snapshot.iteration,
2361
+ interventionId: id,
2362
+ templateId,
2363
+ }), { iteration: snapshot.iteration })
2364
+ if (this.sessionEnded || this.appliedIterationInterventionIds.has(id)) {
2365
+ await this.emitEvent("iteration_intervention_abstained", () =>
2366
+ this.eventFeed.emit("iteration_intervention_abstained", {
2367
+ iteration: snapshot.iteration,
2368
+ interventionId: id,
2369
+ reason: this.sessionEnded ? "session-ended" : "duplicate",
2370
+ }), { iteration: snapshot.iteration })
2371
+ return
2372
+ }
2373
+ this.appliedIterationInterventionIds.add(id)
2374
+ messages.push({ role: "user", content: ITERATION_INTERVENTION_TEMPLATES[templateId] })
2375
+ await this.emitEvent("iteration_intervention_applied", () =>
2376
+ this.eventFeed.emit("iteration_intervention_applied", {
2377
+ iteration: snapshot.iteration,
2378
+ interventionId: id,
2379
+ templateId,
2380
+ }), { iteration: snapshot.iteration })
2381
+ } catch (error) {
2382
+ await this.emitEvent("iteration_intervention_abstained", () =>
2383
+ this.eventFeed.emit("iteration_intervention_abstained", {
2384
+ iteration: snapshot.iteration,
2385
+ reason: "error",
2386
+ }), { iteration: snapshot.iteration })
2387
+ this.logger.warn("[monitor] iteration-boundary hook abstained", {
2388
+ iteration: snapshot.iteration,
2389
+ error: error instanceof Error ? error.message : String(error),
2390
+ })
2391
+ } finally {
2392
+ if (timer !== undefined) clearTimeout(timer)
2393
+ }
2394
+ }
2395
+
2283
2396
  /**
2284
2397
  * (S2) Fire-and-forget auxiliary write dispatcher with an ordering guard:
2285
2398
  * appends `fn` to a per-session promise chain so a later-scheduled write
@@ -3303,6 +3416,7 @@ export class HeadlessSession {
3303
3416
  let consecutiveMistakes = 0
3304
3417
  let lastCallSignature: string | undefined
3305
3418
  let toolCalls = 0
3419
+ let previousIterationTools: readonly IterationBoundaryToolRecord[] = []
3306
3420
  // Commit-before-finishing guardrail: whether the one-per-session
3307
3421
  // "uncommitted changes" nudge has already been given. A retried
3308
3422
  // attempt_completion is always accepted afterwards (see
@@ -3463,6 +3577,25 @@ export class HeadlessSession {
3463
3577
  // report the count at iteration start, not after.
3464
3578
  const historyMessageCount = messages.length
3465
3579
  this.scheduleAux(() => this.emitEvent("iteration_start", () => this.eventFeed.iterationStart(iteration, historyMessageCount)))
3580
+ if (iteration > 1) {
3581
+ const boundarySnapshot = Object.freeze({
3582
+ schemaVersion: 1,
3583
+ sessionId: this.sessionId,
3584
+ iteration,
3585
+ maxIterations,
3586
+ toolCalls: previousIterationTools,
3587
+ consecutiveMistakes,
3588
+ readOnlyStreak,
3589
+ identicalCallStreak,
3590
+ repeatedToolFailureStreak: toolFailureStreakCount,
3591
+ inputTokens: this.totalInputTokens,
3592
+ outputTokens: this.totalOutputTokens,
3593
+ cachedTokens: this.totalCachedTokens,
3594
+ totalToolCalls: toolCalls,
3595
+ reasoningAvailable: messages.some((message) => typeof message.reasoning === "string" && message.reasoning.length > 0),
3596
+ })
3597
+ await this.runIterationBoundaryHook(boundarySnapshot, messages)
3598
+ }
3466
3599
 
3467
3600
  // Pause/resume (dashboard control): block between iterations (never
3468
3601
  // mid-tool-call — same safe boundary as the budget check below)
@@ -4554,6 +4687,7 @@ export class HeadlessSession {
4554
4687
  // turn — the executor re-reads the file per call, so the failure
4555
4688
  // mechanism is stale SEARCH text, not a stale executor).
4556
4689
  const batchEditedPaths = new Map<string, string>()
4690
+ const boundaryTools: IterationBoundaryToolRecord[] = []
4557
4691
 
4558
4692
  // (S3) Partition this turn's calls: pure read-only/exploration calls
4559
4693
  // (read_file/list_files/search_files, the TS code-intel reads, and a
@@ -4769,6 +4903,14 @@ export class HeadlessSession {
4769
4903
  name: call.name,
4770
4904
  content: resultContent,
4771
4905
  })
4906
+ boundaryTools.push({
4907
+ id: call.id,
4908
+ name: call.name,
4909
+ arguments: Object.freeze({ ...call.args }),
4910
+ isError,
4911
+ result: resultContent,
4912
+ ...(isError ? { normalizedError: resultContent.split("\n", 1)[0]?.slice(0, 200) ?? "tool_error" } : {}),
4913
+ })
4772
4914
 
4773
4915
  // Record successful edits so a later sibling call on the same
4774
4916
  // path can be diagnosed (see above).
@@ -4877,6 +5019,7 @@ export class HeadlessSession {
4877
5019
  lastCallSignature = undefined
4878
5020
  }
4879
5021
  }
5022
+ previousIterationTools = Object.freeze(boundaryTools.map((record) => Object.freeze(record)))
4880
5023
 
4881
5024
  // Blind tree-walking guardrail (P1.6): after N consecutive
4882
5025
  // iterations whose every tool call is a plain read/exploration
@@ -151,6 +151,48 @@ export interface ToolResult {
151
151
  isError: boolean
152
152
  }
153
153
 
154
+ /** Fixed corrections available to an iteration-boundary monitor. */
155
+ export type IterationInterventionTemplateId = "specific-correction" | "generic-review"
156
+
157
+ /** One completed tool call and its result at an iteration boundary. */
158
+ export interface IterationBoundaryToolRecord {
159
+ readonly id: string
160
+ readonly name: string
161
+ readonly arguments: Readonly<Record<string, unknown>>
162
+ readonly isError: boolean
163
+ readonly result: string
164
+ readonly normalizedError?: string
165
+ }
166
+
167
+ /** Immutable, versioned observation passed to an optional monitor hook. */
168
+ export interface IterationBoundarySnapshot {
169
+ readonly schemaVersion: 1
170
+ readonly sessionId: string
171
+ readonly iteration: number
172
+ readonly maxIterations: number
173
+ readonly toolCalls: readonly IterationBoundaryToolRecord[]
174
+ readonly consecutiveMistakes: number
175
+ readonly readOnlyStreak: number
176
+ readonly identicalCallStreak: number
177
+ readonly repeatedToolFailureStreak: number
178
+ readonly inputTokens: number
179
+ readonly outputTokens: number
180
+ readonly cachedTokens: number
181
+ readonly totalToolCalls: number
182
+ readonly reasoningAvailable: boolean
183
+ }
184
+
185
+ /** Advisory response accepted from an iteration-boundary monitor. */
186
+ export interface IterationBoundaryIntervention {
187
+ readonly id: string
188
+ readonly templateId: IterationInterventionTemplateId
189
+ }
190
+
191
+ /** Optional monitor callback. It has no authority beyond requesting a fixed correction. */
192
+ export type IterationBoundaryHook = (
193
+ snapshot: IterationBoundarySnapshot,
194
+ ) => IterationBoundaryIntervention | null | undefined | Promise<IterationBoundaryIntervention | null | undefined>
195
+
154
196
  /**
155
197
  * Usage of one LLM call made OUTSIDE the main loop's request path — e.g. the
156
198
  * cloud vision captioning in src/vision/describe.ts (browser screenshots and
@@ -0,0 +1,73 @@
1
+ import type {
2
+ FrozenPredictorArtifact,
3
+ InterventionDecision,
4
+ IterationInterventionTemplateId,
5
+ PilotArm,
6
+ PilotBoundary,
7
+ Prediction,
8
+ } from "./types.js"
9
+ import { predict, predictedIntervention } from "./predictor.js"
10
+
11
+ export const SPECIFIC_CORRECTION =
12
+ "Check the latest tool result. Re-read the affected configuration if necessary, identify the failed precondition, and change the next action before retrying."
13
+ export const GENERIC_REVIEW = "Pause and review your approach before continuing."
14
+ export const TEMPLATE_TEXT: Readonly<Record<IterationInterventionTemplateId, string>> = {
15
+ "specific-correction": SPECIFIC_CORRECTION,
16
+ "generic-review": GENERIC_REVIEW,
17
+ }
18
+
19
+ export interface ControllerState {
20
+ readonly arm: PilotArm
21
+ readonly disarmed: boolean
22
+ readonly appliedInterventionIds: ReadonlySet<string>
23
+ }
24
+
25
+ export function initialControllerState(arm: PilotArm): ControllerState {
26
+ return { arm, disarmed: false, appliedInterventionIds: new Set() }
27
+ }
28
+
29
+ export function decide(
30
+ artifact: FrozenPredictorArtifact | undefined,
31
+ boundary: PilotBoundary,
32
+ state: ControllerState,
33
+ prediction?: Prediction,
34
+ ): { prediction?: Prediction; decision?: InterventionDecision; state: ControllerState } {
35
+ if (state.disarmed || state.arm === "B") return { prediction, state }
36
+ const nextPrediction = prediction ?? (artifact === undefined ? undefined : predict(artifact, boundary))
37
+ if (nextPrediction === undefined || artifact === undefined || !predictedIntervention(artifact, nextPrediction)) {
38
+ return { prediction: nextPrediction, state }
39
+ }
40
+ const templateId: IterationInterventionTemplateId = state.arm === "G" ? "generic-review" : "specific-correction"
41
+ const interventionId = `${boundary.attemptId}:${boundary.sequence}:${templateId}`
42
+ if (state.appliedInterventionIds.has(interventionId)) return { prediction: nextPrediction, state }
43
+ const applied = new Set(state.appliedInterventionIds)
44
+ applied.add(interventionId)
45
+ return {
46
+ prediction: nextPrediction,
47
+ decision: { interventionId, templateId, reason: "threshold" },
48
+ state: { ...state, disarmed: true, appliedInterventionIds: applied },
49
+ }
50
+ }
51
+
52
+ export function randomDecision(
53
+ boundary: PilotBoundary,
54
+ state: ControllerState,
55
+ probability: number,
56
+ random: () => number,
57
+ ): { decision?: InterventionDecision; state: ControllerState } {
58
+ if (state.disarmed || state.arm !== "R" || probability <= 0 || random() >= probability) return { state }
59
+ const interventionId = `${boundary.attemptId}:${boundary.sequence}:specific-correction`
60
+ if (state.appliedInterventionIds.has(interventionId)) return { state }
61
+ const applied = new Set(state.appliedInterventionIds)
62
+ applied.add(interventionId)
63
+ return {
64
+ decision: { interventionId, templateId: "specific-correction", reason: "random-schedule" },
65
+ state: { ...state, disarmed: true, appliedInterventionIds: applied },
66
+ }
67
+ }
68
+
69
+ export function applyFixedTemplate(messages: string[], decision: InterventionDecision): void {
70
+ const text = TEMPLATE_TEXT[decision.templateId]
71
+ if (typeof text !== "string") throw new Error("unknown intervention template")
72
+ messages.push(text)
73
+ }