headlesscode 1.0.2 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -56
- package/package.json +1 -1
- package/src/cli.ts +115 -4
- package/src/engine/claims.ts +503 -0
- package/src/engine/events.ts +32 -0
- package/src/engine/lazy-tools.ts +30 -9
- package/src/engine/loop.ts +462 -7
- package/src/engine/prompt.ts +5 -1
- package/src/engine/types.ts +36 -0
- package/src/rsi/archive.ts +129 -0
- package/src/rsi/config.ts +312 -0
- package/src/rsi/controller.ts +268 -0
- package/src/rsi/curriculum.ts +68 -0
- package/src/rsi/evaluator.ts +106 -0
- package/src/rsi/fitness.ts +64 -0
- package/src/rsi/index.ts +16 -0
- package/src/rsi/models.ts +89 -0
- package/src/rsi/mutation.ts +77 -0
- package/src/rsi/reports.ts +47 -0
- package/src/rsi/roles.ts +37 -0
- package/src/rsi/sandbox.ts +10 -0
- package/src/rsi/search.ts +32 -0
- package/src/rsi/selection.ts +132 -0
- package/src/rsi/trajectory.ts +143 -0
- package/src/rsi/types.ts +317 -0
- package/src/rsi/workspace.ts +96 -0
- package/src/tools/executor.ts +232 -7
package/README.md
CHANGED
|
@@ -1,66 +1,63 @@
|
|
|
1
1
|
# headlesscode
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
- **
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
>
|
|
39
|
-
>
|
|
40
|
-
>
|
|
41
|
-
>
|
|
42
|
-
>
|
|
43
|
-
>
|
|
44
|
-
>
|
|
45
|
-
|
|
46
|
-
## Purpose
|
|
47
|
-
|
|
48
|
-
Drive a coding agent headlessly against real git repos: read a target repo's
|
|
49
|
-
`.roomodes` and `.roo/rules-<slug>/` files, build a system prompt from mode +
|
|
50
|
-
rules, call an LLM API (OpenRouter in Phase 1) with the tool schema, execute
|
|
51
|
-
tool calls via plain `fs`/`child_process` (no `vscode.*` anywhere), and loop
|
|
52
|
-
until completion. Non-interactive by design.
|
|
3
|
+
[](https://github.com/Capsize-Games/headlesscode/actions/workflows/ci.yml)
|
|
4
|
+
[](https://www.npmjs.com/package/headlesscode)
|
|
5
|
+
[](https://nodejs.org/)
|
|
6
|
+
[](./LICENSE)
|
|
7
|
+
|
|
8
|
+
Run a coding agent where the work actually happens: in a terminal, a CI job,
|
|
9
|
+
or an isolated git worktree. `headlesscode` brings the useful parts of
|
|
10
|
+
[Zoo Code](https://github.com/Zoo-Code-Org/Zoo-Code) to a plain Node.js process,
|
|
11
|
+
without requiring a VS Code window.
|
|
12
|
+
|
|
13
|
+
It is made for real repositories and real engineering loops. The harness reads
|
|
14
|
+
the target project's modes and rules, gives the model a controlled tool set,
|
|
15
|
+
keeps work separated in git worktrees, and leaves behind logs and state that a
|
|
16
|
+
person can inspect.
|
|
17
|
+
|
|
18
|
+
## Why use it?
|
|
19
|
+
|
|
20
|
+
- **Run repeatable repository tasks.** Give it a task or GitHub issue and let
|
|
21
|
+
it work through files, commands, and tests from a non-interactive process.
|
|
22
|
+
- **Keep parallel work organized.** Split issues into worker groups, run them
|
|
23
|
+
in separate worktrees, review the results, and optionally run QA before a
|
|
24
|
+
human-approved deploy.
|
|
25
|
+
- **Remember the project.** Opt-in local memory stores project facts and
|
|
26
|
+
rolling session summaries, with a pluggable storage boundary for a future
|
|
27
|
+
remote backend.
|
|
28
|
+
- **Bound the expensive parts.** Per-session cost, duration, iteration, and
|
|
29
|
+
fleet-concurrency limits are built into the orchestration and watcher paths.
|
|
30
|
+
- **Experiment with improvement.** The `improve` command runs a bounded
|
|
31
|
+
recursive self-improvement loop against an external evaluator, with the
|
|
32
|
+
default worker using a local Qwen 3.5 9B model through Ollama.
|
|
33
|
+
|
|
34
|
+
The project also includes a GitHub issue watcher and an evaluation-only cloud
|
|
35
|
+
provider interface. No live cloud resources are launched by the current
|
|
36
|
+
implementation.
|
|
37
|
+
|
|
38
|
+
> **Security warning: default-allow arbitrary command execution.**
|
|
39
|
+
> By default, `headlesscode` runs arbitrary shell commands with the invoking
|
|
40
|
+
> user's privileges. It can read and modify files, including credentials such
|
|
41
|
+
> as `~/.ssh` and `~/.aws`, without approval prompts. Use it only with trusted
|
|
42
|
+
> tasks and isolate it with a container, VM, or dedicated user when untrusted
|
|
43
|
+
> content is involved. The optional permissions layer is defense in depth, not
|
|
44
|
+
> a security boundary. See [`SECURITY.md`](./SECURITY.md).
|
|
53
45
|
|
|
54
46
|
## Quick start
|
|
55
47
|
|
|
56
|
-
|
|
57
|
-
checkout:
|
|
48
|
+
Install from npm:
|
|
58
49
|
|
|
59
50
|
```bash
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
51
|
+
npm install -g headlesscode
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Or run it without installing, via npx:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
npx headlesscode --task "Fix the bug in src/index.ts" --workspace /path/to/target/repo
|
|
58
|
+
```
|
|
63
59
|
|
|
60
|
+
```bash
|
|
64
61
|
# Required (except for --dry-run):
|
|
65
62
|
export HEADLESSCODE_OPENROUTER_API_KEY=sk-or-...
|
|
66
63
|
|
|
@@ -70,7 +67,15 @@ export OPENROUTER_HTTP_REFERER=https://example.com # OpenRouter app header
|
|
|
70
67
|
export OPENROUTER_APP_TITLE="headlesscode" # OpenRouter X-Title header
|
|
71
68
|
export HEADLESSCODE_WORKSPACE_ROOT=/path/to/target/repo # default workspace root
|
|
72
69
|
|
|
73
|
-
|
|
70
|
+
headlesscode --task "Fix the bug in src/index.ts" --workspace /path/to/target/repo
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
To work from a source checkout instead (for contributing):
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
git clone https://github.com/Capsize-Games/headlesscode.git
|
|
77
|
+
cd headlesscode
|
|
78
|
+
npm install
|
|
74
79
|
node bin/headlesscode.mjs --task "Fix the bug in src/index.ts" --workspace /path/to/target/repo
|
|
75
80
|
```
|
|
76
81
|
|
|
@@ -134,6 +139,45 @@ printed to stdout. Exit code 0 = success; 1 = task failed (max iterations or
|
|
|
134
139
|
consecutive-mistake limit); 2 = usage/config error (e.g. missing
|
|
135
140
|
`HEADLESSCODE_OPENROUTER_API_KEY`).
|
|
136
141
|
|
|
142
|
+
## Recursive self-improvement
|
|
143
|
+
|
|
144
|
+
`headlesscode improve` is a bounded research loop for improving the harness
|
|
145
|
+
itself. A supervisor creates isolated candidate worktrees, asks the local Qwen
|
|
146
|
+
3.5 9B worker to make focused changes, runs regression plus visible and hidden
|
|
147
|
+
evaluations, and keeps the archive, score, and selection decision outside the
|
|
148
|
+
candidate worktree. Each generation also produces a report for human review.
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
# Inspect the planned experiment without creating worktrees or calling a model:
|
|
152
|
+
npx tsx src/cli.ts improve --repo . --dry-run
|
|
153
|
+
|
|
154
|
+
# Run one small generation with the local Ollama model:
|
|
155
|
+
npx tsx src/cli.ts improve --repo . --population 2 --generations 1
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
This is an experiment, not unattended model training and not a security
|
|
159
|
+
boundary. The current loop records model-candidate and training interfaces,
|
|
160
|
+
trajectory datasets, parent-selection policies, and resource-tagged jobs, but
|
|
161
|
+
does not yet train adapters, schedule multiple trajectories, or provide
|
|
162
|
+
OS-level candidate isolation. Those limits are tracked in [the RSI roadmap](#rsi-roadmap).
|
|
163
|
+
|
|
164
|
+
## RSI roadmap
|
|
165
|
+
|
|
166
|
+
The implemented loop is intentionally honest about what remains. Follow-up
|
|
167
|
+
work is tracked in GitHub:
|
|
168
|
+
|
|
169
|
+
- [OS-level candidate sandbox](https://github.com/Capsize-Games/headlesscode/issues/3)
|
|
170
|
+
- [Cryptographically verifiable evaluator and artifacts](https://github.com/Capsize-Games/headlesscode/issues/4)
|
|
171
|
+
- [Resource-aware resumable scheduler](https://github.com/Capsize-Games/headlesscode/issues/5)
|
|
172
|
+
- [Adaptive multi-trajectory search](https://github.com/Capsize-Games/headlesscode/issues/6)
|
|
173
|
+
- [Validated curriculum fixtures](https://github.com/Capsize-Games/headlesscode/issues/7)
|
|
174
|
+
- [Adversarial evaluation and cross-model supervision](https://github.com/Capsize-Games/headlesscode/issues/8)
|
|
175
|
+
- [Real LoRA or QLoRA backend](https://github.com/Capsize-Games/headlesscode/issues/9)
|
|
176
|
+
|
|
177
|
+
See [`docs/recursive-self-improvement.md`](./docs/recursive-self-improvement.md)
|
|
178
|
+
for the design and [`docs/rsi-progress.md`](./docs/rsi-progress.md) for the
|
|
179
|
+
record of the first bounded runs.
|
|
180
|
+
|
|
137
181
|
## Registering a new project
|
|
138
182
|
|
|
139
183
|
To use headlesscode against a project that isn't set up as a headlesscode
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "headlesscode",
|
|
3
|
-
"version": "1.0
|
|
3
|
+
"version": "1.2.0",
|
|
4
4
|
"description": "Standalone headless coding-agent harness: runs the Zoo Code agent loop (prompts, tools, modes) without a VS Code UI, driven by CLI, HTTP, and parallel worktree orchestration.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"repository": {
|
package/src/cli.ts
CHANGED
|
@@ -43,6 +43,7 @@ import { analyzeCliMain } from "./orchestrator/analyze-cli.js"
|
|
|
43
43
|
import { costHistoryCliMain } from "./orchestrator/cost-history-cli.js"
|
|
44
44
|
import { resolvePermissions, type PermissionsConfig } from "./permissions/config.js"
|
|
45
45
|
import { resolveModelForMode, resolveReasoningEffortForMode } from "./config/mode-models.js"
|
|
46
|
+
import { improveMain } from "./rsi/controller.js"
|
|
46
47
|
|
|
47
48
|
const VERSION = "0.1.0"
|
|
48
49
|
|
|
@@ -166,6 +167,16 @@ interface CliOptions {
|
|
|
166
167
|
autoApproveModeSwitch: boolean
|
|
167
168
|
/** switch_mode: hard cap on total in-place mode switches per session (default 5). */
|
|
168
169
|
maxModeSwitches?: number
|
|
170
|
+
/**
|
|
171
|
+
* Evidence-gated completion (fabrication fix, 2026-09-01): when set,
|
|
172
|
+
* attempt_completion is refused unless every machine-checkable claim in
|
|
173
|
+
* its result is independently verified against ground truth (file
|
|
174
|
+
* existence, real command re-runs, serial logs, git history — see
|
|
175
|
+
* src/engine/claims.ts). Default ON for the local code backend; this
|
|
176
|
+
* flag forces it on for cloud sessions too. Also settable via
|
|
177
|
+
* HEADLESSCODE_REQUIRE_EVIDENCE.
|
|
178
|
+
*/
|
|
179
|
+
requireEvidence: boolean
|
|
169
180
|
}
|
|
170
181
|
|
|
171
182
|
const USAGE = `headlesscode — headless coding-agent harness (Phase 1 engine)
|
|
@@ -176,6 +187,10 @@ Usage:
|
|
|
176
187
|
headlesscode --dry-run [options] # build system prompt + validate config, no LLM call
|
|
177
188
|
|
|
178
189
|
Subcommands:
|
|
190
|
+
headlesscode improve --repo <path> [--model <id>] [--population <n>] [--dry-run]
|
|
191
|
+
Run a bounded recursive self-improvement generation. Candidates
|
|
192
|
+
use isolated git worktrees; evaluator/scoring paths are protected.
|
|
193
|
+
The default worker is the local Qwen 3.5 9B model.
|
|
179
194
|
headlesscode orchestrate --repo <path> --issue <n>... [--qa] [--deploy] [--dry-run]
|
|
180
195
|
Run a full parallel orchestration round
|
|
181
196
|
(split → spawn → review → QA → deploy gate). See
|
|
@@ -389,6 +404,15 @@ Options:
|
|
|
389
404
|
granting itself a broader mode's edit
|
|
390
405
|
permissions. Also settable via
|
|
391
406
|
$HEADLESSCODE_AUTO_APPROVE_MODE_SWITCH
|
|
407
|
+
--require-evidence Evidence-gated completion (fabrication fix):
|
|
408
|
+
refuse attempt_completion unless every
|
|
409
|
+
machine-checkable claim in its result is
|
|
410
|
+
independently verified against ground truth
|
|
411
|
+
(file existence, real command re-runs, serial
|
|
412
|
+
logs, git history). Default ON for the local
|
|
413
|
+
code backend; this forces it on for cloud
|
|
414
|
+
sessions too. Also settable via
|
|
415
|
+
$HEADLESSCODE_REQUIRE_EVIDENCE
|
|
392
416
|
--max-mode-switches <n> switch_mode: hard cap on total in-place mode
|
|
393
417
|
switches per session (default: 5 — see
|
|
394
418
|
DEFAULT_MAX_MODE_SWITCHES in src/engine/loop.ts).
|
|
@@ -415,6 +439,7 @@ Environment:
|
|
|
415
439
|
HEADLESSCODE_STREAM Opt-in SSE streaming ("1"/"true"/"yes"/"on")
|
|
416
440
|
HEADLESSCODE_LOCAL_EXPLORE Opt-in local exploration phase ("1"/"true")
|
|
417
441
|
HEADLESSCODE_AUTO_APPROVE_MODE_SWITCH Auto-approve switch_mode calls ("1"/"true"/"yes"/"on")
|
|
442
|
+
HEADLESSCODE_REQUIRE_EVIDENCE Force evidence-gated completion ("1"/"true"/"yes"/"on")
|
|
418
443
|
HEADLESSCODE_MAX_MODE_SWITCHES switch_mode: hard cap on total in-place
|
|
419
444
|
mode switches per session (positive int)
|
|
420
445
|
HEADLESSCODE_LOCAL_EXPLORE_MODEL Local model (default qwen3.5:9b)
|
|
@@ -441,6 +466,7 @@ export function parseArgs(argv: string[]): { options: CliOptions; error?: string
|
|
|
441
466
|
stream: false,
|
|
442
467
|
localExplore: false,
|
|
443
468
|
autoApproveModeSwitch: false,
|
|
469
|
+
requireEvidence: false,
|
|
444
470
|
}
|
|
445
471
|
|
|
446
472
|
for (let i = 0; i < argv.length; i++) {
|
|
@@ -659,6 +685,9 @@ export function parseArgs(argv: string[]): { options: CliOptions; error?: string
|
|
|
659
685
|
case "--auto-approve-mode-switch":
|
|
660
686
|
options.autoApproveModeSwitch = true
|
|
661
687
|
break
|
|
688
|
+
case "--require-evidence":
|
|
689
|
+
options.requireEvidence = true
|
|
690
|
+
break
|
|
662
691
|
case "--checkpoint-dir": {
|
|
663
692
|
const value = next()
|
|
664
693
|
if (value === undefined) {
|
|
@@ -692,6 +721,13 @@ export function parseArgs(argv: string[]): { options: CliOptions; error?: string
|
|
|
692
721
|
}
|
|
693
722
|
|
|
694
723
|
export async function main(argv: string[] = process.argv.slice(2)): Promise<number> {
|
|
724
|
+
// Bounded recursive self-improvement: the supervisor owns the evaluator,
|
|
725
|
+
// archive, and selection logic while each candidate runs in its own
|
|
726
|
+
// worktree. See docs/recursive-self-improvement.md.
|
|
727
|
+
if (argv[0] === "improve") {
|
|
728
|
+
return improveMain(argv.slice(1))
|
|
729
|
+
}
|
|
730
|
+
|
|
695
731
|
// Phase 2 subcommand: `headlesscode orchestrate ...` — delegates to the
|
|
696
732
|
// orchestrator module (split → spawn → watch → review). Keeps the Phase 1
|
|
697
733
|
// run path untouched.
|
|
@@ -1155,6 +1191,26 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise<numb
|
|
|
1155
1191
|
// loop.ts's HeadlessSessionConfig.verifyBeforeCompletion doc comment.
|
|
1156
1192
|
const verifyBeforeCompletion =
|
|
1157
1193
|
useLocalCodeBackend && !envBoolean("HEADLESSCODE_ALLOW_UNVERIFIED_COMPLETION")
|
|
1194
|
+
// Evidence-gated completion (fabrication fix, 2026-09-01): when on,
|
|
1195
|
+
// attempt_completion is refused unless every machine-checkable claim in
|
|
1196
|
+
// its result (a file exists, a specific command passed, serial markers
|
|
1197
|
+
// appear, a PR exists) is independently verified against ground truth —
|
|
1198
|
+
// the real filesystem, a real re-run of the exact command, the newest
|
|
1199
|
+
// serial log, and real git history (see src/engine/claims.ts). This is
|
|
1200
|
+
// the structural backstop for the FINAL_REPORT's central finding (§4): a
|
|
1201
|
+
// session claimed "all three hard gates pass" with a fabricated serial
|
|
1202
|
+
// excerpt when the driver was never merged and the claimed target didn't
|
|
1203
|
+
// exist. Default ON for the local code backend (the finetune harness and
|
|
1204
|
+
// real acceptance gates run local) unless explicitly disabled via
|
|
1205
|
+
// HEADLESSCODE_ALLOW_UNVERIFIED_COMPLETION (same opt-in-override pattern
|
|
1206
|
+
// as verifyBeforeCompletion); explicitly forceable via --require-evidence
|
|
1207
|
+
// OR HEADLESSCODE_REQUIRE_EVIDENCE for cloud sessions too (the pure
|
|
1208
|
+
// resolver also honors the env var — see resolveEvidenceRequiredCompletion).
|
|
1209
|
+
const evidenceRequiredCompletion = resolveEvidenceRequiredCompletion(
|
|
1210
|
+
options.requireEvidence,
|
|
1211
|
+
process.env,
|
|
1212
|
+
useLocalCodeBackend,
|
|
1213
|
+
)
|
|
1158
1214
|
// A local (Qwen3.5-9B) session was observed live 2026-08-27 doing the
|
|
1159
1215
|
// actual work correctly (a real, correct edit_file call) and then dying
|
|
1160
1216
|
// anyway: its first attempt_completion was deferred (a prior
|
|
@@ -1208,6 +1264,19 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise<numb
|
|
|
1208
1264
|
// doc comment and executor.ts's largeOverwriteRefusal.
|
|
1209
1265
|
const guardLargeOverwrites =
|
|
1210
1266
|
useLocalCodeBackend && !envBoolean("HEADLESSCODE_ALLOW_UNGUARDED_OVERWRITES")
|
|
1267
|
+
// 2026-09-02: real, confirmed, live-observed failure -- the read_file/
|
|
1268
|
+
// list_files session cache's "[cache] unchanged, reuse the earlier
|
|
1269
|
+
// result" short-circuit (src/tools/executor.ts) saves real token cost
|
|
1270
|
+
// against a remote model's per-token bill, but against the local
|
|
1271
|
+
// backend it directly produced a fabricated attempt_completion: an
|
|
1272
|
+
// edit_file failure told a session to re-read and retry, it DID call
|
|
1273
|
+
// read_file again exactly as instructed, got the cache-hit notice
|
|
1274
|
+
// instead of real content, never made the SECOND identical call that
|
|
1275
|
+
// would have returned real content again (the mechanism's own safety
|
|
1276
|
+
// valve), and gave up with a false success claim instead. See loop.ts's
|
|
1277
|
+
// HeadlessSessionConfig.disableReadFileCache doc comment.
|
|
1278
|
+
const disableReadFileCache =
|
|
1279
|
+
useLocalCodeBackend && !envBoolean("HEADLESSCODE_ALLOW_READ_FILE_CACHE_HIT")
|
|
1211
1280
|
// The cloud-tuned condensation defaults (0.75 hard threshold, 0.6 early-fire
|
|
1212
1281
|
// — see condense.ts) were measured firing needlessly aggressively against a
|
|
1213
1282
|
// local model: a trial condensing at 38 messages against a 16384-token
|
|
@@ -1359,7 +1428,16 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise<numb
|
|
|
1359
1428
|
// pattern already used for condenseThresholdFraction below: a runaway
|
|
1360
1429
|
// generation's blast radius should be a small fraction of the REAL
|
|
1361
1430
|
// local context window, not up to half of it.
|
|
1362
|
-
|
|
1431
|
+
// 2026-09-02: 8192 was still too generous in practice -- verified live,
|
|
1432
|
+
// repeatedly (joeos_finetune_data's eval_verifier.py ground-truth runs,
|
|
1433
|
+
// same day): a stuck local-model turn reliably ran to the FULL 8192-token
|
|
1434
|
+
// cap every time, taking 8-9 real minutes at this model's ~15 tok/s and
|
|
1435
|
+
// consuming the entire session's remaining time budget without ever
|
|
1436
|
+
// producing a tool call. Real, productive turns in the same logs (a tool
|
|
1437
|
+
// call + a few sentences of reasoning) topped out around 1000-2000
|
|
1438
|
+
// output tokens. Lowered so a stuck turn gets cut off in well under a
|
|
1439
|
+
// minute instead of silently eating the whole budget.
|
|
1440
|
+
const LOCAL_MAX_TOKENS = 2048
|
|
1363
1441
|
const maxTokens =
|
|
1364
1442
|
options.maxTokens ?? envNumber("HEADLESSCODE_MAX_TOKENS") ?? (useLocalCodeBackend ? LOCAL_MAX_TOKENS : undefined)
|
|
1365
1443
|
|
|
@@ -1406,12 +1484,14 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise<numb
|
|
|
1406
1484
|
requireExplicitCompletion,
|
|
1407
1485
|
patchLocalToolSchemas,
|
|
1408
1486
|
verifyBeforeCompletion,
|
|
1487
|
+
evidenceRequiredCompletion,
|
|
1409
1488
|
trackCost,
|
|
1410
1489
|
requireArtifactBeforeCompletion,
|
|
1411
1490
|
requireArtifactPathPattern: options.requireArtifactPath,
|
|
1412
1491
|
requireArtifactMinCitations: options.requireArtifactMinCitations,
|
|
1413
1492
|
requireArtifactSections: options.requireArtifactSections,
|
|
1414
1493
|
guardLargeOverwrites,
|
|
1494
|
+
disableReadFileCache,
|
|
1415
1495
|
llmClient: client,
|
|
1416
1496
|
logger,
|
|
1417
1497
|
memory,
|
|
@@ -1496,15 +1576,46 @@ function envNumber(name: string): number | undefined {
|
|
|
1496
1576
|
return Number.isFinite(n) && n > 0 ? n : undefined
|
|
1497
1577
|
}
|
|
1498
1578
|
|
|
1499
|
-
/**
|
|
1500
|
-
export function
|
|
1501
|
-
const raw =
|
|
1579
|
+
/** Pure env-boolean parse: "1"/"true"/"yes"/"on" → true; anything else (incl. unset) → false. */
|
|
1580
|
+
export function envBooleanValue(name: string, env: NodeJS.ProcessEnv): boolean {
|
|
1581
|
+
const raw = env[name]
|
|
1502
1582
|
if (raw === undefined || raw === "") {
|
|
1503
1583
|
return false
|
|
1504
1584
|
}
|
|
1505
1585
|
return ["1", "true", "yes", "on"].includes(raw.trim().toLowerCase())
|
|
1506
1586
|
}
|
|
1507
1587
|
|
|
1588
|
+
/** Parse an env boolean opt-in from process.env: "1"/"true"/"yes"/"on" → true; anything else (incl. unset) → false. */
|
|
1589
|
+
export function envBoolean(name: string): boolean {
|
|
1590
|
+
return envBooleanValue(name, process.env)
|
|
1591
|
+
}
|
|
1592
|
+
|
|
1593
|
+
/**
|
|
1594
|
+
* Resolve whether evidence-gated completion is ON for a session. Pure
|
|
1595
|
+
* (flag + env + local-backend in, boolean out) so the full wiring is
|
|
1596
|
+
* testable without spinning up a session.
|
|
1597
|
+
*
|
|
1598
|
+
* Order of precedence:
|
|
1599
|
+
* 1. the explicit --require-evidence flag always wins;
|
|
1600
|
+
* 2. HEADLESSCODE_REQUIRE_EVIDENCE env forces it on (cloud sessions
|
|
1601
|
+
* included — this is the finetune-harness/acceptance-gate hook, plan
|
|
1602
|
+
* A5: "force evidence-gated completion without code changes");
|
|
1603
|
+
* 3. the local code backend defaults it ON unless explicitly disabled
|
|
1604
|
+
* via HEADLESSCODE_ALLOW_UNVERIFIED_COMPLETION (same opt-in-override
|
|
1605
|
+
* pattern as verifyBeforeCompletion).
|
|
1606
|
+
*/
|
|
1607
|
+
export function resolveEvidenceRequiredCompletion(
|
|
1608
|
+
requireEvidenceFlag: boolean,
|
|
1609
|
+
env: NodeJS.ProcessEnv,
|
|
1610
|
+
useLocalCodeBackend: boolean,
|
|
1611
|
+
): boolean {
|
|
1612
|
+
return (
|
|
1613
|
+
requireEvidenceFlag ||
|
|
1614
|
+
envBooleanValue("HEADLESSCODE_REQUIRE_EVIDENCE", env) ||
|
|
1615
|
+
(useLocalCodeBackend && !envBooleanValue("HEADLESSCODE_ALLOW_UNVERIFIED_COMPLETION", env))
|
|
1616
|
+
)
|
|
1617
|
+
}
|
|
1618
|
+
|
|
1508
1619
|
/**
|
|
1509
1620
|
* Resolve the Phase 3 memory store. Default OFF (null) to preserve existing
|
|
1510
1621
|
* behavior; enabled by --memory-dir or $HEADLESSCODE_MEMORY_DIR. --no-memory
|