repo-harness 0.9.1 → 0.9.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.es.md CHANGED
@@ -85,7 +85,7 @@ artifacts.
85
85
  ## Novedades
86
86
 
87
87
  Las notas de versión viven en [`docs/CHANGELOG.md`](docs/CHANGELOG.md). La línea
88
- actual es `0.9.1`.
88
+ actual es `0.9.2`.
89
89
 
90
90
  ## Cómo funciona
91
91
 
@@ -220,7 +220,7 @@ bun add -g repo-harness
220
220
  repo-harness install
221
221
 
222
222
  # Fallback con npx, con Bun ya en PATH porque el CLI corre sobre Bun
223
- npx -y repo-harness install
223
+ npx -y repo-harness@latest install
224
224
  ```
225
225
 
226
226
  </details>
@@ -419,8 +419,8 @@ Guards habituales:
419
419
 
420
420
  ## Release actual
421
421
 
422
- - npm package: `repo-harness@0.9.1`
423
- - Generated workflow stamp: `repo-harness@0.9.1+template@0.9.1`
422
+ - npm package: `repo-harness@0.9.2`
423
+ - Generated workflow stamp: `repo-harness@0.9.2+template@0.9.2`
424
424
  - GitHub repository: `Ancienttwo/repo-harness`
425
425
  - Release history: [`docs/CHANGELOG.md`](docs/CHANGELOG.md)
426
426
 
package/README.fr.md CHANGED
@@ -85,7 +85,7 @@ l'emportent.
85
85
  ## Nouveautés
86
86
 
87
87
  Les notes de version vivent dans [`docs/CHANGELOG.md`](docs/CHANGELOG.md). La
88
- ligne actuelle est `0.9.1`.
88
+ ligne actuelle est `0.9.2`.
89
89
 
90
90
  ## Comment ça marche
91
91
 
@@ -224,7 +224,7 @@ bun add -g repo-harness
224
224
  repo-harness install
225
225
 
226
226
  # Fallback npx, avec Bun déjà sur PATH car le CLI s'exécute sur Bun
227
- npx -y repo-harness install
227
+ npx -y repo-harness@latest install
228
228
  ```
229
229
 
230
230
  </details>
@@ -424,8 +424,8 @@ Guards courants :
424
424
 
425
425
  ## Release actuelle
426
426
 
427
- - npm package : `repo-harness@0.9.1`
428
- - Generated workflow stamp : `repo-harness@0.9.1+template@0.9.1`
427
+ - npm package : `repo-harness@0.9.2`
428
+ - Generated workflow stamp : `repo-harness@0.9.2+template@0.9.2`
429
429
  - GitHub repository : `Ancienttwo/repo-harness`
430
430
  - Release history : [`docs/CHANGELOG.md`](docs/CHANGELOG.md)
431
431
 
package/README.ja.md CHANGED
@@ -75,7 +75,7 @@ review、checks、handoff と食い違う場合は、source artifacts を優先
75
75
  ## What's New
76
76
 
77
77
  リリースノートは [`docs/CHANGELOG.md`](docs/CHANGELOG.md) にあります。現在の
78
- ラインは `0.9.1` です。
78
+ ラインは `0.9.2` です。
79
79
 
80
80
  ## 仕組み
81
81
 
@@ -201,7 +201,7 @@ bun add -g repo-harness
201
201
  repo-harness install
202
202
 
203
203
  # npx fallback。CLI runtime は Bun なので、Bun が PATH 上に必要です。
204
- npx -y repo-harness install
204
+ npx -y repo-harness@latest install
205
205
  ```
206
206
 
207
207
  </details>
@@ -398,8 +398,8 @@ hook がブロックしたときは、まず terminal の構造化された出
398
398
 
399
399
  ## 現在の Release
400
400
 
401
- - npm package:`repo-harness@0.9.1`
402
- - Generated workflow stamp:`repo-harness@0.9.1+template@0.9.1`
401
+ - npm package:`repo-harness@0.9.2`
402
+ - Generated workflow stamp:`repo-harness@0.9.2+template@0.9.2`
403
403
  - GitHub repository:`Ancienttwo/repo-harness`
404
404
  - Release history:[`docs/CHANGELOG.md`](docs/CHANGELOG.md)
405
405
 
package/README.md CHANGED
@@ -85,7 +85,7 @@ active plan, contract, review, checks, or handoff, the source artifacts win.
85
85
  ## What's New
86
86
 
87
87
  Release notes live in [`docs/CHANGELOG.md`](docs/CHANGELOG.md). The current line
88
- is `0.9.1`.
88
+ is `0.9.2`.
89
89
 
90
90
  ## How It Works
91
91
 
@@ -234,6 +234,10 @@ concrete sprint instead of reinterpreting the original chat.
234
234
 
235
235
  ## First 5 Minutes
236
236
 
237
+ <p align="center">
238
+ <img src="docs/images/repo-harness-install-donkey-carrot.png" alt="Pixel art donkey following a carrot for repo-harness installation" width="900">
239
+ </p>
240
+
237
241
  This is the fastest path for an AI tooling owner evaluating whether the workflow is
238
242
  safe to adopt in a real repo. It separates the machine-level runtime bootstrap
239
243
  from the repo-local contract install, so a dry run can show exactly what will
@@ -256,20 +260,20 @@ curl -fsSL https://raw.githubusercontent.com/Ancienttwo/repo-harness/main/instal
256
260
  irm https://raw.githubusercontent.com/Ancienttwo/repo-harness/main/install.ps1 | iex
257
261
  ```
258
262
 
259
- <details>
260
- <summary>Already have Bun? Use Bun directly, or npx as a fallback</summary>
263
+ If Bun is already on PATH, you can skip the shell installer:
261
264
 
262
265
  ```bash
263
- # Bun (recommended)
266
+ # Bun one-shot bootstrap
267
+ bunx repo-harness@latest install
268
+
269
+ # Or install the persistent CLI first
264
270
  bun add -g repo-harness
265
271
  repo-harness install
266
272
 
267
273
  # npx fallback, with Bun already on PATH because the CLI runs on Bun
268
- npx -y repo-harness install
274
+ npx -y repo-harness@latest install
269
275
  ```
270
276
 
271
- </details>
272
-
273
277
  ### 2. Bootstrap the host runtime once
274
278
 
275
279
  ```bash
@@ -279,15 +283,17 @@ repo-harness install
279
283
  `install` is the first-run global bootstrap path. It installs the current npm
280
284
  package as the global CLI, refreshes repo-harness skill aliases, installs
281
285
  user-level hook adapters, configures Waza runtime skills, persists a brain root
282
- under `~/.repo-harness/config.json`, and configures CodeGraph MCP. In an
283
- interactive terminal it asks Y/n before installing the external skills and
284
- CodeGraph pieces (Enter keeps today's default of installing both); non-TTY
286
+ under `~/.repo-harness/config.json`, and configures CodeGraph MCP. The command
287
+ is idempotent: when the CLI is already installed from Bun's global package
288
+ source, it skips the CLI reinstall and still refreshes the host runtime pieces.
289
+ In an interactive terminal it asks Y/n before installing the external skills
290
+ and CodeGraph pieces (Enter keeps today's default of installing both); non-TTY
285
291
  runs and `--json` stay unprompted with the same default-on behavior. Passing
286
292
  `--no-external-skills` or `--no-codegraph` explicitly also skips that item's
287
- prompt unprompted, which is the escape hatch for PTY-allocating CI (for
288
- example `docker run -t`). It does not apply repo-local workflow files to the
289
- current directory. `repo-harness init` remains a compatibility alias for
290
- existing scripts.
293
+ prompt unprompted, which is the escape hatch for PTY-allocating CI (for example
294
+ `docker run -t`). It does not apply repo-local workflow files to the current
295
+ directory. `repo-harness init` remains a compatibility alias for existing
296
+ scripts.
291
297
 
292
298
  For an Agent-owned, read-only bootstrap audit, run `repo-harness setup check
293
299
  --json` or add `--check-updates` for version and adopted-repo refresh
@@ -627,8 +633,8 @@ Most common guards:
627
633
 
628
634
  ## Current Release
629
635
 
630
- - npm package: `repo-harness@0.9.1`
631
- - Generated workflow stamp: `repo-harness@0.9.1+template@0.9.1`
636
+ - npm package: `repo-harness@0.9.2`
637
+ - Generated workflow stamp: `repo-harness@0.9.2+template@0.9.2`
632
638
  - GitHub repository: `Ancienttwo/repo-harness`
633
639
  - Release history: [`docs/CHANGELOG.md`](docs/CHANGELOG.md)
634
640
 
package/README.zh-CN.md CHANGED
@@ -75,7 +75,7 @@ review、checks 或 handoff 冲突,以 source artifacts 为准。
75
75
 
76
76
  ## What's New
77
77
 
78
- Release notes 见 [`docs/CHANGELOG.md`](docs/CHANGELOG.md),当前版本线是 `0.9.1`。
78
+ Release notes 见 [`docs/CHANGELOG.md`](docs/CHANGELOG.md),当前版本线是 `0.9.2`。
79
79
 
80
80
  ## 工作原理
81
81
 
@@ -198,6 +198,10 @@ source of truth,Codex Goal mode 只围绕具体 sprint 恢复和推进,而
198
198
 
199
199
  ## 前 5 分钟
200
200
 
201
+ <p align="center">
202
+ <img src="docs/images/repo-harness-install-donkey-carrot.png" alt="repo-harness 安装引导的 pixel art 驴和萝卜 banner" width="900">
203
+ </p>
204
+
201
205
  这是评估一个真实仓库是否适合接入该 workflow 的最快路径。它把机器级 runtime
202
206
  bootstrap 和 repo-local contract install 分开,所以 dry-run 能先展示会改什么,
203
207
  再决定是否应用。
@@ -218,20 +222,20 @@ curl -fsSL https://raw.githubusercontent.com/Ancienttwo/repo-harness/main/instal
218
222
  irm https://raw.githubusercontent.com/Ancienttwo/repo-harness/main/install.ps1 | iex
219
223
  ```
220
224
 
221
- <details>
222
- <summary>已经有 Bun?优先直接用 Bun,也可以把 npx 作为备选</summary>
225
+ 如果 Bun 已经在 `PATH` 上,可以跳过 shell installer:
223
226
 
224
227
  ```bash
225
- # Bun(推荐)
228
+ # Bun 一步 bootstrap
229
+ bunx repo-harness@latest install
230
+
231
+ # 或者先安装持久化 CLI
226
232
  bun add -g repo-harness
227
233
  repo-harness install
228
234
 
229
235
  # npx 备选;仍要求 Bun 已在 PATH 上,因为 CLI runtime 是 Bun
230
- npx -y repo-harness install
236
+ npx -y repo-harness@latest install
231
237
  ```
232
238
 
233
- </details>
234
-
235
239
  ### 2. 先做一次 host runtime bootstrap
236
240
 
237
241
  ```bash
@@ -240,8 +244,10 @@ repo-harness install
240
244
 
241
245
  `install` 是首次全局引导入口。它把当前 npm 包安装成全局 CLI,刷新 repo-harness
242
246
  skill aliases,安装 user-level hook adapters,配置 Waza runtime skills,把 brain
243
- root 持久化到 `~/.repo-harness/config.json`,并配置 CodeGraph MCP。它不会把当前目录
244
- 默认迁移成 repo-local workflow。`repo-harness init` 保留为兼容 alias,给已有脚本用。
247
+ root 持久化到 `~/.repo-harness/config.json`,并配置 CodeGraph MCP。这个命令是
248
+ 幂等的:如果 CLI 已经来自 Bun global package source,它会跳过 CLI 重装,但仍继续
249
+ 刷新 host runtime pieces。它不会把当前目录默认迁移成 repo-local workflow。
250
+ `repo-harness init` 保留为兼容 alias,给已有脚本用。
245
251
 
246
252
  如果要让 Agent 做只读 bootstrap audit,运行 `repo-harness setup check
247
253
  --json`;需要版本和已接入仓库刷新提示时加 `--check-updates`。`setup check`
@@ -452,8 +458,8 @@ hook block 工作时,先看 terminal 里的结构化输出。核心字段是
452
458
 
453
459
  ## 当前 Release
454
460
 
455
- - npm package:`repo-harness@0.9.1`
456
- - Generated workflow stamp:`repo-harness@0.9.1+template@0.9.1`
461
+ - npm package:`repo-harness@0.9.2`
462
+ - Generated workflow stamp:`repo-harness@0.9.2+template@0.9.2`
457
463
  - GitHub repository:`Ancienttwo/repo-harness`
458
464
  - Release history:[`docs/CHANGELOG.md`](docs/CHANGELOG.md)
459
465
 
@@ -49,8 +49,8 @@ Owns the runtime-harness-hook-adapters capability boundary declared in .ai/conte
49
49
  - Architecture domain: `runtime-harness`
50
50
  - Architecture capability: `hook-adapters`
51
51
  - Architecture module: `docs/architecture/modules/runtime-harness/hook-adapters.md`
52
- - Last architecture event: 2026-07-06T13:54:06+0800
53
- - Last changed path: `assets/hooks/codex-delegation-advisor.sh`
52
+ - Last architecture event: 2026-07-09T11:39:06+0800
53
+ - Last changed path: `assets/hooks/prompt-guard.sh`
54
54
  - Severity: high
55
55
  - Change type: workflow-surface
56
56
  - Module responsibility: Keep this block aligned with the local boundary described by surrounding human-owned context.
@@ -49,8 +49,8 @@ Owns the runtime-harness-hook-adapters capability boundary declared in .ai/conte
49
49
  - Architecture domain: `runtime-harness`
50
50
  - Architecture capability: `hook-adapters`
51
51
  - Architecture module: `docs/architecture/modules/runtime-harness/hook-adapters.md`
52
- - Last architecture event: 2026-07-06T13:54:06+0800
53
- - Last changed path: `assets/hooks/codex-delegation-advisor.sh`
52
+ - Last architecture event: 2026-07-09T11:39:06+0800
53
+ - Last changed path: `assets/hooks/prompt-guard.sh`
54
54
  - Severity: high
55
55
  - Change type: workflow-surface
56
56
  - Module responsibility: Keep this block aligned with the local boundary described by surrounding human-owned context.
@@ -1054,7 +1054,7 @@ prompt_guard_engine_call
1054
1054
  if [[ "$PG_ENGINE_STATE" != "ok" ]]; then
1055
1055
  if [[ "$PG_ENGINE_STATE" == "legacy" ]]; then
1056
1056
  echo "[PromptGuard] Advisory: the installed repo-harness CLI predates the prompt-verdict protocol; prompt intent gates are degraded to advisory for this prompt."
1057
- echo "[PromptGuard] Refresh the CLI with: bun add -g repo-harness@latest && repo-harness install. Fallback: npx -y repo-harness install."
1057
+ echo "[PromptGuard] Refresh the CLI with: bun add -g repo-harness@latest && repo-harness install. Fallback: npx -y repo-harness@latest install."
1058
1058
  else
1059
1059
  echo "[PromptGuard] Advisory: prompt-guard decision engine is unavailable (repo-harness CLI or bun not found); prompt intent gates are degraded to advisory for this prompt."
1060
1060
  echo "[PromptGuard] Edit-layer guards still enforce plan and contract scope. Install the repo-harness CLI to restore prompt decisions."
@@ -495,10 +495,15 @@ guessed mapping.
495
495
 
496
496
  | Upstream frontmatter | Codex TOML |
497
497
  |---|---|
498
- | `model: opus`, `effort: max` | `model = "gpt-5.5"`, `model_reasoning_effort = "xhigh"` |
499
- | `model: sonnet`, `effort: max` | `model = "gpt-5.5"`, `model_reasoning_effort = "medium"` |
498
+ | `model: opus`, `effort: max` | `model = "gpt-5.6-sol"`, `model_reasoning_effort = "xhigh"` |
499
+ | `model: sonnet`, `effort: max` | `model = "gpt-5.6-terra"`, `model_reasoning_effort = "medium"` |
500
500
  | `tools: [...]` present | `sandbox_mode = "read-only"` |
501
501
 
502
+ The Codex generator also rewrites the exact upstream provider label in the
503
+ description (`Opus 4.8 at max effort` or `Sonnet 5 at max effort`) to the
504
+ mapped GPT-5.6 model and reasoning level. A missing label fails closed so the
505
+ installed metadata cannot claim a different model from the TOML settings.
506
+
502
507
  `developer_instructions` is the upstream `.md` body plus the canonical
503
508
  EXECUTION_BOUNDARY anti-extras clause, kept byte-identical to the
504
509
  `EXECUTION_BOUNDARY` constant in `scripts/contract-run.ts` so every generated
@@ -5,6 +5,12 @@ Use this content for user-level `~/.codex/AGENTS.md` and `~/.claude/CLAUDE.md` w
5
5
  ```md
6
6
  # Global Working Rules
7
7
 
8
+ Rule 0: You may spend as much time as needed thinking. Do not send optional commentary progress messages. Use tools only when they are required. For tasks that do not require tools, complete the reasoning first, then answer in final.
9
+
10
+ Reasoning: Prefer first principles over pattern matching. Before solving, first identify the observable and controllable conditions. For quantitative logic problems, before the final answer, you must prove the strategy is sufficient in the worst case. Numeric answers must have their arithmetic rechecked.
11
+
12
+ Generality: These are general working rules. Do not tailor behavior to any specific evaluation or expected answer.
13
+
8
14
  - Use Chinese by default for this user; keep technical terms in English. If the user writes in another language, mirror that language.
9
15
  - Act as an engineering collaborator: finish the concrete task, verify it, then report conclusion, actual change, reason, verification, and residual risk.
10
16
  - Prefer direct execution over repeated confirmation. Stop to ask only when continuing would likely produce output contrary to the user's intent.
@@ -1,6 +1,6 @@
1
1
  {
2
- "version": "0.9.1",
3
- "templateVersion": "0.9.1",
2
+ "version": "0.9.2",
3
+ "templateVersion": "0.9.2",
4
4
  "skillName": "repo-harness",
5
5
  "contractId": "tasks-first-harness-v1",
6
6
  "compatibility": {
@@ -195,6 +195,10 @@
195
195
  {
196
196
  "version": "0.9.1",
197
197
  "description": "Hardens the Bun-first installer and setup-check path with bunx skills probes, PATH readback hints, and explicit Codex delegation-mode selection"
198
+ },
199
+ {
200
+ "version": "0.9.2",
201
+ "description": "Hardens CLI environment diagnostics and hook execution intent while aligning the generated GPT-5.6 agent fleet with current Codex runtime guidance"
198
202
  }
199
203
  ],
200
204
  "generatedProjectStamp": {
@@ -61,11 +61,26 @@ const EXECUTION_BOUNDARY = [
61
61
  "If the requested outcome cannot be completed without expanding scope, fail closed: stop, name the missing decision, and cite the exact file/section that blocks execution.",
62
62
  ].join("\n");
63
63
 
64
- // Only (opus, max) and (sonnet, max) are recognized. Any other model or a
65
- // non-"max" effort is a fail-closed error -- the generator never guesses a mapping.
64
+ // Only (opus, max) and (sonnet, max) are recognized. Each mapping also owns the
65
+ // exact provider label rewritten in Codex metadata so the generated role never
66
+ // claims to run a different model. Any mismatch is a fail-closed error.
66
67
  const MODEL_EFFORT_MAP = {
67
- opus: { max: { model: "gpt-5.5", effort: "xhigh" } },
68
- sonnet: { max: { model: "gpt-5.5", effort: "medium" } },
68
+ opus: {
69
+ max: {
70
+ model: "gpt-5.6-sol",
71
+ effort: "xhigh",
72
+ sourceDescription: "Opus 4.8 at max effort",
73
+ targetDescription: "GPT-5.6 Sol at extra high reasoning",
74
+ },
75
+ },
76
+ sonnet: {
77
+ max: {
78
+ model: "gpt-5.6-terra",
79
+ effort: "medium",
80
+ sourceDescription: "Sonnet 5 at max effort",
81
+ targetDescription: "GPT-5.6 Terra at medium reasoning",
82
+ },
83
+ },
69
84
  };
70
85
 
71
86
  function fetchSource(agent) {
@@ -133,6 +148,9 @@ function validateFrontmatter(parsed) {
133
148
  if (!mapped) {
134
149
  return { ok: false, reason: `unmapped model/effort combination: ${parsed.model}/${parsed.effort}` };
135
150
  }
151
+ if (!parsed.description.includes(mapped.sourceDescription)) {
152
+ return { ok: false, reason: `description missing expected model label: ${mapped.sourceDescription}` };
153
+ }
136
154
  return { ok: true, mapped };
137
155
  }
138
156
 
@@ -142,8 +160,9 @@ function tomlBasicString(value) {
142
160
 
143
161
  function generateToml(parsed, mapped) {
144
162
  const lines = [];
163
+ const description = parsed.description.replace(mapped.sourceDescription, mapped.targetDescription);
145
164
  lines.push(`name = ${tomlBasicString(parsed.name)}`);
146
- lines.push(`description = ${tomlBasicString(parsed.description)}`);
165
+ lines.push(`description = ${tomlBasicString(description)}`);
147
166
  lines.push(`model = ${tomlBasicString(mapped.model)}`);
148
167
  lines.push(`model_reasoning_effort = ${tomlBasicString(mapped.effort)}`);
149
168
  if (parsed.hasTools) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "repo-harness",
3
- "version": "0.9.1",
3
+ "version": "0.9.2",
4
4
  "description": "Installs, migrates, audits, and repairs repo-local agentic development harnesses",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -61,11 +61,26 @@ const EXECUTION_BOUNDARY = [
61
61
  "If the requested outcome cannot be completed without expanding scope, fail closed: stop, name the missing decision, and cite the exact file/section that blocks execution.",
62
62
  ].join("\n");
63
63
 
64
- // Only (opus, max) and (sonnet, max) are recognized. Any other model or a
65
- // non-"max" effort is a fail-closed error -- the generator never guesses a mapping.
64
+ // Only (opus, max) and (sonnet, max) are recognized. Each mapping also owns the
65
+ // exact provider label rewritten in Codex metadata so the generated role never
66
+ // claims to run a different model. Any mismatch is a fail-closed error.
66
67
  const MODEL_EFFORT_MAP = {
67
- opus: { max: { model: "gpt-5.5", effort: "xhigh" } },
68
- sonnet: { max: { model: "gpt-5.5", effort: "medium" } },
68
+ opus: {
69
+ max: {
70
+ model: "gpt-5.6-sol",
71
+ effort: "xhigh",
72
+ sourceDescription: "Opus 4.8 at max effort",
73
+ targetDescription: "GPT-5.6 Sol at extra high reasoning",
74
+ },
75
+ },
76
+ sonnet: {
77
+ max: {
78
+ model: "gpt-5.6-terra",
79
+ effort: "medium",
80
+ sourceDescription: "Sonnet 5 at max effort",
81
+ targetDescription: "GPT-5.6 Terra at medium reasoning",
82
+ },
83
+ },
69
84
  };
70
85
 
71
86
  function fetchSource(agent) {
@@ -133,6 +148,9 @@ function validateFrontmatter(parsed) {
133
148
  if (!mapped) {
134
149
  return { ok: false, reason: `unmapped model/effort combination: ${parsed.model}/${parsed.effort}` };
135
150
  }
151
+ if (!parsed.description.includes(mapped.sourceDescription)) {
152
+ return { ok: false, reason: `description missing expected model label: ${mapped.sourceDescription}` };
153
+ }
136
154
  return { ok: true, mapped };
137
155
  }
138
156
 
@@ -142,8 +160,9 @@ function tomlBasicString(value) {
142
160
 
143
161
  function generateToml(parsed, mapped) {
144
162
  const lines = [];
163
+ const description = parsed.description.replace(mapped.sourceDescription, mapped.targetDescription);
145
164
  lines.push(`name = ${tomlBasicString(parsed.name)}`);
146
- lines.push(`description = ${tomlBasicString(parsed.description)}`);
165
+ lines.push(`description = ${tomlBasicString(description)}`);
147
166
  lines.push(`model = ${tomlBasicString(mapped.model)}`);
148
167
  lines.push(`model_reasoning_effort = ${tomlBasicString(mapped.effort)}`);
149
168
  if (parsed.hasTools) {
@@ -2278,6 +2278,14 @@ pi_root_context_content() {
2278
2278
  This is the root routing contract for Claude Code and Codex. Keep it short:
2279
2279
  load this first, then follow the repo-owned artifacts it names.
2280
2280
 
2281
+ ## General Working Rules
2282
+
2283
+ Rule 0: You may spend as much time as needed thinking. Do not send optional commentary progress messages. Use tools only when they are required. For tasks that do not require tools, complete the reasoning first, then answer in final.
2284
+
2285
+ Reasoning: Prefer first principles over pattern matching. Before solving, first identify the observable and controllable conditions. For quantitative logic problems, before the final answer, you must prove the strategy is sufficient in the worst case. Numeric answers must have their arithmetic rechecked.
2286
+
2287
+ Generality: These are general working rules. Do not tailor behavior to any specific evaluation or expected answer.
2288
+
2281
2289
  ## Root Workflow Contract
2282
2290
 
2283
2291
  - Keep sibling `CLAUDE.md` and `AGENTS.md` files aligned. Claude Code consumes `CLAUDE.md`; Codex consumes `AGENTS.md`.
@@ -10,6 +10,7 @@ import * as fs from 'fs';
10
10
  import * as os from 'os';
11
11
  import * as path from 'path';
12
12
  import { spawnSync } from 'child_process';
13
+ import { fileURLToPath } from 'url';
13
14
  import { ALL_TARGETS } from '../installer/targets/registry';
14
15
  import { checkCodegraph, type CodegraphCheckResult } from '../tools/codegraph';
15
16
  import { CLI_VERSION } from './status';
@@ -19,6 +20,8 @@ import { ROUTES } from '../hook/route-registry';
19
20
 
20
21
  const TRUST_STATE_LINE = /^\[hooks\.state\."[^"]+\/\.codex\/hooks\.json:/;
21
22
  const PACKAGE_NAME = 'repo-harness';
23
+ const PACKAGE_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..', '..');
24
+ const MIN_CODEX_CLI_VERSION = '0.144.0';
22
25
  const UPDATE_CHECK_ENV = 'REPO_HARNESS_CHECK_UPDATES';
23
26
  const LATEST_VERSION_ENV = 'REPO_HARNESS_LATEST_VERSION';
24
27
 
@@ -105,7 +108,7 @@ function parseVersion(value: string): number[] | null {
105
108
  return match.slice(1).map((part) => Number(part));
106
109
  }
107
110
 
108
- function compareVersions(a: string, b: string): number | null {
111
+ export function compareVersions(a: string, b: string): number | null {
109
112
  const left = parseVersion(a);
110
113
  const right = parseVersion(b);
111
114
  if (!left || !right) return null;
@@ -116,14 +119,70 @@ function compareVersions(a: string, b: string): number | null {
116
119
  return 0;
117
120
  }
118
121
 
119
- function readLatestPackageVersion(): { version?: string; error?: string } {
120
- if (process.env[LATEST_VERSION_ENV]) {
121
- return { version: process.env[LATEST_VERSION_ENV] };
122
+ function checkCodexCliVersion(): DoctorCheckResult {
123
+ const id = 'codex-cli-version';
124
+ const describe = `Codex CLI supports generated GPT-5.6 agent profiles (>= ${MIN_CODEX_CLI_VERSION})`;
125
+ const resolved = findCommandOnPath('codex');
126
+ if (!resolved) {
127
+ return { id, describe, status: 'na', detail: 'codex not found on PATH' };
122
128
  }
123
129
 
124
- const result = spawnSync('npm', ['view', PACKAGE_NAME, 'version', '--json'], {
130
+ const result = spawnSync(resolved, ['--version'], {
125
131
  encoding: 'utf-8',
126
132
  timeout: 5000,
133
+ env: process.env,
134
+ });
135
+ if (result.status !== 0 || result.error) {
136
+ const error = result.stderr || result.stdout || String(result.error?.message ?? result.error ?? 'codex --version failed');
137
+ return { id, describe, status: 'warn', detail: `path=${resolved}; ${error.trim()}` };
138
+ }
139
+
140
+ const output = result.stdout.trim();
141
+ const match = output.match(/^codex-cli (\d+\.\d+\.\d+)$/);
142
+ const version = match?.[1] ?? '';
143
+ const comparison = compareVersions(version, MIN_CODEX_CLI_VERSION);
144
+ if (comparison === null) {
145
+ return {
146
+ id,
147
+ describe,
148
+ status: 'warn',
149
+ detail: `path=${resolved}; unable to parse version from ${JSON.stringify(output)}`,
150
+ };
151
+ }
152
+ if (comparison < 0) {
153
+ return {
154
+ id,
155
+ describe,
156
+ status: 'warn',
157
+ detail: `path=${resolved}; current=${version}; minimum=${MIN_CODEX_CLI_VERSION}`,
158
+ };
159
+ }
160
+ return {
161
+ id,
162
+ describe,
163
+ status: 'ok',
164
+ detail: `path=${resolved}; current=${version}; minimum=${MIN_CODEX_CLI_VERSION}`,
165
+ };
166
+ }
167
+
168
+ export function readLatestPackageVersion(env?: NodeJS.ProcessEnv): { version?: string; error?: string } {
169
+ const activeEnv = env ?? process.env;
170
+ if (activeEnv[LATEST_VERSION_ENV]) {
171
+ return { version: activeEnv[LATEST_VERSION_ENV] };
172
+ }
173
+
174
+ const result = spawnSync(process.execPath, [
175
+ 'pm',
176
+ 'view',
177
+ PACKAGE_NAME,
178
+ 'version',
179
+ '--json',
180
+ '--registry=https://registry.npmjs.org',
181
+ ], {
182
+ encoding: 'utf-8',
183
+ timeout: 5000,
184
+ cwd: PACKAGE_ROOT,
185
+ env: activeEnv,
127
186
  });
128
187
  if (result.status !== 0 || result.error) {
129
188
  return { error: result.stderr || result.stdout || String(result.error?.message ?? result.error ?? 'npm view failed') };
@@ -439,6 +498,7 @@ export function runDoctor(cwd: string = process.cwd()): DoctorReport {
439
498
  const securityReport = runSecurityScan({ cwd });
440
499
  checks.push(checkPath());
441
500
  checks.push(checkVersion());
501
+ checks.push(checkCodexCliVersion());
442
502
  checks.push(checkCliUpdate());
443
503
  for (const target of ALL_TARGETS) {
444
504
  if (target.supportsLocation('global')) {
@@ -5,6 +5,7 @@ import { fileURLToPath } from "url";
5
5
  import { configureBrainRoot, defaultBrainRootChoice, expandHomePath } from "./brain-root";
6
6
  import { syncCrossReviewSkills } from "./init";
7
7
  import { runInstall, type InstallTargetSpec } from "./install";
8
+ import { compareVersions, readLatestPackageVersion } from "./doctor";
8
9
  import { configureCodegraph } from "../tools/codegraph";
9
10
  import { runProcess as runBoundedProcess } from "../../effects/process-runner";
10
11
 
@@ -175,16 +176,31 @@ function isBunGlobalPackageSource(sourceRoot: string, env?: NodeJS.ProcessEnv):
175
176
  }
176
177
  }
177
178
 
179
+ // Best-effort: readLatestPackageVersion() already swallows offline/npm-missing/
180
+ // timeout failures into `.error`, so any lookup failure just yields no hint —
181
+ // this must never turn the "skipped" step into a "failed" one.
182
+ function updateAvailableHint(version: string | null, env?: NodeJS.ProcessEnv): string {
183
+ if (!version) return "";
184
+ const activeEnv = env ?? process.env;
185
+ if (activeEnv.REPO_HARNESS_CHECK_UPDATES !== "1") return "";
186
+ const latest = readLatestPackageVersion(env);
187
+ if (!latest.version) return "";
188
+ const comparison = compareVersions(version, latest.version);
189
+ if (comparison === null || comparison >= 0) return "";
190
+ return `; latest=${latest.version} available — run: repo-harness update`;
191
+ }
192
+
178
193
  function installCli(sourceRoot: string, cwd: string, env?: NodeJS.ProcessEnv, installSpec?: string): GlobalRuntimeStep {
179
194
  const version = packageVersion(sourceRoot);
180
195
  const name = packageName(sourceRoot);
181
196
  if (installSpec === undefined && isBunGlobalPackageSource(sourceRoot, env)) {
197
+ const base = version
198
+ ? `already installed from Bun global package source; version=${version}`
199
+ : "already installed from Bun global package source";
182
200
  return {
183
201
  step: "install repo-harness CLI",
184
202
  status: "skipped",
185
- detail: version
186
- ? `already installed from Bun global package source; version=${version}`
187
- : "already installed from Bun global package source",
203
+ detail: `${base}${updateAvailableHint(version, env)}`,
188
204
  };
189
205
  }
190
206
  const spec = installSpec ?? (existsSync(join(sourceRoot, "package.json")) ? sourceRoot : "repo-harness");
@@ -311,6 +311,7 @@ function doctorChecks(
311
311
  ): InitHookCheck[] {
312
312
  const checks: InitHookCheck[] = [];
313
313
  for (const entry of report.checks) {
314
+ if (target === 'claude' && entry.id === 'codex-cli-version') continue;
314
315
  const source: InitHookCheckSource = entry.id === 'security-config' ? 'security' : 'doctor';
315
316
  let checkStatus: InitHookCheckStatus = entry.status;
316
317
 
@@ -82,12 +82,40 @@ const EXPLICIT_EXECUTION_LINE = re(
82
82
  String.raw`^${SP}*(please\s+)?(implement\s+(this|the)|execute\s+(this|the)|start\s+(implementation|executing|coding)|go ahead|proceed|ship it|开始(实现|执行|落实|写)|执行计划|落实计划|批准执行|批准|直接(改|做|实现|执行|落地)|动手|开干|可以(开始|执行|干)|可以干|干吧|做吧)(${SP}|$)`,
83
83
  );
84
84
 
85
+ // Unlike the legacy execution verbs above, this newly supported phrase must
86
+ // remain an actual command at the start of a line. A whitespace-only prefix
87
+ // keeps quoted examples, questions, and negations out of the execution path.
88
+ const DIRECT_MODIFICATION_LINE = re(
89
+ String.raw`^\s*(请\s*)?直接修改`,
90
+ );
91
+ const DIRECT_MODIFICATION_INLINE_PAYLOAD =
92
+ /“[^”\n]*”|「[^」\n]*」|『[^』\n]*』|"[^"\n]*"|`[^`\n]*`/gu;
93
+ const DIRECT_MODIFICATION_QUESTION = re(
94
+ String.raw`([??]|[吗么呢]${SP}*$|是不是|能不能|可不可以|要不要|应不应该|该不该|会不会|行不行|好不好|对不对|合不合适)`,
95
+ );
96
+ const DIRECT_MODIFICATION_NON_COMMAND = re(
97
+ String.raw`(不合适(?:吧|${SP}*($|[,,;;。]))|(?:是)?不(?:对|应该|行)(?:吧|的?${SP}*($|[,,;;。]))|不要这么做|不是(我|我们)?的?要求|只是(一个|个)?示例|仅作示例|作为示例)`,
98
+ );
99
+
100
+ function isDirectModificationCommandLine(line: string): boolean {
101
+ const topLevelText = line.replace(DIRECT_MODIFICATION_INLINE_PAYLOAD, '');
102
+ return (
103
+ DIRECT_MODIFICATION_LINE.test(line) &&
104
+ !DIRECT_MODIFICATION_QUESTION.test(topLevelText) &&
105
+ !DIRECT_MODIFICATION_NON_COMMAND.test(topLevelText)
106
+ );
107
+ }
108
+
109
+ function hasDirectModificationCommandLine(ctx: PromptIntentContext): boolean {
110
+ return isDirectModificationCommandLine(ctx.firstLine);
111
+ }
112
+
85
113
  export function promptHasExplicitExecutionCommandLine(ctx: PromptIntentContext): boolean {
86
- return EXPLICIT_EXECUTION_LINE.test(ctx.text);
114
+ return EXPLICIT_EXECUTION_LINE.test(ctx.text) || hasDirectModificationCommandLine(ctx);
87
115
  }
88
116
 
89
117
  export function isExplicitExecutionStartLine(ctx: PromptIntentContext): boolean {
90
- return EXPLICIT_EXECUTION_LINE.test(ctx.firstLine);
118
+ return EXPLICIT_EXECUTION_LINE.test(ctx.firstLine) || hasDirectModificationCommandLine(ctx);
91
119
  }
92
120
 
93
121
  const PLAN_EXECUTION_PROJECTION_LINE = re(
@@ -104,6 +132,7 @@ const TRIGGER_QUESTION = re(
104
132
  );
105
133
 
106
134
  export function isTriggerQuestionPrompt(ctx: PromptIntentContext): boolean {
135
+ if (hasDirectModificationCommandLine(ctx)) return false;
107
136
  return TRIGGER_QUESTION.test(ctx.firstLine);
108
137
  }
109
138
 
@@ -130,6 +159,7 @@ const PLAN_REFINEMENT = re(
130
159
  );
131
160
 
132
161
  export function isPlanRefinementIntent(ctx: PromptIntentContext): boolean {
162
+ if (hasDirectModificationCommandLine(ctx)) return false;
133
163
  if (PLAN_REFINEMENT_EXEC.test(ctx.firstLine)) return false;
134
164
  return PLAN_REFINEMENT.test(ctx.firstLine);
135
165
  }
@@ -179,6 +209,7 @@ export function isDiagnosticQuestionIntent(ctx: PromptIntentContext): boolean {
179
209
  if (isExecutionApprovalIntent(ctx)) return false;
180
210
  if (isEmbeddedApprovedPlanIntent(ctx)) return false;
181
211
  if (isPlanShapedMarkdownIntent(ctx)) return false;
212
+ if (hasDirectModificationCommandLine(ctx)) return false;
182
213
  if (DIAGNOSTIC_DIRECT.test(ctx.text)) return true;
183
214
  return DIAGNOSTIC_TOPIC.test(ctx.text) && DIAGNOSTIC_QUESTION.test(ctx.text);
184
215
  }
@@ -198,6 +229,7 @@ export function isReviewReleaseAdvisoryIntent(ctx: PromptIntentContext): boolean
198
229
  if (isEmbeddedApprovedPlanIntent(ctx)) return false;
199
230
  if (isPlanShapedMarkdownIntent(ctx)) return false;
200
231
  if (isExecutionApprovalIntent(ctx)) return false;
232
+ if (hasDirectModificationCommandLine(ctx)) return false;
201
233
  // Review/check prompts often say "execute /check" or "执行 checklist". Those
202
234
  // route to evaluator evidence, not implementation.
203
235
  if (REVIEW_RELEASE_CODING_VERB.test(ctx.text)) return false;
@@ -266,7 +298,9 @@ export function isNextSliceOrStatusAdvisoryIntent(ctx: PromptIntentContext): boo
266
298
  return false;
267
299
  }
268
300
 
269
- const IMPLEMENT_VERB = re('(implement|execute|build it|do it|go ahead|proceed|ship it|实现|执行|开始写|动手|开干)');
301
+ const IMPLEMENT_VERB = re(
302
+ '(implement|execute|build it|do it|go ahead|proceed|ship it|实现|执行|开始写|动手|开干)',
303
+ );
270
304
 
271
305
  export function isImplementIntent(ctx: PromptIntentContext): boolean {
272
306
  if (isTriggerQuestionPrompt(ctx)) return false;
@@ -280,6 +314,7 @@ export function isImplementIntent(ctx: PromptIntentContext): boolean {
280
314
  if (isPassiveWorktreeStatusIntent(ctx)) return false;
281
315
  return (
282
316
  IMPLEMENT_VERB.test(ctx.text) ||
317
+ hasDirectModificationCommandLine(ctx) ||
283
318
  isExecutionApprovalIntent(ctx) ||
284
319
  isEmbeddedApprovedPlanIntent(ctx) ||
285
320
  isPlanShapedMarkdownIntent(ctx)