tickmarkr 2.4.1 → 2.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/README.md +110 -24
  2. package/dist/adapters/claude-code.js +7 -2
  3. package/dist/adapters/codex.d.ts +1 -0
  4. package/dist/adapters/codex.js +66 -5
  5. package/dist/adapters/types.d.ts +4 -0
  6. package/dist/adapters/types.js +33 -0
  7. package/dist/cli/commands/approve.d.ts +42 -0
  8. package/dist/cli/commands/approve.js +80 -13
  9. package/dist/cli/commands/doctor.d.ts +234 -0
  10. package/dist/cli/commands/doctor.js +139 -5
  11. package/dist/cli/commands/report.js +15 -44
  12. package/dist/cli/commands/scope.js +36 -6
  13. package/dist/cli/commands/stats.d.ts +2 -0
  14. package/dist/cli/commands/stats.js +42 -23
  15. package/dist/cli/commands/status.js +20 -4
  16. package/dist/cli/commands/ui.js +42 -53
  17. package/dist/cli/commands/unlock.d.ts +1 -1
  18. package/dist/cli/commands/unlock.js +59 -9
  19. package/dist/cli/help.d.ts +219 -0
  20. package/dist/cli/help.js +212 -0
  21. package/dist/cli/index.d.ts +42 -2
  22. package/dist/cli/index.js +23 -8
  23. package/dist/drivers/herdr.d.ts +7 -2
  24. package/dist/drivers/herdr.js +78 -48
  25. package/dist/drivers/orca.d.ts +3 -2
  26. package/dist/drivers/orca.js +43 -4
  27. package/dist/drivers/subprocess.d.ts +1 -0
  28. package/dist/drivers/subprocess.js +3 -0
  29. package/dist/drivers/types.d.ts +14 -0
  30. package/dist/gates/artifact-manifest.d.ts +50 -0
  31. package/dist/gates/artifact-manifest.js +23 -0
  32. package/dist/plan/scope.d.ts +25 -0
  33. package/dist/plan/scope.js +92 -12
  34. package/dist/report/operator-record.d.ts +49 -0
  35. package/dist/report/operator-record.js +137 -0
  36. package/dist/run/daemon.d.ts +3 -4
  37. package/dist/run/daemon.js +18 -10
  38. package/dist/run/lock.d.ts +59 -2
  39. package/dist/run/lock.js +184 -26
  40. package/dist/run/operator-state.d.ts +86 -0
  41. package/dist/run/operator-state.js +165 -0
  42. package/dist/run/supervision.d.ts +32 -0
  43. package/dist/run/supervision.js +138 -17
  44. package/dist/tui/cockpit/capture.d.ts +19 -0
  45. package/dist/tui/cockpit/capture.js +89 -1
  46. package/dist/tui/cockpit/components.d.ts +15 -1
  47. package/dist/tui/cockpit/components.js +79 -9
  48. package/dist/tui/cockpit/decision-actions.d.ts +147 -0
  49. package/dist/tui/cockpit/decision-actions.js +315 -0
  50. package/dist/tui/cockpit/derive.d.ts +1 -1
  51. package/dist/tui/cockpit/derive.js +2 -0
  52. package/dist/tui/cockpit/evidence-view.d.ts +119 -0
  53. package/dist/tui/cockpit/evidence-view.js +210 -0
  54. package/dist/tui/cockpit/home-view.d.ts +88 -0
  55. package/dist/tui/cockpit/home-view.js +240 -0
  56. package/dist/tui/cockpit/keys.d.ts +125 -0
  57. package/dist/tui/cockpit/keys.js +31 -0
  58. package/dist/tui/cockpit/layout.d.ts +14 -0
  59. package/dist/tui/cockpit/layout.js +15 -0
  60. package/dist/tui/cockpit/live-runtime.d.ts +46 -0
  61. package/dist/tui/cockpit/live-runtime.js +683 -0
  62. package/dist/tui/cockpit/live-store.d.ts +289 -0
  63. package/dist/tui/cockpit/live-store.js +308 -0
  64. package/dist/tui/cockpit/live.d.ts +21 -1
  65. package/dist/tui/cockpit/live.js +12 -1
  66. package/dist/tui/cockpit/run-view.d.ts +100 -0
  67. package/dist/tui/cockpit/run-view.js +202 -0
  68. package/dist/tui/cockpit/shell.d.ts +50 -0
  69. package/dist/tui/cockpit/shell.js +74 -0
  70. package/dist/tui/cockpit/theme.d.ts +27 -0
  71. package/dist/tui/cockpit/theme.js +21 -0
  72. package/package.json +1 -1
  73. package/skills/tickmarkr-auto/SKILL.md +10 -1
  74. package/skills/tickmarkr-loop/SKILL.md +72 -1
  75. package/skills/tickmarkr-overseer/SKILL.md +12 -0
package/README.md CHANGED
@@ -79,7 +79,7 @@ tickmarkr init # guided setup + doctor; scaffolds config and spe
79
79
  tickmarkr compile tickmarkr.spec.md # spec → task graph (fails without acceptance criteria)
80
80
  tickmarkr plan # dry-run routing decisions + cost estimate
81
81
  tickmarkr run # execute, route to best CLI, gate every result (--concurrency N)
82
- tickmarkr report <runId> --md # engagement record in Markdown
82
+ tickmarkr report <runId> --md # Markdown on stdout; redirect to save beside the spec
83
83
  ```
84
84
 
85
85
  That's the flow: `init` scaffolds config, you write tasks with `acceptance[]` criteria, `compile`
@@ -113,16 +113,51 @@ Consent rules — every write is additive, never destructive:
113
113
 
114
114
  ## Monitor and supervise
115
115
 
116
+ `tickmarkr ui [runId]` opens one cockpit with **1 Home, 4 Run, 5 Evidence**. Without a
117
+ run ID it selects the latest journal; an empty repository opens Home. Use
118
+ `tickmarkr ui <runId> --view run` or `--view evidence` to open a delivered view directly.
119
+ `tickmarkr ui --setup <runId>` opens Run Parks and preserves the requested run identity.
120
+ Fleet/Bootstrap and Plan/Health are explicit follow-ons: keys 2/3/6 are not installed.
121
+ Use the existing `tickmarkr fleet`, `tickmarkr init`, `tickmarkr plan` and `tickmarkr doctor`
122
+ commands for those workflows.
123
+
124
+ `?` opens the shortcut sheet; Tab/Shift-Tab move focus through visible regions, Enter opens
125
+ selection, and Esc closes the deepest overlay. `q` quits; text-entry mode keeps `q1?`
126
+ literal. Run shows every task in the matching graph, its recorded attempt/path/pane/alarm,
127
+ and current-attempt gate evidence. `o` requests focus of the recorded owned pane when the
128
+ driver supports it; unavailable panes leave an evidence diagnostic. Evidence keeps original
129
+ journal `#L` identities, full verdict paging, and a stable selection with Follow off.
130
+
116
131
  ```bash
117
- tickmarkr status # engagement state (--watch to follow live)
132
+ tickmarkr status <runId> # preserved printed engagement state
133
+ tickmarkr status <runId> --oneline # compact snapshot, then exit
134
+ tickmarkr status <runId> --watch # TTY: Run cockpit; non-TTY: line output
135
+ tickmarkr status <runId> --watch --plain # preserved line/ANSI fallback, including on a TTY
118
136
  tickmarkr resume <runId> # continue an engagement from the local execution log
119
- tickmarkr approve <runId> <taskId> # approve a parked task (--reason to document)
137
+ tickmarkr approve <runId> <taskId> # append permission for a non-gate park; see below
120
138
  tickmarkr report <runId> # cost/quality report
139
+ tickmarkr report <runId> --md > feature.record.md # explicit file write beside your spec
121
140
  tickmarkr profile # show the learned routing profile
122
141
  tickmarkr profile --explain <shape> <channel> # why a channel ranks where it does for a shape
123
142
  ```
124
143
 
125
- Green tasks land on `tickmarkr/<runId>`; merge to your mainline is always your call.
144
+ For machines, `tickmarkr status <runId> --watch --events` replays and follows projected
145
+ decision events as one JSON document per stdout line. `--jsonl` and `--decision-events`
146
+ are aliases; keep stderr keepalives separate (never `2>&1`). This projection is distinct
147
+ from raw `journal.jsonl`. Webhook delivery remains opt-in via `--webhook <url>`.
148
+ Preserved printed twins also include separate `fleet --print` and `fleet --why` outputs,
149
+ `plan`, `doctor`, `report --compare <baseline-runId>`, `report --bundle <path>` (an explicit
150
+ proof-bundle write), and `stats` (all runs, no run ID). Report retains its learning preview
151
+ and comparison warnings; absent metering stays “not measurable,” never an invented $0.
152
+
153
+ A manual cockpit keeps its final receipt at run-end and follows a later resume of that run.
154
+ By default, the daemon-owned board requests graceful shutdown at run-end and closes only its owned pane
155
+ after checking its watch presence stood down; unconfirmed cleanup is reported. The existing
156
+ `visibility.keepPanes: forever` debug override preserves panes. Herdr keeps
157
+ task/gate grouping, short titles, and the board beside its caller without taking focus.
158
+ Quitting or orderly signals restore raw mode, pointer tracking, title and alternate screen,
159
+ and release only that observer's presence. Green tasks land on `tickmarkr/<runId>`;
160
+ merge to your mainline is always your call.
126
161
 
127
162
  ### Escalation and consults
128
163
 
@@ -133,15 +168,57 @@ away from a failing adapter will never retry it in subsequent `tickmarkr resume`
133
168
 
134
169
  ### Approving tasks
135
170
 
136
- `tickmarkr approve` unblocks two task states:
137
-
138
- **Human gates** (attempt budget ≥ 1):
139
- - The task finished with a result but gates require human judgment (`humanGate: true` in the spec)
140
- - Approving records the Partner's verdict and the task proceeds to merge
141
-
142
- **Attempt-budget exhaustion**:
143
- - The task has burned its full attempt budget without reaching a conclusive result
144
- - Approving grants a fresh attempt budget, routing around all previously-failed channels and adapters
171
+ The recorded partial-human-park case has **1/3 merged**, human T2, blocked T3, and a
172
+ passed tip verify. It is **PARTIAL**, not green: T2's `humanGate: true` parks it **before
173
+ dispatch**. In Run (`4`, or `tickmarkr ui --setup <runId>`), select T2, press `a` for
174
+ Actions, choose Approve with Enter, and review the confirmation: run/task, original park
175
+ `#L`, actor/reason, exact `tickmarkr approve` argv, append-only consequence and enactor.
176
+ Only `y` confirms; `n`/Esc cancel, and Enter never confirms.
177
+
178
+ Read the receipt's newly appended `task-approved` line, actor/reason and disposition back
179
+ from the journal. Approval records permission; it does not dispatch work, pass a gate or
180
+ mark the task done. With no live owner the receipt says **approved; resume required**.
181
+ Exit the observer and run `tickmarkr resume <runId>` explicitly. A matching live daemon
182
+ can enact the release at its next task boundary; a different live run must end before this
183
+ one resumes. Keep any resume refusal and its remediation visible, including a deny/prefer
184
+ config conflict; repair the source/config as directed, never edit the compiled graph to
185
+ force success. After resume, CURRENT TIP is PENDING until fresh evidence arrives. Completion
186
+ requires the latest run-end, a nonfailed known tip result, and empty `failed`, `human`,
187
+ `blocked` and `pending` buckets; the completed case has 3/3 recorded merges. An unrelated
188
+ graph says “not comparable” and supplies no borrowed denominator. Historical GATES RAN
189
+ does not establish current completion.
190
+
191
+ The CLI twin for the same decision is
192
+ `tickmarkr approve <runId> T2 --by operator --reason 'ready to proceed'`, followed by the
193
+ receipt check and explicit resume above. Other parks have different permitted decisions:
194
+
195
+ | Park | Decision and effect |
196
+ |---|---|
197
+ | Human gate / other non-gate park | Plain approve records permission to dispatch. |
198
+ | Attempt cap | Plain approve grants a fresh attempt budget; prior routing exclusions remain. |
199
+ | Infrastructure | Plain approve or `--recheck`; recheck reruns the declared battery and satisfies no gate. |
200
+ | Failed review gate | `--waive` satisfies only that identified gate; `--uphold` funds one fixed attempt carrying findings; `--recheck` reruns the battery. |
201
+ | Other failed gate | `--waive` or `--recheck`; plain approve refuses. |
202
+ | Tombstone / gate failure without identifying evidence | Diagnostic only; no invented decision. |
203
+
204
+ Decisions are append-only and cannot be undone. Unknown tasks, duplicate decisions and
205
+ changed parks refuse; no success receipt is claimed without reading back the append.
206
+
207
+ ### Help and recovery
208
+
209
+ `tickmarkr ui --help`, `tickmarkr eval --help`, `tickmarkr unlock --help` and
210
+ `tickmarkr profile reset --help` print guidance without opening a UI, seeding fixtures,
211
+ unlocking or resetting history. Help flags are recognized before `--`; arguments after
212
+ it are literal data. Help examples describe operations; printing them never executes them.
213
+
214
+ Recovery remains explicit: `unlock <runId>` targets a matching provably dead lock;
215
+ `unlock --garbage` handles malformed lock bytes without inventing a run ID. Both require
216
+ confirmation (`--yes` for non-TTY) and refuse live, inaccessible or changed holders.
217
+ `doctor --cached` reads cached diagnostics; `doctor --probe-preflight` discloses probe
218
+ counts/files; `doctor --fix-only` repairs locally without model probes or catalog refresh.
219
+ Default doctor and `--fix` still probe. `doctor --refresh-catalog` refreshes only the catalog.
220
+ `scope <intent-file> --preview` is a local, non-writing preview; authoring requires TTY
221
+ confirmation or `--yes` and can make model calls.
145
222
 
146
223
  ## Choosing your fleet: `tickmarkr fleet`
147
224
 
@@ -155,7 +232,7 @@ tickmarkr plan # lint the resolved routing table against your spec
155
232
  tickmarkr run # dispatch with the fleet you confirmed
156
233
  ```
157
234
 
158
- `tickmarkr doctor` is a pure sensor; `tickmarkr fleet` is the actuator. The browser is one
235
+ `tickmarkr doctor` probes and records health; `tickmarkr fleet` edits routing config. The browser is one
159
236
  two-pane surface: the left rail lists views (**All models**, **Shapes**, **Steering**) and every
160
237
  installed agent CLI with its auth state and model count; the right pane is a searchable model list
161
238
  with tier, context, price, and probe-latency columns. `Space` allows/denies, `Enter` classifies an
@@ -283,7 +360,7 @@ the frontier-model consult is the *National Office*. The terms below use that vo
283
360
  - (bare token) — member is running normally
284
361
  - **cleanup · <taskId>**: overflow/teardown generation tabs. When a new generation starts (on retry escalation), a new cleanup tab
285
362
  opens labeled with the newest live member's task ID; it auto-closes when the generation completes
286
- - **watch**: a single pane running `tickmarkr status --watch` — the senior's glanceable engagement monitor
363
+ - **watch**: a single owned pane running `tickmarkr ui <runId> --view run` — the same Run cockpit opened by TTY `status --watch`
287
364
 
288
365
  ### Pane naming (when visibility.llm = pane)
289
366
 
@@ -311,13 +388,17 @@ This keeps the Partner focused on decisions that require attention, not noise.
311
388
 
312
389
  tickmarkr closes exactly what it owns and no longer needs, no matter how any process died.
313
390
 
314
- Every pane and tab tickmarkr creates receives a **parseable ownership name** encoding the pane's role, task, attempt, and run:
391
+ Tabs use short human labels (at most 20 characters); panes carry durable ownership names:
315
392
  - `<taskId>` — the task's tab, holding its worker and its judge/review/consult panes
316
393
  - `cleanup · <taskId>` — teardown generation tab for overflow attempts
317
- - `watch` — status monitor pane
318
- - `<role> · <taskId> · A<attempt> · R<runId>` — judge, review, consult, and worker panes (formats like `judge · task-abc123 · A1 · Rrun-20260713-175532`)
394
+ - `tickmarkr:watch:run:0:<runId>` — daemon-owned Run board
395
+ - `tickmarkr:<role>:<taskId>:<attempt>:<runId>` — durable judge, review, consult and worker pane names; display titles may be shorter
319
396
 
320
- tickmarkr creates all owned panes only within the run's workspace; any tickmarkr-owned panes discovered outside the run's workspace (from prior runs or placement bugs) are reconciled and closed. Any pane not matching the ownership contract is **foreign** — created by you or another tool — and is never closed automatically.
397
+ Reconciliation stays within the run's workspace. It can retire another run's panes only
398
+ when this repository's journal evidence proves that run ended; live or unknown runs and
399
+ other workspaces remain protected. Pane names outside the ownership contract are **foreign**.
400
+ Board replacement additionally checks repository/run ownership and its acknowledged watch
401
+ presence; matching a short tab label is never sufficient.
321
402
 
322
403
  **Desired-state reconciliation**: A pure function computes the exact set of panes that should exist from the local journal at any moment:
323
404
  - Worker panes for all in-flight task attempts
@@ -326,12 +407,14 @@ tickmarkr creates all owned panes only within the run's workspace; any tickmarkr
326
407
  - Empty set (after engagement end)
327
408
 
328
409
  The daemon reconciles at every safe point:
329
- 1. **Run start** — clean up any orphaned panes from crashed earlier runs of the same repo
410
+ 1. **Run start** — reconcile this run and older runs proven ended in this repository
330
411
  2. **Resume** — reconcile the restarted journal state and close panes for superseded attempts
331
412
  3. **After terminal events** (task done, failed, human gate) — close the corresponding worker/gate pane and its emptied tab
332
- 4. **At engagement end** — close all remaining owned panes and tabs
413
+ 4. **At engagement end** — close remaining owned panes and tabs, with graceful board shutdown
333
414
 
334
- Reconciliation failures (herdr unavailable, a pane vanished mid-sweep) never fail the engagement — visibility is cosmetic, gates are law.
415
+ `visibility.keepPanes: forever` disables this sweep. Reconciliation failures (herdr
416
+ unavailable, a pane vanished mid-sweep) do not replace gate verdicts; unconfirmed board
417
+ cleanup is journaled and reported.
335
418
 
336
419
  ### Workspace trust
337
420
 
@@ -372,10 +455,13 @@ them; worker-declared deviations are recorded as notes, not authority.
372
455
 
373
456
  If you clone this repo and use Claude Code, project skills are installed in `.claude/skills/`:
374
457
 
375
- - **`/tickmarkr-loop`** — compile a spec, review the routing plan, run the engagement, and commit the Markdown record
376
- - **`/tickmarkr-auto`** — autonomous multi-phase runs (GSD milestones, etc.)
458
+ - **[/tickmarkr-loop](skills/tickmarkr-loop/SKILL.md)** — compile a spec, review the routing plan, run the engagement, and commit the Markdown record
459
+ - **[/tickmarkr-auto](skills/tickmarkr-auto/SKILL.md)** — autonomous multi-phase runs (GSD milestones, etc.)
377
460
 
378
461
  These are optional — the CLI works standalone. Skills are repo-scoped and ship in the npm tarball for agents working in projects that have run `tickmarkr init --agent`.
462
+ The canonical sources live in `skills/`; this repository's installed `.claude/skills/` links
463
+ resolve there. The [overseer skill](skills/tickmarkr-overseer/SKILL.md) links to the same
464
+ loop walkthrough for cockpit and decision guidance; skill names remain unchanged.
379
465
 
380
466
  ## Contributing
381
467
 
@@ -3,7 +3,7 @@ import { readdirSync, readFileSync, realpathSync, statSync } from "node:fs";
3
3
  import { homedir } from "node:os";
4
4
  import { join } from "node:path";
5
5
  import { parseWorkerResult } from "./prompt.js";
6
- import { channelsFromConfig, declareInputBox, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
6
+ import { channelsFromConfig, declareInputBox, MODEL_ID_RE, promptFitsArgv, shq, TokenUsageSchema } from "./types.js";
7
7
  // SPEND-01/SPEND-11: claude writes a per-session JSONL to ~/.claude/projects/<slug>/ where slug is the
8
8
  // realpath'd cwd with every non-alphanumeric char replaced by "-" (verified 114/114 — 36-DIAGNOSIS.md).
9
9
  // The old `/`-only formula missed the "." in `.tickmarkr/worktrees/…` — ENOENT on every worktree dispatch.
@@ -239,7 +239,12 @@ export const claudeCode = {
239
239
  // live check ate the prompt), and --prompt-suggestions takes an OPTIONAL value — appended directly
240
240
  // before the prompt it would swallow it the same way. So the setting's value is always followed by
241
241
  // another flag, never by the prompt positional.
242
- interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
242
+ // OBS-931: the same ONE-argv-string hazard as codex (OBS-930) — over promptArgvCeiling() the TUI
243
+ // launch would E2BIG on Linux, so it returns null → worker-mode-fallback → the headless form.
244
+ // resumeCommand keeps the shape: its contract returns a string (composer delivery is 2.4.3 work).
245
+ interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
246
+ ? `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`
247
+ : null,
243
248
  trustDialog: CLAUDE_TRUST_DIALOG,
244
249
  inputBox: CLAUDE_INPUT_BOX,
245
250
  // A resumed attempt lands in the same painted editor, so it carries the same ghost-text suppression
@@ -4,6 +4,7 @@ export declare function readCodexModelsCache(path?: string): {
4
4
  fetchedAt?: string;
5
5
  };
6
6
  export declare const CODEX_TRUST_DIALOG: TrustDialog;
7
+ export declare const CODEX_INPUT_BOX: import("./types.js").InputBox;
7
8
  export declare function seedCodexTrust(repoRoot: string, configPath?: string): TrustVerdict;
8
9
  export declare function hasCodexTrustedProject(text: string, root: string): boolean;
9
10
  export declare function codexConfigMcpServerNames(configPath?: string): string[];
@@ -3,7 +3,7 @@ import { homedir } from "node:os";
3
3
  import { dirname, join } from "node:path";
4
4
  import { probeVersion } from "./claude-code.js";
5
5
  import { parseWorkerResult } from "./prompt.js";
6
- import { channelsFromConfig, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
6
+ import { channelsFromConfig, declareInputBox, MODEL_ID_RE, promptFitsArgv, shq, TokenUsageSchema } from "./types.js";
7
7
  // SPEND-07: codex writes per-session JSONL to ~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl — date-partitioned,
8
8
  // NOT cwd-keyed. session_meta.payload.cwd is FILE-SCOPED (one codex exec per cwd). token_count events carry
9
9
  // per-turn DELTAS in payload.info.last_token_usage; we read POST-HOC (never the pane, never the trailer).
@@ -95,6 +95,56 @@ export const CODEX_TRUST_DIALOG = {
95
95
  fingerprint: "Do you trust the contents of this directory?",
96
96
  key: "Enter",
97
97
  };
98
+ // OBS-930 / OBS-136: the codex TUI's composer, CAPTURED (codex 0.153.4, herdr pane read, 2026-09-05
99
+ // 15:48Z — tests/fixtures/codex-input-box, provenance in its README), never guessed:
100
+ // (background row)
101
+ // › Ask Codex to do anything ← U+203A, ASCII space, then the DIM placeholder (empty) or the draft
102
+ // (background row)
103
+ // gpt-5.6-luna medium · /private/tmp/tkr-obs930-smoke ← footer: <model> <effort> · <cwd>
104
+ // Two facts a fingerprint alone would get wrong: (1) a submitted turn is echoed into the transcript
105
+ // with the SAME caret (`› You are a smoke test.`), and so is the trust dialog's cursor (`› 1. Yes,
106
+ // continue`) — only the composer is followed by the footer row, so the footer is the anchor, exactly
107
+ // as claude's editor is anchored by its rules; (2) an EMPTY composer paints its placeholder as text,
108
+ // so "empty" is a closed allowlist of captured placeholders (0.153.4 above; 0.144.5's
109
+ // `Run /review on my current changes` from tests/fixtures/codex-mcp-spinner). An unknown placeholder
110
+ // reads as occupied and a submit onto it fails closed by name (OBS-140) — a fixture-capture chore,
111
+ // never a drive-by widening. `match` is THE COMPOSER IS PAINTED (true while empty, true mid-turn);
112
+ // `emptyMatch` is painted AND carrying nothing — the only positive evidence a submit registered.
113
+ const CODEX_ANSI_SGR_RE = /\u001B\[[0-9;]*m/g;
114
+ const CODEX_CARET_RE = /^› /;
115
+ const CODEX_PLACEHOLDER_RE = /^› (?:Ask Codex to do anything|Run \/review on my current changes)$/;
116
+ const CODEX_FOOTER_RE = /^\S+(?: \S+)? · \S/;
117
+ // A wrapped or multi-line draft grows the composer downward before the footer.
118
+ // ponytail: a fixed window, not a parser — raise it if a real capture ever shows a taller composer.
119
+ const CODEX_MAX_COMPOSER_ROWS = 8;
120
+ function matchesCodexComposer(paneText, empty) {
121
+ const lines = paneText.replace(CODEX_ANSI_SGR_RE, "").split("\n").map((l) => l.trim());
122
+ const caret = empty ? CODEX_PLACEHOLDER_RE : CODEX_CARET_RE;
123
+ return lines.some((line, i) => {
124
+ if (!caret.test(line))
125
+ return false;
126
+ for (let below = i + 1; below < lines.length && below <= i + CODEX_MAX_COMPOSER_ROWS; below++) {
127
+ if (CODEX_FOOTER_RE.test(lines[below]))
128
+ return true;
129
+ // only the composer's own rows may sit between the caret row and the footer: the background
130
+ // rows (blank in a text read) and a draft's continuation rows — never another caret row
131
+ if (lines[below] !== "" && CODEX_CARET_RE.test(lines[below]))
132
+ return false;
133
+ }
134
+ return false;
135
+ });
136
+ }
137
+ export const CODEX_INPUT_BOX = declareInputBox("codex", {
138
+ fingerprint: "› ",
139
+ match: (paneText) => matchesCodexComposer(paneText, false),
140
+ emptyMatch: (paneText) => matchesCodexComposer(paneText, true),
141
+ // As for claude (OBS-342): a fresh codex worker slot is a shell awaiting its launch line; every
142
+ // later delivery is a TUI turn awaiting this composer.
143
+ firstDeliveryIsLaunch: true,
144
+ // The 2026-09-05 capture painted the composer ~10 s after the trust answer with MCP suppressed;
145
+ // claude's bound, kept for the same cold-start reasons.
146
+ readinessTimeoutMs: 30_000,
147
+ });
98
148
  // v1.22 T5 / OBS-16: codex keys trust on absolute path under [projects."<root>"] trust_level="trusted"
99
149
  // in ~/.codex/config.toml (CODEX_HOME relocates the dir). Worktrees inherit parent-project trust when
100
150
  // the REPO ROOT is trusted — seed the root once, cover every future worktree. Idempotent: a second
@@ -186,14 +236,24 @@ export const codex = {
186
236
  channels: (cfg) => channelsFromConfig("codex", cfg),
187
237
  // v1.65 T3: every flag the command builder below hardcodes (incl. codexMcpSuppressionFlags' -c/
188
238
  // --disable and GITDIR_WRITABLE's -c) — all listed by top-level `codex --help`, verified 2026-07-22.
189
- hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
239
+ hardcodedFlags: { binary: "codex", flags: ["--sandbox", "-a", "-s", "--model", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
190
240
  // --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
191
241
  // MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
192
242
  // CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
193
243
  headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
194
- // OBS-889: Codex's TUI has no file/stdin prompt form. Returning null makes the daemon journal
195
- // worker-mode-fallback before it runs the argv-safe headless command in the visible pane.
196
- interactiveCommand: () => null,
244
+ // OBS-930: the visible pane runs the REAL TUI. Codex's TUI takes its prompt only as the [PROMPT]
245
+ // positional (`codex --help`, 0.153.4 — no file/stdin form), so the launch inlines the file exactly
246
+ // as the claude adapter does: the prompt is the LAST positional and every flag value is followed by
247
+ // a flag, never by the prompt. The argv hazard that once forbade this (OBS-889: `countLiveSuites`
248
+ // matched a suite word 140 KB into a finished worker's argv) is closed on the counter side — the
249
+ // census reads a command's first four tokens only — and those four never carry a suite word here.
250
+ // Same sandbox, hook trust and MCP suppression as the headless form; `-a never` is the TUI's
251
+ // autonomous approval policy (exec has no approvals to configure).
252
+ // OBS-930 (Linux): the inlined prompt is ONE argv string and Linux caps one at 131072 bytes, so a
253
+ // prompt over promptArgvCeiling() returns null → worker-mode-fallback → the headless form (types.ts).
254
+ interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
255
+ ? `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`
256
+ : null,
197
257
  invoke(task, _cwd, a, ctx) {
198
258
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
199
259
  },
@@ -202,6 +262,7 @@ export const codex = {
202
262
  // "Do you trust this directory?" (OBS-16). doctor-only side effect.
203
263
  trust: (repoRoot) => seedCodexTrust(repoRoot),
204
264
  trustDialog: CODEX_TRUST_DIALOG,
265
+ inputBox: CODEX_INPUT_BOX,
205
266
  // v1.5 MODEL-01: file read only (no `codex models` subcommand exists, verified 2026-07-10).
206
267
  // Already fails OPEN to [] internally — advisory detection, unlike gates' fail-closed.
207
268
  listModels: async () => readCodexModelsCache().models,
@@ -185,5 +185,9 @@ export declare function channelKey(c: {
185
185
  model: string;
186
186
  }): string;
187
187
  export declare function shq(s: string): string;
188
+ export declare const PROMPT_ARGV_CEILING_LINUX = 120000;
189
+ export declare const PROMPT_ARGV_CEILING_DEFAULT = 900000;
190
+ export declare function promptArgvCeiling(platform?: string): number;
191
+ export declare function promptFitsArgv(promptFile: string, platform?: string): boolean;
188
192
  export declare const QUOTA_RE: RegExp;
189
193
  export declare const MODEL_ID_RE: RegExp;
@@ -1,3 +1,4 @@
1
+ import { statSync } from "node:fs";
1
2
  import { z } from "zod";
2
3
  // SPEND-01/06: normalized token counts — the measurable fact. NO cost field, ever: CLIs report
3
4
  // cost:0 on sub plans and notional list prices on others (LIVE-CHECK finding 3); money is Phase 18's
@@ -232,6 +233,38 @@ export function channelKey(c) {
232
233
  export function shq(s) {
233
234
  return `'${s.replaceAll("'", `'\\''`)}'`;
234
235
  }
236
+ // OBS-930 (Linux) / OBS-931: a TUI launch inlines the prompt file as ONE argv string — "$(cat prompt)"
237
+ // as the last positional. Linux caps a single argv string at MAX_ARG_STRLEN = PAGE_SIZE × 32 =
238
+ // 131072 bytes (E2BIG: CI run 33979013874, ubuntu, `codex: Argument list too long` on a 140 KB
239
+ // prompt); darwin enforces only the 1 MB total ARG_MAX, which is why the macOS export proof passed.
240
+ // A real worker prompt is 60–150 KB (OBS-889 measured 149,417 bytes), so on a Linux host the launch
241
+ // fails in production, not only in the test. Ceilings are named BY PLATFORM — never a probe of the
242
+ // running kernel at dispatch time:
243
+ // linux 120_000 — the cap is per STRING and every flag is its own argv entry, so only the prompt
244
+ // counts against it; "$(cat …)" strips nothing but trailing newlines, so the
245
+ // file's byte size IS the string's. 131072 − 120000 leaves ~11 KB of headroom.
246
+ // others 900_000 — under the 1 MB total that darwin/BSD enforce across argv + envp.
247
+ // Over the ceiling the adapter returns null: the daemon journals worker-mode-fallback
248
+ // {reason:"adapter"} and runs the headless form in the visible pane — the pre-OBS-930 behaviour,
249
+ // now only for oversized prompts. An unreadable file is "not proven oversized" and keeps the TUI
250
+ // rendering: the daemon writes the prompt before it builds the launch, a missing file fails the
251
+ // same way in either form, and docs-truth renders the command against a placeholder path.
252
+ // OBS-931 (2.4.3): the real fix is prompt delivery through the composer for large prompts.
253
+ export const PROMPT_ARGV_CEILING_LINUX = 120_000;
254
+ export const PROMPT_ARGV_CEILING_DEFAULT = 900_000;
255
+ export function promptArgvCeiling(platform = process.platform) {
256
+ return platform === "linux" ? PROMPT_ARGV_CEILING_LINUX : PROMPT_ARGV_CEILING_DEFAULT;
257
+ }
258
+ export function promptFitsArgv(promptFile, platform = process.platform) {
259
+ let bytes;
260
+ try {
261
+ bytes = statSync(promptFile).size;
262
+ }
263
+ catch {
264
+ return true;
265
+ }
266
+ return bytes <= promptArgvCeiling(platform);
267
+ }
235
268
  // Quota exhaustion is detected from CLI errors, never predicted (spec §4).
236
269
  // ZAI coding-plan exhaustion text: "Insufficient balance or no resource package. Please recharge."
237
270
  // Anchor the distinctive full phrase, not the two-word "insufficient balance" fragment — that fires
@@ -1,3 +1,4 @@
1
+ import { Journal, type JournalEvent } from "../../run/journal.js";
1
2
  export declare const APPROVAL_DISPOSITIONS: readonly ["dispatch", "waive-gate", "re-dispatch", "fund-fixed-attempt", "fresh-budget"];
2
3
  export type ApprovalDisposition = (typeof APPROVAL_DISPOSITIONS)[number];
3
4
  /**
@@ -8,6 +9,47 @@ export type ApprovalDisposition = (typeof APPROVAL_DISPOSITIONS)[number];
8
9
  */
9
10
  export declare const APPROVAL_ENACTS: Record<ApprovalDisposition, string>;
10
11
  export declare function approvalDispositionForRelease(release: unknown): ApprovalDisposition;
12
+ /** The closed verb set every decision surface may name. Nothing outside it reaches this command. */
13
+ export declare const DECISION_VERBS: readonly ["approve", "waive", "uphold", "recheck"];
14
+ export type DecisionVerb = (typeof DECISION_VERBS)[number];
15
+ /** The newest park a decision binds to, read the one way this command reads it. */
16
+ export interface NewestPark {
17
+ /** Index into the journal's event array; `line` is the physical 1-based journal line. */
18
+ index: number;
19
+ line: number;
20
+ ts: string | undefined;
21
+ /** The daemon-recorded kind (task-human data.kind), never inferred from prose. */
22
+ kind: string | undefined;
23
+ reason: string | undefined;
24
+ /** The newest failed gate before the park — the gate a waive would satisfy. */
25
+ failedGate: string | undefined;
26
+ /** A pre-dispatch human gate whose reason marks it permanent by design (see isTombstonePark). */
27
+ tombstone: boolean;
28
+ }
29
+ /**
30
+ * There is no closed park kind for a declaration-shaped retirement, so the evidence is the one the
31
+ * daemon recorded: a pre-dispatch human-gate park whose reason carries the task's own title, where the
32
+ * spec declares the tombstone. Read narrowly on purpose — every other kind is actionable regardless of prose.
33
+ */
34
+ export declare function isTombstonePark(kind: string | undefined, reason: string | undefined): boolean;
35
+ export declare function newestPark(events: readonly JournalEvent[], taskId: string,
36
+ /** Physical zero-based source indexes corresponding one-for-one with `events`. */
37
+ sourceIndexes?: readonly number[]): NewestPark | undefined;
38
+ /** Parsed events paired with their immutable physical JSONL identities. */
39
+ export declare function readJournalEvents(journal: Journal): {
40
+ events: JournalEvent[];
41
+ sourceIndexes: number[];
42
+ };
43
+ /**
44
+ * FINAL §3.3's decision menu as data: human-gate/attempt-cap/other non-gate parks → approve; infra →
45
+ * approve or recheck; review gate-fail → waive/uphold/recheck; other gate-fail → waive/recheck; a
46
+ * gate-fail park with no failed-gate evidence, or a tombstone → nothing (a diagnostic, never a
47
+ * fabricated verb). The refusals in `approve` below enforce the same table; this is the one place a
48
+ * surface may read it from, so what a menu offers and what the command accepts cannot drift.
49
+ */
50
+ export declare function permittedDecisionVerbs(park: Pick<NewestPark, "kind" | "failedGate" | "tombstone"> | undefined): readonly DecisionVerb[];
51
+ /** The release marker this command appends for a verb on a park — the fact a read-back must match. */
52
+ export declare function releaseForDecision(verb: DecisionVerb, park: Pick<NewestPark, "kind" | "failedGate">): string | undefined;
11
53
  export type ApprovalStatus = "deferred-live" | "recorded-no-owner";
12
54
  /** The requested run plus the different live run currently blocking its repository, when present. */
13
55
  export interface ApprovalRunOwner {
@@ -27,6 +27,73 @@ export function approvalDispositionForRelease(release) {
27
27
  return "dispatch";
28
28
  }
29
29
  import { acquireApprovalSerialization, runLockOwner } from "../../run/lock.js";
30
+ /** The closed verb set every decision surface may name. Nothing outside it reaches this command. */
31
+ export const DECISION_VERBS = ["approve", "waive", "uphold", "recheck"];
32
+ /**
33
+ * There is no closed park kind for a declaration-shaped retirement, so the evidence is the one the
34
+ * daemon recorded: a pre-dispatch human-gate park whose reason carries the task's own title, where the
35
+ * spec declares the tombstone. Read narrowly on purpose — every other kind is actionable regardless of prose.
36
+ */
37
+ export function isTombstonePark(kind, reason) {
38
+ // Declaration retirements use an explicit title marker: either an em-dash-delimited
39
+ // `— tombstone` suffix or the canonical `tombstone, never dispatched` phrase. A task title
40
+ // that merely discusses tombstones is still an ordinary, actionable human gate.
41
+ return kind === "human-gate"
42
+ && /(?:\s—\s+tombstone\b|\btombstone,\s*never dispatched\b)/iu.test(reason ?? "");
43
+ }
44
+ export function newestPark(events, taskId,
45
+ /** Physical zero-based source indexes corresponding one-for-one with `events`. */
46
+ sourceIndexes) {
47
+ for (let i = events.length - 1; i >= 0; i--) {
48
+ const e = events[i];
49
+ if (e.event !== "task-human" || e.taskId !== taskId)
50
+ continue;
51
+ const kind = typeof e.data.kind === "string" ? e.data.kind : undefined;
52
+ const reason = typeof e.data.reason === "string" ? e.data.reason : undefined;
53
+ return {
54
+ index: i, line: (sourceIndexes?.[i] ?? i) + 1, ts: typeof e.ts === "string" ? e.ts : undefined, kind, reason,
55
+ failedGate: failedGateForNewestPark(events, taskId, i), tombstone: isTombstonePark(kind, reason),
56
+ };
57
+ }
58
+ return undefined;
59
+ }
60
+ /** Parsed events paired with their immutable physical JSONL identities. */
61
+ export function readJournalEvents(journal) {
62
+ const tracked = journal.readTracked();
63
+ return {
64
+ events: tracked.map((row) => row.raw),
65
+ sourceIndexes: tracked.map((row) => row.sourceIndex),
66
+ };
67
+ }
68
+ /**
69
+ * FINAL §3.3's decision menu as data: human-gate/attempt-cap/other non-gate parks → approve; infra →
70
+ * approve or recheck; review gate-fail → waive/uphold/recheck; other gate-fail → waive/recheck; a
71
+ * gate-fail park with no failed-gate evidence, or a tombstone → nothing (a diagnostic, never a
72
+ * fabricated verb). The refusals in `approve` below enforce the same table; this is the one place a
73
+ * surface may read it from, so what a menu offers and what the command accepts cannot drift.
74
+ */
75
+ export function permittedDecisionVerbs(park) {
76
+ if (!park || park.tombstone)
77
+ return [];
78
+ if (park.kind === "gate-fail") {
79
+ if (park.failedGate === undefined)
80
+ return [];
81
+ return park.failedGate === "review" ? ["waive", "uphold", "recheck"] : ["waive", "recheck"];
82
+ }
83
+ if (park.kind === "infra")
84
+ return ["approve", "recheck"];
85
+ return ["approve"];
86
+ }
87
+ /** The release marker this command appends for a verb on a park — the fact a read-back must match. */
88
+ export function releaseForDecision(verb, park) {
89
+ if (verb === "waive")
90
+ return GATE_SATISFIED_RELEASE;
91
+ if (verb === "uphold")
92
+ return REVIEW_UPHELD_RELEASE;
93
+ if (verb === "recheck")
94
+ return RECHECK_RELEASE;
95
+ return park.kind === ATTEMPT_CAP_RELEASE ? ATTEMPT_CAP_RELEASE : undefined;
96
+ }
30
97
  // THE liveness rule, written once. The lock is REPOSITORY-wide, so a live owner of some OTHER run is
31
98
  // not an owner of this one and sweeps none of its approvals — claiming otherwise is the same
32
99
  // falsehood in a new shape. Liveness itself comes from lock.ts's runLockOwner (the same inspect() the
@@ -74,19 +141,13 @@ export async function approve(argv, cwd = process.cwd()) {
74
141
  }
75
142
  // OBS-18: only the most recent task-human for this task decides whether this approval grants a
76
143
  // fresh attempt budget. The closed daemon-issued kind, never a human prose string, controls release.
77
- const events = journal.read();
78
- let lastHumanIndex = -1;
79
- for (let i = events.length - 1; i >= 0; i--) {
80
- if (events[i].event === "task-human" && events[i].taskId === taskId) {
81
- lastHumanIndex = i;
82
- break;
83
- }
84
- }
85
- const lastHuman = events[lastHumanIndex];
86
- const capPark = lastHuman?.data.kind === ATTEMPT_CAP_RELEASE;
87
- const gateFailPark = lastHuman?.data.kind === "gate-fail";
88
- const infraPark = lastHuman?.data.kind === "infra";
89
- const failedGate = gateFailPark ? failedGateForNewestPark(events, taskId, lastHumanIndex) : undefined;
144
+ const { events, sourceIndexes } = readJournalEvents(journal);
145
+ const park = newestPark(events, taskId, sourceIndexes);
146
+ const lastHuman = park === undefined ? undefined : events[park.index];
147
+ const capPark = park?.kind === ATTEMPT_CAP_RELEASE;
148
+ const gateFailPark = park?.kind === "gate-fail";
149
+ const infraPark = park?.kind === "infra";
150
+ const failedGate = gateFailPark ? park?.failedGate : undefined;
90
151
  if (gateFailPark && !failedGate) {
91
152
  throw new Error(`task ${taskId} is parked on gate-fail but has no failed gate result on the newest park — refusing to infer one`);
92
153
  }
@@ -137,6 +198,12 @@ export async function approve(argv, cwd = process.cwd()) {
137
198
  choices.push(`--uphold (disposition fund-fixed-attempt)`);
138
199
  throw new Error(`task ${taskId} is parked on failed gate ${failedGate}; plain approve has disposition only for non-gate parks — pass ${choices.join(" or ")}`);
139
200
  }
201
+ // A tombstone has no verb in permittedDecisionVerbs (the named decisions above already refused it
202
+ // as a non-gate park); the command enforces the same table for plain approve, so a surface that
203
+ // skips the helper still cannot release it.
204
+ if (park?.tombstone) {
205
+ throw new Error(`task ${taskId}'s newest park is a tombstone (${park.reason ?? "no reason"}) — permanent by design; no verb releases it`);
206
+ }
140
207
  journal.append("task-approved", taskId, {
141
208
  by,
142
209
  ...(reason ? { reason } : {}),