tickmarkr 2.6.0 → 2.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/adapters/prompt.d.ts +2 -1
  2. package/dist/adapters/prompt.js +10 -0
  3. package/dist/adapters/types.d.ts +1 -0
  4. package/dist/adapters/types.js +10 -0
  5. package/dist/cli/commands/fleet.js +26 -1
  6. package/dist/cli/commands/plan.js +7 -3
  7. package/dist/cli/commands/status.js +13 -3
  8. package/dist/cli/help.d.ts +2 -0
  9. package/dist/cli/help.js +2 -0
  10. package/dist/drivers/orca.js +6 -1
  11. package/dist/gates/cache.d.ts +3 -1
  12. package/dist/gates/cache.js +10 -3
  13. package/dist/gates/llm.js +4 -1
  14. package/dist/gates/review.js +7 -11
  15. package/dist/gates/run-gates.d.ts +4 -0
  16. package/dist/gates/run-gates.js +7 -4
  17. package/dist/gates/test-manifest.d.ts +12 -0
  18. package/dist/gates/test-manifest.js +26 -6
  19. package/dist/graph/graph.d.ts +6 -2
  20. package/dist/graph/graph.js +15 -4
  21. package/dist/route/role-pick.d.ts +16 -0
  22. package/dist/route/role-pick.js +15 -0
  23. package/dist/route/router.d.ts +14 -0
  24. package/dist/route/router.js +9 -1
  25. package/dist/run/consult.js +5 -9
  26. package/dist/run/daemon.d.ts +9 -0
  27. package/dist/run/daemon.js +395 -59
  28. package/dist/run/git.d.ts +36 -1
  29. package/dist/run/git.js +89 -5
  30. package/dist/run/host-health.d.ts +20 -0
  31. package/dist/run/host-health.js +64 -0
  32. package/dist/run/journal.js +9 -0
  33. package/dist/run/operator-state.d.ts +24 -2
  34. package/dist/run/operator-state.js +41 -5
  35. package/dist/run/stall.d.ts +38 -2
  36. package/dist/run/stall.js +276 -6
  37. package/dist/tui/cockpit/board.d.ts +1 -1
  38. package/dist/tui/cockpit/board.js +27 -19
  39. package/dist/tui/cockpit/derive.d.ts +2 -0
  40. package/dist/tui/cockpit/derive.js +4 -0
  41. package/dist/tui/cockpit/live-store.d.ts +1 -0
  42. package/dist/tui/cockpit/live-store.js +31 -8
  43. package/dist/tui/cockpit/run-cockpit.js +2 -1
  44. package/dist/tui/cockpit/run-view.d.ts +2 -4
  45. package/dist/tui/cockpit/run-view.js +9 -8
  46. package/package.json +1 -1
  47. package/skills/tickmarkr-overseer/SKILL.md +43 -6
@@ -53,7 +53,7 @@ export function evidenceLookup(rows, page) {
53
53
  return lookup;
54
54
  }
55
55
  export const GATE_CELL_LETTERS = {
56
- passed: "P", failed: "F", running: "R", "not-run": "-", disabled: "D", unknown: "?",
56
+ passed: "P", failed: "F", queued: "Q", running: "R", "not-run": "-", disabled: "D", unknown: "?",
57
57
  };
58
58
  /** The outcome selector's vocabulary — a classification of the row, never a word search. */
59
59
  export const OUTCOME_FILTERS = ["all", "infra failure", "work failure", "pass", "unknown"];
@@ -81,10 +81,8 @@ function outcomeLabel(outcome) {
81
81
  case "unavailable": return `unknown — ${outcome.reason}`;
82
82
  }
83
83
  }
84
- /**
85
- * The current attempt's seven cells in declaration order. `rows` supplies the later journal rows
86
- * that give a cell its inherited/satisfied label; only rows for this task after the evidence line count.
87
- */
84
+ /** Current-attempt cells in declaration order; later task rows supply inherited/satisfied labels.
85
+ * Only rows after each cell's own evidence line count. */
88
86
  export function runGateCells(task, evidence, rows = []) {
89
87
  return GATE_NAMES.map((gate) => {
90
88
  const cell = task.gates[gate] ?? { state: "unknown" };
@@ -94,12 +92,15 @@ export function runGateCells(task, evidence, rows = []) {
94
92
  const labels = [];
95
93
  let outcome;
96
94
  if (data === undefined) {
97
- labels.push(line === undefined ? { "not-run": "not run", disabled: "disabled by policy", running: "running", unknown: "unknown", passed: "passed", failed: "failed" }[cell.state] : `evidence #L${line} unavailable`);
95
+ labels.push(line === undefined ? { "not-run": "not run", disabled: "disabled by policy", queued: "queued", running: "running", unknown: "unknown", passed: "passed", failed: "failed" }[cell.state] : `evidence #L${line} unavailable`);
96
+ }
97
+ else if (cell.state === "queued") {
98
+ labels.push(`queued — ${row?.event?.event ?? "wait"}${typeof data.count === "number" ? ` (${data.count} suites)` : ""}`);
98
99
  }
99
100
  else if (data.disabled === true) {
100
101
  labels.push("disabled by policy");
101
102
  }
102
- else if (row?.event?.event === "gate-start") {
103
+ else if (["gate-start", "gate-phase-start", "phase-start"].includes(row?.event?.event ?? "")) {
103
104
  labels.push("running");
104
105
  }
105
106
  else {
@@ -124,7 +125,7 @@ export function runGateCells(task, evidence, rows = []) {
124
125
  if (e.event === "task-approved" && e.data.release === "gate-satisfied" && e.data.gate === gate)
125
126
  labels.push(`satisfied by approval #L${later.line}`);
126
127
  }
127
- const verdict = typeof data?.details === "string" ? data.details.split("\n") : [];
128
+ const verdict = row?.event?.event === "gate-result" && typeof data?.details === "string" ? data.details.split("\n") : [];
128
129
  return { gate, state: cell.state, letter: GATE_CELL_LETTERS[cell.state], ...(line === undefined ? {} : { line }), ...(outcome ? { outcome } : {}), outcomeClass: outcomeClassOf(outcome, cell.state), labels, verdict };
129
130
  });
130
131
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.6.0",
3
+ "version": "2.6.1",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -140,11 +140,11 @@ through brief lineage. **An executor choice nobody made is still an executor cho
140
140
  with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
141
141
  updates it). Never long context strings or ✓-chains.
142
142
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Do not load the `herdr` skill and do not map a Herdr workspace. Work from the current Orca terminal context. Keep the overseer in the launching terminal, name it in the same act — `orca terminal rename --terminal "$ORCA_TERMINAL_HANDLE" --title "OVERSEER · <version>"` (tab title; see the seat-name law under Seat-spawn recipes) — and inspect terminals with `orca terminal list --json`. The daemon self-places the watch board as a horizontal split of the launching terminal (`ORCA_TERMINAL_HANDLE`).
143
- 2. **Orchestrator**: Launch the orchestrator with your agent host.
144
- - **On herdr (`HERDR_ENV=1`)**: Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify a model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
143
+ 2. **Orchestrator**: First consume a successful `tickmarkr fleet --pick consult` in the seat repository, per the Fleet selection contract below; then launch the returned adapter/model with your agent host.
144
+ - **On herdr (`HERDR_ENV=1`)**: Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <picked-model>` after the `--` from the Fleet receipt). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <picked-model>` from the Fleet receipt). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
145
145
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create` on a path worktree selector with a command:
146
146
  `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "<agent-cmd>" --json`
147
- For Claude Code: `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "claude --permission-mode bypassPermissions" --json`. For Codex: `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "codex --dangerously-bypass-approvals-and-sandbox" --json`. Parse `result.terminal.handle` from the create receipt.
147
+ For Claude Code: `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "claude --model <picked-model> --permission-mode bypassPermissions --settings '{\"promptSuggestionEnabled\":false}'" --json`. For Codex: `orca terminal create --worktree path:<repo> --title "ORCH · <version>" --command "codex --model <picked-model> --dangerously-bypass-approvals-and-sandbox" --json`. Parse `result.terminal.handle` from the create receipt.
148
148
  3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
149
149
  truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
150
150
  (inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then announce the brief file:
@@ -202,9 +202,46 @@ Inventories retain the **full suite log**, not a tail or summary, and no one run
202
202
 
203
203
  ### Seat-spawn and Leg-2 recipes
204
204
 
205
+ **Fleet selection contract (OBS-1165), on BOTH hosts:** Before every supervisor-opened seat,
206
+ including every respawn, replacement, delegated spawn and Leg-2 dispatch, run the applicable
207
+ `tickmarkr fleet --pick <role>` in that seat's repository. Never reuse a prior pick at respawn.
208
+ The purpose-to-role mapping is explicit:
209
+
210
+ | Seat purpose | Fleet role | Vendor exclusions |
211
+ | --- | --- | --- |
212
+ | orchestrator | consult | Any vendors excluded for this mission |
213
+ | records | consult | Any vendors excluded for this mission |
214
+ | author (including planner, executor and scout) | consult | Any vendors excluded for this mission |
215
+ | consultant | consult | Any vendors excluded for this mission or consultation round |
216
+ | lab-rater | consult | Any vendors excluded for this rating round |
217
+ | independent reviewer (including checker and verifier) | review | Author vendor plus every vendor already used by independent reviewers in this round, and mission exclusions |
218
+
219
+ For example, an independent reviewer runs
220
+ `tickmarkr fleet --pick review --exclude-vendor <author-vendor> --exclude-vendor <prior-reviewer-vendor>`;
221
+ omit the prior-reviewer argument only for the first reviewer. Repeat `--exclude-vendor <vendor>`
222
+ for every applicable exclusion on either role. Track the author's actual vendor and each reviewer's
223
+ returned vendor in the seat record; if the author vendor is unknown, stop before review selection.
224
+ Unknown seat purposes require an explicit mapping decision; never silently map them to consult.
225
+
226
+ Consume only exit status zero and one complete JSON identity with `role`, `adapter`, `model`, `vendor`
227
+ and `channel`. Validate the role and exclusions against the request and record this receipt with the
228
+ seat. A nonzero refusal (including missing `<role>.prefer`, exhausted eligible preferences, or stale
229
+ probe data) stops the spawn: report the named reason, never invent a default, auto-write preferences,
230
+ or fall back to a remembered model. `fleet --print` and a preference list are not a resolved pick.
231
+
232
+ Build the launch command from the returned `adapter` and `model`, shell-quoting the model as one
233
+ argument: `claude-code` maps to executable/kind `claude` with `--model <picked-model>`, `codex` to
234
+ `codex --model <picked-model>`, `kimi` to `kimi --model <picked-model>`, and `grok` to
235
+ `grok -m <picked-model>`. Select only the matching adapter recipe below; examples do not choose
236
+ models. If the returned adapter has no documented visible TUI recipe and approval flags, stop and
237
+ report that transport limitation rather than substitute another adapter. Preserve the existing
238
+ visible-seat transport safeguards: named interactive TUI seats, host-specific create receipts,
239
+ orchestrator tab isolation, role-specific sandbox/approval flags, Claude prompt suggestions disabled,
240
+ Kimi `-y`, artifact plus terminal marker, and verified file-brief delivery. Fleet pick never starts a seat.
241
+
205
242
  - **On herdr (`HERDR_ENV=1`)**: Every mission to a Claude or Grok seat is delivered only with `herdr pane run <pane> "<message>"` and
206
243
  verified by reading the pane back; never use `agent prompt` for mission delivery. Launch a Grok seat with
207
- `herdr agent start <seat> --kind grok --pane <pane> -- -m grok-4.6`.
244
+ `herdr agent start <seat> --kind grok --pane <pane> -- -m <picked-model>`.
208
245
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create --worktree path:<repo> --command "<cmd>" --json`.
209
246
  **`--worktree path:` resolves only an Orca-MANAGED worktree** (`orca worktree list`); on any other checkout —
210
247
  a `git worktree add` the overseer made for a spec branch, a throwaway clone — `terminal create` hangs and
@@ -790,14 +827,14 @@ they are left implicit:
790
827
  stall watcher to catch — silent-time equals lifetime. Standing operator rule since 2026-07-13:
791
828
  consults and one-off LLM calls run as the CLI's real interactive TUI in a visible named pane.
792
829
  Headless is for exit-code probes — a quota check that wants `rc`, never work anyone must watch.
793
- 3. **Buy seat diversity from the live capability matrix, at every dispatch.** When one vendor's model
830
+ 3. **Buy seat diversity through Fleet, at every dispatch and respawn.** Resolve authors with `tickmarkr fleet --pick consult` and independent checkers/verifiers with `tickmarkr fleet --pick review --exclude-vendor <author-vendor>`, adding every applicable vendor exclusion under the Fleet selection contract. Consume the successful JSON adapter/model before using the visible-seat recipes; never select directly from doctor data. When one vendor's model
794
831
  quota collapses, the reflex is to collapse every seat onto the surviving model and hold the
795
832
  cross-vendor CLI back for a late probe — P97 ran planner, checker and verifier as one family that
796
833
  way, and three same-family passes confirmed one wrong anchored conclusion with the refuting fact in
797
834
  the room. `<state-dir>/doctor.json` already lists every installed+authed adapter and its models (nine
798
835
  were authed on 2026-08-17 while every seat ran claude). Priority when independence is scarce:
799
836
  **verifier > checker > planner > executors**.
800
- - **On herdr (`HERDR_ENV=1`)** the independent seat goes cross-vendor with `herdr agent start … --kind codex`;
837
+ - **On herdr (`HERDR_ENV=1`)** the independent seat uses its Fleet-selected adapter kind and model with `herdr agent start … --kind <picked-kind> -- …`;
801
838
  - **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)** it uses the Orca terminal-create
802
839
  seat recipe. The choice is ruled at dispatch, never debated under time pressure.
803
840
  **A codex seat inside a git WORKTREE cannot commit and cannot write outside the worktree** (OBS-824, measured