tickmarkr 2.6.1 → 2.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/README.md +16 -3
  2. package/dist/adapters/catalog-remote.js +89 -47
  3. package/dist/adapters/claude-code.js +9 -6
  4. package/dist/adapters/codex.js +7 -4
  5. package/dist/adapters/prompt.d.ts +1 -0
  6. package/dist/adapters/prompt.js +14 -6
  7. package/dist/adapters/registry.js +3 -3
  8. package/dist/adapters/types.d.ts +12 -4
  9. package/dist/adapters/types.js +6 -0
  10. package/dist/cli/commands/approve.d.ts +11 -4
  11. package/dist/cli/commands/approve.js +82 -27
  12. package/dist/cli/commands/compile.js +13 -3
  13. package/dist/cli/commands/doctor.d.ts +8 -2
  14. package/dist/cli/commands/doctor.js +11 -3
  15. package/dist/cli/commands/fleet.js +87 -11
  16. package/dist/cli/commands/plan.js +13 -8
  17. package/dist/cli/commands/report.d.ts +2 -1
  18. package/dist/cli/commands/report.js +74 -8
  19. package/dist/cli/commands/resume.js +4 -2
  20. package/dist/cli/commands/status.js +43 -20
  21. package/dist/cli/help.d.ts +2 -0
  22. package/dist/cli/help.js +9 -2
  23. package/dist/compile/native.js +7 -0
  24. package/dist/config/config.d.ts +35 -2
  25. package/dist/config/config.js +86 -10
  26. package/dist/config/fleet-overlay.d.ts +13 -2
  27. package/dist/config/fleet-overlay.js +60 -0
  28. package/dist/drivers/herdr.d.ts +12 -0
  29. package/dist/drivers/herdr.js +51 -0
  30. package/dist/drivers/orca.d.ts +35 -2
  31. package/dist/drivers/orca.js +222 -67
  32. package/dist/drivers/types.d.ts +2 -0
  33. package/dist/drivers/types.js +2 -2
  34. package/dist/eval/canary.d.ts +2 -1
  35. package/dist/eval/canary.js +2 -2
  36. package/dist/eval/dispatch.js +1 -0
  37. package/dist/gates/acceptance.d.ts +9 -1
  38. package/dist/gates/acceptance.js +31 -4
  39. package/dist/gates/baseline.d.ts +32 -2
  40. package/dist/gates/baseline.js +111 -24
  41. package/dist/gates/cache.d.ts +8 -0
  42. package/dist/gates/cache.js +12 -2
  43. package/dist/gates/llm.d.ts +11 -4
  44. package/dist/gates/llm.js +40 -21
  45. package/dist/gates/review.d.ts +14 -1
  46. package/dist/gates/review.js +160 -34
  47. package/dist/gates/run-gates.d.ts +56 -4
  48. package/dist/gates/run-gates.js +358 -58
  49. package/dist/gates/test-manifest.d.ts +45 -1
  50. package/dist/gates/test-manifest.js +78 -12
  51. package/dist/graph/schema.d.ts +2 -0
  52. package/dist/graph/schema.js +2 -0
  53. package/dist/plan/scope.js +2 -2
  54. package/dist/route/preference.d.ts +20 -2
  55. package/dist/route/preference.js +48 -13
  56. package/dist/route/router.d.ts +12 -1
  57. package/dist/route/router.js +56 -24
  58. package/dist/run/consult.d.ts +15 -1
  59. package/dist/run/consult.js +18 -7
  60. package/dist/run/daemon.d.ts +38 -2
  61. package/dist/run/daemon.js +895 -192
  62. package/dist/run/git.d.ts +8 -0
  63. package/dist/run/git.js +14 -0
  64. package/dist/run/interactive-seed.d.ts +4 -0
  65. package/dist/run/interactive-seed.js +35 -9
  66. package/dist/run/journal.d.ts +152 -3
  67. package/dist/run/journal.js +551 -50
  68. package/dist/run/lease.d.ts +13 -0
  69. package/dist/run/lease.js +45 -0
  70. package/dist/run/merge.d.ts +3 -1
  71. package/dist/run/merge.js +3 -2
  72. package/dist/run/operator-summary.d.ts +3 -0
  73. package/dist/run/operator-summary.js +3 -1
  74. package/dist/run/protocol.d.ts +46 -1
  75. package/dist/run/protocol.js +14 -2
  76. package/dist/run/receipt-resolver.d.ts +22 -0
  77. package/dist/run/receipt-resolver.js +40 -1
  78. package/dist/run/repair-selection.d.ts +11 -1
  79. package/dist/run/repair-selection.js +17 -9
  80. package/dist/run/supervision.d.ts +7 -1
  81. package/dist/run/supervision.js +5 -2
  82. package/dist/run/wall-budget.d.ts +48 -0
  83. package/dist/run/wall-budget.js +280 -0
  84. package/dist/tui/cockpit/board.js +3 -3
  85. package/dist/tui/cockpit/decision-actions.d.ts +8 -5
  86. package/dist/tui/cockpit/decision-actions.js +55 -32
  87. package/dist/tui/cockpit/derive.js +13 -2
  88. package/dist/tui/cockpit/live-runtime.d.ts +10 -0
  89. package/dist/tui/cockpit/live-runtime.js +50 -3
  90. package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
  91. package/dist/tui/cockpit/run-cockpit.js +27 -2
  92. package/dist/tui/cockpit/run-view.d.ts +9 -2
  93. package/dist/tui/cockpit/run-view.js +66 -9
  94. package/dist/tui/cockpit/setup-cockpit.d.ts +6 -0
  95. package/dist/tui/cockpit/setup-cockpit.js +10 -3
  96. package/dist/tui/ink/fleet-app.d.ts +15 -3
  97. package/dist/tui/ink/fleet-app.js +91 -22
  98. package/package.json +3 -1
  99. package/schema/config.schema.json +825 -0
  100. package/skills/tickmarkr-loop/SKILL.md +15 -3
  101. package/skills/tickmarkr-overseer/SKILL.md +42 -0
  102. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +91 -0
  103. package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
  104. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
  105. package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/README.md CHANGED
@@ -134,7 +134,7 @@ tickmarkr status <runId> --oneline # compact snapshot, then exit
134
134
  tickmarkr status <runId> --watch # TTY: Run cockpit; non-TTY: line output
135
135
  tickmarkr status <runId> --watch --plain # preserved line/ANSI fallback, including on a TTY
136
136
  tickmarkr resume <runId> # continue an engagement from the local execution log
137
- tickmarkr approve <runId> <taskId> # append permission for a non-gate park; see below
137
+ tickmarkr approve <runId> <taskId> --park <line>@<ts> # decide the park status printed; see below
138
138
  tickmarkr report <runId> # cost/quality report
139
139
  tickmarkr report <runId> --md > feature.record.md # explicit file write beside your spec
140
140
  tickmarkr profile # show the learned routing profile
@@ -189,14 +189,27 @@ graph says “not comparable” and supplies no borrowed denominator. Historical
189
189
  does not establish current completion.
190
190
 
191
191
  The CLI twin for the same decision is
192
- `tickmarkr approve <runId> T2 --by operator --reason 'ready to proceed'`, followed by the
193
- receipt check and explicit resume above. Other parks have different permitted decisions:
192
+ `tickmarkr approve <runId> T2 --park <line>@<ts> --by operator --reason 'ready to proceed'`,
193
+ followed by the receipt check and explicit resume above. Every decision is bound to one park:
194
+ status, the park notice and the Run confirmation print its `<line>@<ts>` token (the park row's
195
+ physical journal line and timestamp), and `--park` names it — without `--park` the park open
196
+ when the command starts is bound. A waive is also bound to that park's failed gate, named
197
+ explicitly with `--gate` (`tickmarkr approve <runId> T2 --waive --park <line>@<ts> --gate review`).
198
+ A decision queued behind another approval is revalidated under serialization, and the daemon
199
+ revalidates it again before enactment: once a newer park opens, a stale token, an unbound row or a
200
+ mismatched gate is refused (the daemon journals `approval-refused`) and never waives the newer gate.
201
+ A failed task's recheck keeps working with its own bound token: status prints
202
+ `failed — T3 — failure <line>@<ts>`, and
203
+ `tickmarkr approve <runId> T3 --recheck --park <line>@<ts>` re-gates its landed commits with no
204
+ worker. Other parks have different permitted decisions:
194
205
 
195
206
  | Park | Decision and effect |
196
207
  |---|---|
197
208
  | Human gate / other non-gate park | Plain approve records permission to dispatch. |
198
209
  | Attempt cap | Plain approve grants a fresh attempt budget; prior routing exclusions remain. |
199
210
  | Infrastructure | Plain approve or `--recheck`; recheck reruns the declared battery and satisfies no gate. |
211
+ | Stall with a recorded `reapFailure` (the worker's process census was unreadable or had survivors) | `--recheck` bound to that park's token re-verifies the parked attempt's owned census: only an explicitly recorded empty survivors array (`[]`) releases its harvested commits to the declared battery, with no worker; a missing, unreadable or surviving census re-parks the stall (new token, still approve/recheck). Plain approve dispatches a worker. |
212
+ | Ordinary stall (no `reapFailure`) | Plain approve only — it dispatches a worker; `--recheck` refuses. |
200
213
  | Failed review gate | `--waive` satisfies only that identified gate; `--uphold` funds one fixed attempt carrying findings; `--recheck` reruns the battery. |
201
214
  | Other failed gate | `--waive` or `--recheck`; plain approve refuses. |
202
215
  | Tombstone / gate failure without identifying evidence | Diagnostic only; no invented decision. |
@@ -91,49 +91,78 @@ export function readCachedCatalog(repoRoot, opts = {}) {
91
91
  };
92
92
  }
93
93
  }
94
- function providerModels(modelsDev, preferred) {
95
- const providers = record(modelsDev);
96
- if (!providers)
97
- return [];
98
- const entries = Object.entries(providers);
99
- const preferredEntries = preferred
100
- ? entries.filter(([key, value]) => key === preferred || record(value)?.id === preferred)
101
- : [];
102
- // Price is provider-specific. A provider-qualified query must never borrow an identically named
103
- // model from another provider, where subscription and metered costs can differ materially.
104
- // D-OBS-11 follow-up: that rule presumes the hinted provider EXISTS in the catalog. A hint that
105
- // matches nothing (kimi's "moonshot" vs the catalog's "moonshotai"/"kimi-for-coding"; cursor has
106
- // no provider at all) used to ZERO the search space and blanket-uncover the whole adapter.
107
- // Fail open to the full scan instead — advisory evidence with a visible models.dev id beats none.
108
- const selected = preferred && preferredEntries.length > 0 ? preferredEntries : entries;
109
- return selected.flatMap(([providerKey, provider]) => {
110
- const models = record(record(provider)?.models);
111
- return models ? [{ providerKey, models }] : [];
112
- });
94
+ // OBS-1148: a model maker's own models.dev provider keys, in tie-break order. Reseller rows (bothub,
95
+ // agentrouter, greenpt …) carry another price or none, and JSON key order moved the pick on a plain
96
+ // cache refresh. ponytail: hand-kept list; a maker missing here only loses its tie to key order.
97
+ const FIRST_PARTY_PROVIDERS = ["anthropic", "openai", "google", "xai", "zai", "zhipuai", "alibaba", "moonshotai", "deepseek", "mistral"];
98
+ // OBS-1147: effort/mode tokens CLIs append to a model id (`-high`, `-thinking`, `-max-effort`) — the
99
+ // union of Fleet's own variant set (model-lints.ts) and LiveBench's. NEVER `preview`: a preview is
100
+ // its own client-side identity; only a LiveBench row may carry it as residue (OBS-1149).
101
+ const EFFORT_MODE_TOKENS = new Set(["max", "xhigh", "high", "medium", "low", "minimal", "none", "thinking", "auto", "effort", "fast"]);
102
+ /** Exact canonical id: case and the `.`/`:`/`_` separators only — version digits are never touched. */
103
+ const canonicalModelId = (id) => id.trim().toLowerCase().replace(/[.:_]+/g, "-");
104
+ /** `c-thinking-high` → [`c-thinking`, `c`]: longest base first, so a real `-thinking` model outranks `c`. */
105
+ function effortBases(id) {
106
+ const tokens = id.split("-");
107
+ const bases = [];
108
+ while (tokens.length > 1 && EFFORT_MODE_TOKENS.has(tokens[tokens.length - 1])) {
109
+ tokens.pop();
110
+ bases.push(tokens.join("-"));
111
+ }
112
+ return bases;
113
113
  }
114
+ const codeUnitOrder = (a, b) => (a < b ? -1 : a > b ? 1 : 0);
114
115
  function findModelsDevModel(catalog, provider, modelId) {
115
- // CLI namespaces prefix their catalog ids (kimi-code/k3 vs catalog key k3; omp's openai/gpt-4):
116
- // after exact key/id misses, retry with the bare segment after the last "/". Deterministic
117
- // provider order; first hit wins — acceptable for advisory evidence, never routing.
118
- const bare = modelId.includes("/") ? modelId.slice(modelId.lastIndexOf("/") + 1) : undefined;
119
- for (const { providerKey, models } of providerModels(catalog.modelsDev, provider)) {
120
- const direct = record(models[modelId]);
121
- if (direct)
122
- return { providerKey, models, recordId: modelId, model: direct };
123
- for (const [recordId, value] of Object.entries(models)) {
124
- const candidate = record(value);
125
- if (candidate?.id === modelId)
126
- return { providerKey, models, recordId, model: candidate };
127
- }
128
- }
129
- if (bare) {
130
- for (const { providerKey, models } of providerModels(catalog.modelsDev, provider)) {
131
- const direct = record(models[bare]);
132
- if (direct)
133
- return { providerKey, models, recordId: bare, model: direct };
116
+ const providers = Object.entries(record(catalog.modelsDev) ?? {});
117
+ const named = (name) => name
118
+ ? providers.filter(([key, value]) => key === name || record(value)?.id === name).map(([key]) => key)
119
+ : [];
120
+ // Price is provider-specific: a provider-qualified query never borrows an identically named model
121
+ // from another provider, where subscription and metered costs differ materially. The CLI's own
122
+ // namespace (`zai-coding-plan/glm-5.2`, `google/…`) is the most explicit qualification and outranks
123
+ // the tier vendor hint. D-OBS-11 follow-up: a qualifier naming NO catalog provider (kimi's "moonshot"
124
+ // vs "moonshotai"; cursor has none) fails open to the full scan rather than blanket-uncovering the
125
+ // adapter — advisory evidence with a visible models.dev id beats none.
126
+ const slash = modelId.indexOf("/");
127
+ const namespace = slash > 0 ? named(modelId.slice(0, slash)) : [];
128
+ const scope = namespace.length > 0 ? namespace : named(provider);
129
+ // Full id, then (inside its own namespace) the rest, then the bare segment after the last "/"
130
+ // (kimi-code/k3 vs catalog key k3); only after every exact canonical miss, the effort-stripped bases.
131
+ const exact = [...new Set([
132
+ modelId,
133
+ ...(namespace.length > 0 ? [modelId.slice(slash + 1)] : []),
134
+ ...(slash >= 0 ? [modelId.slice(modelId.lastIndexOf("/") + 1)] : []),
135
+ ].map(canonicalModelId))];
136
+ const candidates = [...new Set([...exact, ...exact.flatMap(effortBases)])];
137
+ // One pass, then a total order: candidate, first-party rank, provider key, record key — never JSON key order.
138
+ const firstParty = (key) => {
139
+ const rank = FIRST_PARTY_PROVIDERS.indexOf(key);
140
+ return rank < 0 ? FIRST_PARTY_PROVIDERS.length : rank;
141
+ };
142
+ let best;
143
+ let bestCandidate = candidates.length;
144
+ for (const [providerKey, value] of providers) {
145
+ if (scope.length > 0 && !scope.includes(providerKey))
146
+ continue;
147
+ for (const [recordId, raw] of Object.entries(record(record(value)?.models) ?? {})) {
148
+ const model = record(raw);
149
+ if (!model)
150
+ continue;
151
+ const candidate = Math.min(...[recordId, model.id]
152
+ .flatMap((id) => (typeof id === "string" ? [candidates.indexOf(canonicalModelId(id))] : []))
153
+ .filter((i) => i >= 0));
154
+ if (!Number.isFinite(candidate))
155
+ continue;
156
+ if (best && (candidate - bestCandidate
157
+ || firstParty(providerKey) - firstParty(best.providerKey)
158
+ || codeUnitOrder(providerKey, best.providerKey)
159
+ || codeUnitOrder(recordId, best.recordId)) >= 0)
160
+ continue;
161
+ best = { providerKey, recordId, model, matchedId: candidates[candidate] };
162
+ bestCandidate = candidate;
134
163
  }
135
164
  }
136
- return undefined;
165
+ return best;
137
166
  }
138
167
  function artificialAnalysisRows(value) {
139
168
  if (Array.isArray(value))
@@ -232,8 +261,14 @@ function assertUsableLiveBench(rows, categories) {
232
261
  // `-thinking-auto-medium-effort`); the highest effort is the model at its best. Rank by
233
262
  // hyphen-delimited token so `xhigh` never reads as `high`.
234
263
  const LIVEBENCH_EFFORT_RANK = { max: 5, xhigh: 4, high: 3, medium: 2, low: 1 };
235
- const LIVEBENCH_VARIANT_RE = /^-(?:max|xhigh|high|medium|low|thinking|auto|effort|fast)(?:-(?:max|xhigh|high|medium|low|thinking|auto|effort|fast))*$/;
236
- const liveBenchEffortRank = (residue) => residue.split("-").reduce((rank, token) => Math.max(rank, LIVEBENCH_EFFORT_RANK[token] ?? 0), 0);
264
+ // OBS-1149: LiveBench benchmarks some models only as their preview (`gemini-3.1-pro-preview-high`),
265
+ // so on THIS side `preview` is residue too. Every residue token must be a word: `.2` never is.
266
+ const isLiveBenchVariant = (residue) => residue.startsWith("-") && residue.slice(1).split("-").every((token) => token === "preview" || EFFORT_MODE_TOKENS.has(token));
267
+ /** A GA row always outranks a preview row of the same model; effort ranks within each. */
268
+ const liveBenchEffortRank = (residue) => {
269
+ const tokens = residue.split("-");
270
+ return (tokens.includes("preview") ? 0 : 10) + tokens.reduce((rank, token) => Math.max(rank, LIVEBENCH_EFFORT_RANK[token] ?? 0), 0);
271
+ };
237
272
  const liveBenchCategoryMean = (row, tasks) => {
238
273
  const scores = (Array.isArray(tasks) ? tasks : [])
239
274
  .map((task) => (typeof task === "string" ? finite(row[task]) : undefined))
@@ -255,13 +290,17 @@ function liveBenchIndex(value, identities) {
255
290
  continue;
256
291
  // Bare startsWith is a false-positive machine: `glm-5` would claim `glm-5.2`. The residue after
257
292
  // the fleet id must be empty or an effort suffix.
258
- const residue = wanted.map((identity) => model.startsWith(identity) ? model.slice(identity.length) : undefined)
259
- .find((rest) => rest === "" || LIVEBENCH_VARIANT_RE.test(rest ?? ""));
260
- if (residue === undefined)
293
+ const residues = wanted.map((identity) => model.startsWith(identity) ? model.slice(identity.length) : undefined);
294
+ const identity = residues.findIndex((rest) => rest === "" || isLiveBenchVariant(rest ?? ""));
295
+ if (identity < 0)
261
296
  continue;
262
- const rank = liveBenchEffortRank(residue);
263
- if (!best || rank > best.rank)
264
- best = { row, rank };
297
+ // Identities are listed most-specific first (the client's `gpt-5.6-high` before its stripped base
298
+ // `gpt-5.6`): a row reached through an earlier identity always beats one reached only through a
299
+ // later one, so a client's exact `-high` row is never outranked by a sibling `-low` row whose
300
+ // residue happens to carry an effort token. Effort rank only orders rows of the SAME identity.
301
+ const rank = liveBenchEffortRank(residues[identity] ?? "");
302
+ if (!best || identity < best.identity || (identity === best.identity && rank > best.rank))
303
+ best = { row, identity, rank };
265
304
  }
266
305
  if (!best)
267
306
  return undefined;
@@ -293,10 +332,13 @@ export function resolveCatalogModel(catalog, query) {
293
332
  modelId,
294
333
  query.model,
295
334
  ...[modelId, query.model].flatMap((identity) => identity.includes("/") ? [identity.slice(identity.lastIndexOf("/") + 1)] : []),
335
+ // OBS-1147/1082: the exact base an effort variant resolved to (`gemini-3-pro-high` → `gemini-3-pro`).
336
+ match.matchedId,
296
337
  ...(typeof model.name === "string" ? [model.name] : []),
297
338
  ];
298
339
  const intelligence = artificialAnalysisIndex(catalog.artificialAnalysis, [
299
340
  modelId,
341
+ match.matchedId,
300
342
  ...(typeof model.name === "string" ? [model.name] : []),
301
343
  ], query.provider);
302
344
  const liveBench = liveBenchIndex(catalog.liveBench, identities);
@@ -210,6 +210,9 @@ export function probeVersion(bin) {
210
210
  note: "auth assumed; verified at dispatch (failover on auth/quota errors)",
211
211
  };
212
212
  }
213
+ // OBS-1182: an absent effort renders nothing, so the CLI keeps its own default. Every builder places
214
+ // it right after --model's value and before another flag, never before the prompt positional.
215
+ const claudeEffortFlag = (effort) => (effort ? ` --effort ${shq(effort)}` : "");
213
216
  export const claudeCode = {
214
217
  id: "claude-code",
215
218
  vendor: "anthropic",
@@ -218,7 +221,7 @@ export const claudeCode = {
218
221
  channels: (cfg) => channelsFromConfig("claude-code", cfg),
219
222
  // v1.65 T3: every flag the command builders below hardcode — doctor checks `claude --help` still
220
223
  // lists each (all present on claude 2.x, verified 2026-07-22). Advisory only, never routing.
221
- hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions", "--settings"] },
224
+ hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions", "--settings", "--effort"] },
222
225
  // --strict-mcp-config --mcp-config '{"mcpServers":{}}': pin the MCP surface to empty so fresh-worktree
223
226
  // workers/gates don't load project .mcp.json servers (herdr scrapes dialogs as idle — v1.4 incident,
224
227
  // memory tickmarkr-worker-mcp-dialog-stall). Live-verified 2026-07-10 on claude 2.1.205 (operator check):
@@ -230,7 +233,7 @@ export const claudeCode = {
230
233
  // flag must always follow the value, never the prompt.
231
234
  // The empty -p argument selects print mode while stdin carries the prompt, keeping its nonce out
232
235
  // of process argv. The redirect path is shell-quoted independently from the model.
233
- headlessCommand: (promptFile, model) => `claude -p '' --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text < ${shq(promptFile)}`,
236
+ headlessCommand: (promptFile, model, effort) => `claude -p '' --model ${shq(model)}${claudeEffortFlag(effort)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text < ${shq(promptFile)}`,
234
237
  // HYG-03 / OBS-137: the residual first-entry dialog is workspace trust, not MCP config loading.
235
238
  // Claude's only store is global last-writer-wins ~/.claude.json, so tickmarkr still does not seed it;
236
239
  // the daemon safely answers only the exact adapter-declared dialog once per slot.
@@ -250,16 +253,16 @@ export const claudeCode = {
250
253
  // OBS-931: the same ONE-argv-string hazard as codex (OBS-930) — over promptArgvCeiling() the TUI
251
254
  // launch would E2BIG on Linux, so it returns null → worker-mode-fallback → the headless form.
252
255
  // resumeCommand keeps the shape: its contract returns a string (composer delivery is 2.4.3 work).
253
- interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
254
- ? `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`
256
+ interactiveCommand: (promptFile, model, effort) => promptFitsArgv(promptFile)
257
+ ? `claude --model ${shq(model)}${claudeEffortFlag(effort)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`
255
258
  : null,
256
259
  trustDialog: CLAUDE_TRUST_DIALOG,
257
260
  inputBox: CLAUDE_INPUT_BOX,
258
261
  // A resumed attempt lands in the same painted editor, so it carries the same ghost-text suppression
259
262
  // and the same value-then-flag placement.
260
- resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
263
+ resumeCommand: (sessionId, promptFile, model, effort) => `claude -r ${shq(sessionId)} --model ${shq(model)}${claudeEffortFlag(effort)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
261
264
  invoke(task, _cwd, a, ctx) {
262
- return { command: this.headlessCommand(ctx.promptFile, a.model) };
265
+ return { command: this.headlessCommand(ctx.promptFile, a.model, a.effort) };
263
266
  },
264
267
  parse: parseWorkerResult,
265
268
  collectUsage(cwd, sinceMs) {
@@ -89,6 +89,9 @@ const GITDIR_WRITABLE = `-c "sandbox_workspace_write.writable_roots=[\\"$(git re
89
89
  // -s/--sandbox workspace-write sandbox (deliberately NOT --dangerously-bypass-approvals-and-sandbox,
90
90
  // which would drop the sandbox). Listed by `codex --help` and `codex exec --help` (verified 2026-07-23).
91
91
  const CODEX_HOOK_TRUST = "--dangerously-bypass-hook-trust";
92
+ // OBS-1182: config override, not a flag — codex has no --effort. Absent renders nothing (CLI default);
93
+ // placed before --model so --model's value stays the last flag before the prompt, as it always was.
94
+ const codexEffortFlag = (effort) => (effort ? ` -c ${shq(`model_reasoning_effort=${effort}`)}` : "");
92
95
  // v1.75 T2 / OBS-137: current Codex workspace-trust prompt (0.144.6). The exact heading
93
96
  // is distinct from normal agent output; Enter accepts the selected "Yes, continue" option.
94
97
  export const CODEX_TRUST_DIALOG = {
@@ -240,7 +243,7 @@ export const codex = {
240
243
  // --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
241
244
  // MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
242
245
  // CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
243
- headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
246
+ headlessCommand: (promptFile, model, effort) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE}${codexEffortFlag(effort)} --model ${shq(model)} - < ${shq(promptFile)}`,
244
247
  // OBS-930: the visible pane runs the REAL TUI. Codex's TUI takes its prompt only as the [PROMPT]
245
248
  // positional (`codex --help`, 0.153.4 — no file/stdin form), so the launch inlines the file exactly
246
249
  // as the claude adapter does: the prompt is the LAST positional and every flag value is followed by
@@ -251,11 +254,11 @@ export const codex = {
251
254
  // autonomous approval policy (exec has no approvals to configure).
252
255
  // OBS-930 (Linux): the inlined prompt is ONE argv string and Linux caps one at 131072 bytes, so a
253
256
  // prompt over promptArgvCeiling() returns null → worker-mode-fallback → the headless form (types.ts).
254
- interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
255
- ? `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`
257
+ interactiveCommand: (promptFile, model, effort) => promptFitsArgv(promptFile)
258
+ ? `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE}${codexEffortFlag(effort)} --model ${shq(model)} "$(cat ${shq(promptFile)})"`
256
259
  : null,
257
260
  invoke(task, _cwd, a, ctx) {
258
- return { command: this.headlessCommand(ctx.promptFile, a.model) };
261
+ return { command: this.headlessCommand(ctx.promptFile, a.model, a.effort) };
259
262
  },
260
263
  parse: parseWorkerResult,
261
264
  // v1.22 T5: seed [projects."<repoRoot>"] trust_level="trusted" so fresh worktrees never stall on
@@ -13,3 +13,4 @@ export declare function parseWorkerResult(raw: string, nonce: string): Classifie
13
13
  export type DeadChannelReason = "auth-required" | "setup-required" | "provider-outage" | "timeout";
14
14
  export declare function classifyTransientCapacity(result: WorkerResult): RegExpExecArray | null;
15
15
  export declare function classifyDeadChannel(result: WorkerResult): DeadChannelReason | undefined;
16
+ export declare const BOOTSTRAP_FAILURE_RE: RegExp;
@@ -68,8 +68,9 @@ function workerVerdictWitness(raw, nonce, positions) {
68
68
  return raw;
69
69
  return `{"nonce":${JSON.stringify(nonce)},"ok":false,`;
70
70
  }
71
- // v1.65 T1: the parse boundary's own no-trailer sentinel summaries. classifyDeadChannel keys on
72
- // these — a result carrying any other summary is a PARSED trailer, i.e. the worker speaking.
71
+ // v1.65 T1: the parse boundary's own no-trailer sentinel summaries — display text only. OBS-1175:
72
+ // classifiers key on the parser's `cause`, never on these: a PARSED trailer may carry either string
73
+ // as its own summary, and it is still the worker speaking.
73
74
  export const NO_TRAILER_SUMMARY = "worker produced no TICKMARKR_RESULT trailer";
74
75
  export const UNPARSEABLE_TRAILER_SUMMARY = "unparseable TICKMARKR_RESULT trailer";
75
76
  // OBS-1062: every interactive TUI echoes the brief into the pane, and the brief carries the trailer
@@ -139,14 +140,12 @@ const TIMEOUT_RE = /\bETIMEDOUT\b|request timed out|connection timed out|deadlin
139
140
  // speaking, so a verdict that QUOTES the capacity phrase stays ordinary work evidence. The caller
140
141
  // hands in the chrome-filtered banner rows as `raw`, exactly as it does for classifyDeadChannel.
141
142
  export function classifyTransientCapacity(result) {
142
- if (result.ok || (result.summary !== NO_TRAILER_SUMMARY && result.summary !== UNPARSEABLE_TRAILER_SUMMARY))
143
- return null;
144
- return CAPACITY_RE.exec(result.raw);
143
+ return unparsed(result) ? CAPACITY_RE.exec(result.raw) : null;
145
144
  }
146
145
  export function classifyDeadChannel(result) {
147
146
  // A parsed trailer — ok:true OR ok:false — is the worker speaking: genuine work outcomes walk
148
147
  // the normal gate/ladder path even when their transcript mentions auth/outage/timeout text.
149
- if (result.ok || (result.summary !== NO_TRAILER_SUMMARY && result.summary !== UNPARSEABLE_TRAILER_SUMMARY))
148
+ if (!unparsed(result))
150
149
  return undefined;
151
150
  if (AUTH_RE.test(result.raw))
152
151
  return "auth-required";
@@ -158,3 +157,12 @@ export function classifyDeadChannel(result) {
158
157
  return "timeout";
159
158
  return undefined;
160
159
  }
160
+ // OBS-1175: the one "no trailer was parsed" test every classifier here shares.
161
+ function unparsed(result) {
162
+ return !result.ok && result.cause !== undefined;
163
+ }
164
+ // OBS-1169: a CLI that died in its own bootstrap (codex: "account/read failed during TUI bootstrap …
165
+ // (code -32603)") never read the brief. Known bootstrap phrasing only — evidence solely inside the
166
+ // startup prefix the daemon's StartupFailureDetector owns (before any tool frame or input box, in an
167
+ // unsaturated read, inside the startup window); the same words in a worker's tool output are work.
168
+ export const BOOTSTRAP_FAILURE_RE = /\b[\w/-]+ failed during TUI bootstrap\b/i;
@@ -5,7 +5,7 @@ import { join } from "node:path";
5
5
  import { modelLints, suggestOverlay } from "./model-lints.js";
6
6
  import { HerdrDriver } from "../drivers/herdr.js";
7
7
  import { tickmarkrDir, stateDirName } from "../graph/graph.js";
8
- import { disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../route/preference.js";
8
+ import { disallowedBy, excludedChannels, exclusionLine, observedIdentity, observedSeat, preferRanks } from "../route/preference.js";
9
9
  import { sh } from "../run/git.js";
10
10
  import { FakeAdapter } from "./fake.js";
11
11
  import { parseWorkerResult } from "./prompt.js";
@@ -581,7 +581,7 @@ export function discoverChannels(cfg, adapters, health, role = "worker") {
581
581
  return a.channels(cfg)
582
582
  .filter((c) => !invalid.has(c.model) && modelAuthed(h, c.model, cfg.routing.allowUnverifiedModels) && (!s || s.includes(c.model)))
583
583
  .map((c) => {
584
- const identity = h?.modelAuth?.[c.model]?.identity ?? h?.modelIdentities?.[c.model];
584
+ const identity = observedIdentity(health, a.id, c.model);
585
585
  return identity ? { ...c, identity } : c;
586
586
  });
587
587
  });
@@ -736,7 +736,7 @@ export function formatDoctorReport(cwd, cfg, health, adapters, opts = {}) {
736
736
  for (const m of models) {
737
737
  const v = h.modelAuth?.[m];
738
738
  const auth = !v ? "unknown" : v.authed ? "authed" : `unauthed: ${trunc(v.reason ?? "probe failed", 40)} (${dateOf(v.probedAt)})`;
739
- const d = disallowedBy({ adapter: a.id, model: m }, cfg.routing);
739
+ const d = disallowedBy(observedSeat(health, a.id, m), cfg.routing);
740
740
  const denied = d?.by === "deny" ? d.entry : "—";
741
741
  const pref = preferRanks({ adapter: a.id, model: m }, cfg).map((p) => `${p.shape}#${p.rank}`).join(",") || "—";
742
742
  statusRows.push(` ${m.padEnd(w)} ${classified[m].padEnd(8)} ${auth} denied=${denied} prefer=${pref}`);
@@ -1,6 +1,7 @@
1
1
  import { z } from "zod";
2
2
  import type { TickmarkrConfig, Tier } from "../config/config.js";
3
- import type { Task } from "../graph/schema.js";
3
+ import type { Effort, Task } from "../graph/schema.js";
4
+ import type { VerdictUnparseableCause } from "../gates/verdict-cause.js";
4
5
  export declare const TokenUsageSchema: z.ZodObject<{
5
6
  input: z.ZodNumber;
6
7
  output: z.ZodNumber;
@@ -15,6 +16,7 @@ export interface Assignment {
15
16
  model: string;
16
17
  channel: "sub" | "api";
17
18
  tier: Tier;
19
+ effort?: Effort;
18
20
  }
19
21
  export interface BillingChannel {
20
22
  adapter: string;
@@ -23,6 +25,7 @@ export interface BillingChannel {
23
25
  channel: "sub" | "api";
24
26
  tier: Tier;
25
27
  identity?: string;
28
+ effort?: Effort;
26
29
  }
27
30
  export declare const MODEL_PROBE_ERRORS: readonly ["EMFILE", "EAGAIN", "ENFILE", "ENOMEM", "ENOSPC"];
28
31
  export type ModelProbeError = typeof MODEL_PROBE_ERRORS[number];
@@ -53,6 +56,7 @@ export interface WorkerResult {
53
56
  summary: string;
54
57
  deviations: string[];
55
58
  raw: string;
59
+ cause?: VerdictUnparseableCause;
56
60
  }
57
61
  export type SeedBannerConfirmResult = {
58
62
  ok: true;
@@ -173,11 +177,11 @@ export interface WorkerAdapter {
173
177
  probeConcurrency?: number;
174
178
  probe(): Promise<AuthHealth>;
175
179
  channels(cfg: TickmarkrConfig): BillingChannel[];
176
- headlessCommand(promptFile: string, model: string): string;
180
+ headlessCommand(promptFile: string, model: string, effort?: Effort): string;
177
181
  harnessBannerRows?: readonly string[];
178
- interactiveCommand(promptFile: string, model: string): string | null;
182
+ interactiveCommand(promptFile: string, model: string, effort?: Effort): string | null;
179
183
  interactiveSeed?: InteractiveSeed;
180
- resumeCommand?(sessionId: string, promptFile: string, model: string): string;
184
+ resumeCommand?(sessionId: string, promptFile: string, model: string, effort?: Effort): string;
181
185
  sessionIdFrom?(output: string): string | undefined;
182
186
  resumeUnknownContext?: boolean;
183
187
  invoke(task: Task, cwd: string, a: Assignment, ctx: {
@@ -201,6 +205,10 @@ export interface WorkerAdapter {
201
205
  };
202
206
  }
203
207
  export declare function channelsFromConfig(adapterId: string, cfg: TickmarkrConfig): BillingChannel[];
208
+ export declare function configuredEffort(cfg: TickmarkrConfig, seat: {
209
+ adapter: string;
210
+ model: string;
211
+ }): Effort | undefined;
204
212
  export declare function channelKey(c: {
205
213
  adapter: string;
206
214
  model: string;
@@ -249,9 +249,15 @@ export function channelsFromConfig(adapterId, cfg) {
249
249
  model,
250
250
  channel: override?.channel ?? e.channel,
251
251
  tier,
252
+ ...(override?.effort ? { effort: override.effort } : {}),
252
253
  }];
253
254
  });
254
255
  }
256
+ // OBS-1182: a config-declared seat's launch effort (judge, consult pin) — the same modelOverrides
257
+ // read channelsFromConfig makes for routed channels. Absent = the CLI's own default.
258
+ export function configuredEffort(cfg, seat) {
259
+ return cfg.tiers[seat.adapter]?.modelOverrides?.[seat.model]?.effort;
260
+ }
255
261
  export function channelKey(c) {
256
262
  return `${c.adapter}:${c.model}`;
257
263
  }
@@ -1,4 +1,4 @@
1
- import { Journal, type JournalEvent } from "../../run/journal.js";
1
+ import { Journal, type DecisionBinding, type JournalEvent } from "../../run/journal.js";
2
2
  export declare const APPROVAL_DISPOSITIONS: readonly ["dispatch", "waive-gate", "re-dispatch", "fund-fixed-attempt", "fresh-budget"];
3
3
  export type ApprovalDisposition = (typeof APPROVAL_DISPOSITIONS)[number];
4
4
  /**
@@ -26,6 +26,8 @@ export interface NewestPark {
26
26
  failedGate: string | undefined;
27
27
  /** A pre-dispatch human gate whose reason marks it permanent by design (see isTombstonePark). */
28
28
  tombstone: boolean;
29
+ /** OBS-1202: a stall park's recorded reap failure (task-human data.reapFailure) — its census is unproven. */
30
+ reapFailure?: string;
29
31
  }
30
32
  /**
31
33
  * There is no closed park kind for a declaration-shaped retirement, so the evidence is the one the
@@ -41,14 +43,19 @@ export declare function readJournalEvents(journal: Journal): {
41
43
  events: JournalEvent[];
42
44
  sourceIndexes: number[];
43
45
  };
46
+ /** OBS-1178: the token a park is decided by — what status, park notices and the cockpit print. */
47
+ export declare const parkToken: (park: Pick<NewestPark, "line" | "ts">) => string | undefined;
48
+ /** OBS-1178: the newest task-failed row for a task — the failure a failed-task recheck binds to. */
49
+ export declare function newestFailure(events: readonly JournalEvent[], taskId: string, sourceIndexes?: readonly number[]): DecisionBinding | undefined;
44
50
  /**
45
- * FINAL §3.3's decision menu as data: human-gate/attempt-cap/other non-gate parks → approve; infra →
46
- * approve or recheck; review gate-fail → waive/uphold/recheck; other gate-fail → waive/recheck; a
51
+ * FINAL §3.3's decision menu as data: human-gate/attempt-cap/other non-gate parks → approve; infra, and
52
+ * a stall park that recorded a reapFailure (OBS-1202), → approve or recheck; review gate-fail →
53
+ * waive/uphold/recheck; other gate-fail → waive/recheck; a
47
54
  * gate-fail park with no failed-gate evidence, or a tombstone → nothing (a diagnostic, never a
48
55
  * fabricated verb). The refusals in `approve` below enforce the same table; this is the one place a
49
56
  * surface may read it from, so what a menu offers and what the command accepts cannot drift.
50
57
  */
51
- export declare function permittedDecisionVerbs(park: Pick<NewestPark, "kind" | "failedGate" | "tombstone"> | undefined): readonly DecisionVerb[];
58
+ export declare function permittedDecisionVerbs(park: Pick<NewestPark, "kind" | "failedGate" | "tombstone" | "reapFailure"> | undefined): readonly DecisionVerb[];
52
59
  /** The release marker this command appends for a verb on a park — the fact a read-back must match. */
53
60
  export declare function releaseForDecision(verb: DecisionVerb, park: Pick<NewestPark, "kind" | "failedGate">): string | undefined;
54
61
  export type ApprovalStatus = "deferred-live" | "recorded-no-owner";