tickmarkr 2.6.1 → 2.6.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -3
- package/dist/adapters/catalog-remote.js +89 -47
- package/dist/adapters/claude-code.js +9 -6
- package/dist/adapters/codex.js +7 -4
- package/dist/adapters/prompt.d.ts +1 -0
- package/dist/adapters/prompt.js +14 -6
- package/dist/adapters/registry.js +3 -3
- package/dist/adapters/types.d.ts +12 -4
- package/dist/adapters/types.js +6 -0
- package/dist/cli/commands/approve.d.ts +11 -4
- package/dist/cli/commands/approve.js +82 -27
- package/dist/cli/commands/compile.js +13 -3
- package/dist/cli/commands/doctor.d.ts +8 -2
- package/dist/cli/commands/doctor.js +11 -3
- package/dist/cli/commands/fleet.js +87 -11
- package/dist/cli/commands/plan.js +13 -8
- package/dist/cli/commands/report.d.ts +2 -1
- package/dist/cli/commands/report.js +74 -8
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/status.js +43 -20
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +9 -2
- package/dist/compile/native.js +7 -0
- package/dist/config/config.d.ts +35 -2
- package/dist/config/config.js +86 -10
- package/dist/config/fleet-overlay.d.ts +13 -2
- package/dist/config/fleet-overlay.js +60 -0
- package/dist/drivers/herdr.d.ts +12 -0
- package/dist/drivers/herdr.js +51 -0
- package/dist/drivers/orca.d.ts +35 -2
- package/dist/drivers/orca.js +222 -67
- package/dist/drivers/types.d.ts +2 -0
- package/dist/drivers/types.js +2 -2
- package/dist/eval/canary.d.ts +2 -1
- package/dist/eval/canary.js +2 -2
- package/dist/eval/dispatch.js +1 -0
- package/dist/gates/acceptance.d.ts +9 -1
- package/dist/gates/acceptance.js +31 -4
- package/dist/gates/baseline.d.ts +32 -2
- package/dist/gates/baseline.js +111 -24
- package/dist/gates/cache.d.ts +8 -0
- package/dist/gates/cache.js +12 -2
- package/dist/gates/llm.d.ts +11 -4
- package/dist/gates/llm.js +40 -21
- package/dist/gates/review.d.ts +14 -1
- package/dist/gates/review.js +160 -34
- package/dist/gates/run-gates.d.ts +56 -4
- package/dist/gates/run-gates.js +358 -58
- package/dist/gates/test-manifest.d.ts +45 -1
- package/dist/gates/test-manifest.js +78 -12
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/plan/scope.js +2 -2
- package/dist/route/preference.d.ts +20 -2
- package/dist/route/preference.js +48 -13
- package/dist/route/router.d.ts +12 -1
- package/dist/route/router.js +56 -24
- package/dist/run/consult.d.ts +15 -1
- package/dist/run/consult.js +18 -7
- package/dist/run/daemon.d.ts +38 -2
- package/dist/run/daemon.js +895 -192
- package/dist/run/git.d.ts +8 -0
- package/dist/run/git.js +14 -0
- package/dist/run/interactive-seed.d.ts +4 -0
- package/dist/run/interactive-seed.js +35 -9
- package/dist/run/journal.d.ts +152 -3
- package/dist/run/journal.js +551 -50
- package/dist/run/lease.d.ts +13 -0
- package/dist/run/lease.js +45 -0
- package/dist/run/merge.d.ts +3 -1
- package/dist/run/merge.js +3 -2
- package/dist/run/operator-summary.d.ts +3 -0
- package/dist/run/operator-summary.js +3 -1
- package/dist/run/protocol.d.ts +46 -1
- package/dist/run/protocol.js +14 -2
- package/dist/run/receipt-resolver.d.ts +22 -0
- package/dist/run/receipt-resolver.js +40 -1
- package/dist/run/repair-selection.d.ts +11 -1
- package/dist/run/repair-selection.js +17 -9
- package/dist/run/supervision.d.ts +7 -1
- package/dist/run/supervision.js +5 -2
- package/dist/run/wall-budget.d.ts +48 -0
- package/dist/run/wall-budget.js +280 -0
- package/dist/tui/cockpit/board.js +3 -3
- package/dist/tui/cockpit/decision-actions.d.ts +8 -5
- package/dist/tui/cockpit/decision-actions.js +55 -32
- package/dist/tui/cockpit/derive.js +13 -2
- package/dist/tui/cockpit/live-runtime.d.ts +10 -0
- package/dist/tui/cockpit/live-runtime.js +50 -3
- package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
- package/dist/tui/cockpit/run-cockpit.js +27 -2
- package/dist/tui/cockpit/run-view.d.ts +9 -2
- package/dist/tui/cockpit/run-view.js +66 -9
- package/dist/tui/cockpit/setup-cockpit.d.ts +6 -0
- package/dist/tui/cockpit/setup-cockpit.js +10 -3
- package/dist/tui/ink/fleet-app.d.ts +15 -3
- package/dist/tui/ink/fleet-app.js +91 -22
- package/package.json +3 -1
- package/schema/config.schema.json +825 -0
- package/skills/tickmarkr-loop/SKILL.md +15 -3
- package/skills/tickmarkr-overseer/SKILL.md +42 -0
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +91 -0
- package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/README.md
CHANGED
|
@@ -134,7 +134,7 @@ tickmarkr status <runId> --oneline # compact snapshot, then exit
|
|
|
134
134
|
tickmarkr status <runId> --watch # TTY: Run cockpit; non-TTY: line output
|
|
135
135
|
tickmarkr status <runId> --watch --plain # preserved line/ANSI fallback, including on a TTY
|
|
136
136
|
tickmarkr resume <runId> # continue an engagement from the local execution log
|
|
137
|
-
tickmarkr approve <runId> <taskId> #
|
|
137
|
+
tickmarkr approve <runId> <taskId> --park <line>@<ts> # decide the park status printed; see below
|
|
138
138
|
tickmarkr report <runId> # cost/quality report
|
|
139
139
|
tickmarkr report <runId> --md > feature.record.md # explicit file write beside your spec
|
|
140
140
|
tickmarkr profile # show the learned routing profile
|
|
@@ -189,14 +189,27 @@ graph says “not comparable” and supplies no borrowed denominator. Historical
|
|
|
189
189
|
does not establish current completion.
|
|
190
190
|
|
|
191
191
|
The CLI twin for the same decision is
|
|
192
|
-
`tickmarkr approve <runId> T2 --by operator --reason 'ready to proceed'`,
|
|
193
|
-
receipt check and explicit resume above.
|
|
192
|
+
`tickmarkr approve <runId> T2 --park <line>@<ts> --by operator --reason 'ready to proceed'`,
|
|
193
|
+
followed by the receipt check and explicit resume above. Every decision is bound to one park:
|
|
194
|
+
status, the park notice and the Run confirmation print its `<line>@<ts>` token (the park row's
|
|
195
|
+
physical journal line and timestamp), and `--park` names it — without `--park` the park open
|
|
196
|
+
when the command starts is bound. A waive is also bound to that park's failed gate, named
|
|
197
|
+
explicitly with `--gate` (`tickmarkr approve <runId> T2 --waive --park <line>@<ts> --gate review`).
|
|
198
|
+
A decision queued behind another approval is revalidated under serialization, and the daemon
|
|
199
|
+
revalidates it again before enactment: once a newer park opens, a stale token, an unbound row or a
|
|
200
|
+
mismatched gate is refused (the daemon journals `approval-refused`) and never waives the newer gate.
|
|
201
|
+
A failed task's recheck keeps working with its own bound token: status prints
|
|
202
|
+
`failed — T3 — failure <line>@<ts>`, and
|
|
203
|
+
`tickmarkr approve <runId> T3 --recheck --park <line>@<ts>` re-gates its landed commits with no
|
|
204
|
+
worker. Other parks have different permitted decisions:
|
|
194
205
|
|
|
195
206
|
| Park | Decision and effect |
|
|
196
207
|
|---|---|
|
|
197
208
|
| Human gate / other non-gate park | Plain approve records permission to dispatch. |
|
|
198
209
|
| Attempt cap | Plain approve grants a fresh attempt budget; prior routing exclusions remain. |
|
|
199
210
|
| Infrastructure | Plain approve or `--recheck`; recheck reruns the declared battery and satisfies no gate. |
|
|
211
|
+
| Stall with a recorded `reapFailure` (the worker's process census was unreadable or had survivors) | `--recheck` bound to that park's token re-verifies the parked attempt's owned census: only an explicitly recorded empty survivors array (`[]`) releases its harvested commits to the declared battery, with no worker; a missing, unreadable or surviving census re-parks the stall (new token, still approve/recheck). Plain approve dispatches a worker. |
|
|
212
|
+
| Ordinary stall (no `reapFailure`) | Plain approve only — it dispatches a worker; `--recheck` refuses. |
|
|
200
213
|
| Failed review gate | `--waive` satisfies only that identified gate; `--uphold` funds one fixed attempt carrying findings; `--recheck` reruns the battery. |
|
|
201
214
|
| Other failed gate | `--waive` or `--recheck`; plain approve refuses. |
|
|
202
215
|
| Tombstone / gate failure without identifying evidence | Diagnostic only; no invented decision. |
|
|
@@ -91,49 +91,78 @@ export function readCachedCatalog(repoRoot, opts = {}) {
|
|
|
91
91
|
};
|
|
92
92
|
}
|
|
93
93
|
}
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
94
|
+
// OBS-1148: a model maker's own models.dev provider keys, in tie-break order. Reseller rows (bothub,
|
|
95
|
+
// agentrouter, greenpt …) carry another price or none, and JSON key order moved the pick on a plain
|
|
96
|
+
// cache refresh. ponytail: hand-kept list; a maker missing here only loses its tie to key order.
|
|
97
|
+
const FIRST_PARTY_PROVIDERS = ["anthropic", "openai", "google", "xai", "zai", "zhipuai", "alibaba", "moonshotai", "deepseek", "mistral"];
|
|
98
|
+
// OBS-1147: effort/mode tokens CLIs append to a model id (`-high`, `-thinking`, `-max-effort`) — the
|
|
99
|
+
// union of Fleet's own variant set (model-lints.ts) and LiveBench's. NEVER `preview`: a preview is
|
|
100
|
+
// its own client-side identity; only a LiveBench row may carry it as residue (OBS-1149).
|
|
101
|
+
const EFFORT_MODE_TOKENS = new Set(["max", "xhigh", "high", "medium", "low", "minimal", "none", "thinking", "auto", "effort", "fast"]);
|
|
102
|
+
/** Exact canonical id: case and the `.`/`:`/`_` separators only — version digits are never touched. */
|
|
103
|
+
const canonicalModelId = (id) => id.trim().toLowerCase().replace(/[.:_]+/g, "-");
|
|
104
|
+
/** `c-thinking-high` → [`c-thinking`, `c`]: longest base first, so a real `-thinking` model outranks `c`. */
|
|
105
|
+
function effortBases(id) {
|
|
106
|
+
const tokens = id.split("-");
|
|
107
|
+
const bases = [];
|
|
108
|
+
while (tokens.length > 1 && EFFORT_MODE_TOKENS.has(tokens[tokens.length - 1])) {
|
|
109
|
+
tokens.pop();
|
|
110
|
+
bases.push(tokens.join("-"));
|
|
111
|
+
}
|
|
112
|
+
return bases;
|
|
113
113
|
}
|
|
114
|
+
const codeUnitOrder = (a, b) => (a < b ? -1 : a > b ? 1 : 0);
|
|
114
115
|
function findModelsDevModel(catalog, provider, modelId) {
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
116
|
+
const providers = Object.entries(record(catalog.modelsDev) ?? {});
|
|
117
|
+
const named = (name) => name
|
|
118
|
+
? providers.filter(([key, value]) => key === name || record(value)?.id === name).map(([key]) => key)
|
|
119
|
+
: [];
|
|
120
|
+
// Price is provider-specific: a provider-qualified query never borrows an identically named model
|
|
121
|
+
// from another provider, where subscription and metered costs differ materially. The CLI's own
|
|
122
|
+
// namespace (`zai-coding-plan/glm-5.2`, `google/…`) is the most explicit qualification and outranks
|
|
123
|
+
// the tier vendor hint. D-OBS-11 follow-up: a qualifier naming NO catalog provider (kimi's "moonshot"
|
|
124
|
+
// vs "moonshotai"; cursor has none) fails open to the full scan rather than blanket-uncovering the
|
|
125
|
+
// adapter — advisory evidence with a visible models.dev id beats none.
|
|
126
|
+
const slash = modelId.indexOf("/");
|
|
127
|
+
const namespace = slash > 0 ? named(modelId.slice(0, slash)) : [];
|
|
128
|
+
const scope = namespace.length > 0 ? namespace : named(provider);
|
|
129
|
+
// Full id, then (inside its own namespace) the rest, then the bare segment after the last "/"
|
|
130
|
+
// (kimi-code/k3 vs catalog key k3); only after every exact canonical miss, the effort-stripped bases.
|
|
131
|
+
const exact = [...new Set([
|
|
132
|
+
modelId,
|
|
133
|
+
...(namespace.length > 0 ? [modelId.slice(slash + 1)] : []),
|
|
134
|
+
...(slash >= 0 ? [modelId.slice(modelId.lastIndexOf("/") + 1)] : []),
|
|
135
|
+
].map(canonicalModelId))];
|
|
136
|
+
const candidates = [...new Set([...exact, ...exact.flatMap(effortBases)])];
|
|
137
|
+
// One pass, then a total order: candidate, first-party rank, provider key, record key — never JSON key order.
|
|
138
|
+
const firstParty = (key) => {
|
|
139
|
+
const rank = FIRST_PARTY_PROVIDERS.indexOf(key);
|
|
140
|
+
return rank < 0 ? FIRST_PARTY_PROVIDERS.length : rank;
|
|
141
|
+
};
|
|
142
|
+
let best;
|
|
143
|
+
let bestCandidate = candidates.length;
|
|
144
|
+
for (const [providerKey, value] of providers) {
|
|
145
|
+
if (scope.length > 0 && !scope.includes(providerKey))
|
|
146
|
+
continue;
|
|
147
|
+
for (const [recordId, raw] of Object.entries(record(record(value)?.models) ?? {})) {
|
|
148
|
+
const model = record(raw);
|
|
149
|
+
if (!model)
|
|
150
|
+
continue;
|
|
151
|
+
const candidate = Math.min(...[recordId, model.id]
|
|
152
|
+
.flatMap((id) => (typeof id === "string" ? [candidates.indexOf(canonicalModelId(id))] : []))
|
|
153
|
+
.filter((i) => i >= 0));
|
|
154
|
+
if (!Number.isFinite(candidate))
|
|
155
|
+
continue;
|
|
156
|
+
if (best && (candidate - bestCandidate
|
|
157
|
+
|| firstParty(providerKey) - firstParty(best.providerKey)
|
|
158
|
+
|| codeUnitOrder(providerKey, best.providerKey)
|
|
159
|
+
|| codeUnitOrder(recordId, best.recordId)) >= 0)
|
|
160
|
+
continue;
|
|
161
|
+
best = { providerKey, recordId, model, matchedId: candidates[candidate] };
|
|
162
|
+
bestCandidate = candidate;
|
|
134
163
|
}
|
|
135
164
|
}
|
|
136
|
-
return
|
|
165
|
+
return best;
|
|
137
166
|
}
|
|
138
167
|
function artificialAnalysisRows(value) {
|
|
139
168
|
if (Array.isArray(value))
|
|
@@ -232,8 +261,14 @@ function assertUsableLiveBench(rows, categories) {
|
|
|
232
261
|
// `-thinking-auto-medium-effort`); the highest effort is the model at its best. Rank by
|
|
233
262
|
// hyphen-delimited token so `xhigh` never reads as `high`.
|
|
234
263
|
const LIVEBENCH_EFFORT_RANK = { max: 5, xhigh: 4, high: 3, medium: 2, low: 1 };
|
|
235
|
-
|
|
236
|
-
|
|
264
|
+
// OBS-1149: LiveBench benchmarks some models only as their preview (`gemini-3.1-pro-preview-high`),
|
|
265
|
+
// so on THIS side `preview` is residue too. Every residue token must be a word: `.2` never is.
|
|
266
|
+
const isLiveBenchVariant = (residue) => residue.startsWith("-") && residue.slice(1).split("-").every((token) => token === "preview" || EFFORT_MODE_TOKENS.has(token));
|
|
267
|
+
/** A GA row always outranks a preview row of the same model; effort ranks within each. */
|
|
268
|
+
const liveBenchEffortRank = (residue) => {
|
|
269
|
+
const tokens = residue.split("-");
|
|
270
|
+
return (tokens.includes("preview") ? 0 : 10) + tokens.reduce((rank, token) => Math.max(rank, LIVEBENCH_EFFORT_RANK[token] ?? 0), 0);
|
|
271
|
+
};
|
|
237
272
|
const liveBenchCategoryMean = (row, tasks) => {
|
|
238
273
|
const scores = (Array.isArray(tasks) ? tasks : [])
|
|
239
274
|
.map((task) => (typeof task === "string" ? finite(row[task]) : undefined))
|
|
@@ -255,13 +290,17 @@ function liveBenchIndex(value, identities) {
|
|
|
255
290
|
continue;
|
|
256
291
|
// Bare startsWith is a false-positive machine: `glm-5` would claim `glm-5.2`. The residue after
|
|
257
292
|
// the fleet id must be empty or an effort suffix.
|
|
258
|
-
const
|
|
259
|
-
|
|
260
|
-
if (
|
|
293
|
+
const residues = wanted.map((identity) => model.startsWith(identity) ? model.slice(identity.length) : undefined);
|
|
294
|
+
const identity = residues.findIndex((rest) => rest === "" || isLiveBenchVariant(rest ?? ""));
|
|
295
|
+
if (identity < 0)
|
|
261
296
|
continue;
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
297
|
+
// Identities are listed most-specific first (the client's `gpt-5.6-high` before its stripped base
|
|
298
|
+
// `gpt-5.6`): a row reached through an earlier identity always beats one reached only through a
|
|
299
|
+
// later one, so a client's exact `-high` row is never outranked by a sibling `-low` row whose
|
|
300
|
+
// residue happens to carry an effort token. Effort rank only orders rows of the SAME identity.
|
|
301
|
+
const rank = liveBenchEffortRank(residues[identity] ?? "");
|
|
302
|
+
if (!best || identity < best.identity || (identity === best.identity && rank > best.rank))
|
|
303
|
+
best = { row, identity, rank };
|
|
265
304
|
}
|
|
266
305
|
if (!best)
|
|
267
306
|
return undefined;
|
|
@@ -293,10 +332,13 @@ export function resolveCatalogModel(catalog, query) {
|
|
|
293
332
|
modelId,
|
|
294
333
|
query.model,
|
|
295
334
|
...[modelId, query.model].flatMap((identity) => identity.includes("/") ? [identity.slice(identity.lastIndexOf("/") + 1)] : []),
|
|
335
|
+
// OBS-1147/1082: the exact base an effort variant resolved to (`gemini-3-pro-high` → `gemini-3-pro`).
|
|
336
|
+
match.matchedId,
|
|
296
337
|
...(typeof model.name === "string" ? [model.name] : []),
|
|
297
338
|
];
|
|
298
339
|
const intelligence = artificialAnalysisIndex(catalog.artificialAnalysis, [
|
|
299
340
|
modelId,
|
|
341
|
+
match.matchedId,
|
|
300
342
|
...(typeof model.name === "string" ? [model.name] : []),
|
|
301
343
|
], query.provider);
|
|
302
344
|
const liveBench = liveBenchIndex(catalog.liveBench, identities);
|
|
@@ -210,6 +210,9 @@ export function probeVersion(bin) {
|
|
|
210
210
|
note: "auth assumed; verified at dispatch (failover on auth/quota errors)",
|
|
211
211
|
};
|
|
212
212
|
}
|
|
213
|
+
// OBS-1182: an absent effort renders nothing, so the CLI keeps its own default. Every builder places
|
|
214
|
+
// it right after --model's value and before another flag, never before the prompt positional.
|
|
215
|
+
const claudeEffortFlag = (effort) => (effort ? ` --effort ${shq(effort)}` : "");
|
|
213
216
|
export const claudeCode = {
|
|
214
217
|
id: "claude-code",
|
|
215
218
|
vendor: "anthropic",
|
|
@@ -218,7 +221,7 @@ export const claudeCode = {
|
|
|
218
221
|
channels: (cfg) => channelsFromConfig("claude-code", cfg),
|
|
219
222
|
// v1.65 T3: every flag the command builders below hardcode — doctor checks `claude --help` still
|
|
220
223
|
// lists each (all present on claude 2.x, verified 2026-07-22). Advisory only, never routing.
|
|
221
|
-
hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions", "--settings"] },
|
|
224
|
+
hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions", "--settings", "--effort"] },
|
|
222
225
|
// --strict-mcp-config --mcp-config '{"mcpServers":{}}': pin the MCP surface to empty so fresh-worktree
|
|
223
226
|
// workers/gates don't load project .mcp.json servers (herdr scrapes dialogs as idle — v1.4 incident,
|
|
224
227
|
// memory tickmarkr-worker-mcp-dialog-stall). Live-verified 2026-07-10 on claude 2.1.205 (operator check):
|
|
@@ -230,7 +233,7 @@ export const claudeCode = {
|
|
|
230
233
|
// flag must always follow the value, never the prompt.
|
|
231
234
|
// The empty -p argument selects print mode while stdin carries the prompt, keeping its nonce out
|
|
232
235
|
// of process argv. The redirect path is shell-quoted independently from the model.
|
|
233
|
-
headlessCommand: (promptFile, model) => `claude -p '' --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text < ${shq(promptFile)}`,
|
|
236
|
+
headlessCommand: (promptFile, model, effort) => `claude -p '' --model ${shq(model)}${claudeEffortFlag(effort)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text < ${shq(promptFile)}`,
|
|
234
237
|
// HYG-03 / OBS-137: the residual first-entry dialog is workspace trust, not MCP config loading.
|
|
235
238
|
// Claude's only store is global last-writer-wins ~/.claude.json, so tickmarkr still does not seed it;
|
|
236
239
|
// the daemon safely answers only the exact adapter-declared dialog once per slot.
|
|
@@ -250,16 +253,16 @@ export const claudeCode = {
|
|
|
250
253
|
// OBS-931: the same ONE-argv-string hazard as codex (OBS-930) — over promptArgvCeiling() the TUI
|
|
251
254
|
// launch would E2BIG on Linux, so it returns null → worker-mode-fallback → the headless form.
|
|
252
255
|
// resumeCommand keeps the shape: its contract returns a string (composer delivery is 2.4.3 work).
|
|
253
|
-
interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
|
|
254
|
-
? `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`
|
|
256
|
+
interactiveCommand: (promptFile, model, effort) => promptFitsArgv(promptFile)
|
|
257
|
+
? `claude --model ${shq(model)}${claudeEffortFlag(effort)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`
|
|
255
258
|
: null,
|
|
256
259
|
trustDialog: CLAUDE_TRUST_DIALOG,
|
|
257
260
|
inputBox: CLAUDE_INPUT_BOX,
|
|
258
261
|
// A resumed attempt lands in the same painted editor, so it carries the same ghost-text suppression
|
|
259
262
|
// and the same value-then-flag placement.
|
|
260
|
-
resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
263
|
+
resumeCommand: (sessionId, promptFile, model, effort) => `claude -r ${shq(sessionId)} --model ${shq(model)}${claudeEffortFlag(effort)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
261
264
|
invoke(task, _cwd, a, ctx) {
|
|
262
|
-
return { command: this.headlessCommand(ctx.promptFile, a.model) };
|
|
265
|
+
return { command: this.headlessCommand(ctx.promptFile, a.model, a.effort) };
|
|
263
266
|
},
|
|
264
267
|
parse: parseWorkerResult,
|
|
265
268
|
collectUsage(cwd, sinceMs) {
|
package/dist/adapters/codex.js
CHANGED
|
@@ -89,6 +89,9 @@ const GITDIR_WRITABLE = `-c "sandbox_workspace_write.writable_roots=[\\"$(git re
|
|
|
89
89
|
// -s/--sandbox workspace-write sandbox (deliberately NOT --dangerously-bypass-approvals-and-sandbox,
|
|
90
90
|
// which would drop the sandbox). Listed by `codex --help` and `codex exec --help` (verified 2026-07-23).
|
|
91
91
|
const CODEX_HOOK_TRUST = "--dangerously-bypass-hook-trust";
|
|
92
|
+
// OBS-1182: config override, not a flag — codex has no --effort. Absent renders nothing (CLI default);
|
|
93
|
+
// placed before --model so --model's value stays the last flag before the prompt, as it always was.
|
|
94
|
+
const codexEffortFlag = (effort) => (effort ? ` -c ${shq(`model_reasoning_effort=${effort}`)}` : "");
|
|
92
95
|
// v1.75 T2 / OBS-137: current Codex workspace-trust prompt (0.144.6). The exact heading
|
|
93
96
|
// is distinct from normal agent output; Enter accepts the selected "Yes, continue" option.
|
|
94
97
|
export const CODEX_TRUST_DIALOG = {
|
|
@@ -240,7 +243,7 @@ export const codex = {
|
|
|
240
243
|
// --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
|
|
241
244
|
// MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
|
|
242
245
|
// CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
|
|
243
|
-
headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
|
|
246
|
+
headlessCommand: (promptFile, model, effort) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE}${codexEffortFlag(effort)} --model ${shq(model)} - < ${shq(promptFile)}`,
|
|
244
247
|
// OBS-930: the visible pane runs the REAL TUI. Codex's TUI takes its prompt only as the [PROMPT]
|
|
245
248
|
// positional (`codex --help`, 0.153.4 — no file/stdin form), so the launch inlines the file exactly
|
|
246
249
|
// as the claude adapter does: the prompt is the LAST positional and every flag value is followed by
|
|
@@ -251,11 +254,11 @@ export const codex = {
|
|
|
251
254
|
// autonomous approval policy (exec has no approvals to configure).
|
|
252
255
|
// OBS-930 (Linux): the inlined prompt is ONE argv string and Linux caps one at 131072 bytes, so a
|
|
253
256
|
// prompt over promptArgvCeiling() returns null → worker-mode-fallback → the headless form (types.ts).
|
|
254
|
-
interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
|
|
255
|
-
? `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`
|
|
257
|
+
interactiveCommand: (promptFile, model, effort) => promptFitsArgv(promptFile)
|
|
258
|
+
? `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE}${codexEffortFlag(effort)} --model ${shq(model)} "$(cat ${shq(promptFile)})"`
|
|
256
259
|
: null,
|
|
257
260
|
invoke(task, _cwd, a, ctx) {
|
|
258
|
-
return { command: this.headlessCommand(ctx.promptFile, a.model) };
|
|
261
|
+
return { command: this.headlessCommand(ctx.promptFile, a.model, a.effort) };
|
|
259
262
|
},
|
|
260
263
|
parse: parseWorkerResult,
|
|
261
264
|
// v1.22 T5: seed [projects."<repoRoot>"] trust_level="trusted" so fresh worktrees never stall on
|
|
@@ -13,3 +13,4 @@ export declare function parseWorkerResult(raw: string, nonce: string): Classifie
|
|
|
13
13
|
export type DeadChannelReason = "auth-required" | "setup-required" | "provider-outage" | "timeout";
|
|
14
14
|
export declare function classifyTransientCapacity(result: WorkerResult): RegExpExecArray | null;
|
|
15
15
|
export declare function classifyDeadChannel(result: WorkerResult): DeadChannelReason | undefined;
|
|
16
|
+
export declare const BOOTSTRAP_FAILURE_RE: RegExp;
|
package/dist/adapters/prompt.js
CHANGED
|
@@ -68,8 +68,9 @@ function workerVerdictWitness(raw, nonce, positions) {
|
|
|
68
68
|
return raw;
|
|
69
69
|
return `{"nonce":${JSON.stringify(nonce)},"ok":false,`;
|
|
70
70
|
}
|
|
71
|
-
// v1.65 T1: the parse boundary's own no-trailer sentinel summaries
|
|
72
|
-
//
|
|
71
|
+
// v1.65 T1: the parse boundary's own no-trailer sentinel summaries — display text only. OBS-1175:
|
|
72
|
+
// classifiers key on the parser's `cause`, never on these: a PARSED trailer may carry either string
|
|
73
|
+
// as its own summary, and it is still the worker speaking.
|
|
73
74
|
export const NO_TRAILER_SUMMARY = "worker produced no TICKMARKR_RESULT trailer";
|
|
74
75
|
export const UNPARSEABLE_TRAILER_SUMMARY = "unparseable TICKMARKR_RESULT trailer";
|
|
75
76
|
// OBS-1062: every interactive TUI echoes the brief into the pane, and the brief carries the trailer
|
|
@@ -139,14 +140,12 @@ const TIMEOUT_RE = /\bETIMEDOUT\b|request timed out|connection timed out|deadlin
|
|
|
139
140
|
// speaking, so a verdict that QUOTES the capacity phrase stays ordinary work evidence. The caller
|
|
140
141
|
// hands in the chrome-filtered banner rows as `raw`, exactly as it does for classifyDeadChannel.
|
|
141
142
|
export function classifyTransientCapacity(result) {
|
|
142
|
-
|
|
143
|
-
return null;
|
|
144
|
-
return CAPACITY_RE.exec(result.raw);
|
|
143
|
+
return unparsed(result) ? CAPACITY_RE.exec(result.raw) : null;
|
|
145
144
|
}
|
|
146
145
|
export function classifyDeadChannel(result) {
|
|
147
146
|
// A parsed trailer — ok:true OR ok:false — is the worker speaking: genuine work outcomes walk
|
|
148
147
|
// the normal gate/ladder path even when their transcript mentions auth/outage/timeout text.
|
|
149
|
-
if (
|
|
148
|
+
if (!unparsed(result))
|
|
150
149
|
return undefined;
|
|
151
150
|
if (AUTH_RE.test(result.raw))
|
|
152
151
|
return "auth-required";
|
|
@@ -158,3 +157,12 @@ export function classifyDeadChannel(result) {
|
|
|
158
157
|
return "timeout";
|
|
159
158
|
return undefined;
|
|
160
159
|
}
|
|
160
|
+
// OBS-1175: the one "no trailer was parsed" test every classifier here shares.
|
|
161
|
+
function unparsed(result) {
|
|
162
|
+
return !result.ok && result.cause !== undefined;
|
|
163
|
+
}
|
|
164
|
+
// OBS-1169: a CLI that died in its own bootstrap (codex: "account/read failed during TUI bootstrap …
|
|
165
|
+
// (code -32603)") never read the brief. Known bootstrap phrasing only — evidence solely inside the
|
|
166
|
+
// startup prefix the daemon's StartupFailureDetector owns (before any tool frame or input box, in an
|
|
167
|
+
// unsaturated read, inside the startup window); the same words in a worker's tool output are work.
|
|
168
|
+
export const BOOTSTRAP_FAILURE_RE = /\b[\w/-]+ failed during TUI bootstrap\b/i;
|
|
@@ -5,7 +5,7 @@ import { join } from "node:path";
|
|
|
5
5
|
import { modelLints, suggestOverlay } from "./model-lints.js";
|
|
6
6
|
import { HerdrDriver } from "../drivers/herdr.js";
|
|
7
7
|
import { tickmarkrDir, stateDirName } from "../graph/graph.js";
|
|
8
|
-
import { disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../route/preference.js";
|
|
8
|
+
import { disallowedBy, excludedChannels, exclusionLine, observedIdentity, observedSeat, preferRanks } from "../route/preference.js";
|
|
9
9
|
import { sh } from "../run/git.js";
|
|
10
10
|
import { FakeAdapter } from "./fake.js";
|
|
11
11
|
import { parseWorkerResult } from "./prompt.js";
|
|
@@ -581,7 +581,7 @@ export function discoverChannels(cfg, adapters, health, role = "worker") {
|
|
|
581
581
|
return a.channels(cfg)
|
|
582
582
|
.filter((c) => !invalid.has(c.model) && modelAuthed(h, c.model, cfg.routing.allowUnverifiedModels) && (!s || s.includes(c.model)))
|
|
583
583
|
.map((c) => {
|
|
584
|
-
const identity =
|
|
584
|
+
const identity = observedIdentity(health, a.id, c.model);
|
|
585
585
|
return identity ? { ...c, identity } : c;
|
|
586
586
|
});
|
|
587
587
|
});
|
|
@@ -736,7 +736,7 @@ export function formatDoctorReport(cwd, cfg, health, adapters, opts = {}) {
|
|
|
736
736
|
for (const m of models) {
|
|
737
737
|
const v = h.modelAuth?.[m];
|
|
738
738
|
const auth = !v ? "unknown" : v.authed ? "authed" : `unauthed: ${trunc(v.reason ?? "probe failed", 40)} (${dateOf(v.probedAt)})`;
|
|
739
|
-
const d = disallowedBy(
|
|
739
|
+
const d = disallowedBy(observedSeat(health, a.id, m), cfg.routing);
|
|
740
740
|
const denied = d?.by === "deny" ? d.entry : "—";
|
|
741
741
|
const pref = preferRanks({ adapter: a.id, model: m }, cfg).map((p) => `${p.shape}#${p.rank}`).join(",") || "—";
|
|
742
742
|
statusRows.push(` ${m.padEnd(w)} ${classified[m].padEnd(8)} ${auth} denied=${denied} prefer=${pref}`);
|
package/dist/adapters/types.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
2
|
import type { TickmarkrConfig, Tier } from "../config/config.js";
|
|
3
|
-
import type { Task } from "../graph/schema.js";
|
|
3
|
+
import type { Effort, Task } from "../graph/schema.js";
|
|
4
|
+
import type { VerdictUnparseableCause } from "../gates/verdict-cause.js";
|
|
4
5
|
export declare const TokenUsageSchema: z.ZodObject<{
|
|
5
6
|
input: z.ZodNumber;
|
|
6
7
|
output: z.ZodNumber;
|
|
@@ -15,6 +16,7 @@ export interface Assignment {
|
|
|
15
16
|
model: string;
|
|
16
17
|
channel: "sub" | "api";
|
|
17
18
|
tier: Tier;
|
|
19
|
+
effort?: Effort;
|
|
18
20
|
}
|
|
19
21
|
export interface BillingChannel {
|
|
20
22
|
adapter: string;
|
|
@@ -23,6 +25,7 @@ export interface BillingChannel {
|
|
|
23
25
|
channel: "sub" | "api";
|
|
24
26
|
tier: Tier;
|
|
25
27
|
identity?: string;
|
|
28
|
+
effort?: Effort;
|
|
26
29
|
}
|
|
27
30
|
export declare const MODEL_PROBE_ERRORS: readonly ["EMFILE", "EAGAIN", "ENFILE", "ENOMEM", "ENOSPC"];
|
|
28
31
|
export type ModelProbeError = typeof MODEL_PROBE_ERRORS[number];
|
|
@@ -53,6 +56,7 @@ export interface WorkerResult {
|
|
|
53
56
|
summary: string;
|
|
54
57
|
deviations: string[];
|
|
55
58
|
raw: string;
|
|
59
|
+
cause?: VerdictUnparseableCause;
|
|
56
60
|
}
|
|
57
61
|
export type SeedBannerConfirmResult = {
|
|
58
62
|
ok: true;
|
|
@@ -173,11 +177,11 @@ export interface WorkerAdapter {
|
|
|
173
177
|
probeConcurrency?: number;
|
|
174
178
|
probe(): Promise<AuthHealth>;
|
|
175
179
|
channels(cfg: TickmarkrConfig): BillingChannel[];
|
|
176
|
-
headlessCommand(promptFile: string, model: string): string;
|
|
180
|
+
headlessCommand(promptFile: string, model: string, effort?: Effort): string;
|
|
177
181
|
harnessBannerRows?: readonly string[];
|
|
178
|
-
interactiveCommand(promptFile: string, model: string): string | null;
|
|
182
|
+
interactiveCommand(promptFile: string, model: string, effort?: Effort): string | null;
|
|
179
183
|
interactiveSeed?: InteractiveSeed;
|
|
180
|
-
resumeCommand?(sessionId: string, promptFile: string, model: string): string;
|
|
184
|
+
resumeCommand?(sessionId: string, promptFile: string, model: string, effort?: Effort): string;
|
|
181
185
|
sessionIdFrom?(output: string): string | undefined;
|
|
182
186
|
resumeUnknownContext?: boolean;
|
|
183
187
|
invoke(task: Task, cwd: string, a: Assignment, ctx: {
|
|
@@ -201,6 +205,10 @@ export interface WorkerAdapter {
|
|
|
201
205
|
};
|
|
202
206
|
}
|
|
203
207
|
export declare function channelsFromConfig(adapterId: string, cfg: TickmarkrConfig): BillingChannel[];
|
|
208
|
+
export declare function configuredEffort(cfg: TickmarkrConfig, seat: {
|
|
209
|
+
adapter: string;
|
|
210
|
+
model: string;
|
|
211
|
+
}): Effort | undefined;
|
|
204
212
|
export declare function channelKey(c: {
|
|
205
213
|
adapter: string;
|
|
206
214
|
model: string;
|
package/dist/adapters/types.js
CHANGED
|
@@ -249,9 +249,15 @@ export function channelsFromConfig(adapterId, cfg) {
|
|
|
249
249
|
model,
|
|
250
250
|
channel: override?.channel ?? e.channel,
|
|
251
251
|
tier,
|
|
252
|
+
...(override?.effort ? { effort: override.effort } : {}),
|
|
252
253
|
}];
|
|
253
254
|
});
|
|
254
255
|
}
|
|
256
|
+
// OBS-1182: a config-declared seat's launch effort (judge, consult pin) — the same modelOverrides
|
|
257
|
+
// read channelsFromConfig makes for routed channels. Absent = the CLI's own default.
|
|
258
|
+
export function configuredEffort(cfg, seat) {
|
|
259
|
+
return cfg.tiers[seat.adapter]?.modelOverrides?.[seat.model]?.effort;
|
|
260
|
+
}
|
|
255
261
|
export function channelKey(c) {
|
|
256
262
|
return `${c.adapter}:${c.model}`;
|
|
257
263
|
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { Journal, type JournalEvent } from "../../run/journal.js";
|
|
1
|
+
import { Journal, type DecisionBinding, type JournalEvent } from "../../run/journal.js";
|
|
2
2
|
export declare const APPROVAL_DISPOSITIONS: readonly ["dispatch", "waive-gate", "re-dispatch", "fund-fixed-attempt", "fresh-budget"];
|
|
3
3
|
export type ApprovalDisposition = (typeof APPROVAL_DISPOSITIONS)[number];
|
|
4
4
|
/**
|
|
@@ -26,6 +26,8 @@ export interface NewestPark {
|
|
|
26
26
|
failedGate: string | undefined;
|
|
27
27
|
/** A pre-dispatch human gate whose reason marks it permanent by design (see isTombstonePark). */
|
|
28
28
|
tombstone: boolean;
|
|
29
|
+
/** OBS-1202: a stall park's recorded reap failure (task-human data.reapFailure) — its census is unproven. */
|
|
30
|
+
reapFailure?: string;
|
|
29
31
|
}
|
|
30
32
|
/**
|
|
31
33
|
* There is no closed park kind for a declaration-shaped retirement, so the evidence is the one the
|
|
@@ -41,14 +43,19 @@ export declare function readJournalEvents(journal: Journal): {
|
|
|
41
43
|
events: JournalEvent[];
|
|
42
44
|
sourceIndexes: number[];
|
|
43
45
|
};
|
|
46
|
+
/** OBS-1178: the token a park is decided by — what status, park notices and the cockpit print. */
|
|
47
|
+
export declare const parkToken: (park: Pick<NewestPark, "line" | "ts">) => string | undefined;
|
|
48
|
+
/** OBS-1178: the newest task-failed row for a task — the failure a failed-task recheck binds to. */
|
|
49
|
+
export declare function newestFailure(events: readonly JournalEvent[], taskId: string, sourceIndexes?: readonly number[]): DecisionBinding | undefined;
|
|
44
50
|
/**
|
|
45
|
-
* FINAL §3.3's decision menu as data: human-gate/attempt-cap/other non-gate parks → approve; infra
|
|
46
|
-
*
|
|
51
|
+
* FINAL §3.3's decision menu as data: human-gate/attempt-cap/other non-gate parks → approve; infra, and
|
|
52
|
+
* a stall park that recorded a reapFailure (OBS-1202), → approve or recheck; review gate-fail →
|
|
53
|
+
* waive/uphold/recheck; other gate-fail → waive/recheck; a
|
|
47
54
|
* gate-fail park with no failed-gate evidence, or a tombstone → nothing (a diagnostic, never a
|
|
48
55
|
* fabricated verb). The refusals in `approve` below enforce the same table; this is the one place a
|
|
49
56
|
* surface may read it from, so what a menu offers and what the command accepts cannot drift.
|
|
50
57
|
*/
|
|
51
|
-
export declare function permittedDecisionVerbs(park: Pick<NewestPark, "kind" | "failedGate" | "tombstone"> | undefined): readonly DecisionVerb[];
|
|
58
|
+
export declare function permittedDecisionVerbs(park: Pick<NewestPark, "kind" | "failedGate" | "tombstone" | "reapFailure"> | undefined): readonly DecisionVerb[];
|
|
52
59
|
/** The release marker this command appends for a verb on a park — the fact a read-back must match. */
|
|
53
60
|
export declare function releaseForDecision(verb: DecisionVerb, park: Pick<NewestPark, "kind" | "failedGate">): string | undefined;
|
|
54
61
|
export type ApprovalStatus = "deferred-live" | "recorded-no-owner";
|