tickmarkr 2.4.0 → 2.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/adapters/catalog-remote.d.ts +11 -0
  2. package/dist/adapters/catalog-remote.js +55 -13
  3. package/dist/adapters/claude-code.js +7 -2
  4. package/dist/adapters/codex.d.ts +1 -0
  5. package/dist/adapters/codex.js +68 -8
  6. package/dist/adapters/qwen.js +3 -3
  7. package/dist/adapters/registry.js +13 -1
  8. package/dist/adapters/types.d.ts +4 -0
  9. package/dist/adapters/types.js +33 -0
  10. package/dist/cli/commands/compile.d.ts +3 -0
  11. package/dist/cli/commands/compile.js +77 -46
  12. package/dist/cli/commands/doctor.js +14 -9
  13. package/dist/cli/commands/fleet.js +11 -5
  14. package/dist/cli/commands/init.js +12 -13
  15. package/dist/cli/commands/plan.js +47 -7
  16. package/dist/cli/commands/run.js +20 -1
  17. package/dist/cli/commands/status.js +20 -21
  18. package/dist/cli/commands/version.js +2 -2
  19. package/dist/compile/collateral.d.ts +14 -5
  20. package/dist/compile/collateral.js +32 -26
  21. package/dist/compile/index.js +17 -6
  22. package/dist/compile/native.js +10 -2
  23. package/dist/compile/ownership.js +15 -9
  24. package/dist/drivers/herdr.d.ts +1 -0
  25. package/dist/drivers/herdr.js +32 -3
  26. package/dist/drivers/orca.d.ts +5 -1
  27. package/dist/drivers/orca.js +51 -2
  28. package/dist/drivers/types.d.ts +1 -1
  29. package/dist/gates/baseline.d.ts +26 -2
  30. package/dist/gates/baseline.js +90 -7
  31. package/dist/gates/review.d.ts +2 -2
  32. package/dist/gates/review.js +2 -18
  33. package/dist/graph/graph.d.ts +20 -0
  34. package/dist/graph/graph.js +66 -1
  35. package/dist/route/preference.d.ts +3 -1
  36. package/dist/route/preference.js +4 -4
  37. package/dist/run/consult.js +1 -0
  38. package/dist/run/daemon.d.ts +5 -0
  39. package/dist/run/daemon.js +151 -22
  40. package/dist/run/git.d.ts +2 -0
  41. package/dist/run/git.js +18 -4
  42. package/dist/run/journal.d.ts +1 -1
  43. package/dist/run/journal.js +1 -1
  44. package/dist/run/lock.d.ts +6 -0
  45. package/dist/run/lock.js +41 -1
  46. package/package.json +1 -1
  47. package/skills/tickmarkr-overseer/SKILL.md +39 -4
  48. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
  49. package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
@@ -11,6 +11,7 @@ export interface CatalogCache {
11
11
  modelsDev: unknown;
12
12
  artificialAnalysis?: unknown;
13
13
  liveBench?: unknown;
14
+ legFetchedAt?: Partial<Record<"modelsDev" | "artificialAnalysis" | "liveBench", string>>;
14
15
  }
15
16
  export interface CatalogReadResult {
16
17
  catalog: CatalogCache;
@@ -47,11 +48,21 @@ export interface RefreshCatalogOptions {
47
48
  timeoutMs?: number;
48
49
  now?: () => Date;
49
50
  }
51
+ export type CatalogLegName = "models.dev" | "Artificial Analysis" | "LiveBench";
52
+ export type CatalogLegStatus = "updated" | "failed" | "skipped";
53
+ export interface CatalogLegResult {
54
+ leg: CatalogLegName;
55
+ status: CatalogLegStatus;
56
+ detail: string;
57
+ retry: string;
58
+ }
50
59
  export interface RefreshCatalogResult {
51
60
  updated: boolean;
52
61
  catalog: CatalogReadResult;
53
62
  warning?: string;
63
+ legs: CatalogLegResult[];
54
64
  }
65
+ export declare const formatCatalogRefreshLegs: (legs: readonly CatalogLegResult[]) => string;
55
66
  export declare const catalogCachePath: (repoRoot: string) => string;
56
67
  /**
57
68
  * Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
@@ -60,9 +60,14 @@ const validModelsDevCatalog = (value) => {
60
60
  });
61
61
  };
62
62
  const staleAt = (fetchedAt, now) => {
63
- const age = now.getTime() - Date.parse(fetchedAt);
63
+ const age = now.getTime() - Date.parse(fetchedAt ?? "");
64
64
  return !Number.isFinite(age) || age > CATALOG_CACHE_MAX_AGE_MS;
65
65
  };
66
+ const legFetchedAt = (catalog, leg) => catalog.legFetchedAt?.[leg] ?? catalog.fetchedAt;
67
+ const catalogStale = (catalog, now, source) => source === "vendored" || staleAt(legFetchedAt(catalog, "modelsDev"), now) || staleAt(legFetchedAt(catalog, "liveBench"), now)
68
+ || (catalog.artificialAnalysis !== undefined && catalog.legFetchedAt?.artificialAnalysis !== undefined
69
+ && staleAt(catalog.legFetchedAt.artificialAnalysis, now));
70
+ export const formatCatalogRefreshLegs = (legs) => `catalog refresh: ${legs.map((leg) => `${leg.leg} ${leg.status} (${leg.detail}; ${leg.retry})`).join("; ")}`;
66
71
  export const catalogCachePath = (repoRoot) => join(repoRoot, ".tickmarkr", "catalog-cache.json");
67
72
  /**
68
73
  * Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
@@ -74,14 +79,14 @@ export function readCachedCatalog(repoRoot, opts = {}) {
74
79
  const parsed = JSON.parse(readFileSync(catalogCachePath(repoRoot), "utf8"));
75
80
  if (!validCache(parsed))
76
81
  throw new Error("catalog cache schema is invalid");
77
- return { catalog: parsed, source: "cache", stale: staleAt(parsed.fetchedAt, now) };
82
+ return { catalog: parsed, source: "cache", stale: catalogStale(parsed, now, "cache") };
78
83
  }
79
84
  catch (error) {
80
85
  const missing = error.code === "ENOENT";
81
86
  return {
82
87
  catalog: VENDORED_CATALOG,
83
88
  source: "vendored",
84
- stale: staleAt(VENDORED_CATALOG.fetchedAt, now),
89
+ stale: true,
85
90
  ...(!missing ? { warning: error instanceof Error ? error.message : String(error) } : {}),
86
91
  };
87
92
  }
@@ -410,6 +415,13 @@ export async function refreshCatalogCommand(opts) {
410
415
  const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
411
416
  const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
412
417
  const warnings = [];
418
+ const legs = [];
419
+ const stamp = now().toISOString();
420
+ const legAt = {
421
+ modelsDev: legFetchedAt(current.catalog, "modelsDev"),
422
+ liveBench: legFetchedAt(current.catalog, "liveBench"),
423
+ ...(current.catalog.artificialAnalysis !== undefined ? { artificialAnalysis: legFetchedAt(current.catalog, "artificialAnalysis") } : {}),
424
+ };
413
425
  let modelsDev = current.catalog.modelsDev;
414
426
  let modelsDevUpdated = false;
415
427
  try {
@@ -417,46 +429,76 @@ export async function refreshCatalogCommand(opts) {
417
429
  if (!validModelsDevCatalog(modelsDev))
418
430
  throw new Error("models.dev catalog schema is invalid");
419
431
  modelsDevUpdated = true;
432
+ legAt.modelsDev = stamp;
433
+ legs.push({ leg: "models.dev", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
420
434
  }
421
435
  catch (error) {
422
- warnings.push(`models.dev refresh failed: ${error instanceof Error ? error.message : String(error)}`);
436
+ const detail = error instanceof Error ? error.message : String(error);
437
+ warnings.push(`models.dev refresh failed: ${detail}`);
438
+ legs.push({ leg: "models.dev", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
423
439
  }
424
440
  let artificialAnalysis = current.catalog.artificialAnalysis;
441
+ let artificialAnalysisUpdated = false;
425
442
  const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
426
443
  if (apiKey) {
427
444
  try {
428
445
  artificialAnalysis = await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs);
446
+ artificialAnalysisUpdated = true;
447
+ legAt.artificialAnalysis = stamp;
448
+ legs.push({ leg: "Artificial Analysis", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
429
449
  }
430
450
  catch (error) {
431
- warnings.push(`Artificial Analysis refresh failed: ${error instanceof Error ? error.message : String(error)}`);
451
+ const detail = error instanceof Error ? error.message : String(error);
452
+ warnings.push(`Artificial Analysis refresh failed: ${detail}`);
453
+ legs.push({ leg: "Artificial Analysis", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
432
454
  }
433
455
  }
456
+ else {
457
+ delete legAt.artificialAnalysis;
458
+ legs.push({ leg: "Artificial Analysis", status: "skipped", detail: "no API key", retry: "retry when ARTIFICIAL_ANALYSIS_API_KEY is set" });
459
+ }
434
460
  let liveBench = current.catalog.liveBench;
435
461
  let liveBenchUpdated = false;
436
462
  try {
437
463
  liveBench = await fetchLiveBench(fetcher, timeoutMs);
438
464
  liveBenchUpdated = true;
465
+ legAt.liveBench = stamp;
466
+ legs.push({ leg: "LiveBench", status: "updated", detail: `table ${LIVEBENCH_TABLE_DATE}`, retry: "next retry after this leg is stale" });
439
467
  }
440
468
  catch (error) {
441
- warnings.push(`LiveBench refresh failed: ${error instanceof Error ? error.message : String(error)}`);
469
+ const detail = error instanceof Error ? error.message : String(error);
470
+ warnings.push(`LiveBench refresh failed: ${detail}`);
471
+ legs.push({ leg: "LiveBench", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
442
472
  }
443
- // models.dev is the required cache spine. Still run every independent leg after it fails, but do
444
- // not write a cache that would make the vendored fallback look fetched.
445
- if (!modelsDevUpdated) {
446
- return { updated: false, catalog: current, warning: warnings.join("; ") };
473
+ const updated = modelsDevUpdated || artificialAnalysisUpdated || liveBenchUpdated;
474
+ // models.dev is the cache spine: do not write a cache that would make the vendored fallback look
475
+ // fetched. Independent legs may only merge onto models.dev fetched now or already held in cache.
476
+ if (!updated || (!modelsDevUpdated && current.source !== "cache")) {
477
+ const reportedLegs = !modelsDevUpdated && current.source !== "cache"
478
+ ? legs.map((leg) => leg.status === "updated"
479
+ ? {
480
+ ...leg,
481
+ status: "failed",
482
+ detail: "fetched but discarded — models.dev spine unavailable",
483
+ retry: "retry next refresh",
484
+ }
485
+ : leg)
486
+ : legs;
487
+ return { updated: false, catalog: current, legs: reportedLegs, ...(warnings.length ? { warning: warnings.join("; ") } : {}) };
447
488
  }
448
489
  const catalog = {
449
490
  schemaVersion: 1,
450
- // A failed keyless leg keeps the old age so doctor/fleet retry it rather than masking it for 7d.
451
- fetchedAt: liveBenchUpdated ? now().toISOString() : current.catalog.fetchedAt,
491
+ fetchedAt: legAt.modelsDev ?? current.catalog.fetchedAt,
452
492
  modelsDev,
453
493
  ...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
454
494
  ...(liveBench !== undefined ? { liveBench } : {}),
495
+ legFetchedAt: legAt,
455
496
  };
456
497
  writeCatalogCache(opts.repoRoot, catalog);
457
498
  return {
458
- updated: true,
499
+ updated,
459
500
  catalog: readCachedCatalog(opts.repoRoot, { now }),
501
+ legs,
460
502
  ...(warnings.length ? { warning: warnings.join("; ") } : {}),
461
503
  };
462
504
  }
@@ -3,7 +3,7 @@ import { readdirSync, readFileSync, realpathSync, statSync } from "node:fs";
3
3
  import { homedir } from "node:os";
4
4
  import { join } from "node:path";
5
5
  import { parseWorkerResult } from "./prompt.js";
6
- import { channelsFromConfig, declareInputBox, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
6
+ import { channelsFromConfig, declareInputBox, MODEL_ID_RE, promptFitsArgv, shq, TokenUsageSchema } from "./types.js";
7
7
  // SPEND-01/SPEND-11: claude writes a per-session JSONL to ~/.claude/projects/<slug>/ where slug is the
8
8
  // realpath'd cwd with every non-alphanumeric char replaced by "-" (verified 114/114 — 36-DIAGNOSIS.md).
9
9
  // The old `/`-only formula missed the "." in `.tickmarkr/worktrees/…` — ENOENT on every worktree dispatch.
@@ -239,7 +239,12 @@ export const claudeCode = {
239
239
  // live check ate the prompt), and --prompt-suggestions takes an OPTIONAL value — appended directly
240
240
  // before the prompt it would swallow it the same way. So the setting's value is always followed by
241
241
  // another flag, never by the prompt positional.
242
- interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
242
+ // OBS-931: the same ONE-argv-string hazard as codex (OBS-930) — over promptArgvCeiling() the TUI
243
+ // launch would E2BIG on Linux, so it returns null → worker-mode-fallback → the headless form.
244
+ // resumeCommand keeps the shape: its contract returns a string (composer delivery is 2.4.3 work).
245
+ interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
246
+ ? `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`
247
+ : null,
243
248
  trustDialog: CLAUDE_TRUST_DIALOG,
244
249
  inputBox: CLAUDE_INPUT_BOX,
245
250
  // A resumed attempt lands in the same painted editor, so it carries the same ghost-text suppression
@@ -4,6 +4,7 @@ export declare function readCodexModelsCache(path?: string): {
4
4
  fetchedAt?: string;
5
5
  };
6
6
  export declare const CODEX_TRUST_DIALOG: TrustDialog;
7
+ export declare const CODEX_INPUT_BOX: import("./types.js").InputBox;
7
8
  export declare function seedCodexTrust(repoRoot: string, configPath?: string): TrustVerdict;
8
9
  export declare function hasCodexTrustedProject(text: string, root: string): boolean;
9
10
  export declare function codexConfigMcpServerNames(configPath?: string): string[];
@@ -3,7 +3,7 @@ import { homedir } from "node:os";
3
3
  import { dirname, join } from "node:path";
4
4
  import { probeVersion } from "./claude-code.js";
5
5
  import { parseWorkerResult } from "./prompt.js";
6
- import { channelsFromConfig, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
6
+ import { channelsFromConfig, declareInputBox, MODEL_ID_RE, promptFitsArgv, shq, TokenUsageSchema } from "./types.js";
7
7
  // SPEND-07: codex writes per-session JSONL to ~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl — date-partitioned,
8
8
  // NOT cwd-keyed. session_meta.payload.cwd is FILE-SCOPED (one codex exec per cwd). token_count events carry
9
9
  // per-turn DELTAS in payload.info.last_token_usage; we read POST-HOC (never the pane, never the trailer).
@@ -95,6 +95,56 @@ export const CODEX_TRUST_DIALOG = {
95
95
  fingerprint: "Do you trust the contents of this directory?",
96
96
  key: "Enter",
97
97
  };
98
+ // OBS-930 / OBS-136: the codex TUI's composer, CAPTURED (codex 0.153.4, herdr pane read, 2026-09-05
99
+ // 15:48Z — tests/fixtures/codex-input-box, provenance in its README), never guessed:
100
+ // (background row)
101
+ // › Ask Codex to do anything ← U+203A, ASCII space, then the DIM placeholder (empty) or the draft
102
+ // (background row)
103
+ // gpt-5.6-luna medium · /private/tmp/tkr-obs930-smoke ← footer: <model> <effort> · <cwd>
104
+ // Two facts a fingerprint alone would get wrong: (1) a submitted turn is echoed into the transcript
105
+ // with the SAME caret (`› You are a smoke test.`), and so is the trust dialog's cursor (`› 1. Yes,
106
+ // continue`) — only the composer is followed by the footer row, so the footer is the anchor, exactly
107
+ // as claude's editor is anchored by its rules; (2) an EMPTY composer paints its placeholder as text,
108
+ // so "empty" is a closed allowlist of captured placeholders (0.153.4 above; 0.144.5's
109
+ // `Run /review on my current changes` from tests/fixtures/codex-mcp-spinner). An unknown placeholder
110
+ // reads as occupied and a submit onto it fails closed by name (OBS-140) — a fixture-capture chore,
111
+ // never a drive-by widening. `match` is THE COMPOSER IS PAINTED (true while empty, true mid-turn);
112
+ // `emptyMatch` is painted AND carrying nothing — the only positive evidence a submit registered.
113
+ const CODEX_ANSI_SGR_RE = /\u001B\[[0-9;]*m/g;
114
+ const CODEX_CARET_RE = /^› /;
115
+ const CODEX_PLACEHOLDER_RE = /^› (?:Ask Codex to do anything|Run \/review on my current changes)$/;
116
+ const CODEX_FOOTER_RE = /^\S+(?: \S+)? · \S/;
117
+ // A wrapped or multi-line draft grows the composer downward before the footer.
118
+ // ponytail: a fixed window, not a parser — raise it if a real capture ever shows a taller composer.
119
+ const CODEX_MAX_COMPOSER_ROWS = 8;
120
+ function matchesCodexComposer(paneText, empty) {
121
+ const lines = paneText.replace(CODEX_ANSI_SGR_RE, "").split("\n").map((l) => l.trim());
122
+ const caret = empty ? CODEX_PLACEHOLDER_RE : CODEX_CARET_RE;
123
+ return lines.some((line, i) => {
124
+ if (!caret.test(line))
125
+ return false;
126
+ for (let below = i + 1; below < lines.length && below <= i + CODEX_MAX_COMPOSER_ROWS; below++) {
127
+ if (CODEX_FOOTER_RE.test(lines[below]))
128
+ return true;
129
+ // only the composer's own rows may sit between the caret row and the footer: the background
130
+ // rows (blank in a text read) and a draft's continuation rows — never another caret row
131
+ if (lines[below] !== "" && CODEX_CARET_RE.test(lines[below]))
132
+ return false;
133
+ }
134
+ return false;
135
+ });
136
+ }
137
+ export const CODEX_INPUT_BOX = declareInputBox("codex", {
138
+ fingerprint: "› ",
139
+ match: (paneText) => matchesCodexComposer(paneText, false),
140
+ emptyMatch: (paneText) => matchesCodexComposer(paneText, true),
141
+ // As for claude (OBS-342): a fresh codex worker slot is a shell awaiting its launch line; every
142
+ // later delivery is a TUI turn awaiting this composer.
143
+ firstDeliveryIsLaunch: true,
144
+ // The 2026-09-05 capture painted the composer ~10 s after the trust answer with MCP suppressed;
145
+ // claude's bound, kept for the same cold-start reasons.
146
+ readinessTimeoutMs: 30_000,
147
+ });
98
148
  // v1.22 T5 / OBS-16: codex keys trust on absolute path under [projects."<root>"] trust_level="trusted"
99
149
  // in ~/.codex/config.toml (CODEX_HOME relocates the dir). Worktrees inherit parent-project trust when
100
150
  // the REPO ROOT is trusted — seed the root once, cover every future worktree. Idempotent: a second
@@ -184,17 +234,26 @@ export const codex = {
184
234
  probeConcurrency: 1,
185
235
  probe: async () => probeVersion("codex"),
186
236
  channels: (cfg) => channelsFromConfig("codex", cfg),
187
- // v1.65 T3: every flag the command builders below hardcode (incl. codexMcpSuppressionFlags' -c/
237
+ // v1.65 T3: every flag the command builder below hardcodes (incl. codexMcpSuppressionFlags' -c/
188
238
  // --disable and GITDIR_WRITABLE's -c) — all listed by top-level `codex --help`, verified 2026-07-22.
189
- hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-a", "-s", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
239
+ hardcodedFlags: { binary: "codex", flags: ["--sandbox", "-a", "-s", "--model", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
190
240
  // --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
191
241
  // MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
192
242
  // CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
193
- headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
194
- // TUI uses expanded -a never -s workspace-write (exec-only flags do not apply)
195
- // (--help 2026-07-09: valid approval policies are untrusted|on-request|never; the previously
196
- // used `on-failure` is invalid and made codex exit 2 pre-inference)
197
- interactiveCommand: (promptFile, model) => `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
243
+ headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
244
+ // OBS-930: the visible pane runs the REAL TUI. Codex's TUI takes its prompt only as the [PROMPT]
245
+ // positional (`codex --help`, 0.153.4 — no file/stdin form), so the launch inlines the file exactly
246
+ // as the claude adapter does: the prompt is the LAST positional and every flag value is followed by
247
+ // a flag, never by the prompt. The argv hazard that once forbade this (OBS-889: `countLiveSuites`
248
+ // matched a suite word 140 KB into a finished worker's argv) is closed on the counter side — the
249
+ // census reads a command's first four tokens only — and those four never carry a suite word here.
250
+ // Same sandbox, hook trust and MCP suppression as the headless form; `-a never` is the TUI's
251
+ // autonomous approval policy (exec has no approvals to configure).
252
+ // OBS-930 (Linux): the inlined prompt is ONE argv string and Linux caps one at 131072 bytes, so a
253
+ // prompt over promptArgvCeiling() returns null → worker-mode-fallback → the headless form (types.ts).
254
+ interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
255
+ ? `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`
256
+ : null,
198
257
  invoke(task, _cwd, a, ctx) {
199
258
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
200
259
  },
@@ -203,6 +262,7 @@ export const codex = {
203
262
  // "Do you trust this directory?" (OBS-16). doctor-only side effect.
204
263
  trust: (repoRoot) => seedCodexTrust(repoRoot),
205
264
  trustDialog: CODEX_TRUST_DIALOG,
265
+ inputBox: CODEX_INPUT_BOX,
206
266
  // v1.5 MODEL-01: file read only (no `codex models` subcommand exists, verified 2026-07-10).
207
267
  // Already fails OPEN to [] internally — advisory detection, unlike gates' fail-closed.
208
268
  listModels: async () => readCodexModelsCache().models,
@@ -58,7 +58,7 @@ function decodeQwenEvents(events) {
58
58
  }
59
59
  }
60
60
  const assistantText = text.join("\n");
61
- const apiError = text.find((line) => line.startsWith("[API Error:"));
61
+ const apiError = assistantText.match(/\[API Error:[^\n]*/)?.[0];
62
62
  if (apiError)
63
63
  failed = true;
64
64
  if (!failed)
@@ -134,8 +134,8 @@ export const qwen = {
134
134
  probeCwd: "neutral",
135
135
  probe: async () => probeQwen(),
136
136
  channels: (cfg) => channelsFromConfig("qwen", cfg),
137
- hardcodedFlags: { binary: "qwen", flags: ["--approval-mode", "-m", "-o", "-p"] },
138
- headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
137
+ hardcodedFlags: { binary: "qwen", flags: ["--safe-mode", "--approval-mode", "-m", "-o", "-p"] },
138
+ headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --safe-mode --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
139
139
  // OBS-905: qwen has NO interactive form. The `-i "$(cat prompt)"` TUI launch put the whole prompt in
140
140
  // argv (the OBS-889 leak-and-census shape) and produced a rendered transcript the JSON decoder above
141
141
  // can never read — under the herdr driver every qwen task read "unparseable" and merged only by harvest.
@@ -399,6 +399,17 @@ function reasonTail(output) {
399
399
  const sp = tail.indexOf(" ");
400
400
  return sp === -1 ? tail : tail.slice(sp + 1);
401
401
  }
402
+ function parsedStartupFailureReason(adapter, output) {
403
+ try {
404
+ const parsed = adapter.parse(output, "tickmarkr-model-probe");
405
+ const cause = parsed.cause;
406
+ if (!parsed.ok && cause === "startup-failure" && parsed.summary.trim()) {
407
+ return reasonTail(parsed.summary.trim().replace(/\s+/g, " "));
408
+ }
409
+ }
410
+ catch { /* adapter parse is advisory for probe diagnostics */ }
411
+ return undefined;
412
+ }
402
413
  function probeFailure(code, stdout, stderr, timedOut, timeoutMs = MODEL_PROBE_TIMEOUT_MS) {
403
414
  // SIGKILL-timeout is not exit-1: report the budget, never the masked kill code (v1.27 T1).
404
415
  if (timedOut)
@@ -456,7 +467,8 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
456
467
  if (r.timedOut && !retry && priorModelAuth?.[model]?.reason?.includes("timed out") === true) {
457
468
  return { verdict: v(false, `probe timed out (repeat — retry skipped) (${MODEL_PROBE_TIMEOUT_MS}ms)`), timedOut: true };
458
469
  }
459
- const reason = probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
470
+ const output = `${r.stderr}\n${r.stdout}`.trim();
471
+ const reason = parsedStartupFailureReason(a, output) ?? probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
460
472
  if (!reason)
461
473
  return { verdict: v(true), timedOut: false };
462
474
  if (!retry)
@@ -185,5 +185,9 @@ export declare function channelKey(c: {
185
185
  model: string;
186
186
  }): string;
187
187
  export declare function shq(s: string): string;
188
+ export declare const PROMPT_ARGV_CEILING_LINUX = 120000;
189
+ export declare const PROMPT_ARGV_CEILING_DEFAULT = 900000;
190
+ export declare function promptArgvCeiling(platform?: string): number;
191
+ export declare function promptFitsArgv(promptFile: string, platform?: string): boolean;
188
192
  export declare const QUOTA_RE: RegExp;
189
193
  export declare const MODEL_ID_RE: RegExp;
@@ -1,3 +1,4 @@
1
+ import { statSync } from "node:fs";
1
2
  import { z } from "zod";
2
3
  // SPEND-01/06: normalized token counts — the measurable fact. NO cost field, ever: CLIs report
3
4
  // cost:0 on sub plans and notional list prices on others (LIVE-CHECK finding 3); money is Phase 18's
@@ -232,6 +233,38 @@ export function channelKey(c) {
232
233
  export function shq(s) {
233
234
  return `'${s.replaceAll("'", `'\\''`)}'`;
234
235
  }
236
+ // OBS-930 (Linux) / OBS-931: a TUI launch inlines the prompt file as ONE argv string — "$(cat prompt)"
237
+ // as the last positional. Linux caps a single argv string at MAX_ARG_STRLEN = PAGE_SIZE × 32 =
238
+ // 131072 bytes (E2BIG: CI run 33979013874, ubuntu, `codex: Argument list too long` on a 140 KB
239
+ // prompt); darwin enforces only the 1 MB total ARG_MAX, which is why the macOS export proof passed.
240
+ // A real worker prompt is 60–150 KB (OBS-889 measured 149,417 bytes), so on a Linux host the launch
241
+ // fails in production, not only in the test. Ceilings are named BY PLATFORM — never a probe of the
242
+ // running kernel at dispatch time:
243
+ // linux 120_000 — the cap is per STRING and every flag is its own argv entry, so only the prompt
244
+ // counts against it; "$(cat …)" strips nothing but trailing newlines, so the
245
+ // file's byte size IS the string's. 131072 − 120000 leaves ~11 KB of headroom.
246
+ // others 900_000 — under the 1 MB total that darwin/BSD enforce across argv + envp.
247
+ // Over the ceiling the adapter returns null: the daemon journals worker-mode-fallback
248
+ // {reason:"adapter"} and runs the headless form in the visible pane — the pre-OBS-930 behaviour,
249
+ // now only for oversized prompts. An unreadable file is "not proven oversized" and keeps the TUI
250
+ // rendering: the daemon writes the prompt before it builds the launch, a missing file fails the
251
+ // same way in either form, and docs-truth renders the command against a placeholder path.
252
+ // OBS-931 (2.4.3): the real fix is prompt delivery through the composer for large prompts.
253
+ export const PROMPT_ARGV_CEILING_LINUX = 120_000;
254
+ export const PROMPT_ARGV_CEILING_DEFAULT = 900_000;
255
+ export function promptArgvCeiling(platform = process.platform) {
256
+ return platform === "linux" ? PROMPT_ARGV_CEILING_LINUX : PROMPT_ARGV_CEILING_DEFAULT;
257
+ }
258
+ export function promptFitsArgv(promptFile, platform = process.platform) {
259
+ let bytes;
260
+ try {
261
+ bytes = statSync(promptFile).size;
262
+ }
263
+ catch {
264
+ return true;
265
+ }
266
+ return bytes <= promptArgvCeiling(platform);
267
+ }
235
268
  // Quota exhaustion is detected from CLI errors, never predicted (spec §4).
236
269
  // ZAI coding-plan exhaustion text: "Insufficient balance or no resource package. Please recharge."
237
270
  // Anchor the distinctive full phrase, not the two-word "insufficient balance" fragment — that fires
@@ -1 +1,4 @@
1
+ import { type SourceScopeFinding } from "../../compile/collateral.js";
2
+ import { compileSource } from "../../compile/index.js";
3
+ export declare function nativeSourceScopeErrors(graph: ReturnType<typeof compileSource>, findings: readonly SourceScopeFinding[]): string[];
1
4
  export declare function compile(argv: string[], cwd?: string, harnessFrom?: string | undefined): Promise<string>;
@@ -1,25 +1,21 @@
1
1
  import { isAbsolute, join } from "node:path";
2
2
  import { parseArgs } from "node:util";
3
- import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
3
+ import { collateralLints, sourceScopeFindings, sourceScopeLints, } from "../../compile/collateral.js";
4
4
  import { CompileError } from "../../compile/common.js";
5
5
  import { compileSource } from "../../compile/index.js";
6
- import { saveGraph, stateDirName } from "../../graph/graph.js";
6
+ import { clearCompileRefusal, saveCompileRefusal, saveGraph, stateDirName } from "../../graph/graph.js";
7
7
  import { formatPriorFindingEvidence, readPriorRunEvidence } from "../../run/journal.js";
8
8
  import { shGit } from "../../run/git.js";
9
9
  import { acquireRunLock, releaseRunLock } from "../../run/lock.js";
10
10
  import { harnessLine, resolveHarness } from "../harness.js";
11
- function nativeSourceScopeErrors(graph, lints) {
11
+ export function nativeSourceScopeErrors(graph, findings) {
12
12
  const byId = new Map(graph.tasks.map((task) => [task.id, task]));
13
13
  const errors = [];
14
- for (const lint of lints) {
15
- const match = lint.match(/^(\S+): criteria implicate out-of-scope source not in files\[\]: (.*)$/);
16
- if (!match)
17
- continue;
18
- const task = byId.get(match[1]);
14
+ for (const finding of findings) {
15
+ const task = byId.get(finding.taskId);
19
16
  if (!task)
20
17
  continue;
21
- const paths = match[2].replace(/ \(capped\)$/, "").split(", ").filter(Boolean);
22
- for (const path of paths) {
18
+ for (const path of finding.paths) {
23
19
  if (task.goal.includes(`scope-waiver: ${path}`))
24
20
  continue;
25
21
  errors.push(`${task.id}: ${path} requires scope-waiver: ${path} in the task goal`);
@@ -60,43 +56,78 @@ export async function compile(argv, cwd = process.cwd(), harnessFrom = process.a
60
56
  const src = positionals[0];
61
57
  if (!src)
62
58
  throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native] [--dry-run] [--strict]");
63
- // resolve against the target repo, not the process cwd (the CLI test passes a tmp repo)
64
- // Both modes reach the same pure compiler; --dry-run only removes the lock/write side effect below.
65
- const g = compileSource(isAbsolute(src) ? src : join(cwd, src), values.type, cwd, // repo root: gsd stores context[0] repo-relative so workers resolve it inside their worktree
66
- (plan) => plan, { strict: values.strict });
67
- const scopeLints = [...collateralLints(g.tasks, cwd), ...sourceScopeLints(g.tasks, cwd)];
68
- const diagnostics = scopeLints.length
69
- ? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
70
- : "";
71
- const unwaived = g.spec.source === "native" ? nativeSourceScopeErrors(g, scopeLints) : [];
72
- if (unwaived.length > 0) {
73
- throw new CompileError(`${src} has unwaived native source-scope authoring errors:\n${unwaived.map((line) => ` - ${line}`).join("\n")}${diagnostics}`);
74
- }
75
- // One bounded read supplies both cross-run surfaces: unresolved findings below and merge facts for
76
- // the ancestry check. Neither fact mutates the compiled graph; status and every readiness predicate
77
- // remain the source compiler's answer.
78
- const prior = readPriorRunEvidence(cwd, g.tasks);
79
- const mergedPending = await mergedPendingDiagnostics(cwd, new Set(g.tasks.filter((task) => task.status === "pending").map((task) => task.id)), prior.merges);
80
- const stateDir = stateDirName(cwd);
81
- if (!values["dry-run"]) {
82
- // HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
83
- // cannot swap graph.json under an active run between the daemon's read and act.
84
- acquireRunLock(cwd, "compile");
85
- try {
86
- saveGraph(cwd, g);
59
+ const sourcePath = isAbsolute(src) ? src : join(cwd, src);
60
+ let stateWriteStarted = false;
61
+ try {
62
+ // Resolve against the target repo, not the process cwd (the CLI test passes a tmp repo).
63
+ // Both modes reach the same pure compiler; --dry-run removes every state write below.
64
+ const g = compileSource(sourcePath, values.type, cwd, // repo root: gsd stores context[0] repo-relative so workers resolve it inside their worktree
65
+ (plan) => plan, { strict: values.strict });
66
+ const sourceFindings = sourceScopeFindings(g.tasks, cwd);
67
+ const scopeLints = [
68
+ ...collateralLints(g.tasks, cwd),
69
+ ...sourceScopeLints(g.tasks, cwd, sourceFindings),
70
+ ];
71
+ const diagnostics = scopeLints.length
72
+ ? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
73
+ : "";
74
+ const unwaived = g.spec.source === "native" ? nativeSourceScopeErrors(g, sourceFindings) : [];
75
+ if (unwaived.length > 0) {
76
+ throw new CompileError(`${src} has unwaived native source-scope authoring errors:\n${unwaived.map((line) => ` - ${line}`).join("\n")}${diagnostics}`);
87
77
  }
88
- finally {
89
- releaseRunLock(cwd);
78
+ // One bounded read supplies both cross-run surfaces: unresolved findings below and merge facts for
79
+ // the ancestry check. Neither fact mutates the compiled graph; status and every readiness predicate
80
+ // remain the source compiler's answer.
81
+ const prior = readPriorRunEvidence(cwd, g.tasks);
82
+ const mergedPending = await mergedPendingDiagnostics(cwd, new Set(g.tasks.filter((task) => task.status === "pending").map((task) => task.id)), prior.merges);
83
+ const stateDir = stateDirName(cwd);
84
+ if (!values["dry-run"]) {
85
+ // HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around both truth records so
86
+ // compile cannot swap graph.json under an active run between the daemon's read and act.
87
+ stateWriteStarted = true;
88
+ acquireRunLock(cwd, "compile");
89
+ try {
90
+ saveGraph(cwd, g);
91
+ clearCompileRefusal(cwd);
92
+ }
93
+ finally {
94
+ releaseRunLock(cwd);
95
+ }
96
+ }
97
+ const summary = values["dry-run"]
98
+ ? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
99
+ : `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
100
+ const priorFindings = prior.findings.length
101
+ ? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
102
+ : "";
103
+ const mergeHistory = mergedPending.length
104
+ ? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
105
+ : "";
106
+ return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
107
+ }
108
+ catch (error) {
109
+ // A dry run is a pure validation query. A real authoring refusal records the negative result
110
+ // without replacing the last good graph; run treats this sibling as newer truth than that graph.
111
+ // State-write failures are excluded: a live daemon's lock refusal is not a verdict on the spec.
112
+ if (!values["dry-run"] && !stateWriteStarted) {
113
+ try {
114
+ acquireRunLock(cwd, "compile-refusal");
115
+ try {
116
+ saveCompileRefusal(cwd, {
117
+ refusedAt: new Date().toISOString(),
118
+ source: sourcePath,
119
+ error: error instanceof Error ? error.message : String(error),
120
+ });
121
+ }
122
+ finally {
123
+ releaseRunLock(cwd);
124
+ }
125
+ }
126
+ catch {
127
+ // The refusal record is best-effort when a live run owns the state boundary. That lock
128
+ // verdict must never replace the compile error that tells the operator what to repair.
129
+ }
90
130
  }
131
+ throw error;
91
132
  }
92
- const summary = values["dry-run"]
93
- ? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
94
- : `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
95
- const priorFindings = prior.findings.length
96
- ? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
97
- : "";
98
- const mergeHistory = mergedPending.length
99
- ? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
100
- : "";
101
- return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
102
133
  }