tickmarkr 2.4.0 → 2.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +11 -0
- package/dist/adapters/catalog-remote.js +55 -13
- package/dist/adapters/claude-code.js +7 -2
- package/dist/adapters/codex.d.ts +1 -0
- package/dist/adapters/codex.js +68 -8
- package/dist/adapters/qwen.js +3 -3
- package/dist/adapters/registry.js +13 -1
- package/dist/adapters/types.d.ts +4 -0
- package/dist/adapters/types.js +33 -0
- package/dist/cli/commands/compile.d.ts +3 -0
- package/dist/cli/commands/compile.js +77 -46
- package/dist/cli/commands/doctor.js +14 -9
- package/dist/cli/commands/fleet.js +11 -5
- package/dist/cli/commands/init.js +12 -13
- package/dist/cli/commands/plan.js +47 -7
- package/dist/cli/commands/run.js +20 -1
- package/dist/cli/commands/status.js +20 -21
- package/dist/cli/commands/version.js +2 -2
- package/dist/compile/collateral.d.ts +14 -5
- package/dist/compile/collateral.js +32 -26
- package/dist/compile/index.js +17 -6
- package/dist/compile/native.js +10 -2
- package/dist/compile/ownership.js +15 -9
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +32 -3
- package/dist/drivers/orca.d.ts +5 -1
- package/dist/drivers/orca.js +51 -2
- package/dist/drivers/types.d.ts +1 -1
- package/dist/gates/baseline.d.ts +26 -2
- package/dist/gates/baseline.js +90 -7
- package/dist/gates/review.d.ts +2 -2
- package/dist/gates/review.js +2 -18
- package/dist/graph/graph.d.ts +20 -0
- package/dist/graph/graph.js +66 -1
- package/dist/route/preference.d.ts +3 -1
- package/dist/route/preference.js +4 -4
- package/dist/run/consult.js +1 -0
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +151 -22
- package/dist/run/git.d.ts +2 -0
- package/dist/run/git.js +18 -4
- package/dist/run/journal.d.ts +1 -1
- package/dist/run/journal.js +1 -1
- package/dist/run/lock.d.ts +6 -0
- package/dist/run/lock.js +41 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +39 -4
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
|
@@ -11,6 +11,7 @@ export interface CatalogCache {
|
|
|
11
11
|
modelsDev: unknown;
|
|
12
12
|
artificialAnalysis?: unknown;
|
|
13
13
|
liveBench?: unknown;
|
|
14
|
+
legFetchedAt?: Partial<Record<"modelsDev" | "artificialAnalysis" | "liveBench", string>>;
|
|
14
15
|
}
|
|
15
16
|
export interface CatalogReadResult {
|
|
16
17
|
catalog: CatalogCache;
|
|
@@ -47,11 +48,21 @@ export interface RefreshCatalogOptions {
|
|
|
47
48
|
timeoutMs?: number;
|
|
48
49
|
now?: () => Date;
|
|
49
50
|
}
|
|
51
|
+
export type CatalogLegName = "models.dev" | "Artificial Analysis" | "LiveBench";
|
|
52
|
+
export type CatalogLegStatus = "updated" | "failed" | "skipped";
|
|
53
|
+
export interface CatalogLegResult {
|
|
54
|
+
leg: CatalogLegName;
|
|
55
|
+
status: CatalogLegStatus;
|
|
56
|
+
detail: string;
|
|
57
|
+
retry: string;
|
|
58
|
+
}
|
|
50
59
|
export interface RefreshCatalogResult {
|
|
51
60
|
updated: boolean;
|
|
52
61
|
catalog: CatalogReadResult;
|
|
53
62
|
warning?: string;
|
|
63
|
+
legs: CatalogLegResult[];
|
|
54
64
|
}
|
|
65
|
+
export declare const formatCatalogRefreshLegs: (legs: readonly CatalogLegResult[]) => string;
|
|
55
66
|
export declare const catalogCachePath: (repoRoot: string) => string;
|
|
56
67
|
/**
|
|
57
68
|
* Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
|
|
@@ -60,9 +60,14 @@ const validModelsDevCatalog = (value) => {
|
|
|
60
60
|
});
|
|
61
61
|
};
|
|
62
62
|
const staleAt = (fetchedAt, now) => {
|
|
63
|
-
const age = now.getTime() - Date.parse(fetchedAt);
|
|
63
|
+
const age = now.getTime() - Date.parse(fetchedAt ?? "");
|
|
64
64
|
return !Number.isFinite(age) || age > CATALOG_CACHE_MAX_AGE_MS;
|
|
65
65
|
};
|
|
66
|
+
const legFetchedAt = (catalog, leg) => catalog.legFetchedAt?.[leg] ?? catalog.fetchedAt;
|
|
67
|
+
const catalogStale = (catalog, now, source) => source === "vendored" || staleAt(legFetchedAt(catalog, "modelsDev"), now) || staleAt(legFetchedAt(catalog, "liveBench"), now)
|
|
68
|
+
|| (catalog.artificialAnalysis !== undefined && catalog.legFetchedAt?.artificialAnalysis !== undefined
|
|
69
|
+
&& staleAt(catalog.legFetchedAt.artificialAnalysis, now));
|
|
70
|
+
export const formatCatalogRefreshLegs = (legs) => `catalog refresh: ${legs.map((leg) => `${leg.leg} ${leg.status} (${leg.detail}; ${leg.retry})`).join("; ")}`;
|
|
66
71
|
export const catalogCachePath = (repoRoot) => join(repoRoot, ".tickmarkr", "catalog-cache.json");
|
|
67
72
|
/**
|
|
68
73
|
* Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
|
|
@@ -74,14 +79,14 @@ export function readCachedCatalog(repoRoot, opts = {}) {
|
|
|
74
79
|
const parsed = JSON.parse(readFileSync(catalogCachePath(repoRoot), "utf8"));
|
|
75
80
|
if (!validCache(parsed))
|
|
76
81
|
throw new Error("catalog cache schema is invalid");
|
|
77
|
-
return { catalog: parsed, source: "cache", stale:
|
|
82
|
+
return { catalog: parsed, source: "cache", stale: catalogStale(parsed, now, "cache") };
|
|
78
83
|
}
|
|
79
84
|
catch (error) {
|
|
80
85
|
const missing = error.code === "ENOENT";
|
|
81
86
|
return {
|
|
82
87
|
catalog: VENDORED_CATALOG,
|
|
83
88
|
source: "vendored",
|
|
84
|
-
stale:
|
|
89
|
+
stale: true,
|
|
85
90
|
...(!missing ? { warning: error instanceof Error ? error.message : String(error) } : {}),
|
|
86
91
|
};
|
|
87
92
|
}
|
|
@@ -410,6 +415,13 @@ export async function refreshCatalogCommand(opts) {
|
|
|
410
415
|
const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
|
|
411
416
|
const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
|
|
412
417
|
const warnings = [];
|
|
418
|
+
const legs = [];
|
|
419
|
+
const stamp = now().toISOString();
|
|
420
|
+
const legAt = {
|
|
421
|
+
modelsDev: legFetchedAt(current.catalog, "modelsDev"),
|
|
422
|
+
liveBench: legFetchedAt(current.catalog, "liveBench"),
|
|
423
|
+
...(current.catalog.artificialAnalysis !== undefined ? { artificialAnalysis: legFetchedAt(current.catalog, "artificialAnalysis") } : {}),
|
|
424
|
+
};
|
|
413
425
|
let modelsDev = current.catalog.modelsDev;
|
|
414
426
|
let modelsDevUpdated = false;
|
|
415
427
|
try {
|
|
@@ -417,46 +429,76 @@ export async function refreshCatalogCommand(opts) {
|
|
|
417
429
|
if (!validModelsDevCatalog(modelsDev))
|
|
418
430
|
throw new Error("models.dev catalog schema is invalid");
|
|
419
431
|
modelsDevUpdated = true;
|
|
432
|
+
legAt.modelsDev = stamp;
|
|
433
|
+
legs.push({ leg: "models.dev", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
|
|
420
434
|
}
|
|
421
435
|
catch (error) {
|
|
422
|
-
|
|
436
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
437
|
+
warnings.push(`models.dev refresh failed: ${detail}`);
|
|
438
|
+
legs.push({ leg: "models.dev", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
423
439
|
}
|
|
424
440
|
let artificialAnalysis = current.catalog.artificialAnalysis;
|
|
441
|
+
let artificialAnalysisUpdated = false;
|
|
425
442
|
const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
|
|
426
443
|
if (apiKey) {
|
|
427
444
|
try {
|
|
428
445
|
artificialAnalysis = await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs);
|
|
446
|
+
artificialAnalysisUpdated = true;
|
|
447
|
+
legAt.artificialAnalysis = stamp;
|
|
448
|
+
legs.push({ leg: "Artificial Analysis", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
|
|
429
449
|
}
|
|
430
450
|
catch (error) {
|
|
431
|
-
|
|
451
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
452
|
+
warnings.push(`Artificial Analysis refresh failed: ${detail}`);
|
|
453
|
+
legs.push({ leg: "Artificial Analysis", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
432
454
|
}
|
|
433
455
|
}
|
|
456
|
+
else {
|
|
457
|
+
delete legAt.artificialAnalysis;
|
|
458
|
+
legs.push({ leg: "Artificial Analysis", status: "skipped", detail: "no API key", retry: "retry when ARTIFICIAL_ANALYSIS_API_KEY is set" });
|
|
459
|
+
}
|
|
434
460
|
let liveBench = current.catalog.liveBench;
|
|
435
461
|
let liveBenchUpdated = false;
|
|
436
462
|
try {
|
|
437
463
|
liveBench = await fetchLiveBench(fetcher, timeoutMs);
|
|
438
464
|
liveBenchUpdated = true;
|
|
465
|
+
legAt.liveBench = stamp;
|
|
466
|
+
legs.push({ leg: "LiveBench", status: "updated", detail: `table ${LIVEBENCH_TABLE_DATE}`, retry: "next retry after this leg is stale" });
|
|
439
467
|
}
|
|
440
468
|
catch (error) {
|
|
441
|
-
|
|
469
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
470
|
+
warnings.push(`LiveBench refresh failed: ${detail}`);
|
|
471
|
+
legs.push({ leg: "LiveBench", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
442
472
|
}
|
|
443
|
-
|
|
444
|
-
// not write a cache that would make the vendored fallback look
|
|
445
|
-
|
|
446
|
-
|
|
473
|
+
const updated = modelsDevUpdated || artificialAnalysisUpdated || liveBenchUpdated;
|
|
474
|
+
// models.dev is the cache spine: do not write a cache that would make the vendored fallback look
|
|
475
|
+
// fetched. Independent legs may only merge onto models.dev fetched now or already held in cache.
|
|
476
|
+
if (!updated || (!modelsDevUpdated && current.source !== "cache")) {
|
|
477
|
+
const reportedLegs = !modelsDevUpdated && current.source !== "cache"
|
|
478
|
+
? legs.map((leg) => leg.status === "updated"
|
|
479
|
+
? {
|
|
480
|
+
...leg,
|
|
481
|
+
status: "failed",
|
|
482
|
+
detail: "fetched but discarded — models.dev spine unavailable",
|
|
483
|
+
retry: "retry next refresh",
|
|
484
|
+
}
|
|
485
|
+
: leg)
|
|
486
|
+
: legs;
|
|
487
|
+
return { updated: false, catalog: current, legs: reportedLegs, ...(warnings.length ? { warning: warnings.join("; ") } : {}) };
|
|
447
488
|
}
|
|
448
489
|
const catalog = {
|
|
449
490
|
schemaVersion: 1,
|
|
450
|
-
|
|
451
|
-
fetchedAt: liveBenchUpdated ? now().toISOString() : current.catalog.fetchedAt,
|
|
491
|
+
fetchedAt: legAt.modelsDev ?? current.catalog.fetchedAt,
|
|
452
492
|
modelsDev,
|
|
453
493
|
...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
|
|
454
494
|
...(liveBench !== undefined ? { liveBench } : {}),
|
|
495
|
+
legFetchedAt: legAt,
|
|
455
496
|
};
|
|
456
497
|
writeCatalogCache(opts.repoRoot, catalog);
|
|
457
498
|
return {
|
|
458
|
-
updated
|
|
499
|
+
updated,
|
|
459
500
|
catalog: readCachedCatalog(opts.repoRoot, { now }),
|
|
501
|
+
legs,
|
|
460
502
|
...(warnings.length ? { warning: warnings.join("; ") } : {}),
|
|
461
503
|
};
|
|
462
504
|
}
|
|
@@ -3,7 +3,7 @@ import { readdirSync, readFileSync, realpathSync, statSync } from "node:fs";
|
|
|
3
3
|
import { homedir } from "node:os";
|
|
4
4
|
import { join } from "node:path";
|
|
5
5
|
import { parseWorkerResult } from "./prompt.js";
|
|
6
|
-
import { channelsFromConfig, declareInputBox, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
|
|
6
|
+
import { channelsFromConfig, declareInputBox, MODEL_ID_RE, promptFitsArgv, shq, TokenUsageSchema } from "./types.js";
|
|
7
7
|
// SPEND-01/SPEND-11: claude writes a per-session JSONL to ~/.claude/projects/<slug>/ where slug is the
|
|
8
8
|
// realpath'd cwd with every non-alphanumeric char replaced by "-" (verified 114/114 — 36-DIAGNOSIS.md).
|
|
9
9
|
// The old `/`-only formula missed the "." in `.tickmarkr/worktrees/…` — ENOENT on every worktree dispatch.
|
|
@@ -239,7 +239,12 @@ export const claudeCode = {
|
|
|
239
239
|
// live check ate the prompt), and --prompt-suggestions takes an OPTIONAL value — appended directly
|
|
240
240
|
// before the prompt it would swallow it the same way. So the setting's value is always followed by
|
|
241
241
|
// another flag, never by the prompt positional.
|
|
242
|
-
|
|
242
|
+
// OBS-931: the same ONE-argv-string hazard as codex (OBS-930) — over promptArgvCeiling() the TUI
|
|
243
|
+
// launch would E2BIG on Linux, so it returns null → worker-mode-fallback → the headless form.
|
|
244
|
+
// resumeCommand keeps the shape: its contract returns a string (composer delivery is 2.4.3 work).
|
|
245
|
+
interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
|
|
246
|
+
? `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`
|
|
247
|
+
: null,
|
|
243
248
|
trustDialog: CLAUDE_TRUST_DIALOG,
|
|
244
249
|
inputBox: CLAUDE_INPUT_BOX,
|
|
245
250
|
// A resumed attempt lands in the same painted editor, so it carries the same ghost-text suppression
|
package/dist/adapters/codex.d.ts
CHANGED
|
@@ -4,6 +4,7 @@ export declare function readCodexModelsCache(path?: string): {
|
|
|
4
4
|
fetchedAt?: string;
|
|
5
5
|
};
|
|
6
6
|
export declare const CODEX_TRUST_DIALOG: TrustDialog;
|
|
7
|
+
export declare const CODEX_INPUT_BOX: import("./types.js").InputBox;
|
|
7
8
|
export declare function seedCodexTrust(repoRoot: string, configPath?: string): TrustVerdict;
|
|
8
9
|
export declare function hasCodexTrustedProject(text: string, root: string): boolean;
|
|
9
10
|
export declare function codexConfigMcpServerNames(configPath?: string): string[];
|
package/dist/adapters/codex.js
CHANGED
|
@@ -3,7 +3,7 @@ import { homedir } from "node:os";
|
|
|
3
3
|
import { dirname, join } from "node:path";
|
|
4
4
|
import { probeVersion } from "./claude-code.js";
|
|
5
5
|
import { parseWorkerResult } from "./prompt.js";
|
|
6
|
-
import { channelsFromConfig, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
|
|
6
|
+
import { channelsFromConfig, declareInputBox, MODEL_ID_RE, promptFitsArgv, shq, TokenUsageSchema } from "./types.js";
|
|
7
7
|
// SPEND-07: codex writes per-session JSONL to ~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl — date-partitioned,
|
|
8
8
|
// NOT cwd-keyed. session_meta.payload.cwd is FILE-SCOPED (one codex exec per cwd). token_count events carry
|
|
9
9
|
// per-turn DELTAS in payload.info.last_token_usage; we read POST-HOC (never the pane, never the trailer).
|
|
@@ -95,6 +95,56 @@ export const CODEX_TRUST_DIALOG = {
|
|
|
95
95
|
fingerprint: "Do you trust the contents of this directory?",
|
|
96
96
|
key: "Enter",
|
|
97
97
|
};
|
|
98
|
+
// OBS-930 / OBS-136: the codex TUI's composer, CAPTURED (codex 0.153.4, herdr pane read, 2026-09-05
|
|
99
|
+
// 15:48Z — tests/fixtures/codex-input-box, provenance in its README), never guessed:
|
|
100
|
+
// (background row)
|
|
101
|
+
// › Ask Codex to do anything ← U+203A, ASCII space, then the DIM placeholder (empty) or the draft
|
|
102
|
+
// (background row)
|
|
103
|
+
// gpt-5.6-luna medium · /private/tmp/tkr-obs930-smoke ← footer: <model> <effort> · <cwd>
|
|
104
|
+
// Two facts a fingerprint alone would get wrong: (1) a submitted turn is echoed into the transcript
|
|
105
|
+
// with the SAME caret (`› You are a smoke test.`), and so is the trust dialog's cursor (`› 1. Yes,
|
|
106
|
+
// continue`) — only the composer is followed by the footer row, so the footer is the anchor, exactly
|
|
107
|
+
// as claude's editor is anchored by its rules; (2) an EMPTY composer paints its placeholder as text,
|
|
108
|
+
// so "empty" is a closed allowlist of captured placeholders (0.153.4 above; 0.144.5's
|
|
109
|
+
// `Run /review on my current changes` from tests/fixtures/codex-mcp-spinner). An unknown placeholder
|
|
110
|
+
// reads as occupied and a submit onto it fails closed by name (OBS-140) — a fixture-capture chore,
|
|
111
|
+
// never a drive-by widening. `match` is THE COMPOSER IS PAINTED (true while empty, true mid-turn);
|
|
112
|
+
// `emptyMatch` is painted AND carrying nothing — the only positive evidence a submit registered.
|
|
113
|
+
const CODEX_ANSI_SGR_RE = /\u001B\[[0-9;]*m/g;
|
|
114
|
+
const CODEX_CARET_RE = /^› /;
|
|
115
|
+
const CODEX_PLACEHOLDER_RE = /^› (?:Ask Codex to do anything|Run \/review on my current changes)$/;
|
|
116
|
+
const CODEX_FOOTER_RE = /^\S+(?: \S+)? · \S/;
|
|
117
|
+
// A wrapped or multi-line draft grows the composer downward before the footer.
|
|
118
|
+
// ponytail: a fixed window, not a parser — raise it if a real capture ever shows a taller composer.
|
|
119
|
+
const CODEX_MAX_COMPOSER_ROWS = 8;
|
|
120
|
+
function matchesCodexComposer(paneText, empty) {
|
|
121
|
+
const lines = paneText.replace(CODEX_ANSI_SGR_RE, "").split("\n").map((l) => l.trim());
|
|
122
|
+
const caret = empty ? CODEX_PLACEHOLDER_RE : CODEX_CARET_RE;
|
|
123
|
+
return lines.some((line, i) => {
|
|
124
|
+
if (!caret.test(line))
|
|
125
|
+
return false;
|
|
126
|
+
for (let below = i + 1; below < lines.length && below <= i + CODEX_MAX_COMPOSER_ROWS; below++) {
|
|
127
|
+
if (CODEX_FOOTER_RE.test(lines[below]))
|
|
128
|
+
return true;
|
|
129
|
+
// only the composer's own rows may sit between the caret row and the footer: the background
|
|
130
|
+
// rows (blank in a text read) and a draft's continuation rows — never another caret row
|
|
131
|
+
if (lines[below] !== "" && CODEX_CARET_RE.test(lines[below]))
|
|
132
|
+
return false;
|
|
133
|
+
}
|
|
134
|
+
return false;
|
|
135
|
+
});
|
|
136
|
+
}
|
|
137
|
+
export const CODEX_INPUT_BOX = declareInputBox("codex", {
|
|
138
|
+
fingerprint: "› ",
|
|
139
|
+
match: (paneText) => matchesCodexComposer(paneText, false),
|
|
140
|
+
emptyMatch: (paneText) => matchesCodexComposer(paneText, true),
|
|
141
|
+
// As for claude (OBS-342): a fresh codex worker slot is a shell awaiting its launch line; every
|
|
142
|
+
// later delivery is a TUI turn awaiting this composer.
|
|
143
|
+
firstDeliveryIsLaunch: true,
|
|
144
|
+
// The 2026-09-05 capture painted the composer ~10 s after the trust answer with MCP suppressed;
|
|
145
|
+
// claude's bound, kept for the same cold-start reasons.
|
|
146
|
+
readinessTimeoutMs: 30_000,
|
|
147
|
+
});
|
|
98
148
|
// v1.22 T5 / OBS-16: codex keys trust on absolute path under [projects."<root>"] trust_level="trusted"
|
|
99
149
|
// in ~/.codex/config.toml (CODEX_HOME relocates the dir). Worktrees inherit parent-project trust when
|
|
100
150
|
// the REPO ROOT is trusted — seed the root once, cover every future worktree. Idempotent: a second
|
|
@@ -184,17 +234,26 @@ export const codex = {
|
|
|
184
234
|
probeConcurrency: 1,
|
|
185
235
|
probe: async () => probeVersion("codex"),
|
|
186
236
|
channels: (cfg) => channelsFromConfig("codex", cfg),
|
|
187
|
-
// v1.65 T3: every flag the command
|
|
237
|
+
// v1.65 T3: every flag the command builder below hardcodes (incl. codexMcpSuppressionFlags' -c/
|
|
188
238
|
// --disable and GITDIR_WRITABLE's -c) — all listed by top-level `codex --help`, verified 2026-07-22.
|
|
189
|
-
hardcodedFlags: { binary: "codex", flags: ["--sandbox", "
|
|
239
|
+
hardcodedFlags: { binary: "codex", flags: ["--sandbox", "-a", "-s", "--model", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
|
|
190
240
|
// --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
|
|
191
241
|
// MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
|
|
192
242
|
// CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
|
|
193
|
-
headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)}
|
|
194
|
-
//
|
|
195
|
-
// (--help
|
|
196
|
-
//
|
|
197
|
-
|
|
243
|
+
headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
|
|
244
|
+
// OBS-930: the visible pane runs the REAL TUI. Codex's TUI takes its prompt only as the [PROMPT]
|
|
245
|
+
// positional (`codex --help`, 0.153.4 — no file/stdin form), so the launch inlines the file exactly
|
|
246
|
+
// as the claude adapter does: the prompt is the LAST positional and every flag value is followed by
|
|
247
|
+
// a flag, never by the prompt. The argv hazard that once forbade this (OBS-889: `countLiveSuites`
|
|
248
|
+
// matched a suite word 140 KB into a finished worker's argv) is closed on the counter side — the
|
|
249
|
+
// census reads a command's first four tokens only — and those four never carry a suite word here.
|
|
250
|
+
// Same sandbox, hook trust and MCP suppression as the headless form; `-a never` is the TUI's
|
|
251
|
+
// autonomous approval policy (exec has no approvals to configure).
|
|
252
|
+
// OBS-930 (Linux): the inlined prompt is ONE argv string and Linux caps one at 131072 bytes, so a
|
|
253
|
+
// prompt over promptArgvCeiling() returns null → worker-mode-fallback → the headless form (types.ts).
|
|
254
|
+
interactiveCommand: (promptFile, model) => promptFitsArgv(promptFile)
|
|
255
|
+
? `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`
|
|
256
|
+
: null,
|
|
198
257
|
invoke(task, _cwd, a, ctx) {
|
|
199
258
|
return { command: this.headlessCommand(ctx.promptFile, a.model) };
|
|
200
259
|
},
|
|
@@ -203,6 +262,7 @@ export const codex = {
|
|
|
203
262
|
// "Do you trust this directory?" (OBS-16). doctor-only side effect.
|
|
204
263
|
trust: (repoRoot) => seedCodexTrust(repoRoot),
|
|
205
264
|
trustDialog: CODEX_TRUST_DIALOG,
|
|
265
|
+
inputBox: CODEX_INPUT_BOX,
|
|
206
266
|
// v1.5 MODEL-01: file read only (no `codex models` subcommand exists, verified 2026-07-10).
|
|
207
267
|
// Already fails OPEN to [] internally — advisory detection, unlike gates' fail-closed.
|
|
208
268
|
listModels: async () => readCodexModelsCache().models,
|
package/dist/adapters/qwen.js
CHANGED
|
@@ -58,7 +58,7 @@ function decodeQwenEvents(events) {
|
|
|
58
58
|
}
|
|
59
59
|
}
|
|
60
60
|
const assistantText = text.join("\n");
|
|
61
|
-
const apiError =
|
|
61
|
+
const apiError = assistantText.match(/\[API Error:[^\n]*/)?.[0];
|
|
62
62
|
if (apiError)
|
|
63
63
|
failed = true;
|
|
64
64
|
if (!failed)
|
|
@@ -134,8 +134,8 @@ export const qwen = {
|
|
|
134
134
|
probeCwd: "neutral",
|
|
135
135
|
probe: async () => probeQwen(),
|
|
136
136
|
channels: (cfg) => channelsFromConfig("qwen", cfg),
|
|
137
|
-
hardcodedFlags: { binary: "qwen", flags: ["--approval-mode", "-m", "-o", "-p"] },
|
|
138
|
-
headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
|
|
137
|
+
hardcodedFlags: { binary: "qwen", flags: ["--safe-mode", "--approval-mode", "-m", "-o", "-p"] },
|
|
138
|
+
headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --safe-mode --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
|
|
139
139
|
// OBS-905: qwen has NO interactive form. The `-i "$(cat prompt)"` TUI launch put the whole prompt in
|
|
140
140
|
// argv (the OBS-889 leak-and-census shape) and produced a rendered transcript the JSON decoder above
|
|
141
141
|
// can never read — under the herdr driver every qwen task read "unparseable" and merged only by harvest.
|
|
@@ -399,6 +399,17 @@ function reasonTail(output) {
|
|
|
399
399
|
const sp = tail.indexOf(" ");
|
|
400
400
|
return sp === -1 ? tail : tail.slice(sp + 1);
|
|
401
401
|
}
|
|
402
|
+
function parsedStartupFailureReason(adapter, output) {
|
|
403
|
+
try {
|
|
404
|
+
const parsed = adapter.parse(output, "tickmarkr-model-probe");
|
|
405
|
+
const cause = parsed.cause;
|
|
406
|
+
if (!parsed.ok && cause === "startup-failure" && parsed.summary.trim()) {
|
|
407
|
+
return reasonTail(parsed.summary.trim().replace(/\s+/g, " "));
|
|
408
|
+
}
|
|
409
|
+
}
|
|
410
|
+
catch { /* adapter parse is advisory for probe diagnostics */ }
|
|
411
|
+
return undefined;
|
|
412
|
+
}
|
|
402
413
|
function probeFailure(code, stdout, stderr, timedOut, timeoutMs = MODEL_PROBE_TIMEOUT_MS) {
|
|
403
414
|
// SIGKILL-timeout is not exit-1: report the budget, never the masked kill code (v1.27 T1).
|
|
404
415
|
if (timedOut)
|
|
@@ -456,7 +467,8 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
|
|
|
456
467
|
if (r.timedOut && !retry && priorModelAuth?.[model]?.reason?.includes("timed out") === true) {
|
|
457
468
|
return { verdict: v(false, `probe timed out (repeat — retry skipped) (${MODEL_PROBE_TIMEOUT_MS}ms)`), timedOut: true };
|
|
458
469
|
}
|
|
459
|
-
const
|
|
470
|
+
const output = `${r.stderr}\n${r.stdout}`.trim();
|
|
471
|
+
const reason = parsedStartupFailureReason(a, output) ?? probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
|
|
460
472
|
if (!reason)
|
|
461
473
|
return { verdict: v(true), timedOut: false };
|
|
462
474
|
if (!retry)
|
package/dist/adapters/types.d.ts
CHANGED
|
@@ -185,5 +185,9 @@ export declare function channelKey(c: {
|
|
|
185
185
|
model: string;
|
|
186
186
|
}): string;
|
|
187
187
|
export declare function shq(s: string): string;
|
|
188
|
+
export declare const PROMPT_ARGV_CEILING_LINUX = 120000;
|
|
189
|
+
export declare const PROMPT_ARGV_CEILING_DEFAULT = 900000;
|
|
190
|
+
export declare function promptArgvCeiling(platform?: string): number;
|
|
191
|
+
export declare function promptFitsArgv(promptFile: string, platform?: string): boolean;
|
|
188
192
|
export declare const QUOTA_RE: RegExp;
|
|
189
193
|
export declare const MODEL_ID_RE: RegExp;
|
package/dist/adapters/types.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { statSync } from "node:fs";
|
|
1
2
|
import { z } from "zod";
|
|
2
3
|
// SPEND-01/06: normalized token counts — the measurable fact. NO cost field, ever: CLIs report
|
|
3
4
|
// cost:0 on sub plans and notional list prices on others (LIVE-CHECK finding 3); money is Phase 18's
|
|
@@ -232,6 +233,38 @@ export function channelKey(c) {
|
|
|
232
233
|
export function shq(s) {
|
|
233
234
|
return `'${s.replaceAll("'", `'\\''`)}'`;
|
|
234
235
|
}
|
|
236
|
+
// OBS-930 (Linux) / OBS-931: a TUI launch inlines the prompt file as ONE argv string — "$(cat prompt)"
|
|
237
|
+
// as the last positional. Linux caps a single argv string at MAX_ARG_STRLEN = PAGE_SIZE × 32 =
|
|
238
|
+
// 131072 bytes (E2BIG: CI run 33979013874, ubuntu, `codex: Argument list too long` on a 140 KB
|
|
239
|
+
// prompt); darwin enforces only the 1 MB total ARG_MAX, which is why the macOS export proof passed.
|
|
240
|
+
// A real worker prompt is 60–150 KB (OBS-889 measured 149,417 bytes), so on a Linux host the launch
|
|
241
|
+
// fails in production, not only in the test. Ceilings are named BY PLATFORM — never a probe of the
|
|
242
|
+
// running kernel at dispatch time:
|
|
243
|
+
// linux 120_000 — the cap is per STRING and every flag is its own argv entry, so only the prompt
|
|
244
|
+
// counts against it; "$(cat …)" strips nothing but trailing newlines, so the
|
|
245
|
+
// file's byte size IS the string's. 131072 − 120000 leaves ~11 KB of headroom.
|
|
246
|
+
// others 900_000 — under the 1 MB total that darwin/BSD enforce across argv + envp.
|
|
247
|
+
// Over the ceiling the adapter returns null: the daemon journals worker-mode-fallback
|
|
248
|
+
// {reason:"adapter"} and runs the headless form in the visible pane — the pre-OBS-930 behaviour,
|
|
249
|
+
// now only for oversized prompts. An unreadable file is "not proven oversized" and keeps the TUI
|
|
250
|
+
// rendering: the daemon writes the prompt before it builds the launch, a missing file fails the
|
|
251
|
+
// same way in either form, and docs-truth renders the command against a placeholder path.
|
|
252
|
+
// OBS-931 (2.4.3): the real fix is prompt delivery through the composer for large prompts.
|
|
253
|
+
export const PROMPT_ARGV_CEILING_LINUX = 120_000;
|
|
254
|
+
export const PROMPT_ARGV_CEILING_DEFAULT = 900_000;
|
|
255
|
+
export function promptArgvCeiling(platform = process.platform) {
|
|
256
|
+
return platform === "linux" ? PROMPT_ARGV_CEILING_LINUX : PROMPT_ARGV_CEILING_DEFAULT;
|
|
257
|
+
}
|
|
258
|
+
export function promptFitsArgv(promptFile, platform = process.platform) {
|
|
259
|
+
let bytes;
|
|
260
|
+
try {
|
|
261
|
+
bytes = statSync(promptFile).size;
|
|
262
|
+
}
|
|
263
|
+
catch {
|
|
264
|
+
return true;
|
|
265
|
+
}
|
|
266
|
+
return bytes <= promptArgvCeiling(platform);
|
|
267
|
+
}
|
|
235
268
|
// Quota exhaustion is detected from CLI errors, never predicted (spec §4).
|
|
236
269
|
// ZAI coding-plan exhaustion text: "Insufficient balance or no resource package. Please recharge."
|
|
237
270
|
// Anchor the distinctive full phrase, not the two-word "insufficient balance" fragment — that fires
|
|
@@ -1 +1,4 @@
|
|
|
1
|
+
import { type SourceScopeFinding } from "../../compile/collateral.js";
|
|
2
|
+
import { compileSource } from "../../compile/index.js";
|
|
3
|
+
export declare function nativeSourceScopeErrors(graph: ReturnType<typeof compileSource>, findings: readonly SourceScopeFinding[]): string[];
|
|
1
4
|
export declare function compile(argv: string[], cwd?: string, harnessFrom?: string | undefined): Promise<string>;
|
|
@@ -1,25 +1,21 @@
|
|
|
1
1
|
import { isAbsolute, join } from "node:path";
|
|
2
2
|
import { parseArgs } from "node:util";
|
|
3
|
-
import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
|
|
3
|
+
import { collateralLints, sourceScopeFindings, sourceScopeLints, } from "../../compile/collateral.js";
|
|
4
4
|
import { CompileError } from "../../compile/common.js";
|
|
5
5
|
import { compileSource } from "../../compile/index.js";
|
|
6
|
-
import { saveGraph, stateDirName } from "../../graph/graph.js";
|
|
6
|
+
import { clearCompileRefusal, saveCompileRefusal, saveGraph, stateDirName } from "../../graph/graph.js";
|
|
7
7
|
import { formatPriorFindingEvidence, readPriorRunEvidence } from "../../run/journal.js";
|
|
8
8
|
import { shGit } from "../../run/git.js";
|
|
9
9
|
import { acquireRunLock, releaseRunLock } from "../../run/lock.js";
|
|
10
10
|
import { harnessLine, resolveHarness } from "../harness.js";
|
|
11
|
-
function nativeSourceScopeErrors(graph,
|
|
11
|
+
export function nativeSourceScopeErrors(graph, findings) {
|
|
12
12
|
const byId = new Map(graph.tasks.map((task) => [task.id, task]));
|
|
13
13
|
const errors = [];
|
|
14
|
-
for (const
|
|
15
|
-
const
|
|
16
|
-
if (!match)
|
|
17
|
-
continue;
|
|
18
|
-
const task = byId.get(match[1]);
|
|
14
|
+
for (const finding of findings) {
|
|
15
|
+
const task = byId.get(finding.taskId);
|
|
19
16
|
if (!task)
|
|
20
17
|
continue;
|
|
21
|
-
const
|
|
22
|
-
for (const path of paths) {
|
|
18
|
+
for (const path of finding.paths) {
|
|
23
19
|
if (task.goal.includes(`scope-waiver: ${path}`))
|
|
24
20
|
continue;
|
|
25
21
|
errors.push(`${task.id}: ${path} requires scope-waiver: ${path} in the task goal`);
|
|
@@ -60,43 +56,78 @@ export async function compile(argv, cwd = process.cwd(), harnessFrom = process.a
|
|
|
60
56
|
const src = positionals[0];
|
|
61
57
|
if (!src)
|
|
62
58
|
throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native] [--dry-run] [--strict]");
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
if (!values["dry-run"]) {
|
|
82
|
-
// HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
|
|
83
|
-
// cannot swap graph.json under an active run between the daemon's read and act.
|
|
84
|
-
acquireRunLock(cwd, "compile");
|
|
85
|
-
try {
|
|
86
|
-
saveGraph(cwd, g);
|
|
59
|
+
const sourcePath = isAbsolute(src) ? src : join(cwd, src);
|
|
60
|
+
let stateWriteStarted = false;
|
|
61
|
+
try {
|
|
62
|
+
// Resolve against the target repo, not the process cwd (the CLI test passes a tmp repo).
|
|
63
|
+
// Both modes reach the same pure compiler; --dry-run removes every state write below.
|
|
64
|
+
const g = compileSource(sourcePath, values.type, cwd, // repo root: gsd stores context[0] repo-relative so workers resolve it inside their worktree
|
|
65
|
+
(plan) => plan, { strict: values.strict });
|
|
66
|
+
const sourceFindings = sourceScopeFindings(g.tasks, cwd);
|
|
67
|
+
const scopeLints = [
|
|
68
|
+
...collateralLints(g.tasks, cwd),
|
|
69
|
+
...sourceScopeLints(g.tasks, cwd, sourceFindings),
|
|
70
|
+
];
|
|
71
|
+
const diagnostics = scopeLints.length
|
|
72
|
+
? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
|
|
73
|
+
: "";
|
|
74
|
+
const unwaived = g.spec.source === "native" ? nativeSourceScopeErrors(g, sourceFindings) : [];
|
|
75
|
+
if (unwaived.length > 0) {
|
|
76
|
+
throw new CompileError(`${src} has unwaived native source-scope authoring errors:\n${unwaived.map((line) => ` - ${line}`).join("\n")}${diagnostics}`);
|
|
87
77
|
}
|
|
88
|
-
|
|
89
|
-
|
|
78
|
+
// One bounded read supplies both cross-run surfaces: unresolved findings below and merge facts for
|
|
79
|
+
// the ancestry check. Neither fact mutates the compiled graph; status and every readiness predicate
|
|
80
|
+
// remain the source compiler's answer.
|
|
81
|
+
const prior = readPriorRunEvidence(cwd, g.tasks);
|
|
82
|
+
const mergedPending = await mergedPendingDiagnostics(cwd, new Set(g.tasks.filter((task) => task.status === "pending").map((task) => task.id)), prior.merges);
|
|
83
|
+
const stateDir = stateDirName(cwd);
|
|
84
|
+
if (!values["dry-run"]) {
|
|
85
|
+
// HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around both truth records so
|
|
86
|
+
// compile cannot swap graph.json under an active run between the daemon's read and act.
|
|
87
|
+
stateWriteStarted = true;
|
|
88
|
+
acquireRunLock(cwd, "compile");
|
|
89
|
+
try {
|
|
90
|
+
saveGraph(cwd, g);
|
|
91
|
+
clearCompileRefusal(cwd);
|
|
92
|
+
}
|
|
93
|
+
finally {
|
|
94
|
+
releaseRunLock(cwd);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
const summary = values["dry-run"]
|
|
98
|
+
? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
|
|
99
|
+
: `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
|
|
100
|
+
const priorFindings = prior.findings.length
|
|
101
|
+
? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
|
|
102
|
+
: "";
|
|
103
|
+
const mergeHistory = mergedPending.length
|
|
104
|
+
? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
|
|
105
|
+
: "";
|
|
106
|
+
return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
|
|
107
|
+
}
|
|
108
|
+
catch (error) {
|
|
109
|
+
// A dry run is a pure validation query. A real authoring refusal records the negative result
|
|
110
|
+
// without replacing the last good graph; run treats this sibling as newer truth than that graph.
|
|
111
|
+
// State-write failures are excluded: a live daemon's lock refusal is not a verdict on the spec.
|
|
112
|
+
if (!values["dry-run"] && !stateWriteStarted) {
|
|
113
|
+
try {
|
|
114
|
+
acquireRunLock(cwd, "compile-refusal");
|
|
115
|
+
try {
|
|
116
|
+
saveCompileRefusal(cwd, {
|
|
117
|
+
refusedAt: new Date().toISOString(),
|
|
118
|
+
source: sourcePath,
|
|
119
|
+
error: error instanceof Error ? error.message : String(error),
|
|
120
|
+
});
|
|
121
|
+
}
|
|
122
|
+
finally {
|
|
123
|
+
releaseRunLock(cwd);
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
catch {
|
|
127
|
+
// The refusal record is best-effort when a live run owns the state boundary. That lock
|
|
128
|
+
// verdict must never replace the compile error that tells the operator what to repair.
|
|
129
|
+
}
|
|
90
130
|
}
|
|
131
|
+
throw error;
|
|
91
132
|
}
|
|
92
|
-
const summary = values["dry-run"]
|
|
93
|
-
? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
|
|
94
|
-
: `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
|
|
95
|
-
const priorFindings = prior.findings.length
|
|
96
|
-
? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
|
|
97
|
-
: "";
|
|
98
|
-
const mergeHistory = mergedPending.length
|
|
99
|
-
? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
|
|
100
|
-
: "";
|
|
101
|
-
return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
|
|
102
133
|
}
|