tickmarkr 2.4.0 → 2.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +11 -0
- package/dist/adapters/catalog-remote.js +55 -13
- package/dist/adapters/codex.js +6 -7
- package/dist/adapters/qwen.js +3 -3
- package/dist/adapters/registry.js +13 -1
- package/dist/cli/commands/compile.d.ts +3 -0
- package/dist/cli/commands/compile.js +77 -46
- package/dist/cli/commands/doctor.js +14 -9
- package/dist/cli/commands/fleet.js +11 -5
- package/dist/cli/commands/init.js +12 -13
- package/dist/cli/commands/plan.js +47 -7
- package/dist/cli/commands/run.js +20 -1
- package/dist/cli/commands/status.js +20 -21
- package/dist/cli/commands/version.js +2 -2
- package/dist/compile/collateral.d.ts +14 -5
- package/dist/compile/collateral.js +32 -26
- package/dist/compile/index.js +17 -6
- package/dist/compile/native.js +10 -2
- package/dist/compile/ownership.js +15 -9
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +32 -3
- package/dist/drivers/orca.d.ts +5 -1
- package/dist/drivers/orca.js +51 -2
- package/dist/drivers/types.d.ts +1 -1
- package/dist/gates/baseline.d.ts +26 -2
- package/dist/gates/baseline.js +90 -7
- package/dist/gates/review.d.ts +2 -2
- package/dist/gates/review.js +2 -18
- package/dist/graph/graph.d.ts +20 -0
- package/dist/graph/graph.js +66 -1
- package/dist/route/preference.d.ts +3 -1
- package/dist/route/preference.js +4 -4
- package/dist/run/consult.js +1 -0
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +151 -22
- package/dist/run/git.d.ts +2 -0
- package/dist/run/git.js +18 -4
- package/dist/run/journal.d.ts +1 -1
- package/dist/run/journal.js +1 -1
- package/dist/run/lock.d.ts +6 -0
- package/dist/run/lock.js +41 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +39 -4
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
|
@@ -11,6 +11,7 @@ export interface CatalogCache {
|
|
|
11
11
|
modelsDev: unknown;
|
|
12
12
|
artificialAnalysis?: unknown;
|
|
13
13
|
liveBench?: unknown;
|
|
14
|
+
legFetchedAt?: Partial<Record<"modelsDev" | "artificialAnalysis" | "liveBench", string>>;
|
|
14
15
|
}
|
|
15
16
|
export interface CatalogReadResult {
|
|
16
17
|
catalog: CatalogCache;
|
|
@@ -47,11 +48,21 @@ export interface RefreshCatalogOptions {
|
|
|
47
48
|
timeoutMs?: number;
|
|
48
49
|
now?: () => Date;
|
|
49
50
|
}
|
|
51
|
+
export type CatalogLegName = "models.dev" | "Artificial Analysis" | "LiveBench";
|
|
52
|
+
export type CatalogLegStatus = "updated" | "failed" | "skipped";
|
|
53
|
+
export interface CatalogLegResult {
|
|
54
|
+
leg: CatalogLegName;
|
|
55
|
+
status: CatalogLegStatus;
|
|
56
|
+
detail: string;
|
|
57
|
+
retry: string;
|
|
58
|
+
}
|
|
50
59
|
export interface RefreshCatalogResult {
|
|
51
60
|
updated: boolean;
|
|
52
61
|
catalog: CatalogReadResult;
|
|
53
62
|
warning?: string;
|
|
63
|
+
legs: CatalogLegResult[];
|
|
54
64
|
}
|
|
65
|
+
export declare const formatCatalogRefreshLegs: (legs: readonly CatalogLegResult[]) => string;
|
|
55
66
|
export declare const catalogCachePath: (repoRoot: string) => string;
|
|
56
67
|
/**
|
|
57
68
|
* Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
|
|
@@ -60,9 +60,14 @@ const validModelsDevCatalog = (value) => {
|
|
|
60
60
|
});
|
|
61
61
|
};
|
|
62
62
|
const staleAt = (fetchedAt, now) => {
|
|
63
|
-
const age = now.getTime() - Date.parse(fetchedAt);
|
|
63
|
+
const age = now.getTime() - Date.parse(fetchedAt ?? "");
|
|
64
64
|
return !Number.isFinite(age) || age > CATALOG_CACHE_MAX_AGE_MS;
|
|
65
65
|
};
|
|
66
|
+
const legFetchedAt = (catalog, leg) => catalog.legFetchedAt?.[leg] ?? catalog.fetchedAt;
|
|
67
|
+
const catalogStale = (catalog, now, source) => source === "vendored" || staleAt(legFetchedAt(catalog, "modelsDev"), now) || staleAt(legFetchedAt(catalog, "liveBench"), now)
|
|
68
|
+
|| (catalog.artificialAnalysis !== undefined && catalog.legFetchedAt?.artificialAnalysis !== undefined
|
|
69
|
+
&& staleAt(catalog.legFetchedAt.artificialAnalysis, now));
|
|
70
|
+
export const formatCatalogRefreshLegs = (legs) => `catalog refresh: ${legs.map((leg) => `${leg.leg} ${leg.status} (${leg.detail}; ${leg.retry})`).join("; ")}`;
|
|
66
71
|
export const catalogCachePath = (repoRoot) => join(repoRoot, ".tickmarkr", "catalog-cache.json");
|
|
67
72
|
/**
|
|
68
73
|
* Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
|
|
@@ -74,14 +79,14 @@ export function readCachedCatalog(repoRoot, opts = {}) {
|
|
|
74
79
|
const parsed = JSON.parse(readFileSync(catalogCachePath(repoRoot), "utf8"));
|
|
75
80
|
if (!validCache(parsed))
|
|
76
81
|
throw new Error("catalog cache schema is invalid");
|
|
77
|
-
return { catalog: parsed, source: "cache", stale:
|
|
82
|
+
return { catalog: parsed, source: "cache", stale: catalogStale(parsed, now, "cache") };
|
|
78
83
|
}
|
|
79
84
|
catch (error) {
|
|
80
85
|
const missing = error.code === "ENOENT";
|
|
81
86
|
return {
|
|
82
87
|
catalog: VENDORED_CATALOG,
|
|
83
88
|
source: "vendored",
|
|
84
|
-
stale:
|
|
89
|
+
stale: true,
|
|
85
90
|
...(!missing ? { warning: error instanceof Error ? error.message : String(error) } : {}),
|
|
86
91
|
};
|
|
87
92
|
}
|
|
@@ -410,6 +415,13 @@ export async function refreshCatalogCommand(opts) {
|
|
|
410
415
|
const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
|
|
411
416
|
const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
|
|
412
417
|
const warnings = [];
|
|
418
|
+
const legs = [];
|
|
419
|
+
const stamp = now().toISOString();
|
|
420
|
+
const legAt = {
|
|
421
|
+
modelsDev: legFetchedAt(current.catalog, "modelsDev"),
|
|
422
|
+
liveBench: legFetchedAt(current.catalog, "liveBench"),
|
|
423
|
+
...(current.catalog.artificialAnalysis !== undefined ? { artificialAnalysis: legFetchedAt(current.catalog, "artificialAnalysis") } : {}),
|
|
424
|
+
};
|
|
413
425
|
let modelsDev = current.catalog.modelsDev;
|
|
414
426
|
let modelsDevUpdated = false;
|
|
415
427
|
try {
|
|
@@ -417,46 +429,76 @@ export async function refreshCatalogCommand(opts) {
|
|
|
417
429
|
if (!validModelsDevCatalog(modelsDev))
|
|
418
430
|
throw new Error("models.dev catalog schema is invalid");
|
|
419
431
|
modelsDevUpdated = true;
|
|
432
|
+
legAt.modelsDev = stamp;
|
|
433
|
+
legs.push({ leg: "models.dev", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
|
|
420
434
|
}
|
|
421
435
|
catch (error) {
|
|
422
|
-
|
|
436
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
437
|
+
warnings.push(`models.dev refresh failed: ${detail}`);
|
|
438
|
+
legs.push({ leg: "models.dev", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
423
439
|
}
|
|
424
440
|
let artificialAnalysis = current.catalog.artificialAnalysis;
|
|
441
|
+
let artificialAnalysisUpdated = false;
|
|
425
442
|
const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
|
|
426
443
|
if (apiKey) {
|
|
427
444
|
try {
|
|
428
445
|
artificialAnalysis = await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs);
|
|
446
|
+
artificialAnalysisUpdated = true;
|
|
447
|
+
legAt.artificialAnalysis = stamp;
|
|
448
|
+
legs.push({ leg: "Artificial Analysis", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
|
|
429
449
|
}
|
|
430
450
|
catch (error) {
|
|
431
|
-
|
|
451
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
452
|
+
warnings.push(`Artificial Analysis refresh failed: ${detail}`);
|
|
453
|
+
legs.push({ leg: "Artificial Analysis", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
432
454
|
}
|
|
433
455
|
}
|
|
456
|
+
else {
|
|
457
|
+
delete legAt.artificialAnalysis;
|
|
458
|
+
legs.push({ leg: "Artificial Analysis", status: "skipped", detail: "no API key", retry: "retry when ARTIFICIAL_ANALYSIS_API_KEY is set" });
|
|
459
|
+
}
|
|
434
460
|
let liveBench = current.catalog.liveBench;
|
|
435
461
|
let liveBenchUpdated = false;
|
|
436
462
|
try {
|
|
437
463
|
liveBench = await fetchLiveBench(fetcher, timeoutMs);
|
|
438
464
|
liveBenchUpdated = true;
|
|
465
|
+
legAt.liveBench = stamp;
|
|
466
|
+
legs.push({ leg: "LiveBench", status: "updated", detail: `table ${LIVEBENCH_TABLE_DATE}`, retry: "next retry after this leg is stale" });
|
|
439
467
|
}
|
|
440
468
|
catch (error) {
|
|
441
|
-
|
|
469
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
470
|
+
warnings.push(`LiveBench refresh failed: ${detail}`);
|
|
471
|
+
legs.push({ leg: "LiveBench", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
442
472
|
}
|
|
443
|
-
|
|
444
|
-
// not write a cache that would make the vendored fallback look
|
|
445
|
-
|
|
446
|
-
|
|
473
|
+
const updated = modelsDevUpdated || artificialAnalysisUpdated || liveBenchUpdated;
|
|
474
|
+
// models.dev is the cache spine: do not write a cache that would make the vendored fallback look
|
|
475
|
+
// fetched. Independent legs may only merge onto models.dev fetched now or already held in cache.
|
|
476
|
+
if (!updated || (!modelsDevUpdated && current.source !== "cache")) {
|
|
477
|
+
const reportedLegs = !modelsDevUpdated && current.source !== "cache"
|
|
478
|
+
? legs.map((leg) => leg.status === "updated"
|
|
479
|
+
? {
|
|
480
|
+
...leg,
|
|
481
|
+
status: "failed",
|
|
482
|
+
detail: "fetched but discarded — models.dev spine unavailable",
|
|
483
|
+
retry: "retry next refresh",
|
|
484
|
+
}
|
|
485
|
+
: leg)
|
|
486
|
+
: legs;
|
|
487
|
+
return { updated: false, catalog: current, legs: reportedLegs, ...(warnings.length ? { warning: warnings.join("; ") } : {}) };
|
|
447
488
|
}
|
|
448
489
|
const catalog = {
|
|
449
490
|
schemaVersion: 1,
|
|
450
|
-
|
|
451
|
-
fetchedAt: liveBenchUpdated ? now().toISOString() : current.catalog.fetchedAt,
|
|
491
|
+
fetchedAt: legAt.modelsDev ?? current.catalog.fetchedAt,
|
|
452
492
|
modelsDev,
|
|
453
493
|
...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
|
|
454
494
|
...(liveBench !== undefined ? { liveBench } : {}),
|
|
495
|
+
legFetchedAt: legAt,
|
|
455
496
|
};
|
|
456
497
|
writeCatalogCache(opts.repoRoot, catalog);
|
|
457
498
|
return {
|
|
458
|
-
updated
|
|
499
|
+
updated,
|
|
459
500
|
catalog: readCachedCatalog(opts.repoRoot, { now }),
|
|
501
|
+
legs,
|
|
460
502
|
...(warnings.length ? { warning: warnings.join("; ") } : {}),
|
|
461
503
|
};
|
|
462
504
|
}
|
package/dist/adapters/codex.js
CHANGED
|
@@ -184,17 +184,16 @@ export const codex = {
|
|
|
184
184
|
probeConcurrency: 1,
|
|
185
185
|
probe: async () => probeVersion("codex"),
|
|
186
186
|
channels: (cfg) => channelsFromConfig("codex", cfg),
|
|
187
|
-
// v1.65 T3: every flag the command
|
|
187
|
+
// v1.65 T3: every flag the command builder below hardcodes (incl. codexMcpSuppressionFlags' -c/
|
|
188
188
|
// --disable and GITDIR_WRITABLE's -c) — all listed by top-level `codex --help`, verified 2026-07-22.
|
|
189
|
-
hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-
|
|
189
|
+
hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
|
|
190
190
|
// --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
|
|
191
191
|
// MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
|
|
192
192
|
// CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
|
|
193
|
-
headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)}
|
|
194
|
-
// TUI
|
|
195
|
-
//
|
|
196
|
-
|
|
197
|
-
interactiveCommand: (promptFile, model) => `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
|
|
193
|
+
headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
|
|
194
|
+
// OBS-889: Codex's TUI has no file/stdin prompt form. Returning null makes the daemon journal
|
|
195
|
+
// worker-mode-fallback before it runs the argv-safe headless command in the visible pane.
|
|
196
|
+
interactiveCommand: () => null,
|
|
198
197
|
invoke(task, _cwd, a, ctx) {
|
|
199
198
|
return { command: this.headlessCommand(ctx.promptFile, a.model) };
|
|
200
199
|
},
|
package/dist/adapters/qwen.js
CHANGED
|
@@ -58,7 +58,7 @@ function decodeQwenEvents(events) {
|
|
|
58
58
|
}
|
|
59
59
|
}
|
|
60
60
|
const assistantText = text.join("\n");
|
|
61
|
-
const apiError =
|
|
61
|
+
const apiError = assistantText.match(/\[API Error:[^\n]*/)?.[0];
|
|
62
62
|
if (apiError)
|
|
63
63
|
failed = true;
|
|
64
64
|
if (!failed)
|
|
@@ -134,8 +134,8 @@ export const qwen = {
|
|
|
134
134
|
probeCwd: "neutral",
|
|
135
135
|
probe: async () => probeQwen(),
|
|
136
136
|
channels: (cfg) => channelsFromConfig("qwen", cfg),
|
|
137
|
-
hardcodedFlags: { binary: "qwen", flags: ["--approval-mode", "-m", "-o", "-p"] },
|
|
138
|
-
headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
|
|
137
|
+
hardcodedFlags: { binary: "qwen", flags: ["--safe-mode", "--approval-mode", "-m", "-o", "-p"] },
|
|
138
|
+
headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --safe-mode --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
|
|
139
139
|
// OBS-905: qwen has NO interactive form. The `-i "$(cat prompt)"` TUI launch put the whole prompt in
|
|
140
140
|
// argv (the OBS-889 leak-and-census shape) and produced a rendered transcript the JSON decoder above
|
|
141
141
|
// can never read — under the herdr driver every qwen task read "unparseable" and merged only by harvest.
|
|
@@ -399,6 +399,17 @@ function reasonTail(output) {
|
|
|
399
399
|
const sp = tail.indexOf(" ");
|
|
400
400
|
return sp === -1 ? tail : tail.slice(sp + 1);
|
|
401
401
|
}
|
|
402
|
+
function parsedStartupFailureReason(adapter, output) {
|
|
403
|
+
try {
|
|
404
|
+
const parsed = adapter.parse(output, "tickmarkr-model-probe");
|
|
405
|
+
const cause = parsed.cause;
|
|
406
|
+
if (!parsed.ok && cause === "startup-failure" && parsed.summary.trim()) {
|
|
407
|
+
return reasonTail(parsed.summary.trim().replace(/\s+/g, " "));
|
|
408
|
+
}
|
|
409
|
+
}
|
|
410
|
+
catch { /* adapter parse is advisory for probe diagnostics */ }
|
|
411
|
+
return undefined;
|
|
412
|
+
}
|
|
402
413
|
function probeFailure(code, stdout, stderr, timedOut, timeoutMs = MODEL_PROBE_TIMEOUT_MS) {
|
|
403
414
|
// SIGKILL-timeout is not exit-1: report the budget, never the masked kill code (v1.27 T1).
|
|
404
415
|
if (timedOut)
|
|
@@ -456,7 +467,8 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
|
|
|
456
467
|
if (r.timedOut && !retry && priorModelAuth?.[model]?.reason?.includes("timed out") === true) {
|
|
457
468
|
return { verdict: v(false, `probe timed out (repeat — retry skipped) (${MODEL_PROBE_TIMEOUT_MS}ms)`), timedOut: true };
|
|
458
469
|
}
|
|
459
|
-
const
|
|
470
|
+
const output = `${r.stderr}\n${r.stdout}`.trim();
|
|
471
|
+
const reason = parsedStartupFailureReason(a, output) ?? probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
|
|
460
472
|
if (!reason)
|
|
461
473
|
return { verdict: v(true), timedOut: false };
|
|
462
474
|
if (!retry)
|
|
@@ -1 +1,4 @@
|
|
|
1
|
+
import { type SourceScopeFinding } from "../../compile/collateral.js";
|
|
2
|
+
import { compileSource } from "../../compile/index.js";
|
|
3
|
+
export declare function nativeSourceScopeErrors(graph: ReturnType<typeof compileSource>, findings: readonly SourceScopeFinding[]): string[];
|
|
1
4
|
export declare function compile(argv: string[], cwd?: string, harnessFrom?: string | undefined): Promise<string>;
|
|
@@ -1,25 +1,21 @@
|
|
|
1
1
|
import { isAbsolute, join } from "node:path";
|
|
2
2
|
import { parseArgs } from "node:util";
|
|
3
|
-
import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
|
|
3
|
+
import { collateralLints, sourceScopeFindings, sourceScopeLints, } from "../../compile/collateral.js";
|
|
4
4
|
import { CompileError } from "../../compile/common.js";
|
|
5
5
|
import { compileSource } from "../../compile/index.js";
|
|
6
|
-
import { saveGraph, stateDirName } from "../../graph/graph.js";
|
|
6
|
+
import { clearCompileRefusal, saveCompileRefusal, saveGraph, stateDirName } from "../../graph/graph.js";
|
|
7
7
|
import { formatPriorFindingEvidence, readPriorRunEvidence } from "../../run/journal.js";
|
|
8
8
|
import { shGit } from "../../run/git.js";
|
|
9
9
|
import { acquireRunLock, releaseRunLock } from "../../run/lock.js";
|
|
10
10
|
import { harnessLine, resolveHarness } from "../harness.js";
|
|
11
|
-
function nativeSourceScopeErrors(graph,
|
|
11
|
+
export function nativeSourceScopeErrors(graph, findings) {
|
|
12
12
|
const byId = new Map(graph.tasks.map((task) => [task.id, task]));
|
|
13
13
|
const errors = [];
|
|
14
|
-
for (const
|
|
15
|
-
const
|
|
16
|
-
if (!match)
|
|
17
|
-
continue;
|
|
18
|
-
const task = byId.get(match[1]);
|
|
14
|
+
for (const finding of findings) {
|
|
15
|
+
const task = byId.get(finding.taskId);
|
|
19
16
|
if (!task)
|
|
20
17
|
continue;
|
|
21
|
-
const
|
|
22
|
-
for (const path of paths) {
|
|
18
|
+
for (const path of finding.paths) {
|
|
23
19
|
if (task.goal.includes(`scope-waiver: ${path}`))
|
|
24
20
|
continue;
|
|
25
21
|
errors.push(`${task.id}: ${path} requires scope-waiver: ${path} in the task goal`);
|
|
@@ -60,43 +56,78 @@ export async function compile(argv, cwd = process.cwd(), harnessFrom = process.a
|
|
|
60
56
|
const src = positionals[0];
|
|
61
57
|
if (!src)
|
|
62
58
|
throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native] [--dry-run] [--strict]");
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
if (!values["dry-run"]) {
|
|
82
|
-
// HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
|
|
83
|
-
// cannot swap graph.json under an active run between the daemon's read and act.
|
|
84
|
-
acquireRunLock(cwd, "compile");
|
|
85
|
-
try {
|
|
86
|
-
saveGraph(cwd, g);
|
|
59
|
+
const sourcePath = isAbsolute(src) ? src : join(cwd, src);
|
|
60
|
+
let stateWriteStarted = false;
|
|
61
|
+
try {
|
|
62
|
+
// Resolve against the target repo, not the process cwd (the CLI test passes a tmp repo).
|
|
63
|
+
// Both modes reach the same pure compiler; --dry-run removes every state write below.
|
|
64
|
+
const g = compileSource(sourcePath, values.type, cwd, // repo root: gsd stores context[0] repo-relative so workers resolve it inside their worktree
|
|
65
|
+
(plan) => plan, { strict: values.strict });
|
|
66
|
+
const sourceFindings = sourceScopeFindings(g.tasks, cwd);
|
|
67
|
+
const scopeLints = [
|
|
68
|
+
...collateralLints(g.tasks, cwd),
|
|
69
|
+
...sourceScopeLints(g.tasks, cwd, sourceFindings),
|
|
70
|
+
];
|
|
71
|
+
const diagnostics = scopeLints.length
|
|
72
|
+
? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
|
|
73
|
+
: "";
|
|
74
|
+
const unwaived = g.spec.source === "native" ? nativeSourceScopeErrors(g, sourceFindings) : [];
|
|
75
|
+
if (unwaived.length > 0) {
|
|
76
|
+
throw new CompileError(`${src} has unwaived native source-scope authoring errors:\n${unwaived.map((line) => ` - ${line}`).join("\n")}${diagnostics}`);
|
|
87
77
|
}
|
|
88
|
-
|
|
89
|
-
|
|
78
|
+
// One bounded read supplies both cross-run surfaces: unresolved findings below and merge facts for
|
|
79
|
+
// the ancestry check. Neither fact mutates the compiled graph; status and every readiness predicate
|
|
80
|
+
// remain the source compiler's answer.
|
|
81
|
+
const prior = readPriorRunEvidence(cwd, g.tasks);
|
|
82
|
+
const mergedPending = await mergedPendingDiagnostics(cwd, new Set(g.tasks.filter((task) => task.status === "pending").map((task) => task.id)), prior.merges);
|
|
83
|
+
const stateDir = stateDirName(cwd);
|
|
84
|
+
if (!values["dry-run"]) {
|
|
85
|
+
// HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around both truth records so
|
|
86
|
+
// compile cannot swap graph.json under an active run between the daemon's read and act.
|
|
87
|
+
stateWriteStarted = true;
|
|
88
|
+
acquireRunLock(cwd, "compile");
|
|
89
|
+
try {
|
|
90
|
+
saveGraph(cwd, g);
|
|
91
|
+
clearCompileRefusal(cwd);
|
|
92
|
+
}
|
|
93
|
+
finally {
|
|
94
|
+
releaseRunLock(cwd);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
const summary = values["dry-run"]
|
|
98
|
+
? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
|
|
99
|
+
: `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
|
|
100
|
+
const priorFindings = prior.findings.length
|
|
101
|
+
? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
|
|
102
|
+
: "";
|
|
103
|
+
const mergeHistory = mergedPending.length
|
|
104
|
+
? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
|
|
105
|
+
: "";
|
|
106
|
+
return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
|
|
107
|
+
}
|
|
108
|
+
catch (error) {
|
|
109
|
+
// A dry run is a pure validation query. A real authoring refusal records the negative result
|
|
110
|
+
// without replacing the last good graph; run treats this sibling as newer truth than that graph.
|
|
111
|
+
// State-write failures are excluded: a live daemon's lock refusal is not a verdict on the spec.
|
|
112
|
+
if (!values["dry-run"] && !stateWriteStarted) {
|
|
113
|
+
try {
|
|
114
|
+
acquireRunLock(cwd, "compile-refusal");
|
|
115
|
+
try {
|
|
116
|
+
saveCompileRefusal(cwd, {
|
|
117
|
+
refusedAt: new Date().toISOString(),
|
|
118
|
+
source: sourcePath,
|
|
119
|
+
error: error instanceof Error ? error.message : String(error),
|
|
120
|
+
});
|
|
121
|
+
}
|
|
122
|
+
finally {
|
|
123
|
+
releaseRunLock(cwd);
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
catch {
|
|
127
|
+
// The refusal record is best-effort when a live run owns the state boundary. That lock
|
|
128
|
+
// verdict must never replace the compile error that tells the operator what to repair.
|
|
129
|
+
}
|
|
90
130
|
}
|
|
131
|
+
throw error;
|
|
91
132
|
}
|
|
92
|
-
const summary = values["dry-run"]
|
|
93
|
-
? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
|
|
94
|
-
: `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
|
|
95
|
-
const priorFindings = prior.findings.length
|
|
96
|
-
? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
|
|
97
|
-
: "";
|
|
98
|
-
const mergeHistory = mergedPending.length
|
|
99
|
-
? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
|
|
100
|
-
: "";
|
|
101
|
-
return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
|
|
102
133
|
}
|
|
@@ -15,13 +15,14 @@ import { HerdrDriver } from "../../drivers/herdr.js";
|
|
|
15
15
|
import { ORCA_FIXTURE_VERSION, parseEnvelope, resolveOrcaCliBinary } from "../../drivers/orca.js";
|
|
16
16
|
import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
|
|
17
17
|
import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
|
|
18
|
-
import { CATALOG_REFRESH_TIMEOUT_MS, LIVEBENCH_TABLE_DATE, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
|
|
18
|
+
import { CATALOG_REFRESH_TIMEOUT_MS, LIVEBENCH_TABLE_DATE, formatCatalogRefreshLegs, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
|
|
19
19
|
import { sh } from "../../run/git.js";
|
|
20
20
|
import { auditNamedTestOracles, listVitestTests } from "../../gates/acceptance.js";
|
|
21
21
|
/** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
|
|
22
22
|
* concatenation and publishes no index, so the release listing is the only enumerable surface. */
|
|
23
23
|
export const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
|
|
24
24
|
export const LIVEBENCH_TABLE_MAX_AGE_DAYS = 90;
|
|
25
|
+
const initialFetch = globalThis.fetch;
|
|
25
26
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
26
27
|
const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
|
|
27
28
|
const attentionRow = (text) => ` ${statusRow("warn", text)}`;
|
|
@@ -348,19 +349,23 @@ export function liveBenchStalenessFinding(now) {
|
|
|
348
349
|
}
|
|
349
350
|
export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(), opts = {}) {
|
|
350
351
|
if (_argv.length === 1 && _argv[0] === "--refresh-catalog") {
|
|
351
|
-
const refreshed = await refreshCatalogCommand({ repoRoot: cwd, now: opts.catalogNow });
|
|
352
|
+
const refreshed = await refreshCatalogCommand({ repoRoot: cwd, fetcher: opts.catalogFetcher, now: opts.catalogNow });
|
|
352
353
|
return refreshed.updated
|
|
353
|
-
? `tickmarkr doctor --refresh-catalog: model catalog refreshed${refreshed.warning ? `; ${refreshed.warning}` : ""}`
|
|
354
|
-
: `tickmarkr doctor --refresh-catalog: catalog refresh unavailable — ${refreshed.warning
|
|
354
|
+
? `tickmarkr doctor --refresh-catalog: model catalog refreshed — ${formatCatalogRefreshLegs(refreshed.legs)}${refreshed.warning ? `; ${refreshed.warning}` : ""}`
|
|
355
|
+
: `tickmarkr doctor --refresh-catalog: catalog refresh unavailable — ${formatCatalogRefreshLegs(refreshed.legs)}${refreshed.warning ? `; ${refreshed.warning}` : ""}`;
|
|
355
356
|
}
|
|
356
357
|
// Operator directive 2026-08-12 (declutter): long per-model lists render only when asked for.
|
|
357
358
|
const listAllModels = _argv.includes("--models");
|
|
358
359
|
const cfg = loadConfig(cwd);
|
|
359
360
|
let catalog = opts.catalog ?? readCachedCatalog(cwd, { now: opts.catalogNow });
|
|
360
|
-
let
|
|
361
|
+
let catalogRefreshLine;
|
|
361
362
|
// RULING-222-17 reverses cache-only for these two operator-facing commands only. The refresh
|
|
362
363
|
// remains age-guarded; compile, plan, and run never import this path.
|
|
363
|
-
|
|
364
|
+
const catalogRefreshAllowed = process.env.VITEST !== "true"
|
|
365
|
+
|| opts.catalogFetcher !== undefined
|
|
366
|
+
|| opts.catalogNow !== undefined
|
|
367
|
+
|| globalThis.fetch !== initialFetch;
|
|
368
|
+
if (!opts.catalog && catalog.stale && catalogRefreshAllowed) {
|
|
364
369
|
const refreshed = await refreshCatalogCommand({
|
|
365
370
|
repoRoot: cwd,
|
|
366
371
|
fetcher: opts.catalogFetcher,
|
|
@@ -368,7 +373,7 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
368
373
|
now: opts.catalogNow,
|
|
369
374
|
});
|
|
370
375
|
catalog = refreshed.catalog;
|
|
371
|
-
|
|
376
|
+
catalogRefreshLine = formatCatalogRefreshLegs(refreshed.legs);
|
|
372
377
|
}
|
|
373
378
|
// banner at START — the logo greets the operator before the ~60s probe wait, never trailing it (operator report 2026-07-17)
|
|
374
379
|
if (opts.banner !== false && visual())
|
|
@@ -466,8 +471,8 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
466
471
|
if (catalog.warning) {
|
|
467
472
|
rows.push(attentionRow(`model catalog cache unreadable — ${catalog.warning}; using vendored fallback (advisory — routing unchanged)`));
|
|
468
473
|
}
|
|
469
|
-
if (
|
|
470
|
-
rows.push(attentionRow(`model catalog auto-refresh
|
|
474
|
+
if (catalogRefreshLine) {
|
|
475
|
+
rows.push(attentionRow(`model catalog auto-refresh — ${catalogRefreshLine}`));
|
|
471
476
|
}
|
|
472
477
|
// v1.48 T1 / v1.86 T12: advisory sweep for known agent CLIs with no drive contract. Presence is
|
|
473
478
|
// resolved through the worker's login shell; advisory targets are never executed or written to health.
|
|
@@ -3,7 +3,7 @@ import { dirname } from "node:path";
|
|
|
3
3
|
import { parseArgs } from "node:util";
|
|
4
4
|
import { allAdapters, discoverChannels, doctorAgeMs, initDoctorReuse, modelAuthExclusions } from "../../adapters/registry.js";
|
|
5
5
|
import { catalogModelAdvisory, catalogTierRanking, declaredModelWindow, fleetUnclassifiedModels } from "../../adapters/model-lints.js";
|
|
6
|
-
import { CATALOG_REFRESH_TIMEOUT_MS, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
|
|
6
|
+
import { CATALOG_REFRESH_TIMEOUT_MS, formatCatalogRefreshLegs, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
|
|
7
7
|
import { CLAUDE_ALIAS_IDENTITY_STAMPS, readClaudeAliasIdentity } from "../../adapters/claude-code.js";
|
|
8
8
|
import { fleetEditableFromConfig, fleetEditableEquals, formatFleetPrint, globalConfigDir, overlayBytesLoadError, renderFleetOverlayWrite, repoOverlayPath, ROUTING_MODES, unifiedYamlDiff, } from "../../config/config.js";
|
|
9
9
|
import { projectFleetWhy, renderFleetWhy } from "../../config/fleet-why.js";
|
|
@@ -14,6 +14,7 @@ import { route } from "../../route/router.js";
|
|
|
14
14
|
import { disallowedBy } from "../../route/preference.js";
|
|
15
15
|
import { resolveRunMode } from "../../run/daemon.js";
|
|
16
16
|
import { loadRoutingProfile } from "../../run/journal.js";
|
|
17
|
+
const initialFetch = globalThis.fetch;
|
|
17
18
|
const NON_TTY_MSG = "tickmarkr fleet: interactive fleet editor requires a TTY — use `tickmarkr fleet --print` for non-interactive output";
|
|
18
19
|
const QUIT = "fleet: quit without writing";
|
|
19
20
|
// v1.60 T3: every preview surface ranks with the SAME exploration setting as the candidate picker
|
|
@@ -98,16 +99,21 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
|
|
|
98
99
|
const interactive = input.isTTY === true && output.isTTY === true;
|
|
99
100
|
let catalog = readCachedCatalog(cwd, { now: io.catalogNow });
|
|
100
101
|
let refreshReason = "";
|
|
101
|
-
|
|
102
|
+
let catalogRefreshAttempted = false;
|
|
103
|
+
const catalogRefreshAllowed = process.env.VITEST !== "true"
|
|
104
|
+
|| io.catalogFetcher !== undefined
|
|
105
|
+
|| io.catalogNow !== undefined
|
|
106
|
+
|| globalThis.fetch !== initialFetch;
|
|
107
|
+
if (catalog.stale && catalogRefreshAllowed) {
|
|
102
108
|
const refreshed = await refreshCatalogCommand({
|
|
103
109
|
repoRoot: cwd,
|
|
104
110
|
fetcher: io.catalogFetcher,
|
|
105
111
|
timeoutMs: CATALOG_REFRESH_TIMEOUT_MS,
|
|
106
112
|
now: io.catalogNow,
|
|
107
113
|
});
|
|
114
|
+
catalogRefreshAttempted = true;
|
|
108
115
|
catalog = refreshed.catalog;
|
|
109
|
-
|
|
110
|
-
refreshReason = `fleet: catalog auto-refresh failed open — ${refreshed.warning}; retained ${catalog.source} catalog`;
|
|
116
|
+
refreshReason = `fleet: catalog auto-refresh — ${formatCatalogRefreshLegs(refreshed.legs)}`;
|
|
111
117
|
}
|
|
112
118
|
if (print) {
|
|
113
119
|
// v1.51 T4: the print surface names the mode and its source layer right under the header —
|
|
@@ -129,7 +135,7 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
|
|
|
129
135
|
if (!why && interactive) {
|
|
130
136
|
const { reuse } = initDoctorReuse(cwd, values.fresh ?? false);
|
|
131
137
|
if (!reuse) {
|
|
132
|
-
output.write(`${await doctor([], cwd, adapters, { banner: false, compact: true })}\n`);
|
|
138
|
+
output.write(`${await doctor([], cwd, adapters, { banner: false, compact: true, ...(catalogRefreshAttempted ? { catalog } : {}), catalogFetcher: io.catalogFetcher, catalogNow: io.catalogNow })}\n`);
|
|
133
139
|
}
|
|
134
140
|
}
|
|
135
141
|
const assembled = await assembleFleetEditor(cwd, adapters, io, { globalDir, catalog });
|
|
@@ -11,6 +11,7 @@ import { BANNER, kvRow, legend, rule, statusRow, title } from "../../brand.js";
|
|
|
11
11
|
import { tickmarkrDir } from "../../graph/graph.js";
|
|
12
12
|
import { orcaHostDetected } from "../../drivers/index.js";
|
|
13
13
|
import { Journal } from "../../run/journal.js";
|
|
14
|
+
import { runLockRunId, runStatusLine } from "../../run/lock.js";
|
|
14
15
|
import { doctor } from "./doctor.js";
|
|
15
16
|
import { assembleFleetEditor } from "./fleet.js";
|
|
16
17
|
const SCAFFOLD_SPEC = "tickmarkr.spec.md";
|
|
@@ -24,20 +25,18 @@ const ENVIRONMENTS_FOOTER = [
|
|
|
24
25
|
" claude code — tickmarkr init --agent installs the /tkr skills + AGENTS.md so Claude Code (or any agent CLI) drives the loop natively",
|
|
25
26
|
" anywhere — no herdr or Orca terminal? same fail-closed gates, headless subprocess driver (legacy: no herdr? same fail-closed gates, headless subprocess driver when Orca markers are absent)",
|
|
26
27
|
].join("\n");
|
|
27
|
-
/**
|
|
28
|
-
function
|
|
29
|
-
const
|
|
28
|
+
/** Current run truth, lock first; falls back to an abandoned journal. */
|
|
29
|
+
function activeRunLine(cwd) {
|
|
30
|
+
const locked = runLockRunId(cwd);
|
|
31
|
+
const runId = locked ?? Journal.latestRunId(cwd, { withJournal: true });
|
|
30
32
|
if (!runId)
|
|
31
33
|
return null;
|
|
34
|
+
let events = [];
|
|
32
35
|
try {
|
|
33
|
-
|
|
34
|
-
if (events.some((e) => e.event === "run-end"))
|
|
35
|
-
return null;
|
|
36
|
-
return runId;
|
|
37
|
-
}
|
|
38
|
-
catch {
|
|
39
|
-
return null;
|
|
36
|
+
events = Journal.open(cwd, runId).read();
|
|
40
37
|
}
|
|
38
|
+
catch { /* lock may exist before first journal row */ }
|
|
39
|
+
return runStatusLine(cwd, runId, events);
|
|
41
40
|
}
|
|
42
41
|
/** Relative paths of specs/*.spec.md already in the repo. */
|
|
43
42
|
function existingSpecs(cwd) {
|
|
@@ -56,9 +55,9 @@ function existingSpecs(cwd) {
|
|
|
56
55
|
}
|
|
57
56
|
/** Context-aware next-steps line (operator-approved 2026-07-17). */
|
|
58
57
|
function nextSteps(cwd, scaffoldedSpec) {
|
|
59
|
-
const
|
|
60
|
-
if (
|
|
61
|
-
return
|
|
58
|
+
const runLine = activeRunLine(cwd);
|
|
59
|
+
if (runLine)
|
|
60
|
+
return `${runLine} — tickmarkr status`;
|
|
62
61
|
const specs = existingSpecs(cwd);
|
|
63
62
|
if (specs.length > 0) {
|
|
64
63
|
const listed = specs.length <= 3 ? specs.join(", ") : `${specs.slice(0, 3).join(", ")}, …`;
|