tickmarkr 1.90.6 → 1.90.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/doctor.d.ts +3 -0
- package/dist/cli/commands/doctor.js +30 -3
- package/dist/cli/commands/fleet.d.ts +18 -0
- package/dist/cli/commands/fleet.js +88 -56
- package/dist/cli/commands/init.d.ts +6 -1
- package/dist/cli/commands/init.js +73 -45
- package/dist/cli/commands/run.js +3 -2
- package/dist/cli/commands/verify.d.ts +1 -0
- package/dist/cli/commands/verify.js +12 -2
- package/dist/config/fleet-overlay.d.ts +4 -0
- package/dist/config/fleet-overlay.js +4 -0
- package/dist/gates/baseline.d.ts +15 -0
- package/dist/gates/baseline.js +77 -4
- package/dist/run/daemon.js +25 -0
- package/dist/tui/ink/components.d.ts +3 -1
- package/dist/tui/ink/components.js +2 -2
- package/dist/tui/ink/fleet-app.d.ts +15 -2
- package/dist/tui/ink/fleet-app.js +178 -41
- package/dist/tui/ink/init-app.d.ts +34 -0
- package/dist/tui/ink/init-app.js +272 -0
- package/package.json +1 -1
package/dist/gates/baseline.js
CHANGED
|
@@ -51,6 +51,14 @@ const TRAILING_FAIL_RE = /^\s*(?:test\s+)?\S+(?:\s+\([^)]*\))?\s+(?:\.{3}|-{3,})
|
|
|
51
51
|
// runner sharing the glyph protocol is read without being enumerated; a status strip drawing ✖
|
|
52
52
|
// mid-line inside chrome matches nothing, the same position rule as the other shapes.
|
|
53
53
|
const GLYPH_FAIL_RE = /^\s*(?:(?:[\w@./-]+:\s*)*)✖\s+\S/;
|
|
54
|
+
// vitest's default reporter draws the same glyph protocol with × (U+00D7, verbatim capture,
|
|
55
|
+
// GATE-FIX-4 drill: `intake-backend:test: × GF4BE deliberate failure — probe only`). It is a
|
|
56
|
+
// fingerprintable SHAPE (isFailureShaped) but deliberately NOT in namesFailure: vitest prints a
|
|
57
|
+
// FAIL twin naming the same test, and the HYG-08 meta.failingTests pin (tests/gates/baseline.test.ts)
|
|
58
|
+
// keeps headline naming to one line per failure — duplicating each failure as glyph + FAIL would
|
|
59
|
+
// re-headline the noise HYG-08 removed, while dropping the × from the fingerprint set would shrink
|
|
60
|
+
// the per-test discriminators DEFECT 4 exists to harvest.
|
|
61
|
+
const X_GLYPH_FAIL_RE = /^\s*(?:(?:[\w@./-]+:\s*)*)×\s+\S/;
|
|
54
62
|
// Q137s (verbatim capture, dossier run-2 consult dossier, turbo 2.9.14 + pnpm 10): monorepo
|
|
55
63
|
// drivers prefix child output and summarize failures in their own grammar. Four shapes, all
|
|
56
64
|
// positional — `<pkg>:<task>:` prefixes reuse GLYPH_FAIL_RE's prefix idiom; `Failed:`/`ERROR`
|
|
@@ -61,11 +69,35 @@ const GLYPH_FAIL_RE = /^\s*(?:(?:[\w@./-]+:\s*)*)✖\s+\S/;
|
|
|
61
69
|
// ` ERROR intake-frontend#lint: command (…) … exited (1)`
|
|
62
70
|
// ` ERROR run failed: command exited (1)`
|
|
63
71
|
const TURBO_FAIL_RE = /^\s*(?:[\w@./-]+:\s*)*ELIFECYCLE\s+Command failed\b|^\s*Failed:\s+\S+#\S+|^\s*ERROR\s+(?:\S+#\S+:|run failed\b)/;
|
|
72
|
+
// GATE-FIX-4 DEFECT 4 (dossier drill, control 3): turbo prefixes every child line with
|
|
73
|
+
// `<pkg>:<task>: `, so the per-test shapes above (FAIL_ANCHOR_RE etc.) — all anchored at line
|
|
74
|
+
// START — never fingerprinted; forgiveness compared only TURBO_FAIL_RE's package-level lines, and a
|
|
75
|
+
// package with one tolerated pre-existing red was blind to every NEW failure inside it (a deliberate
|
|
76
|
+
// failing test came back green). The counts route is closed too: normalizeLine masks digits, so
|
|
77
|
+
// `Tests 3 failed` vs `Tests 2 failed` is not a discriminator — per-test lines, which carry file
|
|
78
|
+
// paths and test names, are. This prefix is deliberately narrower than TURBO_FAIL_RE's label idiom:
|
|
79
|
+
// at least TWO colon-joined segments (`<pkg>:<task>:`, tasks may nest — `pkg:test:unit:`), every
|
|
80
|
+
// segment after the first starting with a letter — so `Error: boom` (one segment), `src/x.ts:12:`
|
|
81
|
+
// (digit segment) and `12:34 error` (eslint stylish) are never stripped.
|
|
82
|
+
const TURBO_PREFIX_RE = /^\s*[\w@./-]+(?::[A-Za-z_][\w.-]*)+:\s+/;
|
|
83
|
+
const stripTurboPrefix = (l) => {
|
|
84
|
+
const m = TURBO_PREFIX_RE.exec(l);
|
|
85
|
+
return m ? l.slice(m[0].length) : undefined;
|
|
86
|
+
};
|
|
64
87
|
// Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
|
|
65
88
|
// and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
|
|
66
89
|
const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
|
|
90
|
+
// The stripped form is a second READ of the same line, for the recognition/headline paths that ask
|
|
91
|
+
// "does anything here name a failure" — verdict classification (isInfraLine/namesRegression) keeps
|
|
92
|
+
// reading the raw line only, so infra/regression verdicts are byte-unchanged by the prefix pass.
|
|
93
|
+
const namesFailureEitherForm = (l) => {
|
|
94
|
+
if (namesFailure(l))
|
|
95
|
+
return true;
|
|
96
|
+
const stripped = stripTurboPrefix(l);
|
|
97
|
+
return stripped !== undefined && namesFailure(stripped);
|
|
98
|
+
};
|
|
67
99
|
const isFailureShaped = (l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l) || ERROR_ANCHOR_RE.test(l)
|
|
68
|
-
|| TSC_ERROR_RE.test(l) || LINTER_ERROR_RE.test(l);
|
|
100
|
+
|| TSC_ERROR_RE.test(l) || LINTER_ERROR_RE.test(l) || X_GLYPH_FAIL_RE.test(l);
|
|
69
101
|
const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
|
|
70
102
|
// T9 — the infra/regression discriminator. A runner that died because the MACHINE ran out of
|
|
71
103
|
// processes, file descriptors or memory never finished asking the question, so its nonzero exit is
|
|
@@ -105,7 +137,18 @@ export function fingerprint(output) {
|
|
|
105
137
|
.split("\n")
|
|
106
138
|
.map((l) => l.replace(ANSI_RE, ""))
|
|
107
139
|
.filter((l) => !PASS_LINE_RE.test(l));
|
|
108
|
-
|
|
140
|
+
// GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
|
|
141
|
+
// prefix removed. A recognized stripped line fingerprints as its STRIPPED text, so the same
|
|
142
|
+
// failure fingerprints identically whether turbo prefixed it or a bare runner printed it; the
|
|
143
|
+
// prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
|
|
144
|
+
const shaped = [];
|
|
145
|
+
for (const l of lines) {
|
|
146
|
+
if (isFailureShaped(l))
|
|
147
|
+
shaped.push(l);
|
|
148
|
+
const stripped = stripTurboPrefix(l);
|
|
149
|
+
if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFailureShaped(stripped))
|
|
150
|
+
shaped.push(stripped);
|
|
151
|
+
}
|
|
109
152
|
if (!shaped.length)
|
|
110
153
|
return lines.some((l) => l.trim()) ? [UNRECOGNIZED_FAILURE] : [];
|
|
111
154
|
return [...new Set(shaped.map(normalizeLine))];
|
|
@@ -180,6 +223,36 @@ export function detectGateCommands(repoRoot, cfg) {
|
|
|
180
223
|
}
|
|
181
224
|
return out;
|
|
182
225
|
}
|
|
226
|
+
/**
|
|
227
|
+
* Dossier GATE-FIX-4 §3 (2026-08-13): turbo schedules per-package tasks and ABORTS the whole
|
|
228
|
+
* run at the first failing package, so every package after the abort goes unverified — and the
|
|
229
|
+
* baseline gate then forgives the truncated output as "matching". The class appeared in three
|
|
230
|
+
* gates and four syntaxes on one repo (turbo's scheduler, `&&`, a JS for-loop process.exit, and
|
|
231
|
+
* turbo nested inside two of them). This preflight names only what it can see textually: the
|
|
232
|
+
* resolved gate command, and — for the synthesized `<pm> run <script>` form — that script's own
|
|
233
|
+
* body. The remedy deliberately demands a forwarding check first: the same repo's lint wrapper
|
|
234
|
+
* branched on argv.length, so a blind `-- --continue` silently replaced the gate with a
|
|
235
|
+
* different command (the trap their overseer caught by reading, not by symmetry).
|
|
236
|
+
*/
|
|
237
|
+
export function turboContinueFindings(repoRoot, cfg) {
|
|
238
|
+
const pkgPath = join(repoRoot, "package.json");
|
|
239
|
+
const scripts = existsSync(pkgPath)
|
|
240
|
+
? (JSON.parse(readFileSync(pkgPath, "utf8")).scripts ?? {})
|
|
241
|
+
: {};
|
|
242
|
+
const findings = [];
|
|
243
|
+
for (const [gate, cmd] of Object.entries(detectGateCommands(repoRoot, cfg))) {
|
|
244
|
+
// The command text plus every package.json script it invokes by name — one level, textual.
|
|
245
|
+
const referenced = Object.keys(scripts).filter((name) => new RegExp(`\\brun\\s+(?:-s\\s+)?${reEscape(name)}(?:\\s|$)`).test(cmd));
|
|
246
|
+
const surface = [cmd, ...referenced.map((name) => scripts[name])].join("\n");
|
|
247
|
+
if (/\bturbo\s+run\b/.test(surface) && !surface.includes("--continue")) {
|
|
248
|
+
findings.push({
|
|
249
|
+
gate,
|
|
250
|
+
detail: `runs turbo without --continue — turbo aborts at the first failing package, so later packages go unverified yet baseline-forgiven. Append --continue where turbo is invoked, and verify forwarding first (\`run ${gate} -- --continue --dry=text\` must echo it): wrappers that branch on argv can silently swap the gated command`,
|
|
251
|
+
});
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
return findings;
|
|
255
|
+
}
|
|
183
256
|
const shellToken = (cmd) => {
|
|
184
257
|
for (const raw of cmd.trim().split(/\s+/)) {
|
|
185
258
|
if (!raw || /^[A-Za-z_][A-Za-z0-9_]*=/.test(raw))
|
|
@@ -302,12 +375,12 @@ function headlineDetails(raw, fresh) {
|
|
|
302
375
|
const headlines = raw
|
|
303
376
|
.split("\n")
|
|
304
377
|
.map((l) => l.replace(ANSI_RE, ""))
|
|
305
|
-
.filter((l) =>
|
|
378
|
+
.filter((l) => namesFailureEitherForm(l) || SUMMARY_FAIL_RE.test(l));
|
|
306
379
|
if (!headlines.length)
|
|
307
380
|
return { details: `new failures vs baseline:\n${fresh.join("\n")}` };
|
|
308
381
|
return {
|
|
309
382
|
details: `failing tests:\n${headlines.join("\n")}\n\nnew failure fingerprints vs baseline (secondary):\n${fresh.join("\n")}`,
|
|
310
|
-
meta: { failingTests: headlines.filter(
|
|
383
|
+
meta: { failingTests: headlines.filter(namesFailureEitherForm) },
|
|
311
384
|
};
|
|
312
385
|
}
|
|
313
386
|
export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
package/dist/run/daemon.js
CHANGED
|
@@ -920,6 +920,31 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
920
920
|
let releaseApprovalSerialization;
|
|
921
921
|
try {
|
|
922
922
|
let graph = loadGraph(repoRoot);
|
|
923
|
+
// GATE-FIX-4 defect 4 (no-op run refusal): a fresh run on a graph with nothing dispatchable used
|
|
924
|
+
// to journal {run-start, run-end} with zero dispatches — and downstream readers (greenness exit,
|
|
925
|
+
// status, notify) treat that run-end as completion, so an all-terminal graph "went green" having
|
|
926
|
+
// done nothing. Refuse HERE, the earliest seam where the graph is known and BEFORE Journal.create
|
|
927
|
+
// makes the run dir: no run-start row, no run dir, and the error exits nonzero through the CLI's
|
|
928
|
+
// ordinary catch. Dispatchability reuses graph.ts's status math rather than restating it: a task
|
|
929
|
+
// is dispatchable iff it is ready now (readyTasks) or can still become ready — pending with a
|
|
930
|
+
// non-parked closure (pendingTasks), or unblocked later by in-flight running/gated residue.
|
|
931
|
+
// FRESH runs only: a resume where everything is terminal already owns its run-end and keeps its
|
|
932
|
+
// current replay-then-quiesce behavior (the run-narrate test pins run-resume → run-end).
|
|
933
|
+
if (!opts.resume && readyTasks(graph).length === 0 && pendingTasks(graph).length === 0
|
|
934
|
+
&& !graph.tasks.some((t) => t.status === "running" || t.status === "gated")) {
|
|
935
|
+
// Every pending task here is blocked-on-terminal by construction (pendingTasks(graph) is empty),
|
|
936
|
+
// so the count is labeled truthfully instead of as still-runnable "pending".
|
|
937
|
+
const counts = new Map();
|
|
938
|
+
for (const t of graph.tasks) {
|
|
939
|
+
const bucket = t.status === "pending" ? "blocked-on-terminal" : t.status;
|
|
940
|
+
counts.set(bucket, (counts.get(bucket) ?? 0) + 1);
|
|
941
|
+
}
|
|
942
|
+
const breakdown = counts.size === 0 ? "graph has zero tasks"
|
|
943
|
+
: [...counts.entries()].map(([s, n]) => `${s} ${n}`).join(", ");
|
|
944
|
+
throw new Error(`nothing to dispatch — no task is ready and none can become ready (${breakdown}). `
|
|
945
|
+
+ "Refusing to start a run that would journal run-start/run-end with zero dispatches. "
|
|
946
|
+
+ "Release parked tasks with `tickmarkr approve <runId> <taskId>`, or compile a fresh graph with `tickmarkr compile`.");
|
|
947
|
+
}
|
|
923
948
|
// v1.51 T2: the routing mode resolves BEFORE any routing input is built — run flag > spec front-matter
|
|
924
949
|
// > repo > global > default. The resolved cfg carries mode-compiled floors; route() never sees the mode.
|
|
925
950
|
const rm = resolveRunMode(repoRoot, { flag: opts.mode, spec: graph.mode, globalDir: opts.globalDir });
|
|
@@ -9,12 +9,14 @@ export declare function TextLines({ lines }: {
|
|
|
9
9
|
export declare function ToggleMark({ active }: {
|
|
10
10
|
active: boolean;
|
|
11
11
|
}): import("react").JSX.Element;
|
|
12
|
-
export declare function FleetListScreen({ title, legend, rows, cursor, details, }: {
|
|
12
|
+
export declare function FleetListScreen({ title, legend, rows, cursor, details, filter, }: {
|
|
13
13
|
title: string;
|
|
14
14
|
legend: string;
|
|
15
15
|
rows: FleetListRow[];
|
|
16
16
|
cursor: number;
|
|
17
17
|
details?: string[];
|
|
18
|
+
/** active type-to-search string; rendered on the legend line so the operator sees the narrowing */
|
|
19
|
+
filter?: string;
|
|
18
20
|
}): import("react").JSX.Element;
|
|
19
21
|
export declare function FleetReviewScreen({ title, legend, diff, }: {
|
|
20
22
|
title: string;
|
|
@@ -17,8 +17,8 @@ export function ToggleMark({ active }) {
|
|
|
17
17
|
? _jsx(Text, { color: "ansi256(41)", children: GLYPHS.toggleActive })
|
|
18
18
|
: _jsx(Text, { dimColor: true, children: GLYPHS.toggleInactive });
|
|
19
19
|
}
|
|
20
|
-
export function FleetListScreen({ title, legend, rows, cursor, details = [], }) {
|
|
21
|
-
return (_jsxs(Box, { flexDirection: "column", children: [_jsx(Text, { bold: true, children: title }), _jsx(Text, { dimColor: true, children: legend }), rows.map((row, index) => (_jsxs(Text, { bold: index === cursor, children: [index === cursor ? `${GLYPHS.pointer} ` : " ", row.content] }, row.id))), _jsx(TextLines, { lines: details })] }));
|
|
20
|
+
export function FleetListScreen({ title, legend, rows, cursor, details = [], filter, }) {
|
|
21
|
+
return (_jsxs(Box, { flexDirection: "column", children: [_jsx(Text, { bold: true, children: title }), _jsx(Text, { dimColor: true, children: filter ? `${legend} · search: ${filter}` : legend }), rows.map((row, index) => (_jsxs(Text, { bold: index === cursor, children: [index === cursor ? `${GLYPHS.pointer} ` : " ", row.content] }, row.id))), _jsx(TextLines, { lines: details })] }));
|
|
22
22
|
}
|
|
23
23
|
export function FleetReviewScreen({ title, legend, diff, }) {
|
|
24
24
|
return (_jsxs(Box, { flexDirection: "column", children: [_jsx(Text, { bold: true, children: title }), _jsx(Text, { dimColor: true, children: legend }), _jsx(Text, { children: diff })] }));
|
|
@@ -8,6 +8,10 @@ export type FleetEditorState = {
|
|
|
8
8
|
classifications: FleetClassification[];
|
|
9
9
|
selectedMode: RoutingMode;
|
|
10
10
|
map: Record<string, MapEntry>;
|
|
11
|
+
judgeSeat?: {
|
|
12
|
+
adapter: string;
|
|
13
|
+
model: string;
|
|
14
|
+
};
|
|
11
15
|
steering: Record<FleetSteeringKey, string[] | undefined>;
|
|
12
16
|
};
|
|
13
17
|
export type FleetOverlayReview = {
|
|
@@ -72,7 +76,7 @@ export type FleetCandidateOption = {
|
|
|
72
76
|
};
|
|
73
77
|
};
|
|
74
78
|
export declare function formatDoctorAge(ageMs: number | null): string;
|
|
75
|
-
export declare function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels, modelGroups, initialMode, modeOptions, initialMap, modePreview, shapeRows, candidatesForShape, preferOptionsForShape, initialSteering, steeringOptionsFor, reviewOverlay, reloadGuard, }: {
|
|
79
|
+
export declare function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels, modelGroups, initialMode, modeOptions, initialMap, modePreview, shapeRows, candidatesForShape, preferOptionsForShape, initialSteering, steeringOptionsFor, reviewOverlay, reloadGuard, entry, initialJudge, judgeSeats, }: {
|
|
76
80
|
ageMs: number | null;
|
|
77
81
|
agents: AgentCli[];
|
|
78
82
|
initialDenyAdapters: string[];
|
|
@@ -89,8 +93,13 @@ export declare function FleetApp({ ageMs, agents, initialDenyAdapters, initialDe
|
|
|
89
93
|
steeringOptionsFor: (which: FleetSteeringKey, current: string[]) => string[];
|
|
90
94
|
reviewOverlay: (state: FleetEditorState) => FleetOverlayReview;
|
|
91
95
|
reloadGuard: (bytes: string) => string | null;
|
|
96
|
+
entry?: "presets" | "probe";
|
|
97
|
+
/** resolved config judge as "adapter:model" — shown on the (keep default) picker row */
|
|
98
|
+
initialJudge?: string;
|
|
99
|
+
/** discovered adapter:model seats — the judge picker universe */
|
|
100
|
+
judgeSeats?: string[];
|
|
92
101
|
}): import("react").JSX.Element;
|
|
93
|
-
export declare function runFleetInkEditor({ ageMs, adapters, health, initialDenyAdapters, initialDenyModels, modelGroups, initialMode, modeOptions, initialMap, modePreview, shapeRows, candidatesForShape, preferOptionsForShape, initialSteering, steeringOptionsFor, reviewOverlay, reloadGuard, initialInput, input, output, debug, }: {
|
|
102
|
+
export declare function runFleetInkEditor({ ageMs, adapters, health, initialDenyAdapters, initialDenyModels, modelGroups, initialMode, modeOptions, initialMap, modePreview, shapeRows, candidatesForShape, preferOptionsForShape, initialSteering, steeringOptionsFor, reviewOverlay, reloadGuard, entry, initialJudge, judgeSeats, initialInput, input, output, debug, }: {
|
|
94
103
|
ageMs: number | null;
|
|
95
104
|
adapters: WorkerAdapter[];
|
|
96
105
|
health: Record<string, AuthHealth>;
|
|
@@ -108,6 +117,10 @@ export declare function runFleetInkEditor({ ageMs, adapters, health, initialDeny
|
|
|
108
117
|
steeringOptionsFor: (which: FleetSteeringKey, current: string[]) => string[];
|
|
109
118
|
reviewOverlay: (state: FleetEditorState) => FleetOverlayReview;
|
|
110
119
|
reloadGuard: (bytes: string) => string | null;
|
|
120
|
+
/** "presets" opens on the routing-mode screen with the extra custom row; default preserves the probe-first walk */
|
|
121
|
+
entry?: "presets" | "probe";
|
|
122
|
+
initialJudge?: string;
|
|
123
|
+
judgeSeats?: string[];
|
|
111
124
|
initialInput?: string[];
|
|
112
125
|
input: NodeJS.ReadStream;
|
|
113
126
|
output: NodeJS.WriteStream;
|