tickmarkr 2.4.0 → 2.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +11 -0
- package/dist/adapters/catalog-remote.js +55 -13
- package/dist/adapters/claude-code.js +7 -2
- package/dist/adapters/codex.d.ts +1 -0
- package/dist/adapters/codex.js +68 -8
- package/dist/adapters/qwen.js +3 -3
- package/dist/adapters/registry.js +13 -1
- package/dist/adapters/types.d.ts +4 -0
- package/dist/adapters/types.js +33 -0
- package/dist/cli/commands/compile.d.ts +3 -0
- package/dist/cli/commands/compile.js +77 -46
- package/dist/cli/commands/doctor.js +14 -9
- package/dist/cli/commands/fleet.js +11 -5
- package/dist/cli/commands/init.js +12 -13
- package/dist/cli/commands/plan.js +47 -7
- package/dist/cli/commands/run.js +20 -1
- package/dist/cli/commands/status.js +20 -21
- package/dist/cli/commands/version.js +2 -2
- package/dist/compile/collateral.d.ts +14 -5
- package/dist/compile/collateral.js +32 -26
- package/dist/compile/index.js +17 -6
- package/dist/compile/native.js +10 -2
- package/dist/compile/ownership.js +15 -9
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +32 -3
- package/dist/drivers/orca.d.ts +5 -1
- package/dist/drivers/orca.js +51 -2
- package/dist/drivers/types.d.ts +1 -1
- package/dist/gates/baseline.d.ts +26 -2
- package/dist/gates/baseline.js +90 -7
- package/dist/gates/review.d.ts +2 -2
- package/dist/gates/review.js +2 -18
- package/dist/graph/graph.d.ts +20 -0
- package/dist/graph/graph.js +66 -1
- package/dist/route/preference.d.ts +3 -1
- package/dist/route/preference.js +4 -4
- package/dist/run/consult.js +1 -0
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +151 -22
- package/dist/run/git.d.ts +2 -0
- package/dist/run/git.js +18 -4
- package/dist/run/journal.d.ts +1 -1
- package/dist/run/journal.js +1 -1
- package/dist/run/lock.d.ts +6 -0
- package/dist/run/lock.js +41 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +39 -4
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
package/dist/compile/index.js
CHANGED
|
@@ -37,12 +37,16 @@ function detect(src) {
|
|
|
37
37
|
function repoOverlayMode(repoRoot) {
|
|
38
38
|
if (!repoRoot)
|
|
39
39
|
return undefined;
|
|
40
|
+
const path = join(repoRoot, ".tickmarkr", "config.yaml");
|
|
41
|
+
if (!existsSync(path))
|
|
42
|
+
return undefined;
|
|
40
43
|
try {
|
|
41
|
-
const cfg = parse(readFileSync(
|
|
44
|
+
const cfg = parse(readFileSync(path, "utf8"));
|
|
42
45
|
return typeof cfg?.routing?.mode === "string" ? cfg.routing.mode : undefined;
|
|
43
46
|
}
|
|
44
|
-
catch {
|
|
45
|
-
|
|
47
|
+
catch (error) {
|
|
48
|
+
throw new CompileError(`${path} does not parse, so compile cannot prove the repository routing overlay: `
|
|
49
|
+
+ `${error instanceof Error ? error.message : String(error)}`);
|
|
46
50
|
}
|
|
47
51
|
}
|
|
48
52
|
function hasModeOverride(src) {
|
|
@@ -58,9 +62,13 @@ function hasModeOverride(src) {
|
|
|
58
62
|
return false;
|
|
59
63
|
}
|
|
60
64
|
function enforceModeOverlay(graph, src, repoRoot) {
|
|
61
|
-
if (graph.spec.source !== "native"
|
|
65
|
+
if (graph.spec.source !== "native")
|
|
62
66
|
return;
|
|
67
|
+
// Read every native compile's overlay before checking whether either side declares a mode. A
|
|
68
|
+
// corrupt file cannot masquerade as an absent mode and silently skip the disagreement gate.
|
|
63
69
|
const overlayMode = repoOverlayMode(repoRoot);
|
|
70
|
+
if (graph.mode === undefined)
|
|
71
|
+
return;
|
|
64
72
|
if (overlayMode === undefined || overlayMode === graph.mode || hasModeOverride(src))
|
|
65
73
|
return;
|
|
66
74
|
throw new CompileError(`${src} front-matter mode ${graph.mode} disagrees with repository routing.mode ${overlayMode}; `
|
|
@@ -75,7 +83,7 @@ function enforceTaskUnitContract(g, src, repoRoot) {
|
|
|
75
83
|
return g;
|
|
76
84
|
}
|
|
77
85
|
export function finalizePlan(plan, src, repoRoot) {
|
|
78
|
-
const graph =
|
|
86
|
+
const graph = validateGraph({
|
|
79
87
|
version: plan.version,
|
|
80
88
|
...(plan.mode !== undefined ? { mode: plan.mode } : {}),
|
|
81
89
|
spec: {
|
|
@@ -85,8 +93,11 @@ export function finalizePlan(plan, src, repoRoot) {
|
|
|
85
93
|
...(plan.base !== undefined ? { base: plan.base } : {}),
|
|
86
94
|
},
|
|
87
95
|
tasks: plan.tasks,
|
|
88
|
-
})
|
|
96
|
+
});
|
|
97
|
+
// Overlay readability is compile truth, not a mode-disagreement optimization. Check it before
|
|
98
|
+
// other repo-dependent lints so malformed native config cannot be hidden by an earlier finding.
|
|
89
99
|
enforceModeOverlay(graph, src, repoRoot);
|
|
100
|
+
enforceTaskUnitContract(graph, src, repoRoot);
|
|
90
101
|
// overseer-217 removal condition, now paid: on this milestone's authored graph the conventional
|
|
91
102
|
// name map emitted 21 raw unowned-test findings; review found 1 real and 20 false, while intersecting
|
|
92
103
|
// with a direct import or command-entry spawn retained the real one and left 0 false positives. That
|
package/dist/compile/native.js
CHANGED
|
@@ -286,11 +286,19 @@ export function authoringLintFindings(tasks, file) {
|
|
|
286
286
|
if (scope)
|
|
287
287
|
findings.push(scope);
|
|
288
288
|
for (const symbol of fencedIdentifiers(text)) {
|
|
289
|
-
|
|
289
|
+
// Empty files[] is the deliberately unrestricted scope and a non-repository programmatic
|
|
290
|
+
// compile has no disk corpus to prove against. But a declared files[] that expands to zero
|
|
291
|
+
// repository files is evidence, not absence: its preservation fence cannot be satisfied.
|
|
292
|
+
if (!root || task.files.length === 0)
|
|
290
293
|
continue;
|
|
294
|
+
if ([...taskTexts.values()].some((body) => body.includes(symbol)))
|
|
295
|
+
continue;
|
|
296
|
+
const matchDetail = taskFiles.length === 0
|
|
297
|
+
? `this task's files[] matched zero files on disk (${task.files.join(", ")})`
|
|
298
|
+
: `that symbol has zero hits in this task's files[] (${taskFiles.join(", ")})`;
|
|
291
299
|
findings.push({
|
|
292
300
|
code: "fence-symbol-absent", fixtureId: id, taskId: task.id, criterion: index + 1,
|
|
293
|
-
detail: `preservation fence cites \`${symbol}\` but
|
|
301
|
+
detail: `preservation fence cites \`${symbol}\` but ${matchDetail} — OBS-604`,
|
|
294
302
|
});
|
|
295
303
|
}
|
|
296
304
|
if (/line[- ]count|physical line|not greater than (?:the|\d)|no larger than|at most \d+ (?:lines|bytes)/i.test(text)) {
|
|
@@ -34,24 +34,29 @@ function sourceFiles(repoRoot) {
|
|
|
34
34
|
return [];
|
|
35
35
|
}
|
|
36
36
|
}
|
|
37
|
-
// OBS-898:
|
|
38
|
-
|
|
39
|
-
|
|
37
|
+
// OBS-898: expand the task's source declarations against ONE source-tree listing, once per compile.
|
|
38
|
+
// Tests then query this in-memory index; neither a test file nor a second corroboration pass can
|
|
39
|
+
// trigger another source walk or glob expansion.
|
|
40
|
+
function namedSourceIndex(tasks, allSources) {
|
|
40
41
|
const matches = new Map();
|
|
41
42
|
for (const task of tasks) {
|
|
42
43
|
for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/"))) {
|
|
43
44
|
const fromGlob = /[*?{[]/.test(entry);
|
|
44
45
|
const entries = fromGlob ? allSources.filter(filesGlob(entry)) : [entry];
|
|
45
46
|
for (const sourcePath of entries) {
|
|
46
|
-
|
|
47
|
-
if (stem === source || stem.startsWith(`${source}-`)) {
|
|
48
|
-
matches.set(`${task.id}:${sourcePath}`, { taskId: task.id, source: sourcePath, fromGlob });
|
|
49
|
-
}
|
|
47
|
+
matches.set(`${task.id}:${sourcePath}`, { taskId: task.id, source: sourcePath, fromGlob });
|
|
50
48
|
}
|
|
51
49
|
}
|
|
52
50
|
}
|
|
53
51
|
return [...matches.values()];
|
|
54
52
|
}
|
|
53
|
+
function namedSources(test, index) {
|
|
54
|
+
const stem = basename(test).replace(/\.test\.ts$/, "");
|
|
55
|
+
return index.filter(({ source }) => {
|
|
56
|
+
const sourceStem = basename(source, extname(source));
|
|
57
|
+
return stem === sourceStem || stem.startsWith(`${sourceStem}-`);
|
|
58
|
+
});
|
|
59
|
+
}
|
|
55
60
|
const moduleKey = (path) => normalize(path).replace(/\.(?:[cm]?[jt]sx?)$/, "");
|
|
56
61
|
function directImportSpecifiers(text) {
|
|
57
62
|
// Comments cannot create an edge. Keep strings intact because they are the import target.
|
|
@@ -167,9 +172,10 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
167
172
|
}
|
|
168
173
|
const globOwnedTests = new Map();
|
|
169
174
|
const allSources = sourceFiles(repoRoot);
|
|
175
|
+
const namedIndex = namedSourceIndex(tasks, allSources);
|
|
170
176
|
for (const source of sources) {
|
|
171
177
|
const ids = predictedBy.get(source.path) ?? new Set();
|
|
172
|
-
for (const named of namedSources(source.path,
|
|
178
|
+
for (const named of namedSources(source.path, namedIndex)) {
|
|
173
179
|
ids.add(named.taskId);
|
|
174
180
|
if (named.fromGlob) {
|
|
175
181
|
const owners = globOwnedTests.get(source.path) ?? new Set();
|
|
@@ -185,7 +191,7 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
185
191
|
if (owners(test).length === 0 && !(globOwnedTests.get(test)?.size)) {
|
|
186
192
|
const ids = [...taskIds].sort();
|
|
187
193
|
const source = sourceByPath.get(test);
|
|
188
|
-
const evidence = corroboration(source, namedSources(test,
|
|
194
|
+
const evidence = corroboration(source, namedSources(test, namedIndex));
|
|
189
195
|
findings.push({
|
|
190
196
|
code: "unowned-test",
|
|
191
197
|
test,
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -71,6 +71,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
71
71
|
private watches;
|
|
72
72
|
constructor(bin?: string, workersPerTab?: number, time?: HerdrTimeSource, journal?: DriverJournal | undefined);
|
|
73
73
|
private appendDispatchRetry;
|
|
74
|
+
private appendPaneClose;
|
|
74
75
|
private openRunJournal;
|
|
75
76
|
private liveSupervisionSeats;
|
|
76
77
|
private journalReconcile;
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -196,6 +196,29 @@ export class HerdrDriver {
|
|
|
196
196
|
// recovery to the file and the pipe while the operator's rail stays silent about it.
|
|
197
197
|
Journal.open(repoRoot, owned.runId, this.narrate).append("dispatch-retry", owned.taskId, data);
|
|
198
198
|
}
|
|
199
|
+
// OBS-906: close() is the harvest path, not a reconcile path, so the sweep's close row cannot
|
|
200
|
+
// describe it. Persist the worker slot identity beside the exact pane/tab address this path used.
|
|
201
|
+
// A driver used outside a daemon has no bound run journal; the injectable sink remains the unit
|
|
202
|
+
// seam there, while every daemon-created worktree is bound by worktree() below.
|
|
203
|
+
appendPaneClose(slot, placement) {
|
|
204
|
+
const owned = parseOwnedName(slot.name);
|
|
205
|
+
if (owned?.role !== "worker")
|
|
206
|
+
return;
|
|
207
|
+
try {
|
|
208
|
+
const data = { slot: slot.name, ...placement };
|
|
209
|
+
if (this.journal) {
|
|
210
|
+
this.journal("pane-close", slot.name, data);
|
|
211
|
+
return;
|
|
212
|
+
}
|
|
213
|
+
const repoRoot = this.journalRoots.get(slot.cwd);
|
|
214
|
+
if (!repoRoot)
|
|
215
|
+
return;
|
|
216
|
+
Journal.open(repoRoot, owned.runId, this.narrate).append("pane-close", owned.taskId, data);
|
|
217
|
+
}
|
|
218
|
+
catch {
|
|
219
|
+
/* close is best-effort; a journal failure must not make a successful pane reap fatal */
|
|
220
|
+
}
|
|
221
|
+
}
|
|
199
222
|
openRunJournal(runId) {
|
|
200
223
|
const roots = new Set(this.journalRoots.values());
|
|
201
224
|
roots.add(process.cwd());
|
|
@@ -1101,11 +1124,15 @@ export class HerdrDriver {
|
|
|
1101
1124
|
return this.serial(() => this.closeGrouped(slot));
|
|
1102
1125
|
}
|
|
1103
1126
|
if (slot.tabId) {
|
|
1104
|
-
await this.herdr(`tab close ${shq(slot.tabId)}`); // reaps the slot's whole tab, best-effort
|
|
1127
|
+
const closed = await this.herdr(`tab close ${shq(slot.tabId)}`); // reaps the slot's whole tab, best-effort
|
|
1128
|
+
if (closed.code === 0)
|
|
1129
|
+
this.appendPaneClose(slot, { tabId: slot.tabId });
|
|
1105
1130
|
return;
|
|
1106
1131
|
}
|
|
1107
1132
|
const pane = await this.paneId(slot);
|
|
1108
|
-
await this.herdr(`pane close ${shq(pane)}`); // best-effort
|
|
1133
|
+
const closed = await this.herdr(`pane close ${shq(pane)}`); // best-effort
|
|
1134
|
+
if (closed.code === 0)
|
|
1135
|
+
this.appendPaneClose(slot, { paneId: pane });
|
|
1109
1136
|
}
|
|
1110
1137
|
// D-08 ref-counted teardown, PER GENERATION (VIS-09 item 2): pane close per member; a generation's
|
|
1111
1138
|
// tab closes only when ITS OWN last member leaves; the group entry dies when all generations are gone.
|
|
@@ -1119,7 +1146,9 @@ export class HerdrDriver {
|
|
|
1119
1146
|
if (!gen)
|
|
1120
1147
|
return; // generation already torn down — its tab is gone
|
|
1121
1148
|
const pane = await this.paneId(slot);
|
|
1122
|
-
await this.herdr(`pane close ${shq(pane)}`); // best-effort
|
|
1149
|
+
const closed = await this.herdr(`pane close ${shq(pane)}`); // best-effort
|
|
1150
|
+
if (closed.code === 0)
|
|
1151
|
+
this.appendPaneClose(slot, { paneId: pane, tabId: gen.tabId });
|
|
1123
1152
|
gen.members = gen.members.filter((m) => m.name !== slot.name);
|
|
1124
1153
|
await this.renameGroupTab(gen);
|
|
1125
1154
|
if (gen.members.length === 0) {
|
package/dist/drivers/orca.d.ts
CHANGED
|
@@ -15,6 +15,8 @@ export declare const NOT_WRITABLE_CODE = "terminal_not_writable";
|
|
|
15
15
|
export declare const RUNNING_STATUS = "running";
|
|
16
16
|
export declare const STATUS_GOVERNED_METHODS: readonly ["read", "waitOutput", "status", "waitAgentStatus"];
|
|
17
17
|
export declare const WORKTREE_ADOPTION_TIMEOUT_MS = 60000;
|
|
18
|
+
/** A missing slot gets the same bounded chance to appear as a reaped shell gets to settle. */
|
|
19
|
+
export declare const PENDING_PROJECT_GRACE_MS = 2000;
|
|
18
20
|
export interface OrcaExec {
|
|
19
21
|
(args: string[], cwd: string, timeoutMs?: number): Promise<ShResult>;
|
|
20
22
|
}
|
|
@@ -136,7 +138,7 @@ export declare class OrcaDriver implements ExecutorDriver {
|
|
|
136
138
|
describe(slot: Slot): {
|
|
137
139
|
surface?: string;
|
|
138
140
|
hostPlatform?: string;
|
|
139
|
-
};
|
|
141
|
+
} | undefined;
|
|
140
142
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
141
143
|
private sendText;
|
|
142
144
|
private sendReceipt;
|
|
@@ -221,6 +223,8 @@ export declare class OrcaDriver implements ExecutorDriver {
|
|
|
221
223
|
* that still reports truncated at totalCount rows is a listing this sweep declines to judge on.
|
|
222
224
|
*/
|
|
223
225
|
private listAll;
|
|
226
|
+
private openRunJournal;
|
|
227
|
+
private dropExpiredProjects;
|
|
224
228
|
/**
|
|
225
229
|
* Sweep tickmarkr-owned terminals down to `desired`. Ownership is decided ONLY by parseOwnedName
|
|
226
230
|
* over the owned TAB title, through the same panesToClose fold herdr uses (drivers/types.ts): an
|
package/dist/drivers/orca.js
CHANGED
|
@@ -58,6 +58,8 @@ export const WORKTREE_ADOPTION_TIMEOUT_MS = 60_000;
|
|
|
58
58
|
const WORKTREE_ADOPTION_POLL_MS = 1_000;
|
|
59
59
|
const WORKTREE_ADOPTION_JOURNAL_MS = 2_000;
|
|
60
60
|
const NUDGE_ECHO_TIMEOUT_MS = 2_000;
|
|
61
|
+
/** A missing slot gets the same bounded chance to appear as a reaped shell gets to settle. */
|
|
62
|
+
export const PENDING_PROJECT_GRACE_MS = 2_000;
|
|
61
63
|
const SYSTEM_TIME = {
|
|
62
64
|
now: () => Date.now(),
|
|
63
65
|
sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
|
|
@@ -366,7 +368,7 @@ export class OrcaDriver {
|
|
|
366
368
|
this.taskWorktrees.set(owned.taskId, worktree);
|
|
367
369
|
const pending = this.pendingProjects.get(owned.taskId);
|
|
368
370
|
if (pending) {
|
|
369
|
-
await this.setWorkspaceStatus(worktree, pending);
|
|
371
|
+
await this.setWorkspaceStatus(worktree, pending.state);
|
|
370
372
|
this.pendingProjects.delete(owned.taskId);
|
|
371
373
|
}
|
|
372
374
|
}
|
|
@@ -1007,7 +1009,7 @@ export class OrcaDriver {
|
|
|
1007
1009
|
if (!worktree) {
|
|
1008
1010
|
// The daemon projects in-progress immediately before it creates the task checkout/slot.
|
|
1009
1011
|
// Hold only that latest state; slot() applies it once the task's own path is known.
|
|
1010
|
-
this.pendingProjects.set(taskId, state);
|
|
1012
|
+
this.pendingProjects.set(taskId, { state, since: this.time.now() });
|
|
1011
1013
|
return;
|
|
1012
1014
|
}
|
|
1013
1015
|
await this.setWorkspaceStatus(worktree, state);
|
|
@@ -1076,6 +1078,52 @@ export class OrcaDriver {
|
|
|
1076
1078
|
throw new OrcaError("list", "terminal list is still truncated at totalCount rows", whole.raw);
|
|
1077
1079
|
return whole;
|
|
1078
1080
|
}
|
|
1081
|
+
openRunJournal(runId) {
|
|
1082
|
+
const roots = new Set(this.journalRoots.values());
|
|
1083
|
+
roots.add(process.cwd());
|
|
1084
|
+
for (const repoRoot of roots) {
|
|
1085
|
+
try {
|
|
1086
|
+
return Journal.open(repoRoot, runId, this.narrate);
|
|
1087
|
+
}
|
|
1088
|
+
catch {
|
|
1089
|
+
/* try the next daemon-bound root */
|
|
1090
|
+
}
|
|
1091
|
+
}
|
|
1092
|
+
return undefined;
|
|
1093
|
+
}
|
|
1094
|
+
// A projection exists only to bridge project() to the worker slot that follows it. After the
|
|
1095
|
+
// grace, desired remains the dispatch oracle: a declared worker is still being placed and must
|
|
1096
|
+
// keep its projection. Make genuine absence durable by appending first and deleting second; with
|
|
1097
|
+
// no writable run journal the entry remains eligible for a later reconcile.
|
|
1098
|
+
dropExpiredProjects(desired, runId) {
|
|
1099
|
+
const now = this.time.now();
|
|
1100
|
+
const desiredTasks = new Set();
|
|
1101
|
+
for (const name of desired) {
|
|
1102
|
+
const owned = parseOwnedName(name);
|
|
1103
|
+
if (owned?.role === "worker")
|
|
1104
|
+
desiredTasks.add(owned.taskId);
|
|
1105
|
+
}
|
|
1106
|
+
const expired = [...this.pendingProjects].filter(([, pending]) => now - pending.since > PENDING_PROJECT_GRACE_MS).filter(([taskId]) => !desiredTasks.has(taskId));
|
|
1107
|
+
if (expired.length === 0)
|
|
1108
|
+
return;
|
|
1109
|
+
const journal = this.openRunJournal(runId);
|
|
1110
|
+
if (!journal)
|
|
1111
|
+
return;
|
|
1112
|
+
for (const [taskId, pending] of expired) {
|
|
1113
|
+
try {
|
|
1114
|
+
const pendingMs = Math.max(0, now - pending.since);
|
|
1115
|
+
journal.append("project-unplaced", taskId, {
|
|
1116
|
+
state: pending.state,
|
|
1117
|
+
pendingMs,
|
|
1118
|
+
graceMs: PENDING_PROJECT_GRACE_MS,
|
|
1119
|
+
});
|
|
1120
|
+
this.pendingProjects.delete(taskId);
|
|
1121
|
+
}
|
|
1122
|
+
catch {
|
|
1123
|
+
/* reconcile is cosmetic; preserve the projection until a later journalled drop */
|
|
1124
|
+
}
|
|
1125
|
+
}
|
|
1126
|
+
}
|
|
1079
1127
|
/**
|
|
1080
1128
|
* Sweep tickmarkr-owned terminals down to `desired`. Ownership is decided ONLY by parseOwnedName
|
|
1081
1129
|
* over the owned TAB title, through the same panesToClose fold herdr uses (drivers/types.ts): an
|
|
@@ -1091,6 +1139,7 @@ export class OrcaDriver {
|
|
|
1091
1139
|
* Cosmetic by contract: every failure is swallowed, per candidate and overall.
|
|
1092
1140
|
*/
|
|
1093
1141
|
async reconcile(desired, runId, opts) {
|
|
1142
|
+
this.dropExpiredProjects(desired, runId);
|
|
1094
1143
|
try {
|
|
1095
1144
|
// Every call below is handle-addressed or explicitly selectored, so the CLI's own cwd selects
|
|
1096
1145
|
// nothing — it only has to exist, which the checkouts being swept no longer need to.
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -76,7 +76,7 @@ export interface ExecutorDriver {
|
|
|
76
76
|
/** The exact terminal read surface used for liveness evidence. */
|
|
77
77
|
readSource?: string;
|
|
78
78
|
/** Placement facts returned by drivers whose terminal host exposes them. */
|
|
79
|
-
describe?(slot: Slot): SlotPlacement | Promise<SlotPlacement
|
|
79
|
+
describe?(slot: Slot): SlotPlacement | Promise<SlotPlacement> | undefined;
|
|
80
80
|
slot(cwd: string, name: string, opts?: SlotOpts): Promise<Slot>;
|
|
81
81
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
82
82
|
waitOutput(slot: Slot, pattern: string, timeoutMs: number, opts?: {
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -24,7 +24,9 @@ export interface BaselineCommand {
|
|
|
24
24
|
fingerprints: string[];
|
|
25
25
|
missingCommand?: boolean;
|
|
26
26
|
/** Why a capture returned no verdict. */
|
|
27
|
-
invalidCause?: "ceiling-kill" | "resource-exhaustion";
|
|
27
|
+
invalidCause?: "ceiling-kill" | "resource-exhaustion" | "infra";
|
|
28
|
+
/** The runner's summary was green and only its teardown fingerprint followed. */
|
|
29
|
+
teardownFingerprint?: true;
|
|
28
30
|
/** What this command actually took at capture, on a pristine tree. Absent in pre-v1.90 baselines. */
|
|
29
31
|
durationMs?: number;
|
|
30
32
|
/** Sum of the per-file durations named by the runner; null when its output names none. */
|
|
@@ -154,4 +156,26 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
|
|
|
154
156
|
id: string;
|
|
155
157
|
acceptance: AcceptanceItem[];
|
|
156
158
|
}>): Promise<VacuousOracleWarning[]>;
|
|
157
|
-
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[]
|
|
159
|
+
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: {
|
|
160
|
+
rerunOf?: HostStarvedRerun;
|
|
161
|
+
}): Promise<GateResult[]>;
|
|
162
|
+
export interface HostStarvedRerun {
|
|
163
|
+
durationMs: number;
|
|
164
|
+
referenceMs: number;
|
|
165
|
+
waitedMs: number;
|
|
166
|
+
}
|
|
167
|
+
export type RunnerVerdict = FailureClassification | "green-teardown" | undefined;
|
|
168
|
+
/** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
|
|
169
|
+
export declare function classifyRunnerOutput(raw: string, code: number): RunnerVerdict;
|
|
170
|
+
export declare const HOST_STARVED_DURATION_FACTOR = 2;
|
|
171
|
+
/** Every fresh failure head is timeout-class and the suite took twice its own baseline measurement. */
|
|
172
|
+
export declare function hostStarved(fresh: string, durationMs: number, referenceMs: number | undefined): boolean;
|
|
173
|
+
interface CalmWindow {
|
|
174
|
+
pollMs: number;
|
|
175
|
+
maxWaitMs: number;
|
|
176
|
+
loadProvider: () => number;
|
|
177
|
+
calmLoad: () => number;
|
|
178
|
+
}
|
|
179
|
+
export declare function setCalmWindowForTests(over: Partial<CalmWindow>): void;
|
|
180
|
+
export declare function resetCalmWindowForTests(): void;
|
|
181
|
+
export {};
|
package/dist/gates/baseline.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { existsSync, readFileSync } from "node:fs";
|
|
2
|
+
import { availableParallelism, loadavg } from "node:os";
|
|
2
3
|
import { join } from "node:path";
|
|
3
4
|
import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
|
|
4
5
|
// incident #2 (run-20260709-104447): a vitest ✓ PASS line with "error" in the test NAME, wrapped in ANSI
|
|
@@ -501,11 +502,20 @@ export async function captureBaseline(cwd, commands) {
|
|
|
501
502
|
base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
|
|
502
503
|
continue;
|
|
503
504
|
}
|
|
505
|
+
// OBS-885/887: capture and gate ask the same classifier. A green summary followed only by the
|
|
506
|
+
// teardown fingerprint is a pass; infrastructure without a summary is no verdict to forgive.
|
|
507
|
+
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
508
|
+
if (runnerVerdict === "infra") {
|
|
509
|
+
console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence and no green summary — it recorded NO verdict; nothing is forgiven for this command`);
|
|
510
|
+
base.commands[name] = invalidCaptureEntry(durationMs, "infra");
|
|
511
|
+
continue;
|
|
512
|
+
}
|
|
504
513
|
base.commands[name] = {
|
|
505
|
-
exitCode: r.code,
|
|
514
|
+
exitCode: runnerVerdict === "green-teardown" ? 0 : r.code,
|
|
515
|
+
...(runnerVerdict === "green-teardown" ? { teardownFingerprint: true } : {}),
|
|
506
516
|
// a command that exits 0 has no failures to fingerprint — recording any would be a lie the
|
|
507
517
|
// compare step then has to forgive
|
|
508
|
-
fingerprints: r.code === 0 ? [] : fingerprint(raw),
|
|
518
|
+
fingerprints: r.code === 0 || runnerVerdict === "green-teardown" ? [] : fingerprint(raw),
|
|
509
519
|
missingCommand: missingConfiguredCommand(cmd, r),
|
|
510
520
|
durationMs,
|
|
511
521
|
...fileTiming(raw, durationMs),
|
|
@@ -570,8 +580,9 @@ function headlineDetails(raw, fresh) {
|
|
|
570
580
|
meta: { failingTests: headlines.filter(namesFailureEitherForm) },
|
|
571
581
|
};
|
|
572
582
|
}
|
|
573
|
-
export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
583
|
+
export async function compareToBaseline(cwd, commands, baseline, enabled, opts = {}) {
|
|
574
584
|
const results = [];
|
|
585
|
+
const rerunOf = opts.rerunOf;
|
|
575
586
|
for (const name of enabled) {
|
|
576
587
|
const cmd = commands[name];
|
|
577
588
|
if (!cmd) {
|
|
@@ -595,7 +606,13 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
595
606
|
// states no capacity — a row that never divided the machine must not claim that it did.
|
|
596
607
|
const record = (g) => {
|
|
597
608
|
const withReap = r.reapedGroup ? { ...g, meta: { ...g.meta, reapedGroup: true } } : g;
|
|
598
|
-
|
|
609
|
+
const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
|
|
610
|
+
const withRerun = rerunOf ? {
|
|
611
|
+
...withReapError,
|
|
612
|
+
details: `host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details}`,
|
|
613
|
+
meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
|
|
614
|
+
} : withReapError;
|
|
615
|
+
results.push(r.capacity ? { ...withRerun, capacity: r.capacity } : withRerun);
|
|
599
616
|
};
|
|
600
617
|
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
601
618
|
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
|
@@ -629,6 +646,12 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
629
646
|
continue;
|
|
630
647
|
}
|
|
631
648
|
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
649
|
+
// OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
|
|
650
|
+
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
651
|
+
if (runnerVerdict === "green-teardown") {
|
|
652
|
+
record({ gate: name, pass: true, details: `exit ${r.code} after a green suite summary; only the runner's teardown fingerprint followed it`, meta: { teardownFingerprint: true } });
|
|
653
|
+
continue;
|
|
654
|
+
}
|
|
632
655
|
// OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
|
|
633
656
|
// unrecognized-output marker, which is evidence for the operator and never grounds to reject.
|
|
634
657
|
// ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
|
|
@@ -640,10 +663,20 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
640
663
|
// T9: classify the FRESH diff before charging it. The complete runner output can legitimately
|
|
641
664
|
// contain a baseline-recorded assertion beside a newly introduced infrastructure death; letting
|
|
642
665
|
// that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
|
|
643
|
-
// `
|
|
666
|
+
// `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
|
|
644
667
|
// retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
|
|
645
|
-
const
|
|
646
|
-
const
|
|
668
|
+
const freshVerdict = failing.length ? classifyRunnerOutput(failing.join("\n"), r.code) : undefined;
|
|
669
|
+
const freshClassification = freshVerdict === "infra" || freshVerdict === "regression" ? freshVerdict : undefined;
|
|
670
|
+
const classification = freshClassification ?? (!failing.length && (runnerVerdict === "infra" || runnerVerdict === "regression") ? runnerVerdict : undefined);
|
|
671
|
+
// OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
|
|
672
|
+
// own baseline measurement. The first read buys one calm rerun here, never a worker repair.
|
|
673
|
+
if (name === "test" && classification !== "infra" && failing.length && !rerunOf
|
|
674
|
+
&& hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
|
|
675
|
+
const waitedMs = await waitForCalmWindow();
|
|
676
|
+
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
677
|
+
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { rerunOf: provenance }));
|
|
678
|
+
continue;
|
|
679
|
+
}
|
|
647
680
|
if (classification === "infra") {
|
|
648
681
|
const evidence = failing.length
|
|
649
682
|
? failing.slice(0, 10).join("\n")
|
|
@@ -708,3 +741,53 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
708
741
|
}
|
|
709
742
|
return results;
|
|
710
743
|
}
|
|
744
|
+
const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
|
|
745
|
+
const TEARDOWN_RE = /\[vitest-worker\]: Timeout calling\b|\[birpc\] rpc is closed, cannot call\b/;
|
|
746
|
+
const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
|
|
747
|
+
/** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
|
|
748
|
+
export function classifyRunnerOutput(raw, code) {
|
|
749
|
+
if (code === 0)
|
|
750
|
+
return undefined;
|
|
751
|
+
const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
|
|
752
|
+
const text = lines.join("\n");
|
|
753
|
+
const summary = SUMMARY_LINE_RE.exec(text);
|
|
754
|
+
if (summary) {
|
|
755
|
+
const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
|
|
756
|
+
const summaryGreen = failed.every((count) => count === 0);
|
|
757
|
+
const summaryLine = text.slice(0, summary.index).split("\n").length - 1;
|
|
758
|
+
const teardownLine = lines.findIndex((line, index) => index > summaryLine && TEARDOWN_RE.test(line));
|
|
759
|
+
const otherFailure = lines.some((line, index) => {
|
|
760
|
+
if (index === summaryLine || TEARDOWN_RE.test(line))
|
|
761
|
+
return false;
|
|
762
|
+
if (PASS_LINE_RE.test(line) || OPERATOR_LINE_RE.test(line))
|
|
763
|
+
return false;
|
|
764
|
+
if (index > summaryLine && UNHANDLED_HEADER_RE.test(line))
|
|
765
|
+
return false;
|
|
766
|
+
return namesRegression(line);
|
|
767
|
+
});
|
|
768
|
+
if (summaryGreen && teardownLine > summaryLine && !otherFailure)
|
|
769
|
+
return "green-teardown";
|
|
770
|
+
}
|
|
771
|
+
return classifyFailureOutput(text);
|
|
772
|
+
}
|
|
773
|
+
const TIMEOUT_CLASS_RE = /\b(?:Test|Hook) timed out in (?:\d+|#) ?ms\b|\bexceeded (?:\d+|#) ?(?:ms|s|seconds)\b|\btimed out after (?:\d+|#)/i;
|
|
774
|
+
const ERROR_HEAD_RE = /^\s*(?:[A-Za-z][A-Za-z0-9]*Error|Error):\s/;
|
|
775
|
+
export const HOST_STARVED_DURATION_FACTOR = 2;
|
|
776
|
+
/** Every fresh failure head is timeout-class and the suite took twice its own baseline measurement. */
|
|
777
|
+
export function hostStarved(fresh, durationMs, referenceMs) {
|
|
778
|
+
if (!referenceMs || durationMs < HOST_STARVED_DURATION_FACTOR * referenceMs)
|
|
779
|
+
return false;
|
|
780
|
+
const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
|
|
781
|
+
return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
|
|
782
|
+
}
|
|
783
|
+
const DEFAULT_CALM = { pollMs: 5_000, maxWaitMs: 600_000, loadProvider: () => loadavg()[0] ?? 0, calmLoad: () => availableParallelism() / 2 };
|
|
784
|
+
let calm = DEFAULT_CALM;
|
|
785
|
+
export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
|
|
786
|
+
export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
|
|
787
|
+
async function waitForCalmWindow() {
|
|
788
|
+
const started = Date.now();
|
|
789
|
+
while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
|
|
790
|
+
await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
|
|
791
|
+
}
|
|
792
|
+
return Date.now() - started;
|
|
793
|
+
}
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { type Assignment, type BillingChannel, type WorkerAdapter } from "../adapters/types.js";
|
|
2
2
|
import { type TickmarkrConfig, type Tier } from "../config/config.js";
|
|
3
3
|
import { type Task } from "../graph/schema.js";
|
|
4
|
+
import { modelProvider } from "../route/preference.js";
|
|
4
5
|
import { type GateVia } from "./llm.js";
|
|
5
6
|
import type { GateResult } from "./types.js";
|
|
6
7
|
import { type VerdictUnparseableCause } from "./verdict-cause.js";
|
|
@@ -48,8 +49,7 @@ export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffM
|
|
|
48
49
|
export declare function isDiffCapPark(result: GateResult): boolean;
|
|
49
50
|
export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
50
51
|
export declare function modelId(model: string): string;
|
|
51
|
-
|
|
52
|
-
export declare function modelProvider(model: string, fallback?: string): string;
|
|
52
|
+
export { modelProvider };
|
|
53
53
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
54
54
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
55
55
|
floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
|
package/dist/gates/review.js
CHANGED
|
@@ -8,6 +8,7 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
9
|
import { redactSecrets } from "../run/redact.js";
|
|
10
10
|
import { marginalCostRank } from "../route/router.js";
|
|
11
|
+
import { modelProvider } from "../route/preference.js";
|
|
11
12
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
12
13
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
13
14
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
@@ -165,24 +166,7 @@ export function diffCapParkReason(results) {
|
|
|
165
166
|
export function modelId(model) {
|
|
166
167
|
return model.slice(model.lastIndexOf("/") + 1);
|
|
167
168
|
}
|
|
168
|
-
|
|
169
|
-
export function modelProvider(model, fallback = "unknown") {
|
|
170
|
-
const id = model.toLowerCase();
|
|
171
|
-
const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
|
|
172
|
-
if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
|
|
173
|
-
return "openai";
|
|
174
|
-
if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
|
|
175
|
-
return "anthropic";
|
|
176
|
-
if (prefix === "google" || /^gemini(?:-|$)/.test(id))
|
|
177
|
-
return "google";
|
|
178
|
-
if (prefix === "xai" || /^grok(?:-|$)/.test(id))
|
|
179
|
-
return "xai";
|
|
180
|
-
if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
|
|
181
|
-
return "zhipu";
|
|
182
|
-
if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
|
|
183
|
-
return "moonshot";
|
|
184
|
-
return fallback;
|
|
185
|
-
}
|
|
169
|
+
export { modelProvider };
|
|
186
170
|
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
187
171
|
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
188
172
|
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
package/dist/graph/graph.d.ts
CHANGED
|
@@ -1,6 +1,26 @@
|
|
|
1
1
|
import { type RunGraph, type Task, type TaskStatus } from "./schema.js";
|
|
2
2
|
export declare function stateDirName(_repoRoot: string): string;
|
|
3
3
|
export declare function graphPath(repoRoot: string): string;
|
|
4
|
+
export interface CompileRefusalRecord {
|
|
5
|
+
refusedAt: string;
|
|
6
|
+
source: string;
|
|
7
|
+
error: string;
|
|
8
|
+
}
|
|
9
|
+
export declare function compileRefusalPath(repoRoot: string): string;
|
|
10
|
+
export declare function readCompileRefusal(repoRoot: string): CompileRefusalRecord | undefined;
|
|
11
|
+
export declare function saveCompileRefusal(repoRoot: string, record: CompileRefusalRecord): void;
|
|
12
|
+
export declare function clearCompileRefusal(repoRoot: string): void;
|
|
13
|
+
export interface OnDiskSpecHash {
|
|
14
|
+
path: string;
|
|
15
|
+
hash: string;
|
|
16
|
+
}
|
|
17
|
+
/**
|
|
18
|
+
* Re-hash the exact single source file recorded by CLI compiles. Native and PRD record the source
|
|
19
|
+
* markdown; Spec Kit records its tasks.md. GSD combines several plan bodies and is deliberately not
|
|
20
|
+
* reconstructed here. A missing file is the one fail-open case: the refusal record is the fail-closed
|
|
21
|
+
* evidence for failed recompiles, while moved/deleted source files need not strand a compiled graph.
|
|
22
|
+
*/
|
|
23
|
+
export declare function onDiskSpecHash(_repoRoot: string, graph: RunGraph): OnDiskSpecHash | undefined;
|
|
4
24
|
export declare function graphDefinitionHash(g: RunGraph): string;
|
|
5
25
|
export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
|
|
6
26
|
export declare function tickmarkrDir(repoRoot: string): string;
|