tickmarkr 1.97.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/brand.d.ts +28 -0
- package/dist/brand.js +41 -0
- package/dist/cli/commands/compile.js +32 -1
- package/dist/cli/commands/resume.js +9 -3
- package/dist/cli/commands/run.d.ts +61 -1
- package/dist/cli/commands/run.js +368 -17
- package/dist/cli/commands/status.js +145 -28
- package/dist/compile/collateral.d.ts +25 -0
- package/dist/compile/collateral.js +46 -11
- package/dist/compile/native.js +10 -0
- package/dist/drivers/herdr.d.ts +19 -13
- package/dist/drivers/herdr.js +88 -26
- package/dist/drivers/types.d.ts +2 -0
- package/dist/gates/acceptance.js +17 -7
- package/dist/gates/llm.d.ts +19 -0
- package/dist/gates/llm.js +104 -6
- package/dist/gates/run-gates.d.ts +18 -0
- package/dist/gates/run-gates.js +195 -29
- package/dist/gates/scope.d.ts +9 -1
- package/dist/gates/scope.js +22 -2
- package/dist/graph/graph.d.ts +1 -0
- package/dist/graph/graph.js +19 -2
- package/dist/report/compare.js +17 -2
- package/dist/run/daemon.d.ts +1 -8
- package/dist/run/daemon.js +231 -246
- package/dist/run/environment.d.ts +18 -1
- package/dist/run/environment.js +19 -2
- package/dist/run/journal.d.ts +50 -3
- package/dist/run/journal.js +181 -5
- package/dist/run/protocol.d.ts +4 -4
- package/dist/run/stall.d.ts +30 -0
- package/dist/run/stall.js +173 -0
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +20 -15
package/dist/gates/run-gates.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
|
-
import { readFileSync } from "node:fs";
|
|
2
|
-
import {
|
|
1
|
+
import { mkdtempSync, readFileSync, rmSync } from "node:fs";
|
|
2
|
+
import { loadavg, tmpdir } from "node:os";
|
|
3
|
+
import { join, posix } from "node:path";
|
|
3
4
|
import { channelKey, shq } from "../adapters/types.js";
|
|
4
5
|
import { TIER_RANK } from "../config/config.js";
|
|
5
6
|
import { getAdapter } from "../adapters/registry.js";
|
|
@@ -14,6 +15,85 @@ import { reviewGate } from "./review.js";
|
|
|
14
15
|
import { scopeGate } from "./scope.js";
|
|
15
16
|
import { shGit } from "../run/git.js";
|
|
16
17
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
18
|
+
const productionLoadProvider = () => loadavg()[0] ?? 0;
|
|
19
|
+
let loadProvider = productionLoadProvider;
|
|
20
|
+
/** Test seam — inject deterministic load samples; production always reads os.loadavg. */
|
|
21
|
+
export function setLoadProviderForTests(provider) {
|
|
22
|
+
loadProvider = provider;
|
|
23
|
+
}
|
|
24
|
+
export function resetLoadProviderForTests() {
|
|
25
|
+
loadProvider = productionLoadProvider;
|
|
26
|
+
}
|
|
27
|
+
/**
|
|
28
|
+
* Instrument the adapter command itself, which is the first adapter-owned operation runLlm performs.
|
|
29
|
+
* acceptanceGate/reviewGate deliberately retain ownership of deterministic oracles, policy checks,
|
|
30
|
+
* diff reads and prompt construction; none of that preprocessing belongs to an LLM invocation span.
|
|
31
|
+
*
|
|
32
|
+
* Both stamps are written by the same shell immediately around the adapter command. That excludes
|
|
33
|
+
* pane-slot acquisition as well as runLlm's scratch cleanup and verdict parsing, without changing
|
|
34
|
+
* llm.ts's output contract. The subshell keeps an adapter command's `exit` from bypassing the end
|
|
35
|
+
* stamp, and the original exit status is preserved.
|
|
36
|
+
*/
|
|
37
|
+
function instrumentLlmAdapter(adapter, clocks) {
|
|
38
|
+
return new Proxy(adapter, {
|
|
39
|
+
get(target, property) {
|
|
40
|
+
if (property === "headlessCommand") {
|
|
41
|
+
return (promptFile, model) => {
|
|
42
|
+
const command = target.headlessCommand(promptFile, model);
|
|
43
|
+
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-gate-invocation-"));
|
|
44
|
+
const startedAtPath = join(dir, "started-at");
|
|
45
|
+
const completedAtPath = join(dir, "completed-at");
|
|
46
|
+
clocks.push({
|
|
47
|
+
channel: channelKey({ adapter: target.id, model }),
|
|
48
|
+
preparedAt: Date.now(),
|
|
49
|
+
startedAtPath,
|
|
50
|
+
completedAtPath,
|
|
51
|
+
dir,
|
|
52
|
+
});
|
|
53
|
+
const stamp = (path) => `${shq(process.execPath)} -e ${shq('require("node:fs").writeFileSync(process.argv[1], String(Date.now()))')} ${shq(path)}`;
|
|
54
|
+
// End with a status-bearing subshell, not `exit`: pane mode appends its nonce-bound
|
|
55
|
+
// completion trailer on the next script line and must remain able to run it.
|
|
56
|
+
return `${stamp(startedAtPath)}; ( ${command} ); __tickmarkr_invocation_status=$?; ${stamp(completedAtPath)}; (exit $__tickmarkr_invocation_status)`;
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
const value = Reflect.get(target, property, target);
|
|
60
|
+
// Real adapters may use private fields; bind their methods to the target rather than the Proxy.
|
|
61
|
+
return typeof value === "function" ? value.bind(target) : value;
|
|
62
|
+
},
|
|
63
|
+
});
|
|
64
|
+
}
|
|
65
|
+
function finishLlmDispatches(clocks) {
|
|
66
|
+
return clocks.map((clock) => {
|
|
67
|
+
let startedAt = clock.preparedAt;
|
|
68
|
+
let completedAt = Date.now();
|
|
69
|
+
try {
|
|
70
|
+
const stampedStart = Number(readFileSync(clock.startedAtPath, "utf8"));
|
|
71
|
+
const stamped = Number(readFileSync(clock.completedAtPath, "utf8"));
|
|
72
|
+
if (Number.isFinite(stampedStart))
|
|
73
|
+
startedAt = stampedStart;
|
|
74
|
+
if (Number.isFinite(stamped) && stamped >= startedAt)
|
|
75
|
+
completedAt = stamped;
|
|
76
|
+
}
|
|
77
|
+
catch {
|
|
78
|
+
// A killed command may never reach its stamp; the gate return is the honest upper boundary.
|
|
79
|
+
}
|
|
80
|
+
finally {
|
|
81
|
+
rmSync(clock.dir, { recursive: true, force: true });
|
|
82
|
+
}
|
|
83
|
+
return { channel: clock.channel, durationMs: completedAt - startedAt };
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
async function captureLlmDispatches(adapters, run) {
|
|
87
|
+
const clocks = [];
|
|
88
|
+
try {
|
|
89
|
+
const captured = await captureLlmOutput(() => run(adapters.map((a) => instrumentLlmAdapter(a, clocks))));
|
|
90
|
+
return { ...captured, invocations: finishLlmDispatches(clocks) };
|
|
91
|
+
}
|
|
92
|
+
catch (error) {
|
|
93
|
+
finishLlmDispatches(clocks);
|
|
94
|
+
throw error;
|
|
95
|
+
}
|
|
96
|
+
}
|
|
17
97
|
const TEST_FILE_RE = /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/;
|
|
18
98
|
// relative specifiers only — `from "./x.js"`, `import("./x.js")`, `require("./x.js")`
|
|
19
99
|
const IMPORT_RE = /(?:\bfrom\s*|\bimport\s*\(\s*|\brequire\s*\(\s*)["'](\.[^"']*)["']/g;
|
|
@@ -124,12 +204,56 @@ export async function runGates(task, ctx) {
|
|
|
124
204
|
// exactly one `test` gate-result ever leaves a round, always carrying which suite spoke for it.
|
|
125
205
|
// (A RED screen IS the verdict: the round ends there, so it is recorded immediately.)
|
|
126
206
|
let heldTest;
|
|
207
|
+
// v2.0 T2 (OBS-554): this round's per-gate measurement. Every interval a gate actually spends
|
|
208
|
+
// executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
|
|
209
|
+
// gate's screen and its full suite) sums to its own cost and never to the span between them.
|
|
210
|
+
const spans = new Map();
|
|
211
|
+
// The test gate's two halves, kept apart as well as summed: `durationMs` alone cannot say whether
|
|
212
|
+
// a slow round was a slow subset or a slow full suite, and the parked scheduler's threshold is
|
|
213
|
+
// defined over the full-suite cost.
|
|
214
|
+
let selectedDurationMs;
|
|
215
|
+
let fullDurationMs;
|
|
216
|
+
const measure = async (gate, run) => {
|
|
217
|
+
const at = Date.now();
|
|
218
|
+
const load1Start = loadProvider();
|
|
219
|
+
try {
|
|
220
|
+
return await run();
|
|
221
|
+
}
|
|
222
|
+
finally {
|
|
223
|
+
const prior = spans.get(gate);
|
|
224
|
+
spans.set(gate, {
|
|
225
|
+
durationMs: (prior?.durationMs ?? 0) + (Date.now() - at),
|
|
226
|
+
load1Start: prior?.load1Start ?? load1Start,
|
|
227
|
+
load1End: loadProvider(),
|
|
228
|
+
});
|
|
229
|
+
}
|
|
230
|
+
};
|
|
231
|
+
// The measurement is attached at the ONE seam every result leaves this function through, so a
|
|
232
|
+
// path that forgets to measure is visibly missing its telemetry rather than carrying a fabricated
|
|
233
|
+
// zero. The daemon lifts these off `meta` onto the gate-result row (src/run/daemon.ts).
|
|
234
|
+
const withTelemetry = (result) => {
|
|
235
|
+
const span = spans.get(result.gate);
|
|
236
|
+
if (!span)
|
|
237
|
+
return result;
|
|
238
|
+
return {
|
|
239
|
+
...result,
|
|
240
|
+
meta: {
|
|
241
|
+
...result.meta,
|
|
242
|
+
...span,
|
|
243
|
+
...(result.gate === "test" && selectedDurationMs !== undefined ? { selectedDurationMs } : {}),
|
|
244
|
+
...(result.gate === "test" && fullDurationMs !== undefined ? { fullDurationMs } : {}),
|
|
245
|
+
},
|
|
246
|
+
};
|
|
247
|
+
};
|
|
127
248
|
const sequence = GATE_NAMES.filter((g) => enabled(g));
|
|
128
249
|
const total = sequence.length;
|
|
129
250
|
const indexOf = (gate) => sequence.indexOf(gate) + 1;
|
|
251
|
+
// The stamp lands here, on the ONE object that is both pushed and published, so the round's record
|
|
252
|
+
// and its event stream carry byte-identical results — an invariant the fixtures pin.
|
|
130
253
|
const record = async (result) => {
|
|
131
|
-
|
|
132
|
-
|
|
254
|
+
const stamped = withTelemetry(result);
|
|
255
|
+
results.push(stamped);
|
|
256
|
+
await ctx.onGate?.({ phase: "end", gate: stamped.gate, result: stamped });
|
|
133
257
|
};
|
|
134
258
|
const emitStart = async (gate, parentAt) => {
|
|
135
259
|
await ctx.onGate?.({ phase: "start", gate, index: indexOf(gate), total, ...(parentAt === undefined ? {} : { parentAt }) });
|
|
@@ -141,7 +265,7 @@ export async function runGates(task, ctx) {
|
|
|
141
265
|
if (heldTest) {
|
|
142
266
|
const held = heldTest;
|
|
143
267
|
heldTest = undefined;
|
|
144
|
-
await ctx.onGate?.({ phase: "end", gate: "test", result: held });
|
|
268
|
+
await ctx.onGate?.({ phase: "end", gate: "test", result: held }); // stamped when it was held
|
|
145
269
|
}
|
|
146
270
|
const sorted = [...results].sort((a, b) => GATE_NAMES.indexOf(a.gate) - GATE_NAMES.indexOf(b.gate));
|
|
147
271
|
// v1.87 T5: no round returns a MERGEABLE GREEN on a dirty tree. The battery is not the only gate
|
|
@@ -155,7 +279,7 @@ export async function runGates(task, ctx) {
|
|
|
155
279
|
if (last && sorted.every((r) => r.pass || r.meta?.skipped === true)) {
|
|
156
280
|
const dirt = await dirtyWorktree();
|
|
157
281
|
if (dirt) {
|
|
158
|
-
const refusal = dirtyRoundRefusal(last.gate, dirt);
|
|
282
|
+
const refusal = withTelemetry(dirtyRoundRefusal(last.gate, dirt));
|
|
159
283
|
results[results.indexOf(last)] = refusal;
|
|
160
284
|
sorted[sorted.length - 1] = refusal;
|
|
161
285
|
await ctx.onGate?.({ phase: "end", gate: refusal.gate, result: refusal });
|
|
@@ -221,7 +345,14 @@ export async function runGates(task, ctx) {
|
|
|
221
345
|
// ponytail: compareToBaseline batches build/test/lint — their starts are emitted at iteration,
|
|
222
346
|
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
223
347
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
348
|
+
// ponytail: legacy runs build/test/lint in ONE compareToBaseline call, so there is one interval
|
|
349
|
+
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
350
|
+
const batchAt = Date.now();
|
|
351
|
+
const batchLoadStart = loadProvider();
|
|
224
352
|
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
|
|
353
|
+
const batch = { durationMs: Date.now() - batchAt, load1Start: batchLoadStart, load1End: loadProvider() };
|
|
354
|
+
for (const g of toolGates)
|
|
355
|
+
spans.set(g, batch);
|
|
225
356
|
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
226
357
|
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
227
358
|
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
@@ -238,7 +369,10 @@ export async function runGates(task, ctx) {
|
|
|
238
369
|
// the full vitest suite before anyone reads its verdict.
|
|
239
370
|
for (const g of toolGates) {
|
|
240
371
|
await emitStart(g);
|
|
241
|
-
const [r] = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]);
|
|
372
|
+
const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
|
|
373
|
+
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
374
|
+
if (g === "test" && selected)
|
|
375
|
+
selectedDurationMs = spans.get("test").durationMs;
|
|
242
376
|
// The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
|
|
243
377
|
// tracked file makes it dirty again, and every gate after it — including the next shell gate,
|
|
244
378
|
// which would then run against bytes HEAD does not hold — inherits that. So re-check after each
|
|
@@ -257,8 +391,8 @@ export async function runGates(task, ctx) {
|
|
|
257
391
|
if (!screened.pass)
|
|
258
392
|
await record(screened);
|
|
259
393
|
else {
|
|
260
|
-
heldTest = screened;
|
|
261
|
-
results.push(
|
|
394
|
+
heldTest = withTelemetry(screened);
|
|
395
|
+
results.push(heldTest);
|
|
262
396
|
}
|
|
263
397
|
}
|
|
264
398
|
else {
|
|
@@ -292,10 +426,10 @@ export async function runGates(task, ctx) {
|
|
|
292
426
|
*/
|
|
293
427
|
const allowDeviations = [...(ctx.cfg.scope?.allowDeviations ?? [])];
|
|
294
428
|
Object.freeze(allowDeviations);
|
|
295
|
-
const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, allowDeviations);
|
|
429
|
+
const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, allowDeviations, ctx.collateral ? { taskId: task.id, predicted: ctx.collateral } : undefined);
|
|
296
430
|
const runGate = async (gate, compute) => {
|
|
297
431
|
await emitStart(gate);
|
|
298
|
-
await record(await compute
|
|
432
|
+
await record(await measure(gate, compute));
|
|
299
433
|
};
|
|
300
434
|
/**
|
|
301
435
|
* T4 (OBS-265): the deterministic git checks run BEFORE the battery, as a screen — they answer
|
|
@@ -319,7 +453,7 @@ export async function runGates(task, ctx) {
|
|
|
319
453
|
for (const [gate, compute] of [["evidence", evidenceResult], ["scope", scopeResult]]) {
|
|
320
454
|
if (!enabled(gate))
|
|
321
455
|
continue;
|
|
322
|
-
screened.push(await compute
|
|
456
|
+
screened.push(await measure(gate, compute));
|
|
323
457
|
if (screened[screened.length - 1].pass)
|
|
324
458
|
continue;
|
|
325
459
|
for (const r of screened) {
|
|
@@ -354,20 +488,30 @@ export async function runGates(task, ctx) {
|
|
|
354
488
|
// v1.19 (T2): testCmd threads the detected test runner to the gate so named-test oracles run
|
|
355
489
|
// deterministically (filtered via -t) before any LLM judge dispatch.
|
|
356
490
|
const invocations = [];
|
|
491
|
+
// v2.0 T2 (OBS-554): one entry per JUDGE DISPATCH — primary and the GATE-09 retry alike. The gate's
|
|
492
|
+
// own durationMs is the pair's envelope and cannot answer what the parked ceiling recalibration
|
|
493
|
+
// asks ("how long does ONE healthy judge invocation take?"), so the invocations are kept apart.
|
|
494
|
+
// Separate from `invocations` above deliberately: that array is transcript evidence and records
|
|
495
|
+
// one entry per CAPTURED OUTPUT, so a dispatch that produced none contributes nothing to it.
|
|
496
|
+
const invocationSpans = [];
|
|
357
497
|
const invokeJudge = async (adapter, model, via) => {
|
|
358
|
-
const
|
|
359
|
-
|
|
360
|
-
|
|
498
|
+
const captured = await captureLlmDispatches([adapter], ([instrumented]) => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter: instrumented, model }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
|
|
499
|
+
// The instrumented adapter is reached only by runLlm. Deterministic oracles and diff-cap exits
|
|
500
|
+
// never call headlessCommand, so they produce no clock and cannot manufacture an invocation.
|
|
501
|
+
invocationSpans.push(...captured.invocations);
|
|
361
502
|
const unparseable = captured.value.meta?.unparseable === true;
|
|
362
503
|
// acceptanceGate has exactly one runLlm call. Keep the map shape so a future deterministic early
|
|
363
504
|
// return (zero outputs) stays telemetry-free instead of manufacturing a judge invocation.
|
|
364
|
-
for (const output of captured.outputs) {
|
|
505
|
+
for (const [index, output] of captured.outputs.entries()) {
|
|
506
|
+
const span = captured.invocations[index];
|
|
507
|
+
if (!span)
|
|
508
|
+
continue;
|
|
365
509
|
invocations.push({
|
|
366
510
|
taskId: task.id,
|
|
367
|
-
channel,
|
|
511
|
+
channel: span.channel,
|
|
368
512
|
outcome: unparseable ? "failed" : "done",
|
|
369
513
|
judgeOutcome: unparseable ? "unparseable" : "parseable",
|
|
370
|
-
durationMs:
|
|
514
|
+
durationMs: span.durationMs,
|
|
371
515
|
...(unparseable ? { transcript: output } : {}),
|
|
372
516
|
});
|
|
373
517
|
}
|
|
@@ -412,11 +556,26 @@ export async function runGates(task, ctx) {
|
|
|
412
556
|
a = await invokeJudge(retryAdapter, retry.model, retryJvia);
|
|
413
557
|
a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
|
|
414
558
|
}
|
|
415
|
-
|
|
559
|
+
// No dispatch, no key: a deterministic-oracle round writes no `invocations` field rather than an
|
|
560
|
+
// empty array a reader could mistake for "measured, and it cost nothing".
|
|
561
|
+
return { result: invocationSpans.length ? { ...a, meta: { ...a.meta, invocations: invocationSpans } } : a, invocations };
|
|
416
562
|
};
|
|
417
563
|
// cross-vendor review
|
|
418
564
|
const runReview = async () => {
|
|
419
|
-
|
|
565
|
+
// v2.0 T2: per-dispatch spans, exactly as the judge keeps them. A round that re-asks a second
|
|
566
|
+
// seat spends two invocations, and one blended span cannot tell a slow reviewer from two.
|
|
567
|
+
// A pick that found NO eligible seat dispatched nothing, so it contributes no invocation.
|
|
568
|
+
const invocations = [];
|
|
569
|
+
// Dispatch is PROVEN, never inferred: captureLlmOutput records one output per runLlm return, so an
|
|
570
|
+
// empty capture means reviewGate returned before asking anyone — a policy skip, a pre-dispatch diff
|
|
571
|
+
// cap, or no eligible seat. Reading `noEligibleReviewer` alone missed the first two and invented
|
|
572
|
+
// an "unknown" span for each.
|
|
573
|
+
const dispatch = async (run) => {
|
|
574
|
+
const captured = await captureLlmDispatches(ctx.adapters, run);
|
|
575
|
+
invocations.push(...captured.invocations);
|
|
576
|
+
return captured.value;
|
|
577
|
+
};
|
|
578
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir));
|
|
420
579
|
// OBS-193: an unparseable review verdict retries the REVIEW exactly once on a different reviewer —
|
|
421
580
|
// never the worker (GATE-09's judge-retry shape: straight-line single `if`, meta-only detection,
|
|
422
581
|
// the flaked verdict never enters results). The exclusion rides reviewGate's own excludeReviewers
|
|
@@ -427,13 +586,13 @@ export async function runGates(task, ctx) {
|
|
|
427
586
|
const retryVia = ctx.via
|
|
428
587
|
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + "-r1" }
|
|
429
588
|
: undefined;
|
|
430
|
-
const second = await reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels,
|
|
589
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir));
|
|
431
590
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
432
591
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
433
592
|
rv = { ...second, meta: { ...second.meta, reviewRetry: { flaked, retried } } };
|
|
434
593
|
}
|
|
435
594
|
}
|
|
436
|
-
return rv;
|
|
595
|
+
return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
|
|
437
596
|
};
|
|
438
597
|
// v1.87 T5: the refusal is the FIRST thing a round does, whatever that round is configured to run.
|
|
439
598
|
// Guarding only the configured build/test/lint commands left the hole this repairs: the battery is
|
|
@@ -442,8 +601,14 @@ export async function runGates(task, ctx) {
|
|
|
442
601
|
// oracles judged uncommitted state, on a commit whose diff nobody had run. One check at the top, and
|
|
443
602
|
// no shell-executing gate path is reachable on a dirty tree. It lands on the first gate of this
|
|
444
603
|
// round's sequence: the round dies there, exactly as it does on a red command.
|
|
604
|
+
// The check runs BEFORE any gate, so on a clean tree it belongs to no gate: charging every round's
|
|
605
|
+
// first gate for it would inflate the one measurement the parked recalibrations key on. It becomes
|
|
606
|
+
// that gate's interval only on the path where it IS what the gate did — the refusal below.
|
|
607
|
+
const entryAt = Date.now();
|
|
608
|
+
const entryLoad = loadProvider();
|
|
445
609
|
const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
|
|
446
610
|
if (entryDirt) {
|
|
611
|
+
spans.set(sequence[0], { durationMs: Date.now() - entryAt, load1Start: entryLoad, load1End: loadProvider() });
|
|
447
612
|
await emitStart(sequence[0]);
|
|
448
613
|
await record(dirtyRefusal(sequence[0], entryDirt));
|
|
449
614
|
return done();
|
|
@@ -481,8 +646,8 @@ export async function runGates(task, ctx) {
|
|
|
481
646
|
await emitStart("acceptance", parentAt);
|
|
482
647
|
if (enabled("review"))
|
|
483
648
|
await emitStart("review", parentAt);
|
|
484
|
-
const judging = enabled("acceptance") ? runAcceptance
|
|
485
|
-
const reviewing = enabled("review") ? runReview
|
|
649
|
+
const judging = enabled("acceptance") ? measure("acceptance", runAcceptance) : undefined;
|
|
650
|
+
const reviewing = enabled("review") ? measure("review", runReview) : undefined;
|
|
486
651
|
// Attach BOTH publication handlers before awaiting either. Dispatch concurrency alone is not
|
|
487
652
|
// enough: an acceptance-first await withholds a completed review behind a slow/hung judge and a
|
|
488
653
|
// process death can lose that already-earned verdict. The returned result is still sorted into
|
|
@@ -497,14 +662,14 @@ export async function runGates(task, ctx) {
|
|
|
497
662
|
// Legacy serial walk — frozen, and reachable only from the fixtures that pin it.
|
|
498
663
|
if (enabled("acceptance")) {
|
|
499
664
|
await emitStart("acceptance");
|
|
500
|
-
const judged = await runAcceptance
|
|
665
|
+
const judged = await measure("acceptance", runAcceptance);
|
|
501
666
|
await withJudgeInvocationEvidence(judged.invocations, () => record(judged.result));
|
|
502
667
|
if (failed())
|
|
503
668
|
return done();
|
|
504
669
|
}
|
|
505
670
|
if (enabled("review")) {
|
|
506
671
|
await emitStart("review");
|
|
507
|
-
await record(await runReview
|
|
672
|
+
await record(await measure("review", runReview));
|
|
508
673
|
}
|
|
509
674
|
return done();
|
|
510
675
|
}
|
|
@@ -518,11 +683,12 @@ export async function runGates(task, ctx) {
|
|
|
518
683
|
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
519
684
|
// may have run one before it, and every gate between the battery and here reads commits only, so
|
|
520
685
|
// a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
|
|
521
|
-
const [full] = await compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]);
|
|
686
|
+
const [full] = await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]));
|
|
687
|
+
fullDurationMs = spans.get("test").durationMs - (selectedDurationMs ?? 0);
|
|
522
688
|
const dirt = full.pass ? await dirtyWorktree() : undefined;
|
|
523
|
-
const merged = dirt
|
|
689
|
+
const merged = withTelemetry(dirt
|
|
524
690
|
? dirtyRefusal("test", dirt, ctx.commands.test)
|
|
525
|
-
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } };
|
|
691
|
+
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
|
|
526
692
|
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
527
693
|
heldTest = undefined;
|
|
528
694
|
await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
|
package/dist/gates/scope.d.ts
CHANGED
|
@@ -6,4 +6,12 @@ export declare function dispositionOffenders(offenders: string[], allowDeviation
|
|
|
6
6
|
hard: string[];
|
|
7
7
|
allowed: string[];
|
|
8
8
|
};
|
|
9
|
-
|
|
9
|
+
/**
|
|
10
|
+
* OBS-547: `collateral` is this task's slice of the run's ONE full prediction map (computed at run
|
|
11
|
+
* start, uncapped — src/compile/collateral.ts). Absent ⇒ no classification, today's behaviour; the
|
|
12
|
+
* gate never computes a map of its own, so what it classifies on is always what the run predicted.
|
|
13
|
+
*/
|
|
14
|
+
export declare function scopeGate(worktree: string, integrationTip: string, files: string[], result: WorkerResult, allowDeviations?: string[], collateral?: {
|
|
15
|
+
taskId: string;
|
|
16
|
+
predicted: ReadonlyArray<string>;
|
|
17
|
+
}): Promise<GateResult>;
|
package/dist/gates/scope.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { classifyScopeOffenders } from "../compile/collateral.js";
|
|
1
2
|
import { filesGlob } from "../graph/files-glob.js";
|
|
2
3
|
import { shGitOk } from "../run/git.js";
|
|
3
4
|
// OBS-61: the live integration tip can advance between attempts (sibling merge) while a resumed
|
|
@@ -22,7 +23,12 @@ export function dispositionOffenders(offenders, allowDeviations) {
|
|
|
22
23
|
}
|
|
23
24
|
return { hard, allowed };
|
|
24
25
|
}
|
|
25
|
-
|
|
26
|
+
/**
|
|
27
|
+
* OBS-547: `collateral` is this task's slice of the run's ONE full prediction map (computed at run
|
|
28
|
+
* start, uncapped — src/compile/collateral.ts). Absent ⇒ no classification, today's behaviour; the
|
|
29
|
+
* gate never computes a map of its own, so what it classifies on is always what the run predicted.
|
|
30
|
+
*/
|
|
31
|
+
export async function scopeGate(worktree, integrationTip, files, result, allowDeviations = [], collateral) {
|
|
26
32
|
if (!files.length)
|
|
27
33
|
return { gate: "scope", pass: true, details: "no file scope declared — unrestricted" };
|
|
28
34
|
const baseRef = await scopeDiffBase(worktree, integrationTip);
|
|
@@ -38,5 +44,19 @@ export async function scopeGate(worktree, integrationTip, files, result, allowDe
|
|
|
38
44
|
if (!hard.length) {
|
|
39
45
|
return { gate: "scope", pass: true, details: `out-of-scope but operator-allowlisted:\n${allowed.join("\n")}${note}` };
|
|
40
46
|
}
|
|
41
|
-
|
|
47
|
+
// OBS-547: cross-reference at the red — the red supplies the paths, the prediction the classification.
|
|
48
|
+
const verdict = collateral
|
|
49
|
+
? classifyScopeOffenders(collateral.taskId, hard, collateral.predicted)
|
|
50
|
+
: undefined;
|
|
51
|
+
const classification = verdict?.authoring
|
|
52
|
+
? `\nauthoring defect (OBS-547): the collateral lint named every one of these paths before dispatch. Repair:\n${verdict.repair}`
|
|
53
|
+
: verdict && verdict.missed.length
|
|
54
|
+
? `\ncollateral lint did not predict: ${verdict.missed.join(", ")}`
|
|
55
|
+
: "";
|
|
56
|
+
return {
|
|
57
|
+
gate: "scope",
|
|
58
|
+
pass: false,
|
|
59
|
+
details: `out-of-scope edits not covered by scope.allowDeviations:\n${hard.join("\n")}${note}${classification}`,
|
|
60
|
+
...(verdict ? { meta: { collateral: verdict } } : {}),
|
|
61
|
+
};
|
|
42
62
|
}
|
package/dist/graph/graph.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { type RunGraph, type Task, type TaskStatus } from "./schema.js";
|
|
|
2
2
|
export declare function stateDirName(_repoRoot: string): string;
|
|
3
3
|
export declare function graphPath(repoRoot: string): string;
|
|
4
4
|
export declare function graphDefinitionHash(g: RunGraph): string;
|
|
5
|
+
export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
|
|
5
6
|
export declare function tickmarkrDir(repoRoot: string): string;
|
|
6
7
|
export declare function loadGraph(repoRoot: string): RunGraph;
|
|
7
8
|
export declare function saveGraph(repoRoot: string, g: RunGraph): void;
|
package/dist/graph/graph.js
CHANGED
|
@@ -19,6 +19,16 @@ export function graphDefinitionHash(g) {
|
|
|
19
19
|
const definitions = g.tasks.map(({ status: _status, evidence: _evidence, ...def }) => def);
|
|
20
20
|
return createHash("sha256").update(JSON.stringify({ version: g.version, spec: g.spec, tasks: definitions })).digest("hex").slice(0, 16);
|
|
21
21
|
}
|
|
22
|
+
// OBS-543: cross-run evidence belongs to the artifact one task describes, not to the whole compiled
|
|
23
|
+
// graph. A sibling task, dependency, routing hint or status change therefore cannot expire a useful
|
|
24
|
+
// finding; changing the goal, write surface or acceptance contract does. Keep the full digest here:
|
|
25
|
+
// unlike graphDefinitionHash this value is persisted beside evidence and is the fail-closed join a
|
|
26
|
+
// later run uses, so there is no benefit in making collision diagnosis less explicit.
|
|
27
|
+
export function taskContentDigest(task) {
|
|
28
|
+
return createHash("sha256")
|
|
29
|
+
.update(JSON.stringify({ goal: task.goal, files: task.files, acceptance: task.acceptance }))
|
|
30
|
+
.digest("hex");
|
|
31
|
+
}
|
|
22
32
|
export function tickmarkrDir(repoRoot) {
|
|
23
33
|
const dir = join(repoRoot, stateDirName(repoRoot));
|
|
24
34
|
mkdirSync(dir, { recursive: true });
|
|
@@ -59,7 +69,14 @@ export function setStatus(g, id, status) {
|
|
|
59
69
|
return { ...g, tasks: g.tasks.map((t) => (t.id === id ? { ...t, status } : t)) };
|
|
60
70
|
}
|
|
61
71
|
export function addEvidence(g, id, patch) {
|
|
62
|
-
getTask(g, id);
|
|
72
|
+
const subject = getTask(g, id);
|
|
73
|
+
const digest = taskContentDigest(subject);
|
|
74
|
+
// This is the graph-evidence boundary where a gate result meets the task it measured. Stamp a
|
|
75
|
+
// copy, never mutate the gate result runGates returned: callers still use that live object for
|
|
76
|
+
// predicates, while durable evidence gains the content identity a later run can compare.
|
|
77
|
+
const gateResults = (patch.gateResults ?? []).map((result) => result !== null && typeof result === "object" && !Array.isArray(result)
|
|
78
|
+
? { ...result, taskContentDigest: digest }
|
|
79
|
+
: result);
|
|
63
80
|
return {
|
|
64
81
|
...g,
|
|
65
82
|
tasks: g.tasks.map((t) => t.id === id
|
|
@@ -68,7 +85,7 @@ export function addEvidence(g, id, patch) {
|
|
|
68
85
|
evidence: {
|
|
69
86
|
commits: [...t.evidence.commits, ...(patch.commits ?? [])],
|
|
70
87
|
artifacts: [...t.evidence.artifacts, ...(patch.artifacts ?? [])],
|
|
71
|
-
gateResults: [...t.evidence.gateResults, ...
|
|
88
|
+
gateResults: [...t.evidence.gateResults, ...gateResults],
|
|
72
89
|
},
|
|
73
90
|
}
|
|
74
91
|
: t),
|
package/dist/report/compare.js
CHANGED
|
@@ -21,7 +21,21 @@ export function recordedEnvironment(events) {
|
|
|
21
21
|
return undefined;
|
|
22
22
|
adapterVersions[k] = v;
|
|
23
23
|
}
|
|
24
|
-
|
|
24
|
+
// Capacity arrived in v2.0 as a pair. Two absent fields are an honest older record; one without
|
|
25
|
+
// the other is a malformed new record and fails closed instead of being compared as legacy.
|
|
26
|
+
const hasCores = Object.hasOwn(o, "cores");
|
|
27
|
+
const hasForkCap = Object.hasOwn(o, "forkCap");
|
|
28
|
+
if (hasCores !== hasForkCap)
|
|
29
|
+
return undefined;
|
|
30
|
+
if (hasCores && (typeof o.cores !== "number" || !Number.isFinite(o.cores) || o.cores <= 0
|
|
31
|
+
|| typeof o.forkCap !== "number" || !Number.isFinite(o.forkCap) || o.forkCap <= 0))
|
|
32
|
+
return undefined;
|
|
33
|
+
return {
|
|
34
|
+
tickmarkrVersion: o.tickmarkrVersion,
|
|
35
|
+
configHash: o.configHash,
|
|
36
|
+
adapterVersions,
|
|
37
|
+
...(hasCores ? { cores: o.cores, forkCap: o.forkCap } : {}),
|
|
38
|
+
};
|
|
25
39
|
}
|
|
26
40
|
return undefined;
|
|
27
41
|
}
|
|
@@ -31,7 +45,8 @@ function envFingerprint(env) {
|
|
|
31
45
|
return env.configHash;
|
|
32
46
|
}
|
|
33
47
|
function envEqual(a, b) {
|
|
34
|
-
if (a.tickmarkrVersion !== b.tickmarkrVersion || a.configHash !== b.configHash
|
|
48
|
+
if (a.tickmarkrVersion !== b.tickmarkrVersion || a.configHash !== b.configHash
|
|
49
|
+
|| a.cores !== b.cores || a.forkCap !== b.forkCap)
|
|
35
50
|
return false;
|
|
36
51
|
const ak = Object.keys(a.adapterVersions).sort();
|
|
37
52
|
const bk = Object.keys(b.adapterVersions).sort();
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -4,6 +4,7 @@ import { type ExecutorDriver } from "../drivers/types.js";
|
|
|
4
4
|
import { type Baseline } from "../gates/baseline.js";
|
|
5
5
|
import type { GateResult } from "../gates/types.js";
|
|
6
6
|
import { Journal, type JournalEvent } from "./journal.js";
|
|
7
|
+
export { harvestCpuFlatWindowMs, resetHarvestCpuFlatMsForTests, setHarvestCpuFlatMsForTests, workerTreeCpuMs, } from "./stall.js";
|
|
7
8
|
export interface RunOptions {
|
|
8
9
|
runId?: string;
|
|
9
10
|
resume?: boolean;
|
|
@@ -117,10 +118,6 @@ export declare function resetDeadChannelFastKillMsForTests(): void;
|
|
|
117
118
|
/** Test seam — shrink the harvest silence gate without minute-long sleeps. */
|
|
118
119
|
export declare function setHarvestSilentMsForTests(ms: number): void;
|
|
119
120
|
export declare function resetHarvestSilentMsForTests(): void;
|
|
120
|
-
export declare function harvestCpuFlatWindowMs(resolutionMs: number): number;
|
|
121
|
-
/** Test seam — pin the flat window so a probe case need not sit through a real one. */
|
|
122
|
-
export declare function setHarvestCpuFlatMsForTests(ms: number): void;
|
|
123
|
-
export declare function resetHarvestCpuFlatMsForTests(): void;
|
|
124
121
|
export declare const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries committed work; the worker emitted no TICKMARKR_RESULT trailer";
|
|
125
122
|
/** T4 (OBS-266): identity of the command SET a tip verify ran — a changed command is a different verify. */
|
|
126
123
|
export declare function commandsHash(commands: Record<string, string>): string;
|
|
@@ -139,10 +136,6 @@ export declare function verifyIntegrationTipCached(intWt: string, commands: Reco
|
|
|
139
136
|
lastMergedTask?: string;
|
|
140
137
|
baseline?: Baseline;
|
|
141
138
|
}): Promise<boolean>;
|
|
142
|
-
export declare function workerTreeCpuMs(marker: string, cwd: string): Promise<{
|
|
143
|
-
ms: number;
|
|
144
|
-
resolutionMs: number;
|
|
145
|
-
} | undefined>;
|
|
146
139
|
/** Test seam — exercise the production observer's total read bound with a small real tree. */
|
|
147
140
|
export declare function setObserveBudgetBytesForTests(bytes: number): void;
|
|
148
141
|
export declare function resetObserveBudgetBytesForTests(): void;
|