tickmarkr 1.97.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,6 @@
1
- import { readFileSync } from "node:fs";
2
- import { posix } from "node:path";
1
+ import { mkdtempSync, readFileSync, rmSync } from "node:fs";
2
+ import { loadavg, tmpdir } from "node:os";
3
+ import { join, posix } from "node:path";
3
4
  import { channelKey, shq } from "../adapters/types.js";
4
5
  import { TIER_RANK } from "../config/config.js";
5
6
  import { getAdapter } from "../adapters/registry.js";
@@ -14,6 +15,85 @@ import { reviewGate } from "./review.js";
14
15
  import { scopeGate } from "./scope.js";
15
16
  import { shGit } from "../run/git.js";
16
17
  import { withJudgeInvocationEvidence } from "../run/journal.js";
18
+ const productionLoadProvider = () => loadavg()[0] ?? 0;
19
+ let loadProvider = productionLoadProvider;
20
+ /** Test seam — inject deterministic load samples; production always reads os.loadavg. */
21
+ export function setLoadProviderForTests(provider) {
22
+ loadProvider = provider;
23
+ }
24
+ export function resetLoadProviderForTests() {
25
+ loadProvider = productionLoadProvider;
26
+ }
27
+ /**
28
+ * Instrument the adapter command itself, which is the first adapter-owned operation runLlm performs.
29
+ * acceptanceGate/reviewGate deliberately retain ownership of deterministic oracles, policy checks,
30
+ * diff reads and prompt construction; none of that preprocessing belongs to an LLM invocation span.
31
+ *
32
+ * Both stamps are written by the same shell immediately around the adapter command. That excludes
33
+ * pane-slot acquisition as well as runLlm's scratch cleanup and verdict parsing, without changing
34
+ * llm.ts's output contract. The subshell keeps an adapter command's `exit` from bypassing the end
35
+ * stamp, and the original exit status is preserved.
36
+ */
37
+ function instrumentLlmAdapter(adapter, clocks) {
38
+ return new Proxy(adapter, {
39
+ get(target, property) {
40
+ if (property === "headlessCommand") {
41
+ return (promptFile, model) => {
42
+ const command = target.headlessCommand(promptFile, model);
43
+ const dir = mkdtempSync(join(tmpdir(), "tickmarkr-gate-invocation-"));
44
+ const startedAtPath = join(dir, "started-at");
45
+ const completedAtPath = join(dir, "completed-at");
46
+ clocks.push({
47
+ channel: channelKey({ adapter: target.id, model }),
48
+ preparedAt: Date.now(),
49
+ startedAtPath,
50
+ completedAtPath,
51
+ dir,
52
+ });
53
+ const stamp = (path) => `${shq(process.execPath)} -e ${shq('require("node:fs").writeFileSync(process.argv[1], String(Date.now()))')} ${shq(path)}`;
54
+ // End with a status-bearing subshell, not `exit`: pane mode appends its nonce-bound
55
+ // completion trailer on the next script line and must remain able to run it.
56
+ return `${stamp(startedAtPath)}; ( ${command} ); __tickmarkr_invocation_status=$?; ${stamp(completedAtPath)}; (exit $__tickmarkr_invocation_status)`;
57
+ };
58
+ }
59
+ const value = Reflect.get(target, property, target);
60
+ // Real adapters may use private fields; bind their methods to the target rather than the Proxy.
61
+ return typeof value === "function" ? value.bind(target) : value;
62
+ },
63
+ });
64
+ }
65
+ function finishLlmDispatches(clocks) {
66
+ return clocks.map((clock) => {
67
+ let startedAt = clock.preparedAt;
68
+ let completedAt = Date.now();
69
+ try {
70
+ const stampedStart = Number(readFileSync(clock.startedAtPath, "utf8"));
71
+ const stamped = Number(readFileSync(clock.completedAtPath, "utf8"));
72
+ if (Number.isFinite(stampedStart))
73
+ startedAt = stampedStart;
74
+ if (Number.isFinite(stamped) && stamped >= startedAt)
75
+ completedAt = stamped;
76
+ }
77
+ catch {
78
+ // A killed command may never reach its stamp; the gate return is the honest upper boundary.
79
+ }
80
+ finally {
81
+ rmSync(clock.dir, { recursive: true, force: true });
82
+ }
83
+ return { channel: clock.channel, durationMs: completedAt - startedAt };
84
+ });
85
+ }
86
+ async function captureLlmDispatches(adapters, run) {
87
+ const clocks = [];
88
+ try {
89
+ const captured = await captureLlmOutput(() => run(adapters.map((a) => instrumentLlmAdapter(a, clocks))));
90
+ return { ...captured, invocations: finishLlmDispatches(clocks) };
91
+ }
92
+ catch (error) {
93
+ finishLlmDispatches(clocks);
94
+ throw error;
95
+ }
96
+ }
17
97
  const TEST_FILE_RE = /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/;
18
98
  // relative specifiers only — `from "./x.js"`, `import("./x.js")`, `require("./x.js")`
19
99
  const IMPORT_RE = /(?:\bfrom\s*|\bimport\s*\(\s*|\brequire\s*\(\s*)["'](\.[^"']*)["']/g;
@@ -124,12 +204,56 @@ export async function runGates(task, ctx) {
124
204
  // exactly one `test` gate-result ever leaves a round, always carrying which suite spoke for it.
125
205
  // (A RED screen IS the verdict: the round ends there, so it is recorded immediately.)
126
206
  let heldTest;
207
+ // v2.0 T2 (OBS-554): this round's per-gate measurement. Every interval a gate actually spends
208
+ // executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
209
+ // gate's screen and its full suite) sums to its own cost and never to the span between them.
210
+ const spans = new Map();
211
+ // The test gate's two halves, kept apart as well as summed: `durationMs` alone cannot say whether
212
+ // a slow round was a slow subset or a slow full suite, and the parked scheduler's threshold is
213
+ // defined over the full-suite cost.
214
+ let selectedDurationMs;
215
+ let fullDurationMs;
216
+ const measure = async (gate, run) => {
217
+ const at = Date.now();
218
+ const load1Start = loadProvider();
219
+ try {
220
+ return await run();
221
+ }
222
+ finally {
223
+ const prior = spans.get(gate);
224
+ spans.set(gate, {
225
+ durationMs: (prior?.durationMs ?? 0) + (Date.now() - at),
226
+ load1Start: prior?.load1Start ?? load1Start,
227
+ load1End: loadProvider(),
228
+ });
229
+ }
230
+ };
231
+ // The measurement is attached at the ONE seam every result leaves this function through, so a
232
+ // path that forgets to measure is visibly missing its telemetry rather than carrying a fabricated
233
+ // zero. The daemon lifts these off `meta` onto the gate-result row (src/run/daemon.ts).
234
+ const withTelemetry = (result) => {
235
+ const span = spans.get(result.gate);
236
+ if (!span)
237
+ return result;
238
+ return {
239
+ ...result,
240
+ meta: {
241
+ ...result.meta,
242
+ ...span,
243
+ ...(result.gate === "test" && selectedDurationMs !== undefined ? { selectedDurationMs } : {}),
244
+ ...(result.gate === "test" && fullDurationMs !== undefined ? { fullDurationMs } : {}),
245
+ },
246
+ };
247
+ };
127
248
  const sequence = GATE_NAMES.filter((g) => enabled(g));
128
249
  const total = sequence.length;
129
250
  const indexOf = (gate) => sequence.indexOf(gate) + 1;
251
+ // The stamp lands here, on the ONE object that is both pushed and published, so the round's record
252
+ // and its event stream carry byte-identical results — an invariant the fixtures pin.
130
253
  const record = async (result) => {
131
- results.push(result);
132
- await ctx.onGate?.({ phase: "end", gate: result.gate, result });
254
+ const stamped = withTelemetry(result);
255
+ results.push(stamped);
256
+ await ctx.onGate?.({ phase: "end", gate: stamped.gate, result: stamped });
133
257
  };
134
258
  const emitStart = async (gate, parentAt) => {
135
259
  await ctx.onGate?.({ phase: "start", gate, index: indexOf(gate), total, ...(parentAt === undefined ? {} : { parentAt }) });
@@ -141,7 +265,7 @@ export async function runGates(task, ctx) {
141
265
  if (heldTest) {
142
266
  const held = heldTest;
143
267
  heldTest = undefined;
144
- await ctx.onGate?.({ phase: "end", gate: "test", result: held });
268
+ await ctx.onGate?.({ phase: "end", gate: "test", result: held }); // stamped when it was held
145
269
  }
146
270
  const sorted = [...results].sort((a, b) => GATE_NAMES.indexOf(a.gate) - GATE_NAMES.indexOf(b.gate));
147
271
  // v1.87 T5: no round returns a MERGEABLE GREEN on a dirty tree. The battery is not the only gate
@@ -155,7 +279,7 @@ export async function runGates(task, ctx) {
155
279
  if (last && sorted.every((r) => r.pass || r.meta?.skipped === true)) {
156
280
  const dirt = await dirtyWorktree();
157
281
  if (dirt) {
158
- const refusal = dirtyRoundRefusal(last.gate, dirt);
282
+ const refusal = withTelemetry(dirtyRoundRefusal(last.gate, dirt));
159
283
  results[results.indexOf(last)] = refusal;
160
284
  sorted[sorted.length - 1] = refusal;
161
285
  await ctx.onGate?.({ phase: "end", gate: refusal.gate, result: refusal });
@@ -221,7 +345,14 @@ export async function runGates(task, ctx) {
221
345
  // ponytail: compareToBaseline batches build/test/lint — their starts are emitted at iteration,
222
346
  // not at true execution start. They are collectively sub-second (measured), so the debounce
223
347
  // suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
348
+ // ponytail: legacy runs build/test/lint in ONE compareToBaseline call, so there is one interval
349
+ // to measure and each of its gates carries it. Split it only if this branch ever stops batching.
350
+ const batchAt = Date.now();
351
+ const batchLoadStart = loadProvider();
224
352
  const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
353
+ const batch = { durationMs: Date.now() - batchAt, load1Start: batchLoadStart, load1End: loadProvider() };
354
+ for (const g of toolGates)
355
+ spans.set(g, batch);
225
356
  // The same refusal AFTER the commands, because a green command can dirty the tree the check
226
357
  // above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
227
358
  // lands on the last gate that had one — the round dies there either way. A red battery is
@@ -238,7 +369,10 @@ export async function runGates(task, ctx) {
238
369
  // the full vitest suite before anyone reads its verdict.
239
370
  for (const g of toolGates) {
240
371
  await emitStart(g);
241
- const [r] = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]);
372
+ const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
373
+ // the screen's interval IS the test gate's first interval, so the split needs no second clock
374
+ if (g === "test" && selected)
375
+ selectedDurationMs = spans.get("test").durationMs;
242
376
  // The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
243
377
  // tracked file makes it dirty again, and every gate after it — including the next shell gate,
244
378
  // which would then run against bytes HEAD does not hold — inherits that. So re-check after each
@@ -257,8 +391,8 @@ export async function runGates(task, ctx) {
257
391
  if (!screened.pass)
258
392
  await record(screened);
259
393
  else {
260
- heldTest = screened;
261
- results.push(screened);
394
+ heldTest = withTelemetry(screened);
395
+ results.push(heldTest);
262
396
  }
263
397
  }
264
398
  else {
@@ -292,10 +426,10 @@ export async function runGates(task, ctx) {
292
426
  */
293
427
  const allowDeviations = [...(ctx.cfg.scope?.allowDeviations ?? [])];
294
428
  Object.freeze(allowDeviations);
295
- const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, allowDeviations);
429
+ const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, allowDeviations, ctx.collateral ? { taskId: task.id, predicted: ctx.collateral } : undefined);
296
430
  const runGate = async (gate, compute) => {
297
431
  await emitStart(gate);
298
- await record(await compute());
432
+ await record(await measure(gate, compute));
299
433
  };
300
434
  /**
301
435
  * T4 (OBS-265): the deterministic git checks run BEFORE the battery, as a screen — they answer
@@ -319,7 +453,7 @@ export async function runGates(task, ctx) {
319
453
  for (const [gate, compute] of [["evidence", evidenceResult], ["scope", scopeResult]]) {
320
454
  if (!enabled(gate))
321
455
  continue;
322
- screened.push(await compute());
456
+ screened.push(await measure(gate, compute));
323
457
  if (screened[screened.length - 1].pass)
324
458
  continue;
325
459
  for (const r of screened) {
@@ -354,20 +488,30 @@ export async function runGates(task, ctx) {
354
488
  // v1.19 (T2): testCmd threads the detected test runner to the gate so named-test oracles run
355
489
  // deterministically (filtered via -t) before any LLM judge dispatch.
356
490
  const invocations = [];
491
+ // v2.0 T2 (OBS-554): one entry per JUDGE DISPATCH — primary and the GATE-09 retry alike. The gate's
492
+ // own durationMs is the pair's envelope and cannot answer what the parked ceiling recalibration
493
+ // asks ("how long does ONE healthy judge invocation take?"), so the invocations are kept apart.
494
+ // Separate from `invocations` above deliberately: that array is transcript evidence and records
495
+ // one entry per CAPTURED OUTPUT, so a dispatch that produced none contributes nothing to it.
496
+ const invocationSpans = [];
357
497
  const invokeJudge = async (adapter, model, via) => {
358
- const started = Date.now();
359
- const captured = await captureLlmOutput(() => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter, model }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
360
- const channel = channelKey({ adapter: adapter.id, model });
498
+ const captured = await captureLlmDispatches([adapter], ([instrumented]) => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter: instrumented, model }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
499
+ // The instrumented adapter is reached only by runLlm. Deterministic oracles and diff-cap exits
500
+ // never call headlessCommand, so they produce no clock and cannot manufacture an invocation.
501
+ invocationSpans.push(...captured.invocations);
361
502
  const unparseable = captured.value.meta?.unparseable === true;
362
503
  // acceptanceGate has exactly one runLlm call. Keep the map shape so a future deterministic early
363
504
  // return (zero outputs) stays telemetry-free instead of manufacturing a judge invocation.
364
- for (const output of captured.outputs) {
505
+ for (const [index, output] of captured.outputs.entries()) {
506
+ const span = captured.invocations[index];
507
+ if (!span)
508
+ continue;
365
509
  invocations.push({
366
510
  taskId: task.id,
367
- channel,
511
+ channel: span.channel,
368
512
  outcome: unparseable ? "failed" : "done",
369
513
  judgeOutcome: unparseable ? "unparseable" : "parseable",
370
- durationMs: Date.now() - started,
514
+ durationMs: span.durationMs,
371
515
  ...(unparseable ? { transcript: output } : {}),
372
516
  });
373
517
  }
@@ -412,11 +556,26 @@ export async function runGates(task, ctx) {
412
556
  a = await invokeJudge(retryAdapter, retry.model, retryJvia);
413
557
  a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
414
558
  }
415
- return { result: a, invocations };
559
+ // No dispatch, no key: a deterministic-oracle round writes no `invocations` field rather than an
560
+ // empty array a reader could mistake for "measured, and it cost nothing".
561
+ return { result: invocationSpans.length ? { ...a, meta: { ...a.meta, invocations: invocationSpans } } : a, invocations };
416
562
  };
417
563
  // cross-vendor review
418
564
  const runReview = async () => {
419
- let rv = await reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, ctx.adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir);
565
+ // v2.0 T2: per-dispatch spans, exactly as the judge keeps them. A round that re-asks a second
566
+ // seat spends two invocations, and one blended span cannot tell a slow reviewer from two.
567
+ // A pick that found NO eligible seat dispatched nothing, so it contributes no invocation.
568
+ const invocations = [];
569
+ // Dispatch is PROVEN, never inferred: captureLlmOutput records one output per runLlm return, so an
570
+ // empty capture means reviewGate returned before asking anyone — a policy skip, a pre-dispatch diff
571
+ // cap, or no eligible seat. Reading `noEligibleReviewer` alone missed the first two and invented
572
+ // an "unknown" span for each.
573
+ const dispatch = async (run) => {
574
+ const captured = await captureLlmDispatches(ctx.adapters, run);
575
+ invocations.push(...captured.invocations);
576
+ return captured.value;
577
+ };
578
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir));
420
579
  // OBS-193: an unparseable review verdict retries the REVIEW exactly once on a different reviewer —
421
580
  // never the worker (GATE-09's judge-retry shape: straight-line single `if`, meta-only detection,
422
581
  // the flaked verdict never enters results). The exclusion rides reviewGate's own excludeReviewers
@@ -427,13 +586,13 @@ export async function runGates(task, ctx) {
427
586
  const retryVia = ctx.via
428
587
  ? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + "-r1" }
429
588
  : undefined;
430
- const second = await reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, ctx.adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir);
589
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir));
431
590
  if (second.meta?.noEligibleReviewer !== true) {
432
591
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
433
592
  rv = { ...second, meta: { ...second.meta, reviewRetry: { flaked, retried } } };
434
593
  }
435
594
  }
436
- return rv;
595
+ return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
437
596
  };
438
597
  // v1.87 T5: the refusal is the FIRST thing a round does, whatever that round is configured to run.
439
598
  // Guarding only the configured build/test/lint commands left the hole this repairs: the battery is
@@ -442,8 +601,14 @@ export async function runGates(task, ctx) {
442
601
  // oracles judged uncommitted state, on a commit whose diff nobody had run. One check at the top, and
443
602
  // no shell-executing gate path is reachable on a dirty tree. It lands on the first gate of this
444
603
  // round's sequence: the round dies there, exactly as it does on a red command.
604
+ // The check runs BEFORE any gate, so on a clean tree it belongs to no gate: charging every round's
605
+ // first gate for it would inflate the one measurement the parked recalibrations key on. It becomes
606
+ // that gate's interval only on the path where it IS what the gate did — the refusal below.
607
+ const entryAt = Date.now();
608
+ const entryLoad = loadProvider();
445
609
  const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
446
610
  if (entryDirt) {
611
+ spans.set(sequence[0], { durationMs: Date.now() - entryAt, load1Start: entryLoad, load1End: loadProvider() });
447
612
  await emitStart(sequence[0]);
448
613
  await record(dirtyRefusal(sequence[0], entryDirt));
449
614
  return done();
@@ -481,8 +646,8 @@ export async function runGates(task, ctx) {
481
646
  await emitStart("acceptance", parentAt);
482
647
  if (enabled("review"))
483
648
  await emitStart("review", parentAt);
484
- const judging = enabled("acceptance") ? runAcceptance() : undefined;
485
- const reviewing = enabled("review") ? runReview() : undefined;
649
+ const judging = enabled("acceptance") ? measure("acceptance", runAcceptance) : undefined;
650
+ const reviewing = enabled("review") ? measure("review", runReview) : undefined;
486
651
  // Attach BOTH publication handlers before awaiting either. Dispatch concurrency alone is not
487
652
  // enough: an acceptance-first await withholds a completed review behind a slow/hung judge and a
488
653
  // process death can lose that already-earned verdict. The returned result is still sorted into
@@ -497,14 +662,14 @@ export async function runGates(task, ctx) {
497
662
  // Legacy serial walk — frozen, and reachable only from the fixtures that pin it.
498
663
  if (enabled("acceptance")) {
499
664
  await emitStart("acceptance");
500
- const judged = await runAcceptance();
665
+ const judged = await measure("acceptance", runAcceptance);
501
666
  await withJudgeInvocationEvidence(judged.invocations, () => record(judged.result));
502
667
  if (failed())
503
668
  return done();
504
669
  }
505
670
  if (enabled("review")) {
506
671
  await emitStart("review");
507
- await record(await runReview());
672
+ await record(await measure("review", runReview));
508
673
  }
509
674
  return done();
510
675
  }
@@ -518,11 +683,12 @@ export async function runGates(task, ctx) {
518
683
  // This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
519
684
  // may have run one before it, and every gate between the battery and here reads commits only, so
520
685
  // a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
521
- const [full] = await compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]);
686
+ const [full] = await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]));
687
+ fullDurationMs = spans.get("test").durationMs - (selectedDurationMs ?? 0);
522
688
  const dirt = full.pass ? await dirtyWorktree() : undefined;
523
- const merged = dirt
689
+ const merged = withTelemetry(dirt
524
690
  ? dirtyRefusal("test", dirt, ctx.commands.test)
525
- : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } };
691
+ : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
526
692
  results[results.findIndex((r) => r.gate === "test")] = merged;
527
693
  heldTest = undefined;
528
694
  await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
@@ -6,4 +6,12 @@ export declare function dispositionOffenders(offenders: string[], allowDeviation
6
6
  hard: string[];
7
7
  allowed: string[];
8
8
  };
9
- export declare function scopeGate(worktree: string, integrationTip: string, files: string[], result: WorkerResult, allowDeviations?: string[]): Promise<GateResult>;
9
+ /**
10
+ * OBS-547: `collateral` is this task's slice of the run's ONE full prediction map (computed at run
11
+ * start, uncapped — src/compile/collateral.ts). Absent ⇒ no classification, today's behaviour; the
12
+ * gate never computes a map of its own, so what it classifies on is always what the run predicted.
13
+ */
14
+ export declare function scopeGate(worktree: string, integrationTip: string, files: string[], result: WorkerResult, allowDeviations?: string[], collateral?: {
15
+ taskId: string;
16
+ predicted: ReadonlyArray<string>;
17
+ }): Promise<GateResult>;
@@ -1,3 +1,4 @@
1
+ import { classifyScopeOffenders } from "../compile/collateral.js";
1
2
  import { filesGlob } from "../graph/files-glob.js";
2
3
  import { shGitOk } from "../run/git.js";
3
4
  // OBS-61: the live integration tip can advance between attempts (sibling merge) while a resumed
@@ -22,7 +23,12 @@ export function dispositionOffenders(offenders, allowDeviations) {
22
23
  }
23
24
  return { hard, allowed };
24
25
  }
25
- export async function scopeGate(worktree, integrationTip, files, result, allowDeviations = []) {
26
+ /**
27
+ * OBS-547: `collateral` is this task's slice of the run's ONE full prediction map (computed at run
28
+ * start, uncapped — src/compile/collateral.ts). Absent ⇒ no classification, today's behaviour; the
29
+ * gate never computes a map of its own, so what it classifies on is always what the run predicted.
30
+ */
31
+ export async function scopeGate(worktree, integrationTip, files, result, allowDeviations = [], collateral) {
26
32
  if (!files.length)
27
33
  return { gate: "scope", pass: true, details: "no file scope declared — unrestricted" };
28
34
  const baseRef = await scopeDiffBase(worktree, integrationTip);
@@ -38,5 +44,19 @@ export async function scopeGate(worktree, integrationTip, files, result, allowDe
38
44
  if (!hard.length) {
39
45
  return { gate: "scope", pass: true, details: `out-of-scope but operator-allowlisted:\n${allowed.join("\n")}${note}` };
40
46
  }
41
- return { gate: "scope", pass: false, details: `out-of-scope edits not covered by scope.allowDeviations:\n${hard.join("\n")}${note}` };
47
+ // OBS-547: cross-reference at the red — the red supplies the paths, the prediction the classification.
48
+ const verdict = collateral
49
+ ? classifyScopeOffenders(collateral.taskId, hard, collateral.predicted)
50
+ : undefined;
51
+ const classification = verdict?.authoring
52
+ ? `\nauthoring defect (OBS-547): the collateral lint named every one of these paths before dispatch. Repair:\n${verdict.repair}`
53
+ : verdict && verdict.missed.length
54
+ ? `\ncollateral lint did not predict: ${verdict.missed.join(", ")}`
55
+ : "";
56
+ return {
57
+ gate: "scope",
58
+ pass: false,
59
+ details: `out-of-scope edits not covered by scope.allowDeviations:\n${hard.join("\n")}${note}${classification}`,
60
+ ...(verdict ? { meta: { collateral: verdict } } : {}),
61
+ };
42
62
  }
@@ -2,6 +2,7 @@ import { type RunGraph, type Task, type TaskStatus } from "./schema.js";
2
2
  export declare function stateDirName(_repoRoot: string): string;
3
3
  export declare function graphPath(repoRoot: string): string;
4
4
  export declare function graphDefinitionHash(g: RunGraph): string;
5
+ export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
5
6
  export declare function tickmarkrDir(repoRoot: string): string;
6
7
  export declare function loadGraph(repoRoot: string): RunGraph;
7
8
  export declare function saveGraph(repoRoot: string, g: RunGraph): void;
@@ -19,6 +19,16 @@ export function graphDefinitionHash(g) {
19
19
  const definitions = g.tasks.map(({ status: _status, evidence: _evidence, ...def }) => def);
20
20
  return createHash("sha256").update(JSON.stringify({ version: g.version, spec: g.spec, tasks: definitions })).digest("hex").slice(0, 16);
21
21
  }
22
+ // OBS-543: cross-run evidence belongs to the artifact one task describes, not to the whole compiled
23
+ // graph. A sibling task, dependency, routing hint or status change therefore cannot expire a useful
24
+ // finding; changing the goal, write surface or acceptance contract does. Keep the full digest here:
25
+ // unlike graphDefinitionHash this value is persisted beside evidence and is the fail-closed join a
26
+ // later run uses, so there is no benefit in making collision diagnosis less explicit.
27
+ export function taskContentDigest(task) {
28
+ return createHash("sha256")
29
+ .update(JSON.stringify({ goal: task.goal, files: task.files, acceptance: task.acceptance }))
30
+ .digest("hex");
31
+ }
22
32
  export function tickmarkrDir(repoRoot) {
23
33
  const dir = join(repoRoot, stateDirName(repoRoot));
24
34
  mkdirSync(dir, { recursive: true });
@@ -59,7 +69,14 @@ export function setStatus(g, id, status) {
59
69
  return { ...g, tasks: g.tasks.map((t) => (t.id === id ? { ...t, status } : t)) };
60
70
  }
61
71
  export function addEvidence(g, id, patch) {
62
- getTask(g, id);
72
+ const subject = getTask(g, id);
73
+ const digest = taskContentDigest(subject);
74
+ // This is the graph-evidence boundary where a gate result meets the task it measured. Stamp a
75
+ // copy, never mutate the gate result runGates returned: callers still use that live object for
76
+ // predicates, while durable evidence gains the content identity a later run can compare.
77
+ const gateResults = (patch.gateResults ?? []).map((result) => result !== null && typeof result === "object" && !Array.isArray(result)
78
+ ? { ...result, taskContentDigest: digest }
79
+ : result);
63
80
  return {
64
81
  ...g,
65
82
  tasks: g.tasks.map((t) => t.id === id
@@ -68,7 +85,7 @@ export function addEvidence(g, id, patch) {
68
85
  evidence: {
69
86
  commits: [...t.evidence.commits, ...(patch.commits ?? [])],
70
87
  artifacts: [...t.evidence.artifacts, ...(patch.artifacts ?? [])],
71
- gateResults: [...t.evidence.gateResults, ...(patch.gateResults ?? [])],
88
+ gateResults: [...t.evidence.gateResults, ...gateResults],
72
89
  },
73
90
  }
74
91
  : t),
@@ -21,7 +21,21 @@ export function recordedEnvironment(events) {
21
21
  return undefined;
22
22
  adapterVersions[k] = v;
23
23
  }
24
- return { tickmarkrVersion: o.tickmarkrVersion, configHash: o.configHash, adapterVersions };
24
+ // Capacity arrived in v2.0 as a pair. Two absent fields are an honest older record; one without
25
+ // the other is a malformed new record and fails closed instead of being compared as legacy.
26
+ const hasCores = Object.hasOwn(o, "cores");
27
+ const hasForkCap = Object.hasOwn(o, "forkCap");
28
+ if (hasCores !== hasForkCap)
29
+ return undefined;
30
+ if (hasCores && (typeof o.cores !== "number" || !Number.isFinite(o.cores) || o.cores <= 0
31
+ || typeof o.forkCap !== "number" || !Number.isFinite(o.forkCap) || o.forkCap <= 0))
32
+ return undefined;
33
+ return {
34
+ tickmarkrVersion: o.tickmarkrVersion,
35
+ configHash: o.configHash,
36
+ adapterVersions,
37
+ ...(hasCores ? { cores: o.cores, forkCap: o.forkCap } : {}),
38
+ };
25
39
  }
26
40
  return undefined;
27
41
  }
@@ -31,7 +45,8 @@ function envFingerprint(env) {
31
45
  return env.configHash;
32
46
  }
33
47
  function envEqual(a, b) {
34
- if (a.tickmarkrVersion !== b.tickmarkrVersion || a.configHash !== b.configHash)
48
+ if (a.tickmarkrVersion !== b.tickmarkrVersion || a.configHash !== b.configHash
49
+ || a.cores !== b.cores || a.forkCap !== b.forkCap)
35
50
  return false;
36
51
  const ak = Object.keys(a.adapterVersions).sort();
37
52
  const bk = Object.keys(b.adapterVersions).sort();
@@ -4,6 +4,7 @@ import { type ExecutorDriver } from "../drivers/types.js";
4
4
  import { type Baseline } from "../gates/baseline.js";
5
5
  import type { GateResult } from "../gates/types.js";
6
6
  import { Journal, type JournalEvent } from "./journal.js";
7
+ export { harvestCpuFlatWindowMs, resetHarvestCpuFlatMsForTests, setHarvestCpuFlatMsForTests, workerTreeCpuMs, } from "./stall.js";
7
8
  export interface RunOptions {
8
9
  runId?: string;
9
10
  resume?: boolean;
@@ -117,10 +118,6 @@ export declare function resetDeadChannelFastKillMsForTests(): void;
117
118
  /** Test seam — shrink the harvest silence gate without minute-long sleeps. */
118
119
  export declare function setHarvestSilentMsForTests(ms: number): void;
119
120
  export declare function resetHarvestSilentMsForTests(): void;
120
- export declare function harvestCpuFlatWindowMs(resolutionMs: number): number;
121
- /** Test seam — pin the flat window so a probe case need not sit through a real one. */
122
- export declare function setHarvestCpuFlatMsForTests(ms: number): void;
123
- export declare function resetHarvestCpuFlatMsForTests(): void;
124
121
  export declare const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries committed work; the worker emitted no TICKMARKR_RESULT trailer";
125
122
  /** T4 (OBS-266): identity of the command SET a tip verify ran — a changed command is a different verify. */
126
123
  export declare function commandsHash(commands: Record<string, string>): string;
@@ -139,10 +136,6 @@ export declare function verifyIntegrationTipCached(intWt: string, commands: Reco
139
136
  lastMergedTask?: string;
140
137
  baseline?: Baseline;
141
138
  }): Promise<boolean>;
142
- export declare function workerTreeCpuMs(marker: string, cwd: string): Promise<{
143
- ms: number;
144
- resolutionMs: number;
145
- } | undefined>;
146
139
  /** Test seam — exercise the production observer's total read bound with a small real tree. */
147
140
  export declare function setObserveBudgetBytesForTests(bytes: number): void;
148
141
  export declare function resetObserveBudgetBytesForTests(): void;