tickmarkr 1.87.0 → 1.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/dist/adapters/catalog.d.ts +18 -1
  2. package/dist/adapters/catalog.js +44 -1
  3. package/dist/adapters/fake.d.ts +2 -1
  4. package/dist/adapters/fake.js +7 -0
  5. package/dist/adapters/grok.js +11 -0
  6. package/dist/adapters/kimi.d.ts +2 -1
  7. package/dist/adapters/kimi.js +36 -0
  8. package/dist/adapters/opencode.js +17 -0
  9. package/dist/adapters/pi.js +11 -0
  10. package/dist/adapters/prompt.js +8 -1
  11. package/dist/adapters/registry.js +76 -57
  12. package/dist/adapters/types.d.ts +34 -3
  13. package/dist/adapters/types.js +99 -1
  14. package/dist/cli/commands/approve.d.ts +2 -0
  15. package/dist/cli/commands/approve.js +104 -84
  16. package/dist/cli/commands/compile.d.ts +1 -1
  17. package/dist/cli/commands/compile.js +29 -12
  18. package/dist/cli/commands/init.js +1 -1
  19. package/dist/cli/commands/plan.d.ts +1 -1
  20. package/dist/cli/commands/plan.js +10 -1
  21. package/dist/cli/commands/report.js +49 -0
  22. package/dist/cli/commands/status.js +298 -96
  23. package/dist/cli/harness.d.ts +13 -0
  24. package/dist/cli/harness.js +50 -0
  25. package/dist/compile/collateral.js +4 -4
  26. package/dist/compile/index.d.ts +14 -3
  27. package/dist/compile/index.js +36 -10
  28. package/dist/compile/native.js +101 -25
  29. package/dist/drivers/subprocess.d.ts +6 -1
  30. package/dist/drivers/subprocess.js +9 -4
  31. package/dist/gates/acceptance.d.ts +21 -1
  32. package/dist/gates/acceptance.js +67 -22
  33. package/dist/gates/artifact-manifest.d.ts +119 -0
  34. package/dist/gates/artifact-manifest.js +357 -0
  35. package/dist/gates/baseline.d.ts +6 -0
  36. package/dist/gates/baseline.js +52 -7
  37. package/dist/gates/llm.js +37 -26
  38. package/dist/gates/review.d.ts +16 -11
  39. package/dist/gates/review.js +44 -150
  40. package/dist/gates/run-gates.js +124 -7
  41. package/dist/graph/schema.d.ts +3 -1
  42. package/dist/graph/schema.js +4 -1
  43. package/dist/run/daemon.d.ts +42 -0
  44. package/dist/run/daemon.js +2321 -1967
  45. package/dist/run/git.d.ts +50 -0
  46. package/dist/run/git.js +113 -2
  47. package/dist/run/interactive-seed.d.ts +6 -2
  48. package/dist/run/interactive-seed.js +72 -5
  49. package/dist/run/journal.d.ts +9 -1
  50. package/dist/run/journal.js +99 -9
  51. package/dist/run/lock.d.ts +11 -0
  52. package/dist/run/lock.js +97 -6
  53. package/dist/run/outcome.d.ts +50 -0
  54. package/dist/run/outcome.js +152 -0
  55. package/dist/run/protocol.d.ts +460 -0
  56. package/dist/run/protocol.js +433 -0
  57. package/dist/run/supervision.d.ts +29 -0
  58. package/dist/run/supervision.js +189 -0
  59. package/fixtures/wrapped-acceptance.native.md +29 -0
  60. package/package.json +1 -1
  61. package/schema/rungraph.schema.json +21 -2
  62. package/skills/tickmarkr-overseer/SKILL.md +257 -5
  63. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +79 -8
  64. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +77 -0
  65. package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
  66. package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
  67. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +183 -0
package/dist/run/git.d.ts CHANGED
@@ -2,6 +2,18 @@ import { ROUTING_ENV_SEAMS } from "../route/router.js";
2
2
  export { ROUTING_ENV_SEAMS };
3
3
  export declare const FORK_CAP_ENV = "VITEST_MAX_FORKS";
4
4
  export declare const DEFAULT_FORK_CAP = "6";
5
+ /**
6
+ * cap = max(1, floor(cores / concurrency)). Total test parallelism is then cap × concurrency, which
7
+ * never exceeds max(cores, concurrency): at or below the core count the floor keeps the product
8
+ * under `cores`, and above it the floor is 0 so the minimum of 1 pins the product to `concurrency`
9
+ * itself. (The tempting `cap × concurrency ≤ cores` is unsatisfiable once concurrency > cores —
10
+ * a run cannot give a suite less than one fork.)
11
+ */
12
+ export declare const deriveForkCap: (concurrency: number, cores?: number) => number;
13
+ /** Run `fn` with the fork budget this run's resolved concurrency implies. */
14
+ export declare const runWithForkBudget: <T>(concurrency: number, fn: () => Promise<T>) => Promise<T>;
15
+ /** The cap owned by the run on this async context; the standalone default outside one. */
16
+ export declare const resolvedForkCap: () => string;
5
17
  export interface ShResult {
6
18
  code: number;
7
19
  stdout: string;
@@ -29,4 +41,42 @@ export declare function linkNodeModules(repo: string, dir: string, { force }?: {
29
41
  force?: boolean | undefined;
30
42
  }): boolean;
31
43
  export declare const WORKTREE_LAYOUT_CONTRACT = "## Worktree layout contract (harness-provisioned \u2014 do not modify)\n- node_modules is a symlink into the main repo's node_modules, provisioned by tickmarkr. Never commit, delete, or replace it \u2014 the harness re-asserts this link before gates run, so modifying it cannot help and may fail your attempt.";
44
+ /** A declared patch this probe could not find in the target's history, with the paths it touched. */
45
+ export interface MissingDeclaredPatch {
46
+ commit: string;
47
+ paths: string[];
48
+ merge: boolean;
49
+ }
50
+ /**
51
+ * `contained` — every declared patch is in the target, by ancestry or by patch identity.
52
+ * `drifted` — at least one declared patch is missing (or is an unproven merge); each is named.
53
+ * `unresolvable` — git could not answer; the raw git output is kept so the caller reports evidence,
54
+ * not a guess. Only `contained` is a clean verdict: the other two never let a caller assume one.
55
+ */
56
+ export type BaseContainment = {
57
+ result: "contained";
58
+ via: "ancestry" | "patch-id";
59
+ } | {
60
+ result: "drifted";
61
+ missing: MissingDeclaredPatch[];
62
+ } | {
63
+ result: "unresolvable";
64
+ ref: string;
65
+ evidence: string;
66
+ };
67
+ /**
68
+ * Is the declared base contained in `targetRef`'s history? Pure git evidence — no journal, no daemon
69
+ * state, no worker claim. Ancestry answers it outright; a recreated branch carries the same patches
70
+ * under new commit identities, which `git cherry` settles by patch-id.
71
+ *
72
+ * The trap this probe exists to refuse: `git cherry` (like `git log -p | git patch-id`) SKIPS merge
73
+ * commits, so a declared tip whose unique commits are all merges produces an EMPTY patch stream, and
74
+ * "no missing patches" reads exactly like "contained". So the verdict is NOT read off that stream.
75
+ * The commits are enumerated from `git rev-list --parents`, which sees merges, and patch identity
76
+ * only ever SUBTRACTS from that list: every unique commit is drift until something proves it, and
77
+ * for a merge the only direct proof git offers is reachability from the target — which would have
78
+ * kept it out of the range to begin with. The `!merge` guard below keeps that fail-closed even if a
79
+ * future `git cherry` ever emitted a `-` line for a merge.
80
+ */
81
+ export declare function declaredBaseContainment(cwd: string, declaredRef: string, targetRef?: string): Promise<BaseContainment>;
32
82
  export declare function removeWorktree(repo: string, dir: string): Promise<void>;
package/dist/run/git.js CHANGED
@@ -1,5 +1,7 @@
1
+ import { AsyncLocalStorage } from "node:async_hooks";
1
2
  import { spawn } from "node:child_process";
2
3
  import { existsSync, lstatSync, mkdirSync, readFileSync, readlinkSync, rmSync, symlinkSync, writeFileSync } from "node:fs";
4
+ import { availableParallelism } from "node:os";
3
5
  import { join, resolve } from "node:path";
4
6
  import { shq } from "../adapters/types.js";
5
7
  import { tickmarkrDir } from "../graph/graph.js";
@@ -11,6 +13,36 @@ export { ROUTING_ENV_SEAMS };
11
13
  // as argv (OBS-55) so child test oracles stay intact. The operator's own export wins.
12
14
  export const FORK_CAP_ENV = "VITEST_MAX_FORKS";
13
15
  export const DEFAULT_FORK_CAP = "6";
16
+ /**
17
+ * The fork budget belongs to ONE run: how many gate suites can be in flight at once is exactly that
18
+ * run's resolved concurrency, so the per-suite cap must divide the machine by THAT number and no
19
+ * other. Two things rule out a module-level variable. A single Node process holds more than one
20
+ * runDaemon call — the suites do it, and so does a supervisor driving two repositories — so a
21
+ * captured-once global hands whichever run started first its cap to every other run, and only a
22
+ * reset seam production never calls could hide it. And re-deriving the number at spawn time reads
23
+ * whatever `process.argv` or the config overlay says NOW, not what the run resolved: `parseArgs`
24
+ * settles `--concurrency 2 --concurrency 8` on the LAST occurrence, and an overlay is a mutable
25
+ * file, so a re-derived cap can disagree with the concurrency the run is actually enforcing.
26
+ *
27
+ * AsyncLocalStorage is the stdlib answer to both. The store is entered once, around the run body,
28
+ * from the single value `runDaemon` resolved; every shell that run spawns — baseline capture, gate
29
+ * batteries, tip verify, worker environments — inherits it through the async context, and a
30
+ * concurrent run's shells inherit their own. There is nothing to reset: leaving the run leaves
31
+ * the store, so sequential runs cannot inherit each other either.
32
+ */
33
+ const forkBudget = new AsyncLocalStorage();
34
+ /**
35
+ * cap = max(1, floor(cores / concurrency)). Total test parallelism is then cap × concurrency, which
36
+ * never exceeds max(cores, concurrency): at or below the core count the floor keeps the product
37
+ * under `cores`, and above it the floor is 0 so the minimum of 1 pins the product to `concurrency`
38
+ * itself. (The tempting `cap × concurrency ≤ cores` is unsatisfiable once concurrency > cores —
39
+ * a run cannot give a suite less than one fork.)
40
+ */
41
+ export const deriveForkCap = (concurrency, cores = availableParallelism()) => Math.max(1, Math.floor(cores / Math.max(1, concurrency)));
42
+ /** Run `fn` with the fork budget this run's resolved concurrency implies. */
43
+ export const runWithForkBudget = (concurrency, fn) => forkBudget.run(String(deriveForkCap(concurrency)), fn);
44
+ /** The cap owned by the run on this async context; the standalone default outside one. */
45
+ export const resolvedForkCap = () => forkBudget.getStore() ?? DEFAULT_FORK_CAP;
14
46
  // stdin "ignore": same class as HARD-05 / SubprocessDriver — never leave an open pipe a child can block on
15
47
  // (pi -p / codex exec wait for stdin EOF). timedOut distinguishes SIGKILL-timeout from a real nonzero exit.
16
48
  function shell(cmd, cwd, timeoutMs, login) {
@@ -21,9 +53,9 @@ function shell(cmd, cwd, timeoutMs, login) {
21
53
  const env = { ...process.env };
22
54
  for (const k of ROUTING_ENV_SEAMS)
23
55
  delete env[k];
24
- // OBS-110: apply the default fork cap only when the operator has not already set one.
56
+ // OBS-110: apply the run's own fork cap only when the operator has not already set one.
25
57
  if (!(FORK_CAP_ENV in env))
26
- env[FORK_CAP_ENV] = DEFAULT_FORK_CAP;
58
+ env[FORK_CAP_ENV] = resolvedForkCap();
27
59
  return new Promise((resolve) => {
28
60
  // detached: bash gets its own process group so a timeout can kill the whole tree —
29
61
  // SIGKILLing bash alone orphans grandchildren (codex/pi) that hold the stdio pipes
@@ -230,6 +262,85 @@ export function linkNodeModules(repo, dir, { force = false } = {}) {
230
262
  // worker from spending an attempt on environment repair.
231
263
  export const WORKTREE_LAYOUT_CONTRACT = `## Worktree layout contract (harness-provisioned — do not modify)
232
264
  - node_modules is a symlink into the main repo's node_modules, provisioned by tickmarkr. Never commit, delete, or replace it — the harness re-asserts this link before gates run, so modifying it cannot help and may fail your attempt.`;
265
+ // First-parent name-only diff: works for ordinary commits, merges (--diff-merges=first-parent, which
266
+ // a bare -m would widen to every parent's diff) and the root commit (--root), so one command names
267
+ // the paths of every commit shape this probe can report. The raw result is returned, never a path
268
+ // list: a failed diff-tree produced NO paths, and folding that into an empty array would ship a
269
+ // drifted verdict whose "touched nothing" evidence git never said — an empty list is only an answer
270
+ // after a successful command, so the caller turns a failure into unresolvable with git's own words.
271
+ const touchedPaths = (cwd, sha) => shGit(`git diff-tree --no-commit-id --name-only -z -r --root --diff-merges=first-parent ${shq(sha)}`, cwd);
272
+ /**
273
+ * Is the declared base contained in `targetRef`'s history? Pure git evidence — no journal, no daemon
274
+ * state, no worker claim. Ancestry answers it outright; a recreated branch carries the same patches
275
+ * under new commit identities, which `git cherry` settles by patch-id.
276
+ *
277
+ * The trap this probe exists to refuse: `git cherry` (like `git log -p | git patch-id`) SKIPS merge
278
+ * commits, so a declared tip whose unique commits are all merges produces an EMPTY patch stream, and
279
+ * "no missing patches" reads exactly like "contained". So the verdict is NOT read off that stream.
280
+ * The commits are enumerated from `git rev-list --parents`, which sees merges, and patch identity
281
+ * only ever SUBTRACTS from that list: every unique commit is drift until something proves it, and
282
+ * for a merge the only direct proof git offers is reachability from the target — which would have
283
+ * kept it out of the range to begin with. The `!merge` guard below keeps that fail-closed even if a
284
+ * future `git cherry` ever emitted a `-` line for a merge.
285
+ */
286
+ export async function declaredBaseContainment(cwd, declaredRef, targetRef = "HEAD") {
287
+ for (const ref of [declaredRef, targetRef]) {
288
+ const r = await shGit(`git rev-parse --verify ${shq(`${ref}^{commit}`)}`, cwd);
289
+ if (r.code !== 0)
290
+ return { result: "unresolvable", ref, evidence: (r.stderr || r.stdout).trim() };
291
+ }
292
+ const ancestor = await shGit(`git merge-base --is-ancestor ${shq(declaredRef)} ${shq(targetRef)}`, cwd);
293
+ if (ancestor.timedOut) {
294
+ const detail = (ancestor.stderr || ancestor.stdout).trim();
295
+ return {
296
+ result: "unresolvable",
297
+ ref: declaredRef,
298
+ evidence: `git merge-base --is-ancestor timed out${detail ? `: ${detail}` : ""}`,
299
+ };
300
+ }
301
+ if (ancestor.code === 0)
302
+ return { result: "contained", via: "ancestry" };
303
+ // 1 is the honest "not an ancestor"; anything else is git failing to answer, never a clean read.
304
+ if (ancestor.code !== 1) {
305
+ return { result: "unresolvable", ref: declaredRef, evidence: (ancestor.stderr || ancestor.stdout).trim() };
306
+ }
307
+ // Ancestry above is a POSITIVE proof and survives a truncated history; everything below reads
308
+ // history as if it were complete, and a shallow clone is exactly the history that is not. Its
309
+ // graft makes the boundary commit parentless, so `rev-list` stops there and `git cherry` offers
310
+ // the boundary's whole tree as ONE synthetic patch — which a single squashed commit on the target
311
+ // matches, hiding every truncated declared patch behind it (a depth-1 clone reports one matched
312
+ // patch where the complete repository reports three missing ones). The same truncation on the
313
+ // target side invents drift just as easily, so neither verdict is available: the repository does
314
+ // not hold the evidence, and saying so is the only honest read.
315
+ const shallow = await shGit("git rev-parse --is-shallow-repository", cwd);
316
+ if (shallow.code !== 0 || shallow.stdout.trim() !== "false") {
317
+ const said = (shallow.stdout || shallow.stderr).trim();
318
+ return { result: "unresolvable", ref: declaredRef, evidence: `git rev-parse --is-shallow-repository: ${said}` };
319
+ }
320
+ const range = `${shq(targetRef)}..${shq(declaredRef)}`;
321
+ const listed = await shGit(`git rev-list --parents ${range}`, cwd);
322
+ if (listed.code !== 0) {
323
+ return { result: "unresolvable", ref: declaredRef, evidence: (listed.stderr || listed.stdout).trim() };
324
+ }
325
+ const cherry = await shGit(`git cherry ${shq(targetRef)} ${shq(declaredRef)}`, cwd);
326
+ if (cherry.code !== 0) {
327
+ return { result: "unresolvable", ref: declaredRef, evidence: (cherry.stderr || cherry.stdout).trim() };
328
+ }
329
+ const proven = new Set(cherry.stdout.split("\n").filter((l) => l.startsWith("- ")).map((l) => l.slice(2).trim()));
330
+ const missing = [];
331
+ for (const line of listed.stdout.split("\n").filter((l) => l.trim())) {
332
+ const [sha, ...parents] = line.trim().split(/\s+/);
333
+ const merge = parents.length > 1;
334
+ if (!merge && proven.has(sha))
335
+ continue;
336
+ const paths = await touchedPaths(cwd, sha);
337
+ if (paths.code !== 0) {
338
+ return { result: "unresolvable", ref: sha, evidence: (paths.stderr || paths.stdout).trim() };
339
+ }
340
+ missing.push({ commit: sha, paths: paths.stdout.split("\0").filter((path) => path.length > 0), merge });
341
+ }
342
+ return missing.length > 0 ? { result: "drifted", missing } : { result: "contained", via: "patch-id" };
343
+ }
233
344
  export async function removeWorktree(repo, dir) {
234
345
  await shGit(`git worktree remove --force ${shq(dir)}`, repo); // best-effort; stale dirs are re-added with -B
235
346
  await shGit(`rm -rf ${shq(dir)}`, repo);
@@ -1,16 +1,20 @@
1
- import type { Assignment, WorkerAdapter } from "../adapters/types.js";
1
+ import { type Assignment, type WorkerAdapter } from "../adapters/types.js";
2
2
  import type { ExecutorDriver, Slot } from "../drivers/types.js";
3
3
  export interface InteractiveSeedResult {
4
4
  output: string;
5
5
  seedFailed: boolean;
6
6
  seedError?: string;
7
7
  sessionId?: string;
8
+ trustAnswered: boolean;
8
9
  }
10
+ type SeedDriver = Pick<ExecutorDriver, "run" | "waitOutput" | "read" | "sendKey">;
9
11
  export declare function runInteractiveSeed(opts: {
10
- driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
12
+ driver: SeedDriver;
11
13
  slot: Slot;
12
14
  adapter: WorkerAdapter;
13
15
  assignment: Assignment;
14
16
  promptFile: string;
15
17
  taskTimeoutMinutes: number;
18
+ onTrustAnswered?: () => void;
16
19
  }): Promise<InteractiveSeedResult>;
20
+ export {};
@@ -1,3 +1,61 @@
1
+ import { matchesTrustDialog } from "../adapters/types.js";
2
+ // The workspace-trust prompt is a STARTUP gate: it renders before the readiness banner and blocks
3
+ // it, so this wait is the only window in which anything can answer it. Observation therefore runs
4
+ // for the WHOLE readiness budget. Review round 7 (material): a 60-second cutoff here was a window
5
+ // in which a real modal is simply never read — with a two-minute deadline and the modal at 61s the
6
+ // seed sent zero keys and failed on readiness. A cycle is a bounded readiness wait plus one pane
7
+ // read, so readiness still returns the instant it appears and the cadence only bounds how long a
8
+ // modal sits unanswered; the loop's total cost is one pane read per second of a launch that is
9
+ // blocked anyway, and it ends at the first of readiness, a modal, or the deadline.
10
+ const TRUST_POLL_MS = 1_000;
11
+ const TRUST_PANE_ROWS = 80;
12
+ // v1.89 T19 / OBS-406: wait for readiness, answering a fingerprint-matched trust modal at most ONCE
13
+ // on the way. The daemon's own trust loop runs only after runInteractiveSeed returns, and this wait
14
+ // is exactly the window the modal blocks in — a declaration consulted only there is unreachable
15
+ // (kimi.ts states that gap at its own `trustDialog`).
16
+ async function awaitReadiness(opts) {
17
+ const { driver, slot, readinessMatch, deadline } = opts;
18
+ const dialog = opts.adapter.trustDialog;
19
+ const left = () => Math.max(0, deadline - Date.now());
20
+ // Nothing this launch could answer (no declaration, an honest {kind:"none"}, or a driver with no
21
+ // keystroke surface): the single long wait, byte-identical to pre-T19 behaviour.
22
+ if (!dialog || dialog.kind === "none" || !driver.sendKey) {
23
+ return { ready: await driver.waitOutput(slot, readinessMatch, left()), trustAnswered: false };
24
+ }
25
+ let trustAnswered = false;
26
+ while (!trustAnswered && left() > 0) {
27
+ if (await driver.waitOutput(slot, readinessMatch, Math.min(TRUST_POLL_MS, left()))) {
28
+ return { ready: true, trustAnswered };
29
+ }
30
+ let paneText;
31
+ try {
32
+ paneText = await driver.read(slot, TRUST_PANE_ROWS);
33
+ }
34
+ catch {
35
+ continue; // a failed read is not a matched modal — keep observing, spend nothing
36
+ }
37
+ if (!matchesTrustDialog(paneText, dialog))
38
+ continue;
39
+ // Review round 7 (material): the latch is spent BEFORE the awaited send, not after it. A send
40
+ // that dispatches the key and THEN rejects is ambiguous — the keystroke may already be in the
41
+ // slot — so a retry is a possible SECOND Enter landing on whatever modal is showing next, the
42
+ // exact harm this contract exists to prevent. Ambiguity fails closed: one attempt per launch,
43
+ // and the attempt (not the resolution) is what the daemon's per-slot latch inherits.
44
+ trustAnswered = true;
45
+ // Review round 8 (material): the RETURN VALUE is not the only path out of this function — the
46
+ // seed-line delivery below it throws DeliveryReadinessError, and a daemon that learns of the
47
+ // answer only from a result it never receives initializes its latch false with the key already
48
+ // sent. So the answer is reported the instant it is spent, before anything that can throw.
49
+ opts.onTrustAnswered?.();
50
+ try {
51
+ await driver.sendKey(slot, dialog.key);
52
+ }
53
+ catch {
54
+ /* dispatched-then-rejected: never retried here, and never re-tried by the daemon either */
55
+ }
56
+ }
57
+ return { ready: await driver.waitOutput(slot, readinessMatch, left()), trustAnswered };
58
+ }
1
59
  // v1.69 T6: launch-then-seed handoff for adapters whose real TUI cannot be argv-seeded.
2
60
  // Both the launch command and the seed line are delivered through the driver's existing `run`
3
61
  // primitive (pane-run on herdr). After the seed line is injected we read the pane back and
@@ -5,16 +63,25 @@
5
63
  export async function runInteractiveSeed(opts) {
6
64
  const seed = opts.adapter.interactiveSeed;
7
65
  await opts.driver.run(opts.slot, seed.launch(opts.assignment.model));
8
- const ready = await opts.driver.waitOutput(opts.slot, seed.readinessMatch, opts.taskTimeoutMinutes * 60_000);
66
+ // `trustAnswered` rides EVERY return below, including both early ones: the daemon initializes its
67
+ // per-slot latch from it, and an omission there reads as "no key was sent" — the second-Enter defect.
68
+ const { ready, trustAnswered } = await awaitReadiness({
69
+ driver: opts.driver,
70
+ slot: opts.slot,
71
+ adapter: opts.adapter,
72
+ readinessMatch: seed.readinessMatch,
73
+ deadline: Date.now() + opts.taskTimeoutMinutes * 60_000,
74
+ ...(opts.onTrustAnswered ? { onTrustAnswered: opts.onTrustAnswered } : {}),
75
+ });
9
76
  const banner = await opts.driver.read(opts.slot, 1000);
10
77
  if (!ready) {
11
- return { output: banner, seedFailed: true, seedError: `readiness pattern not seen: ${seed.readinessMatch}` };
78
+ return { output: banner, seedFailed: true, seedError: `readiness pattern not seen: ${seed.readinessMatch}`, trustAnswered };
12
79
  }
13
80
  let sessionId;
14
81
  if (seed.confirmBanner) {
15
82
  const confirm = seed.confirmBanner(banner, opts.assignment.model);
16
83
  if (!confirm.ok) {
17
- return { output: banner, seedFailed: true, seedError: confirm.error };
84
+ return { output: banner, seedFailed: true, seedError: confirm.error, trustAnswered };
18
85
  }
19
86
  sessionId = confirm.sessionId;
20
87
  }
@@ -26,8 +93,8 @@ export async function runInteractiveSeed(opts) {
26
93
  output = await opts.driver.read(opts.slot, 1000);
27
94
  const bottom = output.trimEnd().split("\n").pop() ?? "";
28
95
  if (!bottom.includes(seedText)) {
29
- return { output, seedFailed: false, ...(sessionId ? { sessionId } : {}) };
96
+ return { output, seedFailed: false, trustAnswered, ...(sessionId ? { sessionId } : {}) };
30
97
  }
31
98
  }
32
- return { output, seedFailed: true, seedError: "seed line never left the input box", ...(sessionId ? { sessionId } : {}) };
99
+ return { output, seedFailed: true, seedError: "seed line never left the input box", trustAnswered, ...(sessionId ? { sessionId } : {}) };
33
100
  }
@@ -3,6 +3,7 @@ import { type Assignment } from "../adapters/types.js";
3
3
  import type { TickmarkrConfig } from "../config/config.js";
4
4
  import { type GateName, type TaskStatus } from "../graph/schema.js";
5
5
  import { type ProfileDiscount, type RoutingProfile } from "../route/profile.js";
6
+ import { type DecisionEventWrite, type TrackedJournalRow } from "./protocol.js";
6
7
  export interface JournalEvent {
7
8
  ts: string;
8
9
  event: string;
@@ -18,6 +19,10 @@ export interface ResumeState {
18
19
  lastAssignment?: Assignment;
19
20
  upheldFeedback?: string;
20
21
  }
22
+ export interface CurrentAttemptGateReplay {
23
+ commit: string;
24
+ results: Map<GateName, boolean>;
25
+ }
21
26
  export declare const CHANNEL_EXCLUSION_KINDS: readonly ["dead-channel"];
22
27
  export type ChannelExclusionKind = (typeof CHANNEL_EXCLUSION_KINDS)[number];
23
28
  export declare const ATTEMPT_CAP_RELEASE: "attempt-cap";
@@ -118,13 +123,13 @@ export declare const TelemetryRowSchema: z.ZodObject<{
118
123
  "attempt-cap": "attempt-cap";
119
124
  "gate-fail": "gate-fail";
120
125
  quota: "quota";
126
+ infra: "infra";
121
127
  dispatch: "dispatch";
122
128
  "human-gate": "human-gate";
123
129
  "reroute-exhausted": "reroute-exhausted";
124
130
  stall: "stall";
125
131
  "merge-conflict": "merge-conflict";
126
132
  "tip-moved": "tip-moved";
127
- infra: "infra";
128
133
  }>>;
129
134
  tokens: z.ZodCatch<z.ZodOptional<z.ZodObject<{
130
135
  input: z.ZodNumber;
@@ -217,12 +222,15 @@ export declare class Journal {
217
222
  withJournal?: boolean;
218
223
  }): string | null;
219
224
  private get journalPath();
225
+ append(decision: DecisionEventWrite): void;
220
226
  append(event: string, taskId?: string, data?: Record<string, unknown>): void;
221
227
  phaseStart(taskId: string, phase: TaskPhase, data?: Record<string, unknown>): void;
222
228
  read(): JournalEvent[];
229
+ readTracked(): TrackedJournalRow[];
223
230
  replayStatuses(): Map<string, TaskStatus>;
224
231
  replayResumeState(): Map<string, ResumeState>;
225
232
  replaySatisfiedGates(): Map<string, GateName>;
233
+ replayCurrentAttemptGateResults(): Map<string, CurrentAttemptGateReplay>;
226
234
  replayExcludedChannels(): Set<string>;
227
235
  telemetry(row: TelemetryRow): void;
228
236
  readTelemetry(): TelemetryRow[];
@@ -6,6 +6,7 @@ import { channelKey, TokenUsageSchema } from "../adapters/types.js";
6
6
  import { stateDirName, tickmarkrDir } from "../graph/graph.js";
7
7
  import { GATE_NAMES, TIERS } from "../graph/schema.js";
8
8
  import { buildProfile, classify } from "../route/profile.js";
9
+ import { DecisionEventSchema, trackJournalRows, } from "./protocol.js";
9
10
  import { redactSecrets } from "./redact.js";
10
11
  export function phaseForGate(gate) {
11
12
  if (gate === "acceptance")
@@ -57,7 +58,8 @@ export const REVIEW_UPHELD_RELEASE = "review-upheld";
57
58
  export const RECHECK_RELEASE = "recheck";
58
59
  // OBS-189: review rounds are scoped to the current ENGAGEMENT — the stretch since the newest operator
59
60
  // approval for the task. A whole-journal count re-parks an upheld task before its funded attempt can
60
- // dispatch (measured live on run-20260726-213539), making a fresh journal the only escape.
61
+ // dispatch (measured live on run-20260726-213539), making a fresh journal the only escape. A T15
62
+ // replayMeasurement re-observes an interrupted round and is audit evidence, not a newly funded round.
61
63
  export function reviewRoundsSinceApproval(events, taskId) {
62
64
  let rounds = 0;
63
65
  for (const e of events) {
@@ -65,7 +67,8 @@ export function reviewRoundsSinceApproval(events, taskId) {
65
67
  continue;
66
68
  if (e.event === "task-approved")
67
69
  rounds = 0;
68
- else if (e.event === "gate-result" && e.data.gate === "review" && e.data.pass === false)
70
+ else if (e.event === "gate-result" && e.data.gate === "review" && e.data.pass === false
71
+ && e.data.replayMeasurement !== true)
69
72
  rounds++;
70
73
  }
71
74
  return rounds;
@@ -200,6 +203,17 @@ const VOLATILE_TOKENS = [
200
203
  // An absolute path INTO the repo keeps its repo-relative tail — that tail IS identity (a defect in
201
204
  // daemon.ts is not a defect in journal.ts); only the machine/worktree prefix ahead of it is volatile.
202
205
  [/\/(?:[\w.@~+%-]+\/)*((?:src|tests|scripts|docs|fixtures|specs|schema|skills|assets)\/[\w.@~+%/-]+)/g, "<path>/$1"],
206
+ // v1.88 T3: a mkdtemp directory name is a chosen prefix plus six characters the RUNTIME generates,
207
+ // and the final-segment rule below keeps the whole name as identity — so a failure that NAMES the
208
+ // temp directory carried its random suffix through, two dispatches of one defect fingerprinted
209
+ // apart, and gate-fingerprint-cap was unreachable (v1.87 T5 re-ran six times where the cap would
210
+ // have parked it at two). Only the generated suffix is volatile; the chosen prefix is identity —
211
+ // a tickmarkr-llm- failure is not a tickmarkr-eval- one and must not spend its retry budget.
212
+ // Gated three ways, because a false cap bans a legitimate retry while a missed cap costs one round:
213
+ // on a tmp ROOT (mkdtemp writes under os.tmpdir()), on the separator that ends the chosen prefix
214
+ // (it is what tells us where the prefix stops), and on the suffix carrying an uppercase or a digit
215
+ // so an ordinary six-letter word — /tmp/tickmarkr-worker-output — stays identity.
216
+ [/(\/(?:private\/)?(?:tmp|var\/folders)\/(?:[\w.@~+%-]+\/)*?[\w.@~+%-]*[-_])(?=[A-Za-z0-9]*[A-Z0-9])[A-Za-z0-9]{6}(?![\w.@~+%-])/g, "$1<tmp>"],
203
217
  // Rule: an absolute diagnostic path's machine/worktree prefix is volatile, but its named file is
204
218
  // identity. Therefore paths outside the repo-marker set keep their final segment: two machines
205
219
  // naming parse.js normalize together, while parse.js and render.js can never spend one another's
@@ -292,7 +306,8 @@ export function normalizeGateFailure(details) {
292
306
  }
293
307
  // Two normalized-identical failures of one gate on one task buy no more rounds (the ladder cannot fix
294
308
  // what it already re-ran verbatim). Engagement-scoped exactly like reviewRoundsSinceApproval: an
295
- // operator approval is a new engagement, and nothing else resets the count.
309
+ // operator approval is a new engagement, and nothing else resets the count. Resume re-measurements
310
+ // are excluded because the interrupted attempt already bought the result they confirm.
296
311
  export const GATE_FINGERPRINT_CAP = 2;
297
312
  export function identicalGateFailures(events, taskId, gate, normalized) {
298
313
  let n = 0;
@@ -302,6 +317,7 @@ export function identicalGateFailures(events, taskId, gate, normalized) {
302
317
  if (e.event === "task-approved")
303
318
  n = 0;
304
319
  else if (e.event === "gate-result" && e.data.gate === gate && e.data.pass === false
320
+ && e.data.replayMeasurement !== true
305
321
  && typeof e.data.details === "string"
306
322
  && normalizeGateFailure(e.data.details) === normalized)
307
323
  n++;
@@ -474,12 +490,14 @@ const PROVIDER_OUTAGE_RE = /Unable to reach the model provider|cannot reach the
474
490
  export function classifyWorkerResultCause(opts) {
475
491
  if (opts.ok && opts.finished)
476
492
  return undefined;
493
+ // v1.89 T7: a provider-outage signature is independent of trailer state and has its own remedy.
477
494
  if (PROVIDER_OUTAGE_RE.test(opts.output))
478
495
  return "provider-death";
496
+ // A stall kill is distinguished by its timeout signal, before any trailer-derived result fields.
497
+ if (opts.timedOut)
498
+ return "stall-timeout";
479
499
  if (opts.summary === "unparseable TICKMARKR_RESULT trailer")
480
500
  return "malformed-trailer";
481
- if (!opts.finished && opts.timedOut)
482
- return "stall-timeout";
483
501
  if (!opts.finished && opts.exitCode !== null)
484
502
  return "clean-exit-no-trailer";
485
503
  if (!opts.finished)
@@ -657,6 +675,24 @@ function readJsonl(path) {
657
675
  }
658
676
  return out;
659
677
  }
678
+ // The tracked compatibility boundary additionally retains physical source identity. Keep it separate
679
+ // from readJsonl so the daemon's hot raw replay path pays no schema or wrapper-allocation cost.
680
+ function readJsonlSource(path) {
681
+ if (!existsSync(path))
682
+ return [];
683
+ const out = [];
684
+ for (const [sourceIndex, line] of readFileSync(path, "utf8").split("\n").entries()) {
685
+ if (!line.trim())
686
+ continue;
687
+ try {
688
+ out.push({ sourceIndex, raw: JSON.parse(line) });
689
+ }
690
+ catch {
691
+ // torn trailing write after a crash — ignore; everything before it is intact
692
+ }
693
+ }
694
+ return out;
695
+ }
660
696
  // Cross-run telemetry for Phase-12 profile derivation: the last K runs' rows, each
661
697
  // tagged with its runId (runIds are zero-padded run-UTCYYYYMMDD-HHMMSS-sequence ⇒ plain .sort() is
662
698
  // chronological, same as latestRunId). Rows are facts, not classifications — Phase 12
@@ -775,17 +811,22 @@ export class Journal {
775
811
  get journalPath() {
776
812
  return join(this.dir, "journal.jsonl");
777
813
  }
778
- append(event, taskId, data = {}) {
814
+ append(eventOrDecision, taskId, data = {}) {
815
+ const decisionRow = typeof eventOrDecision === "string"
816
+ ? undefined
817
+ : DecisionEventSchema.parse({ ...eventOrDecision, ts: new Date().toISOString() });
818
+ const event = decisionRow?.event ?? eventOrDecision;
819
+ const inputData = decisionRow?.data ?? data;
779
820
  const evidence = judgePersistence.getStore();
780
821
  const failed = evidence?.invocations.filter((invocation) => invocation.transcript !== undefined) ?? [];
781
822
  const persistedData = event === "judge-retry" && failed.length > 0
782
823
  ? {
783
- ...data,
824
+ ...inputData,
784
825
  transcript: failed[0].transcript,
785
826
  ...(failed[1] ? { retryTranscript: failed[1].transcript } : {}),
786
827
  }
787
- : data;
788
- const row = {
828
+ : inputData;
829
+ const row = decisionRow ?? {
789
830
  ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data: persistedData,
790
831
  };
791
832
  // T3 secret redaction: only the persisted bytes are masked — the caller's data stays untouched in
@@ -825,6 +866,9 @@ export class Journal {
825
866
  read() {
826
867
  return readJsonl(this.journalPath);
827
868
  }
869
+ readTracked() {
870
+ return trackJournalRows(this.runId, readJsonlSource(this.journalPath));
871
+ }
828
872
  replayStatuses() {
829
873
  const s = new Map();
830
874
  for (const e of this.read()) {
@@ -956,6 +1000,52 @@ export class Journal {
956
1000
  }
957
1001
  return satisfied;
958
1002
  }
1003
+ // T15: replay completed measurements from the CURRENT worker attempt. A later task-dispatch starts
1004
+ // a new attempt and erases every older result; a result naming a different commit starts a new
1005
+ // commit-scoped set. Last result per gate wins, so a later failure retracts a pass and a later pass
1006
+ // can replace a failure. Missing/malformed commit or gate data is inert and therefore never skipped.
1007
+ // This fold never derives satisfaction from task-approved: failed-gate authority belongs
1008
+ // exclusively to replaySatisfiedGates(), whose typed release-marker contract above is unchanged.
1009
+ replayCurrentAttemptGateResults() {
1010
+ const replay = new Map();
1011
+ for (const e of this.read()) {
1012
+ if (!e.taskId)
1013
+ continue;
1014
+ if (e.event === "task-dispatch") {
1015
+ replay.delete(e.taskId);
1016
+ continue;
1017
+ }
1018
+ // Releases that buy another worker deliberately end the attempt whose measurements preceded
1019
+ // them. An untyped approval and gate-satisfied stay in their own authority semantics: neither
1020
+ // can turn a failure into observed green here.
1021
+ if (e.event === "task-approved"
1022
+ && (e.data.release === ATTEMPT_CAP_RELEASE || e.data.release === RECHECK_RELEASE
1023
+ || e.data.release === REVIEW_UPHELD_RELEASE)) {
1024
+ replay.delete(e.taskId);
1025
+ continue;
1026
+ }
1027
+ if (e.event !== "gate-result")
1028
+ continue;
1029
+ if (typeof e.data.commit !== "string" || typeof e.data.gate !== "string"
1030
+ || !GATE_NAMES.includes(e.data.gate)) {
1031
+ replay.delete(e.taskId); // a newer unattributable verdict invalidates the older candidate
1032
+ continue;
1033
+ }
1034
+ let state = replay.get(e.taskId);
1035
+ if (!state || state.commit !== e.data.commit) {
1036
+ state = { commit: e.data.commit, results: new Map() };
1037
+ replay.set(e.taskId, state);
1038
+ }
1039
+ // A selected-test green is explicitly a screen, never the merge verdict. Only its later
1040
+ // fullSuite replacement may survive a restart as a satisfied test gate.
1041
+ const incompleteTest = e.data.gate === "test" && Array.isArray(e.data.selectedTests)
1042
+ && e.data.fullSuite !== true;
1043
+ const satisfied = !incompleteTest
1044
+ && (e.data.pass === true || e.data.skipped === true) && e.data.infra !== true;
1045
+ state.results.set(e.data.gate, satisfied);
1046
+ }
1047
+ return replay;
1048
+ }
959
1049
  // v1.71 OBS-119: run-wide exclusion fold — same replay discipline as replayResumeState().
960
1050
  // channel-exclusion is the typed event; dead-channel-failover.from is the pre-v1.71 compat seam.
961
1051
  replayExcludedChannels() {
@@ -17,6 +17,17 @@ export declare function acquireRunLock(repoRoot: string, runId: string): {
17
17
  };
18
18
  };
19
19
  export declare function releaseRunLock(repoRoot: string): void;
20
+ export interface ApprovalSerialization {
21
+ /** true when another approval or run-end already owned the boundary before this caller */
22
+ contended: boolean;
23
+ release(): void;
24
+ }
25
+ export declare function acquireApprovalSerialization(repoRoot: string, runId: string): Promise<ApprovalSerialization>;
26
+ export declare function runLockOwner(repoRoot: string): {
27
+ pid?: number;
28
+ runId?: string;
29
+ live: boolean;
30
+ } | undefined;
20
31
  export declare function isRunLockLive(repoRoot: string): boolean;
21
32
  export declare function unlockRun(repoRoot: string): {
22
33
  held: false;