tickmarkr 2.6.2 → 2.6.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +12 -2
  2. package/dist/adapters/codex.d.ts +2 -1
  3. package/dist/adapters/codex.js +26 -5
  4. package/dist/adapters/model-lints.d.ts +1 -0
  5. package/dist/adapters/model-lints.js +48 -1
  6. package/dist/adapters/model-windows.d.ts +2 -1
  7. package/dist/adapters/model-windows.js +18 -7
  8. package/dist/adapters/types.d.ts +6 -0
  9. package/dist/cli/commands/approve.d.ts +6 -3
  10. package/dist/cli/commands/approve.js +43 -6
  11. package/dist/cli/commands/doctor.d.ts +8 -2
  12. package/dist/cli/commands/doctor.js +6 -1
  13. package/dist/cli/commands/fleet.js +264 -59
  14. package/dist/cli/commands/init.js +5 -3
  15. package/dist/cli/commands/plan.js +13 -8
  16. package/dist/cli/commands/report.d.ts +40 -2
  17. package/dist/cli/commands/report.js +278 -11
  18. package/dist/cli/commands/resume.js +2 -4
  19. package/dist/cli/commands/run.d.ts +11 -0
  20. package/dist/cli/commands/run.js +33 -5
  21. package/dist/cli/commands/status.js +49 -4
  22. package/dist/cli/commands/verify.js +332 -121
  23. package/dist/compile/native.js +137 -0
  24. package/dist/config/config.d.ts +40 -9
  25. package/dist/config/config.js +133 -15
  26. package/dist/config/fleet-overlay.d.ts +13 -3
  27. package/dist/config/fleet-overlay.js +12 -8
  28. package/dist/drivers/herdr.d.ts +12 -0
  29. package/dist/drivers/herdr.js +51 -0
  30. package/dist/drivers/orca.d.ts +9 -1
  31. package/dist/drivers/orca.js +29 -7
  32. package/dist/drivers/types.d.ts +2 -0
  33. package/dist/drivers/types.js +2 -2
  34. package/dist/gates/acceptance.d.ts +7 -0
  35. package/dist/gates/acceptance.js +27 -5
  36. package/dist/gates/baseline.d.ts +20 -1
  37. package/dist/gates/baseline.js +100 -20
  38. package/dist/gates/cache.d.ts +8 -0
  39. package/dist/gates/cache.js +12 -2
  40. package/dist/gates/llm.d.ts +6 -0
  41. package/dist/gates/llm.js +27 -8
  42. package/dist/gates/review.d.ts +6 -1
  43. package/dist/gates/review.js +122 -32
  44. package/dist/gates/run-gates.d.ts +54 -3
  45. package/dist/gates/run-gates.js +331 -45
  46. package/dist/gates/test-manifest.d.ts +42 -0
  47. package/dist/gates/test-manifest.js +69 -10
  48. package/dist/route/router.d.ts +23 -1
  49. package/dist/route/router.js +54 -16
  50. package/dist/run/consult.d.ts +3 -1
  51. package/dist/run/consult.js +4 -2
  52. package/dist/run/daemon.d.ts +2 -1
  53. package/dist/run/daemon.js +349 -79
  54. package/dist/run/interactive-seed.d.ts +4 -0
  55. package/dist/run/interactive-seed.js +35 -9
  56. package/dist/run/journal.d.ts +123 -1
  57. package/dist/run/journal.js +480 -17
  58. package/dist/run/lease.d.ts +13 -0
  59. package/dist/run/lease.js +45 -0
  60. package/dist/run/protocol.d.ts +15 -0
  61. package/dist/run/protocol.js +11 -1
  62. package/dist/run/receipt-resolver.d.ts +22 -0
  63. package/dist/run/receipt-resolver.js +40 -1
  64. package/dist/run/repair-selection.d.ts +11 -1
  65. package/dist/run/repair-selection.js +17 -9
  66. package/dist/run/wall-budget.d.ts +48 -0
  67. package/dist/run/wall-budget.js +280 -0
  68. package/dist/tui/cockpit/live-store.d.ts +6 -0
  69. package/dist/tui/cockpit/live-store.js +36 -11
  70. package/dist/tui/cockpit/run-cockpit.js +2 -2
  71. package/dist/tui/cockpit/run-view.d.ts +2 -1
  72. package/dist/tui/cockpit/run-view.js +13 -9
  73. package/dist/tui/cockpit/setup-cockpit.d.ts +2 -0
  74. package/dist/tui/cockpit/setup-cockpit.js +4 -0
  75. package/package.json +2 -1
  76. package/schema/config.schema.json +8 -1
  77. package/skills/tickmarkr-auto/SKILL.md +2 -2
  78. package/skills/tickmarkr-loop/SKILL.md +17 -5
  79. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +4 -1
@@ -7,6 +7,8 @@ import { isPidLive, STALE_MS } from "../../run/lock.js";
7
7
  import { readTierLiveness, SUPERVISION_TIERS } from "../../run/supervision.js";
8
8
  import { OperatorStateFold } from "../../run/operator-state.js";
9
9
  export const OBSERVATION_INTERVAL_MS = 1_000;
10
+ /** Stat-to-fstat observations one poll spends on a journal a writer keeps appending to. */
11
+ export const APPEND_RACE_OBSERVATIONS = 3;
10
12
  // Graph declarations have a separate 16 MiB bound; journal retention remains unchanged.
11
13
  export const STORE_LIMITS = { graphBytes: 16 * 1024 * 1024, history: 256, historyBytes: 2 * 1024 * 1024, recordBytes: 1024 * 1024, readBytes: 1024 * 1024, subscribers: 64, metrics: 12, errors: 32 };
12
14
  const errorText = (e) => e instanceof Error ? e.message : String(e);
@@ -48,6 +50,7 @@ export class JournalTail {
48
50
  bytesRead = 0;
49
51
  lastSuccessfulReadAt;
50
52
  failure;
53
+ stale = false;
51
54
  constructor(source, hooks = {}) {
52
55
  this.source = source;
53
56
  this.hooks = hooks;
@@ -65,18 +68,45 @@ export class JournalTail {
65
68
  this.generation++;
66
69
  this.hooks.reset?.();
67
70
  }
71
+ /** A same-inode append between stat and fstat is retried; exhaustion keeps the last good snapshot, pending. */
68
72
  poll(now = Date.now()) {
73
+ try {
74
+ for (let observation = 1; !this.observe(); observation++) {
75
+ if (observation === APPEND_RACE_OBSERVATIONS) {
76
+ this.stale = true;
77
+ this.failure = undefined;
78
+ return this.snapshot();
79
+ }
80
+ }
81
+ this.stale = false;
82
+ this.failure = undefined;
83
+ this.lastSuccessfulReadAt = now;
84
+ }
85
+ catch (e) {
86
+ this.failure = { source: this.source, error: errorText(e) };
87
+ }
88
+ return this.snapshot();
89
+ }
90
+ /** One stat-to-fstat observation. False when an append raced it; nothing was consumed or reset. */
91
+ observe() {
69
92
  let fd;
70
93
  try {
71
94
  const st = statSync(this.source, { bigint: true });
72
95
  if (!st.isFile())
73
96
  throw new Error("journal source is not a regular file");
74
- if (!this.st || fileIdentity(st) !== fileIdentity(this.st) || st.size < this.st.size || (st.size === this.st.size && stamp(st) !== stamp(this.st)))
75
- this.reset();
76
- if (Number(st.size) > this.offset) {
97
+ const replaced = !this.st || fileIdentity(st) !== fileIdentity(this.st) || st.size < this.st.size || (st.size === this.st.size && stamp(st) !== stamp(this.st));
98
+ if (Number(st.size) > (replaced ? 0 : this.offset)) {
77
99
  fd = openSync(this.source, "r");
78
- if (stamp(fstatSync(fd, { bigint: true })) !== stamp(st))
100
+ const opened = fstatSync(fd, { bigint: true });
101
+ if (stamp(opened) !== stamp(st)) {
102
+ if (fileIdentity(opened) === fileIdentity(st) && opened.size > st.size)
103
+ return false;
79
104
  throw new Error("journal changed before read; retry observation");
105
+ }
106
+ }
107
+ if (replaced)
108
+ this.reset();
109
+ if (fd !== undefined) {
80
110
  let remaining = STORE_LIMITS.readBytes;
81
111
  const buffer = Buffer.allocUnsafe(Math.min(64 * 1024, Number(st.size) - this.offset));
82
112
  while (this.offset < Number(st.size) && remaining > 0) {
@@ -113,17 +143,12 @@ export class JournalTail {
113
143
  }
114
144
  }
115
145
  this.st = st;
116
- this.failure = undefined;
117
- this.lastSuccessfulReadAt = now;
118
- }
119
- catch (e) {
120
- this.failure = { source: this.source, error: errorText(e) };
146
+ return true;
121
147
  }
122
148
  finally {
123
149
  if (fd !== undefined)
124
150
  closeSync(fd);
125
151
  }
126
- return this.snapshot();
127
152
  }
128
153
  appendCarry(bytes) {
129
154
  this.carryBytes += bytes.length;
@@ -137,7 +162,7 @@ export class JournalTail {
137
162
  history: this.history.slice(), errors: this.errors.slice(), malformedCount: this.malformedCount,
138
163
  pending: this.carryBytes ? { line: this.line + 1, bytes: this.carryBytes } : undefined,
139
164
  backlogBytes: Math.max(0, Number(this.st?.size ?? 0) - this.offset),
140
- status: this.failure ? "unreadable" : this.malformedCount ? "corrupt" : this.carryBytes ? "pending" : "readable",
165
+ status: this.failure ? "unreadable" : this.malformedCount ? "corrupt" : this.carryBytes || this.stale ? "pending" : "readable",
141
166
  error: this.failure, lastSuccessfulReadAt: this.lastSuccessfulReadAt, bytesRead: this.bytesRead,
142
167
  };
143
168
  }
@@ -334,7 +334,7 @@ export function taskProjectionText(row, journal = []) {
334
334
  `blocker ${blocker ? `${blocker.kind} ${at(evidence.blocker)}` : "none"}`,
335
335
  `next ${blocker?.nextAction == null ? fieldReading(undefined) : `${blocker.nextAction} ${at(evidence.nextAction)}`}`,
336
336
  ...(harvest?.suspectedStalledHarvest ? [`${STALL_MARKER} · launch ${at(evidence.launch)} · ${harvest.nudgeFailures} nudge failed ${at(evidence.nudge)} · ${harvest.pageCount} paged ${at(evidence.page)} · no worker-result`] : []),
337
- ].reduce((text, part) => `${text} · ${part}`);
337
+ ].join(" · ");
338
338
  }
339
339
  /**
340
340
  * Identity, phase, and build are unrecorded until a dispatch. Append the blocker and next action
@@ -354,7 +354,7 @@ function undispatchedProjectionText(row, journal) {
354
354
  MISSING_EVIDENCE,
355
355
  ...(showBlocker ? [`blocker ${blocker.kind} ${at(evidence.blocker)}`] : []),
356
356
  ...(showNext ? [`next ${blocker.nextAction} ${at(evidence.nextAction)}`] : []),
357
- ].reduce((text, part) => `${text} · ${part}`);
357
+ ].join(" · ");
358
358
  }
359
359
  /**
360
360
  * Correlate each already-derived field with the row that can have produced that value. This is
@@ -93,8 +93,9 @@ export interface TaskProjection {
93
93
  readonly build: ProjectionField;
94
94
  readonly blocker: ProjectionField;
95
95
  readonly nextAction: ProjectionField;
96
- /** OBS-1048: present only when the newest launched attempt has failed nudges, pages and no worker-result. OBS-1104: neverDispatched collapses that task's line. */
96
+ /** OBS-1048: present only when the newest launched attempt has failed nudges, pages and no worker-result. */
97
97
  readonly stalled?: ProjectionField;
98
+ /** OBS-1104: a task with no recorded dispatch collapses its line to one missing-evidence clause. */
98
99
  readonly neverDispatched?: true;
99
100
  }
100
101
  export declare const STALL_MARKER = "\u26A0 stalled harvest suspected";
@@ -251,8 +251,11 @@ export function projectRunTasks(snapshot, rows, graph, decisions, runId) {
251
251
  const blocker = { label: `blocker ${blk ? `${blk.kind}${blk.diagnostic ? ` · ${blk.diagnostic}` : ""}` : "none"}`, ...(blk && blockerRow ? { line: blockerRow.line } : {}) };
252
252
  const nextAction = { label: `next ${blk?.nextAction ?? "none"}`, ...(blk?.nextAction && blockerRow ? { line: blockerRow.line } : {}) };
253
253
  // OBS-1048 harvest, chronological: the newest worker-launch opens the attempt; a later worker-result retires it.
254
- let launched, launchedAttempt, returned = false;
255
- const nudges = [], pages = [];
254
+ let launched;
255
+ let launchedAttempt;
256
+ let returned = false;
257
+ const nudges = [];
258
+ const pages = [];
256
259
  // Finding 1: only rows of the launched attempt count; a row without an attempt belongs to it (derive.ts attemptHarvests).
257
260
  const ofLaunched = (d) => launched !== undefined && (ordinal(d.attempt) ?? launchedAttempt) === launchedAttempt;
258
261
  for (const r of own) {
@@ -389,15 +392,16 @@ function undispatchedProjectionLine(p) {
389
392
  ].join(" · ");
390
393
  }
391
394
  /**
392
- * Rows the projection panel paints. A never-dispatched clause is shorter than the field line it
393
- * replaces; the shell's content counter is the panel's wrapped height, and the pinned run frames
394
- * record that counter. Blank rows keep the block as tall as the field lines were.
395
+ * Rows the projection panel paints: each task's clause at its own wrapped height (OBS-1193). A
396
+ * never-dispatched clause is shorter than the field line it replaces, so the block is shorter too;
397
+ * the shell's content counter reads that real height and the pinned run frames record it.
395
398
  */
396
399
  function projectionBlockRows(projections, wrap) {
397
- const shown = projections.flatMap((p) => wrap(projectionLine(p)).map((line, i) => ({ key: `${p.taskId}:${i}`, line, strong: p.stalled !== undefined && i === 0 })));
398
- const prior = projections.flatMap((p) => wrap(recordedProjectionLine(p)));
399
- const pad = Math.max(0, prior.length - shown.length);
400
- return [...shown, ...Array.from({ length: pad }, (_, i) => ({ key: `projection-pad:${i}`, line: " ", strong: false }))];
400
+ return projections.flatMap((p) => wrap(projectionLine(p)).map((line, i) => ({
401
+ key: `${p.taskId}:${i}`,
402
+ line,
403
+ strong: p.stalled !== undefined && i === 0,
404
+ })));
401
405
  }
402
406
  /** The Run body: the approved board (BD-1) over the fold, then the selected task's detail panels. */
403
407
  export function RunView({ snapshot, rows, page, graph, decisions, session, columns, run, now = Date.now }) {
@@ -73,6 +73,8 @@ export type ParkedDecision = {
73
73
  readonly tombstone: boolean;
74
74
  /** OBS-1178: the park's `<line>@<ts>` token — the confirmed write binds to it via `--park`. */
75
75
  readonly park?: string;
76
+ /** OBS-1202: a stall park's recorded reap failure — the one stall park recheck may release. */
77
+ readonly reapFailure?: string;
76
78
  };
77
79
  /** The parks a verb can release — the rows the surface draws live verbs on. */
78
80
  export declare function actionableDecisions(decisions: readonly ParkedDecision[]): readonly ParkedDecision[];
@@ -607,6 +607,7 @@ export function deriveParkedDecisions(journal) {
607
607
  failedGate,
608
608
  tombstone: isTombstonePark(kind, reason),
609
609
  ...(typeof parked?.ts === "string" ? { park: bindingToken({ line: lines[parkedIndex], ts: parked.ts }) } : {}),
610
+ ...(kind === "stall" && typeof parked?.data.reapFailure === "string" ? { reapFailure: parked.data.reapFailure } : {}),
610
611
  });
611
612
  }
612
613
  return decisions;
@@ -623,6 +624,9 @@ export function initialSetupDecisionsSession() {
623
624
  export function setupDecisionVerbs(decision) {
624
625
  if (decision.tombstone)
625
626
  return [];
627
+ // OBS-1202: the production table's census-recovery row; an ordinary stall stays approve-only.
628
+ if (decision.kind === "stall" && decision.reapFailure !== undefined)
629
+ return ["approve", "recheck"];
626
630
  if (decision.kind !== "gate-fail")
627
631
  return ["approve"];
628
632
  if (decision.failedGate === undefined)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.6.2",
3
+ "version": "2.6.4",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -81,6 +81,7 @@
81
81
  "scripts/emit-schema.ts",
82
82
  "scripts/probe-rig.mjs",
83
83
  "scripts/run-ci-vitest.sh",
84
+ "scripts/vitest-lease.ts",
84
85
  "specs/export-selftest.spec.md"
85
86
  ],
86
87
  "prefixes": [
@@ -10,7 +10,6 @@
10
10
  "type": "boolean"
11
11
  },
12
12
  "repairSelection": {
13
- "default": false,
14
13
  "type": "boolean"
15
14
  },
16
15
  "taskExecutionLimitMs": {
@@ -619,6 +618,14 @@
619
618
  "exclusiveMinimum": 0,
620
619
  "maximum": 9007199254740991
621
620
  },
621
+ "evidenceQuotaBytes": {
622
+ "type": "integer",
623
+ "minimum": 0,
624
+ "maximum": 9007199254740991
625
+ },
626
+ "repairSelection": {
627
+ "type": "boolean"
628
+ },
622
629
  "byShape": {
623
630
  "type": "object",
624
631
  "propertyNames": {
@@ -92,7 +92,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
92
92
  1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
93
93
  2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
94
94
  3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
95
- 4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
96
- 5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
95
+ 4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every execution clause for you; the debt clause is yours — read CURRENT `tickmarkr status <runId>` (step 5). Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
96
+ 5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty, and CURRENT `tickmarkr status <runId>` reads its owed checks outstanding empty AND known (`outstanding 0`) — a run with a parked task is partial, not green. Empty execution buckets alone are not green either (D-660): every bucket empty and the tip passed, but an operator waived one review, leaves one accepted-risk review check owed — status reads `outstanding 1 (T7 review)` and `run`/`resume` still exit 0 on execution alone, so the run is execution complete, not green. It turns green only when a `tickmarkr verify --record <runId>` discharge moves CURRENT status to `outstanding 0`, including a discharge landing after run-end — the historical run-end record keeps the old count, so never read debt from it; `outstanding unknown` is never green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
97
97
  6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout; redirect explicitly beside the spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records.
98
98
  7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
@@ -88,8 +88,8 @@ When spawning consultants (agents gathering synthesis input for decisions like S
88
88
  1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
89
89
  2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
90
90
  3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
91
- 4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
92
- 5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
91
+ 4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every execution clause for you; the debt clause is yours — read CURRENT `tickmarkr status <runId>` (step 5). Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
92
+ 5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty, and CURRENT `tickmarkr status <runId>` reads its owed checks outstanding empty AND known (`outstanding 0`) — a run with a parked task is partial, not green. Empty execution buckets alone are not green either (D-660): every bucket empty and the tip passed, but an operator waived one review, leaves one accepted-risk review check owed — status reads `outstanding 1 (T7 review)` and `run`/`resume` still exit 0 on execution alone, so the run is execution complete, not green. It turns green only when a `tickmarkr verify --record <runId>` discharge moves CURRENT status to `outstanding 0`, including a discharge landing after run-end — the historical run-end record keeps the old count, so never read debt from it; `outstanding unknown` is never green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
93
93
  6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout. Redirect it explicitly beside the source spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).
94
94
 
95
95
  ## Cockpit, parked decisions and printed twins
@@ -122,14 +122,26 @@ own bound failure token: status prints `failed — T3 — failure <line>@<ts>`,
122
122
  `tickmarkr approve <runId> T3 --recheck --park <line>@<ts>` re-gates its landed commits. Preserve resume refusals and repair the named source/config issue
123
123
  (including deny/prefer conflicts); never edit the compiled graph to force a result. Resume
124
124
  makes CURRENT TIP PENDING; historical GATES RAN does not prove completion. The completed
125
- case has 3/3 recorded merges, a latest run-end, a nonfailed known tip result and empty
126
- `failed`, `human`, `blocked`, `pending` buckets. A mismatched graph is “not comparable.”
125
+ case has 3/3 recorded merges, a latest run-end, a nonfailed known tip result, empty
126
+ `failed`, `human`, `blocked`, `pending` buckets, and CURRENT status reading `outstanding 0`
127
+ — owed checks outstanding empty AND known. Replay D-660 before calling it green: the same
128
+ 3/3 with every bucket empty, after T2's review was waived, reads `outstanding 1 (T2 review)`;
129
+ `resume` exits 0 and its final line says `execution complete; outstanding 1 (T2 review)`, not
130
+ `verified`. It is green only once `tickmarkr verify --record <runId>` discharges that check and
131
+ CURRENT status reads `outstanding 0` — even after run-end, whose record still carries 1.
132
+ `outstanding unknown` (a legacy waiver) is never green. A mismatched graph is “not comparable.”
127
133
 
128
134
  Run offers only validated park verbs: human/attempt-cap/other non-gate parks allow approve;
129
135
  infra allows approve or `--recheck`; review gate-fail allows `--waive`, `--uphold` or
130
136
  `--recheck`; other gate-fail allows waive/recheck. Waive satisfies only the identified
131
137
  failed gate, uphold funds a fixed attempt carrying review findings, and recheck reruns the
132
- declared battery without satisfying a gate. Attempt-cap approval resets the budget while
138
+ declared battery without satisfying a gate. A stall park that recorded a `reapFailure`
139
+ (unreadable or surviving worker census) allows approve or `--recheck --park <line>@<ts>`:
140
+ recheck re-verifies that attempt's owned census and gates its harvested commits with no
141
+ worker only with an explicitly recorded empty survivors array (`[]`) — a missing,
142
+ unreadable or surviving census
143
+ re-parks the stall under a new token — while plain approve dispatches a worker. An
144
+ ordinary stall park (no `reapFailure`) stays approve-only; `--recheck` refuses it. Attempt-cap approval resets the budget while
133
145
  retaining routing exclusions. Tombstones and failures without identified gate evidence
134
146
  are diagnostic-only. Decisions cannot be undone; stale or duplicate decisions refuse.
135
147
 
@@ -44,7 +44,10 @@ awk -v esc="$esc" '
44
44
  gsub(/\^\[\[[0-9;]*m/, "", line); gsub(esc "\\[[0-9;]*m", "", line)
45
45
  sub(/^[^\t]*\t[^\t]*\t[0-9T:.-]+Z[ ]?/, "", line)
46
46
  }
47
- line ~ /Test timed out|Error: Hook timed out/ { timedout++ }
47
+ # The Vitest timeout message carries its millisecond count; a test that PRINTS a fingerprint normalized
48
+ # to #ms (the daemon OBS-1106 notifications) is output, not a timed-out test. A real timeout also fails
49
+ # its test, so the summary failed count still reds it.
50
+ line ~ /(Test|Hook) timed out in [0-9]+ ?ms/ { timedout++ }
48
51
  line ~ /ERROR: Coverage for .* does not meet .*threshold/ { coverage++ }
49
52
  line ~ /^npm (error|ERR!) signal / { signals++ }
50
53
  line ~ /^(⎯)+ Unhandled Errors (⎯)+[ \t]*$/ { if (armed) close_block(); opening = 1; next }