tickmarkr 2.5.6 → 2.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/adapters/prompt.js +21 -1
  2. package/dist/cli/commands/approve.js +5 -5
  3. package/dist/cli/commands/status.js +95 -34
  4. package/dist/cli/commands/verify.js +6 -4
  5. package/dist/gates/baseline.d.ts +8 -3
  6. package/dist/gates/baseline.js +6 -3
  7. package/dist/gates/review.js +12 -1
  8. package/dist/gates/run-gates.d.ts +4 -1
  9. package/dist/gates/run-gates.js +65 -9
  10. package/dist/gates/test-manifest.d.ts +3 -0
  11. package/dist/gates/test-manifest.js +20 -2
  12. package/dist/gates/test-reporter.js +6 -1
  13. package/dist/graph/graph.d.ts +2 -0
  14. package/dist/graph/graph.js +5 -0
  15. package/dist/run/activity.d.ts +28 -0
  16. package/dist/run/activity.js +194 -0
  17. package/dist/run/daemon.d.ts +9 -0
  18. package/dist/run/daemon.js +200 -46
  19. package/dist/run/git.d.ts +6 -1
  20. package/dist/run/git.js +63 -11
  21. package/dist/run/journal.d.ts +20 -4
  22. package/dist/run/journal.js +60 -9
  23. package/dist/run/operator-page-summary.d.ts +56 -0
  24. package/dist/run/operator-page-summary.js +68 -0
  25. package/dist/run/operator-summary.d.ts +69 -0
  26. package/dist/run/operator-summary.js +77 -0
  27. package/dist/run/protocol.d.ts +71 -0
  28. package/dist/run/protocol.js +32 -0
  29. package/dist/tui/cockpit/board.d.ts +9 -0
  30. package/dist/tui/cockpit/board.js +10 -0
  31. package/dist/tui/cockpit/derive.d.ts +35 -0
  32. package/dist/tui/cockpit/derive.js +152 -10
  33. package/dist/tui/cockpit/evidence-view.d.ts +2 -0
  34. package/dist/tui/cockpit/evidence-view.js +42 -12
  35. package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
  36. package/dist/tui/cockpit/run-cockpit.js +95 -1
  37. package/dist/tui/cockpit/run-view.d.ts +37 -0
  38. package/dist/tui/cockpit/run-view.js +189 -2
  39. package/package.json +1 -1
  40. package/skills/tickmarkr-overseer/SKILL.md +55 -3
@@ -71,9 +71,29 @@ function workerVerdictWitness(raw, nonce, positions) {
71
71
  // these — a result carrying any other summary is a PARSED trailer, i.e. the worker speaking.
72
72
  export const NO_TRAILER_SUMMARY = "worker produced no TICKMARKR_RESULT trailer";
73
73
  export const UNPARSEABLE_TRAILER_SUMMARY = "unparseable TICKMARKR_RESULT trailer";
74
+ // OBS-1062: every interactive TUI echoes the brief into the pane, and the brief carries the trailer
75
+ // TEMPLATE above — so the nonce token is on screen before the worker has said anything. A token
76
+ // whose object opens with the template's literal `"ok":true|false` (not a bool) is the brief being
77
+ // displayed, never the worker speaking; it must not count as trailer participation, or the daemon
78
+ // reaps the attempt as malformed at its first poll (v2.5.7 run …115246, both workers at 34 s).
79
+ const TEMPLATE_BODY = '{"ok":true|false';
80
+ // Bounded to THIS occurrence (review finding, Leg 0a R2): the `{` must sit before the next nonce
81
+ // token, or a malformed worker token followed by a later template redraw would be erased as echo.
82
+ function isTemplateEcho(raw, at, end) {
83
+ const open = raw.indexOf("{", at);
84
+ if (open === -1 || open >= end)
85
+ return false;
86
+ const joined = raw
87
+ .slice(open, open + 64)
88
+ .split("\n")
89
+ .map((l) => l.replace(/^[\s│|]+/, "").replace(/[\s│|]+$/, ""))
90
+ .join("");
91
+ return joined.startsWith(TEMPLATE_BODY);
92
+ }
74
93
  export function parseWorkerResult(raw, nonce) {
75
94
  const fail = (summary, cause) => ({ ok: false, summary, deviations: [], raw, cause });
76
- const positions = trailerTokenPositions(raw, nonce);
95
+ const all = trailerTokenPositions(raw, nonce);
96
+ const positions = all.filter((at, i) => !isTemplateEcho(raw, at, all[i + 1] ?? raw.length));
77
97
  // TUIs echo the prompt template, redraw lines, and HARD-wrap the JSON with per-line margins
78
98
  // (cursor does; recent-unwrapped can't rejoin hard newlines). Scan occurrences backward — last
79
99
  // parseable wins — joining wrapped lines, stripping margin/box chrome, and growing the candidate
@@ -1,11 +1,11 @@
1
- import { graphDefinitionHash, loadGraph, saveGraph } from "../../graph/graph.js";
1
+ import { graphDefinitionHash, loadGraph, saveGraph, taskDefinitionFingerprint } from "../../graph/graph.js";
2
2
  import { execFileSync } from "node:child_process";
3
3
  import { userInfo } from "node:os";
4
4
  import { loadConfig } from "../../config/config.js";
5
5
  import { integrationBranch } from "../../run/merge.js";
6
6
  import { separabilityErrors } from "../../compile/collateral.js";
7
7
  import { GATE_NAMES } from "../../graph/schema.js";
8
- import { applyScopeAmendments, engagementComparable, ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE } from "../../run/journal.js";
8
+ import { applyScopeAmendments, engagementComparable, engagementReleased, ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE } from "../../run/journal.js";
9
9
  export const APPROVAL_DISPOSITIONS = ["dispatch", "waive-gate", "re-dispatch", "fund-fixed-attempt", "fresh-budget"];
10
10
  /**
11
11
  * What each disposition's release actually buys — the clause every operator-facing sentence about
@@ -178,7 +178,7 @@ export async function approve(argv, cwd = process.cwd()) {
178
178
  throw new Error(`scope-request for ${taskId} requires --files <glob,…>`);
179
179
  if (decisions)
180
180
  throw new Error("--files cannot be combined with --waive, --uphold or --recheck");
181
- const graph = applyScopeAmendments(loadGraph(cwd), journal);
181
+ const graph = applyScopeAmendments(loadGraph(cwd), journal, false, engagementReleased(events));
182
182
  const from = graphDefinitionHash(graph);
183
183
  if (!engagementComparable(journal.read(), from).comparable
184
184
  || (lastHuman?.data.graphDefinitionHash !== undefined && lastHuman.data.graphDefinitionHash !== from)) {
@@ -196,11 +196,11 @@ export async function approve(argv, cwd = process.cwd()) {
196
196
  throw new Error(`refusing scope-request approval for ${taskId}: ${conflicts.join("\n")}`);
197
197
  journal.append("task-approved", taskId, {
198
198
  by, ...(reason ? { reason } : {}), via: "cli", release: "scope-request",
199
- amendment: { from, to: graphDefinitionHash(amended), beforeFiles: task.files, files: amendedFiles, parkLine: park.line },
199
+ amendment: { from, to: graphDefinitionHash(amended), beforeFiles: task.files, files: amendedFiles, parkLine: park.line, definition: taskDefinitionFingerprint(task) },
200
200
  });
201
201
  // Do not write graph.json from this process while the daemon owns it: its sweep materializes
202
202
  // the amendment without replacing a sibling's running state with this command's snapshot.
203
- const projected = applyScopeAmendments(graph, journal);
203
+ const projected = applyScopeAmendments(graph, journal, false, engagementReleased(events));
204
204
  const owner = approvalRunOwner(cwd, runId);
205
205
  if (!owner.live && !owner.blockingRunId)
206
206
  saveGraph(cwd, projected);
@@ -7,8 +7,11 @@ import { HerdrDriver } from "../../drivers/herdr.js";
7
7
  import { formatOwnedName, parseOwnedName, } from "../../drivers/types.js";
8
8
  import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../../graph/graph.js";
9
9
  import { GATE_NAMES } from "../../graph/schema.js";
10
- import { foldActivity } from "../../run/activity.js";
11
- import { Journal, engagementComparable, isQualityFailureParkKind, parseRunId, preservedRefsByTask, recordedTaskFailureKind, runHasEnded, upheldFeedbackByTask, } from "../../run/journal.js";
10
+ import { projectActivity } from "../../run/activity.js";
11
+ import { projectOperatorSummary } from "../../run/operator-summary.js";
12
+ import { trackJournalRows } from "../../run/protocol.js";
13
+ import { newestPark, permittedDecisionVerbs } from "./approve.js";
14
+ import { Journal, formatJournalNarration, engagementComparable, isQualityFailureParkKind, parseRunId, preservedRefsByTask, recordedTaskFailureKind, runHasEnded, upheldFeedbackByTask, } from "../../run/journal.js";
12
15
  import { isPidLive, runLockRunId, runStatusLine } from "../../run/lock.js";
13
16
  import { normalizeGateOutcome } from "../../run/outcome.js";
14
17
  import { desiredPanes } from "../../run/reconcile.js";
@@ -96,8 +99,8 @@ const taskClockReference = (events, taskId, content, fallback) => {
96
99
  }
97
100
  return fallback;
98
101
  };
99
- // Every attempt number a cell phrase carries is the engagement's own ladder (foldActivity counts
100
- // dispatch labels from 1). Name that ruler wherever a bare one appears, not only in the one phrase
102
+ // Every attempt number a cell phrase carries is the engagement's own ladder (journal dispatch
103
+ // labels are displayed from 1). Name that ruler wherever a bare one appears, not only in the one phrase
101
104
  // shipping today — a phrase added upstream must not reach the operator unruled.
102
105
  const labelAttemptRuler = (text) => text.replace(/(?<!engagement |run )\battempt (?=\d)/gu, "engagement attempt ");
103
106
  // The timer must keep the process ALIVE: an unref'd timer here let the event loop drain after the
@@ -907,9 +910,58 @@ const readRunRecord = (cwd, graph, namedRunId) => {
907
910
  * catch an empty or torn first write. Those snapshots own no task fold yet; once run-start exists,
908
911
  * however, a fold failure is unexpected and must surface instead of becoming a false empty run.
909
912
  */
910
- const recordTaskRows = (record, graph) => record.comparable && record.events.some((event) => event.event === "run-start")
911
- ? new Map(deriveRunCockpitData({ fileName: `${record.runId}.journal.jsonl`, raw: record.raw }, "status", { graph }).taskRows.map((row) => [row.taskId, row]))
913
+ const recordTaskRows = (record, graph, isDaemonAlive) => record.comparable && record.events.some((event) => event.event === "run-start")
914
+ ? new Map(deriveRunCockpitData({ fileName: `${record.runId}.journal.jsonl`, raw: record.raw }, "status", { graph, isDaemonAlive }).taskRows.map((row) => [row.taskId, row]))
912
915
  : new Map();
916
+ /** Adapt the single journal snapshot to the shared, evidence-only projections. */
917
+ const recordProjection = (record, graph, isDaemonAlive) => {
918
+ const rows = record ? recordTaskRows(record, graph, isDaemonAlive) : new Map();
919
+ const events = record?.comparable ? record.events : [];
920
+ const activity = projectActivity(record?.runId ?? "", trackJournalRows(record?.runId ?? "", events.map((raw, sourceIndex) => ({ raw, sourceIndex }))), graph.tasks);
921
+ const tasks = graph.tasks.map(task => {
922
+ const row = rows.get(task.id);
923
+ const evidence = events.filter(event => event.taskId === task.id);
924
+ const last = evidence.at(-1);
925
+ const recorded = activity.get(task.id);
926
+ let phase = last ? recorded.state : undefined;
927
+ // A completed battery is evidence of readiness, never evidence that merge started.
928
+ const gates = gateSnapshot(task, events, record?.rehashAt);
929
+ const boundary = events.reduce((index, event, at) => ["run-start", "run-resume", "run-end"].includes(event.event)
930
+ || (event.taskId === task.id && (event.event === "task-dispatch"
931
+ || (event.event === "phase-start" && event.data.phase === "gates"))) ? at : index, -1);
932
+ // Read the latest outcome, including non-verdicts. The gate rail retains earlier
933
+ // verdicts while a screen or infrastructure retry is pending; that historical
934
+ // pass must not make the current battery look ready to merge.
935
+ const currentResults = new Map(events.slice(boundary + 1).filter(event => event.taskId === task.id && event.event === "gate-result")
936
+ .map(event => [event.data.gate, normalizeGateOutcome(event.data).kind]));
937
+ if (recorded.state === "unconfirmed" && task.gates.length > 0 && !gates.priorGraph
938
+ && task.gates.every(gate => currentResults.get(gate) === "passed")) {
939
+ phase = "awaiting merge phase";
940
+ }
941
+ const responsibility = [...evidence].reverse().find(event => typeof event.data.role === "string" || typeof event.data.agent === "string");
942
+ return {
943
+ id: task.id, deps: task.deps, status: row?.state ?? task.status,
944
+ phase, lastEvidenceAt: last?.ts,
945
+ responsible: responsibility ? {
946
+ role: typeof responsibility.data.role === "string" ? responsibility.data.role : undefined,
947
+ agent: typeof responsibility.data.agent === "string" ? responsibility.data.agent : undefined,
948
+ } : undefined,
949
+ };
950
+ });
951
+ const decisions = tasks.flatMap(task => {
952
+ const park = task.status === "human" ? newestPark(events, task.id) : undefined;
953
+ return park ? [{ taskId: task.id, park, verbs: permittedDecisionVerbs(park) }] : [];
954
+ });
955
+ return { rows, activity, summaries: new Map(projectOperatorSummary(tasks, decisions).map(task => [task.taskId, task])) };
956
+ };
957
+ const summaryText = (summary) => sanitizeTaskText([
958
+ `phase ${summary.phase ?? "unrecorded"}`,
959
+ `last evidence ${summary.lastEvidenceAt ?? "unrecorded"}`,
960
+ `responsible ${[summary.responsible?.role, summary.responsible?.agent].filter(Boolean).join(" / ") || "unrecorded"}`,
961
+ `blocker ${summary.blocker?.kind ?? "none"}`,
962
+ `next action ${summary.blocker?.nextAction ?? "unrecorded"}`,
963
+ `human-decision ${summary.blocker?.decisionRequired ? "yes" : "no"}`,
964
+ ].join(" · "));
913
965
  /** The graph tasks this run's record actually speaks about — a row with no recorded fact is silence. */
914
966
  const recordedTasks = (graph, rows) => graph.tasks.filter((task) => {
915
967
  const row = rows.get(task.id);
@@ -936,14 +988,11 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
936
988
  const comparable = record?.comparable ?? false;
937
989
  const rehashAt = record?.rehashAt;
938
990
  const assignments = new Map();
939
- let replayed = null;
940
991
  const contexts = new Map();
941
992
  // v1.53 T5: this run is dead — a newer run replaced it
942
993
  const supersededBy = [...events].reverse()
943
994
  .find((e) => e.event === "superseded" && typeof e.data.by === "string")?.data.by;
944
995
  if (record && comparable) {
945
- if (!journalRowsOnly)
946
- replayed = Journal.open(cwd, record.runId).replayStatuses();
947
996
  for (const e of events) {
948
997
  if (!journalRowsOnly && e.event === "task-dispatch" && e.taskId) {
949
998
  const a = e.data.assignment;
@@ -956,7 +1005,12 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
956
1005
  }
957
1006
  }
958
1007
  }
959
- const taskRows = record && journalRowsOnly ? recordTaskRows(record, g) : new Map();
1008
+ const projection = recordProjection(record, g, () => daemon.state !== "dead");
1009
+ const taskRows = projection.rows;
1010
+ // Preserve the plain table's compact status vocabulary using the same snapshot as its evidence.
1011
+ const replayed = new Map([...taskRows].flatMap(([id, row]) => row.state === undefined ? [] : [[id,
1012
+ row.state === "running" || row.state === "interrupted" ? "pending" : graphTaskStatus(row.state, "pending"),
1013
+ ]]));
960
1014
  const clockFallback = new Date(now);
961
1015
  const recordedClock = new Date(events.at(-1)?.ts ?? now);
962
1016
  const zoneReference = Number.isFinite(recordedClock.getTime()) ? recordedClock : clockFallback;
@@ -973,14 +1027,7 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
973
1027
  };
974
1028
  const renderedTasks = journalRowsOnly ? recordedTasks(g, taskRows) : g.tasks;
975
1029
  const starved = new Set(blockedTasks(effective).map((t) => t.id));
976
- // OBS-104: ONE activity fold feeds both surfaces — never re-derived here. Comparable events only
977
- // (a recompiled graph's journal must not animate the wrong tasks); with no or stale journal the
978
- // dep-waiting cells still derive from the effective graph statuses.
979
- const activity = foldActivity(comparable ? events : [], effective.tasks);
980
- // RULING-v189-scope §14c: a card answers BOTH supervisor questions. `foldActivity` names what a
981
- // task waits FOR; this names what waits ON it. Read from `effective.tasks` — whose statuses are
982
- // the journal's replay, never the compiled graph's — so a dependent the record says is done stops
983
- // being named, and a task all of whose dependents are done acquires no entry at all.
1030
+ // Reverse dependencies share the effective task statuses with the projection.
984
1031
  const dependents = new Map();
985
1032
  for (const task of effective.tasks) {
986
1033
  if (task.status === "done")
@@ -1039,7 +1086,18 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1039
1086
  || (st === "failed" && failureKind === undefined));
1040
1087
  const livePhase = phases.get(t.id);
1041
1088
  const isStarved = !livePhase && starved.has(t.id);
1042
- const rawPhrase = livePhase ? phaseDetail(livePhase, now, workerLiveness.get(t.id)) : isStarved ? undefined : activity.cells.get(t.id);
1089
+ const summary = projection.summaries.get(t.id);
1090
+ // Attempt labels can restart at zero in legacy engagements. Pair the preparing caption's
1091
+ // counter and timestamp from the same dispatch, rather than retaining a prior higher label.
1092
+ const dispatch = [...events].reverse().find(event => event.taskId === t.id && event.event === "task-dispatch");
1093
+ const projectedPhrase = summary.blocker?.kind === "dependency-wait"
1094
+ ? `dep-waiting on ${summary.blocker.prerequisites.join(", ")}`
1095
+ : summary.phase === "terminal"
1096
+ ? st === "human" ? `parked${failureKind ? ` (${failureKind})` : ""}` : undefined
1097
+ : summary.phase === "preparing" && dispatch
1098
+ ? `preparing · attempt ${(Number.isInteger(dispatch.data.attempt) ? dispatch.data.attempt : 0) + 1} since ${dispatch.ts.slice(11, 19)}`
1099
+ : summary.phase ?? undefined;
1100
+ const rawPhrase = livePhase ? phaseDetail(livePhase, now, workerLiveness.get(t.id)) : projectedPhrase;
1043
1101
  const phrase = rawPhrase === undefined
1044
1102
  ? undefined
1045
1103
  : journalRowsOnly
@@ -1053,12 +1111,12 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1053
1111
  ? gateSnapshot(t, events, rehashAt)
1054
1112
  : { states: defaultGateStates(t), priorGraph: false };
1055
1113
  const pane = panes.get(t.id);
1056
- return { t, st, merged, failureKind, redTier, label, assignCol, isStarved, phrase, channel, ctx, livePhase, pane, ...gates };
1114
+ return { t, st, merged, failureKind, redTier, label, assignCol, isStarved, phrase, channel, ctx, livePhase, pane, summary, ...gates };
1057
1115
  });
1058
1116
  if (!unicode) {
1059
- // machine/CI surface — journals without phase-start stay byte-identical; new phase-aware frames
1060
- // use an ASCII spinner so pipes never receive terminal-only braille/ANSI.
1061
- const rows = cells.flatMap(({ t, st, merged, label, assignCol, livePhase, states, priorGraph, pane }) => {
1117
+ // Keep the machine task-title columns intact; projection details occupy their own line.
1118
+ // Phase-aware frames use ASCII spinners so pipes never receive terminal-only braille/ANSI.
1119
+ const rows = cells.flatMap(({ t, st, merged, label, assignCol, livePhase, states, priorGraph, pane, summary }) => {
1062
1120
  const chain = gateChain(states, false);
1063
1121
  const prefix = livePhase ? ` ${ASCII_SPINNER[animationFrame % ASCII_SPINNER.length]} ${t.id} ` : ` ${surfaceTaskBox(st, merged)} ${t.id} `;
1064
1122
  const suffix = ` ${chain}${priorGraph ? ` ${PRIOR_GRAPH_MARKER}` : ""} ${livePhase ? "running" : surfaceStatusWord(st)}${label} ${assignCol}${pane ? ` pane ${pane}` : ""}`;
@@ -1069,6 +1127,7 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1069
1127
  // (a pane name is 60 columns on its own), so the floor costs wrapping, never the graph.
1070
1128
  return [
1071
1129
  `${prefix}${shortGoal(t.title, Math.max(MACHINE_TITLE_FLOOR, width - prefix.length - suffix.length))}${suffix}`,
1130
+ ` ${summaryText(summary)}`,
1072
1131
  ...recoveryLinesForTask(t.id).map((line) => ` ${line}`),
1073
1132
  ];
1074
1133
  });
@@ -1142,7 +1201,8 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1142
1201
  besideLockup(fillLockup(dim(versionText), versionCells), factLines[1]),
1143
1202
  ...factLines.slice(2).map((line) => `${" ".repeat(lockupCells + lockupGapCells)}${line}`),
1144
1203
  ];
1145
- const nowLine = activity.now ? [legend(` now: ${activity.now}`)] : [];
1204
+ const lastEvent = comparable ? events.at(-1) : undefined;
1205
+ const nowLine = lastEvent ? [legend(` now: ${formatJournalNarration(lastEvent)}`)] : [];
1146
1206
  const supervisionLegend = legend(` ${supervisionText(supervision)}`);
1147
1207
  const taskSectionSummary = `${done} of ${total} merged · rows in graph order · gates left→right in pipeline order`;
1148
1208
  const taskSection = ` ${ok("▌")} ${title("TASKS")} ${dim(taskSectionSummary)}`;
@@ -1220,6 +1280,7 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1220
1280
  ` ${tone(cell.t.id)} ${dim(taskArea(cell.t))} ${sanitizeTaskText(cell.t.title)}`,
1221
1281
  ...wrapCells(` ${dim(machinery)}`, boardColumns, { continuationPrefix: " " }),
1222
1282
  ...wrapCells(` ${dim(noteFor(cell))}`, boardColumns, { continuationPrefix: " " }).filter((line) => line.trim()),
1283
+ ...wrapCells(` ${dim(summaryText(cell.summary))}`, boardColumns, { continuationPrefix: " " }),
1223
1284
  "",
1224
1285
  ];
1225
1286
  });
@@ -1249,15 +1310,15 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1249
1310
  ? running
1250
1311
  : dim;
1251
1312
  const titleTone = cell.redTier || cell.livePhase || (effort?.parks ?? 0) >= 3 ? title : dim;
1252
- return Array.from({ length: height }, (_, index) => ` ${idTone(fitColumn(columns[0][index] ?? "", idWidth, 1))}`
1253
- + dim(fitColumn(columns[1][index] ?? "", areaWidth))
1254
- + dim(fitColumn(columns[2][index] ?? "", depsWidth))
1255
- + titleTone(fitColumn(columns[3][index] ?? "", taskWidth))
1256
- + fitCells(columns[4][index] ?? "", gatesWidth)
1257
- + " "
1258
- + dim(fitColumn(columns[5][index] ?? "", channelWidth))
1259
- + dim(fitColumn(columns[6][index] ?? "", attemptWidth, 1))
1260
- + dim(fitCells(columns[7][index] ?? "", noteWidth)));
1313
+ return [...Array.from({ length: height }, (_, index) => ` ${idTone(fitColumn(columns[0][index] ?? "", idWidth, 1))}`
1314
+ + dim(fitColumn(columns[1][index] ?? "", areaWidth))
1315
+ + dim(fitColumn(columns[2][index] ?? "", depsWidth))
1316
+ + titleTone(fitColumn(columns[3][index] ?? "", taskWidth))
1317
+ + fitCells(columns[4][index] ?? "", gatesWidth)
1318
+ + " "
1319
+ + dim(fitColumn(columns[5][index] ?? "", channelWidth))
1320
+ + dim(fitColumn(columns[6][index] ?? "", attemptWidth, 1))
1321
+ + dim(fitCells(columns[7][index] ?? "", noteWidth))), ...wrapCells(` ${dim(summaryText(cell.summary))}`, boardColumns, { continuationPrefix: " " })];
1261
1322
  });
1262
1323
  return [legend(headerRow), dim(ruleRow), ...rows];
1263
1324
  };
@@ -1305,7 +1366,7 @@ const oneLine = (cwd, namedRunId) => {
1305
1366
  const record = readRunRecord(cwd, graph, namedRunId);
1306
1367
  if (!record)
1307
1368
  return sanitizeTaskText(`tickmarkr · no runs yet · 0/${graph.tasks.length} done`);
1308
- const rows = recordTaskRows(record, graph);
1369
+ const { rows } = recordProjection(record, graph);
1309
1370
  const claims = record.comparable
1310
1371
  ? [
1311
1372
  `${recordedDone(recordedTasks(graph, rows), rows)}/${graph.tasks.length} done`,
@@ -150,10 +150,12 @@ export async function verify(argv, cwd = process.cwd()) {
150
150
  allowPositionals: false,
151
151
  });
152
152
  const stateRoot = await verifyStateRoot(cwd);
153
+ const fileRoot = (file) => existsSync(join(cwd, ".tickmarkr", file)) ? cwd : stateRoot;
153
154
  if (resolve(stateRoot) !== resolve(cwd)) {
154
- console.error(`verify: state files graph.json, doctor.json and config.yaml resolved read-only from ${join(stateRoot, ".tickmarkr")} (linked worktree)`);
155
+ console.error(`verify: state files resolved read-only from ${join(stateRoot, ".tickmarkr")} (linked worktree; per-file origins: `
156
+ + STATE_FILES.map(file => `${file}: ${join(fileRoot(file), ".tickmarkr", file)} (${fileRoot(file) === cwd ? "local" : "common root"})`).join("; ") + ")");
155
157
  }
156
- const cfg = loadConfig(stateRoot);
158
+ const cfg = loadConfig(fileRoot("config.yaml"));
157
159
  const head = (await shGitOk("git rev-parse HEAD", cwd)).trim();
158
160
  const baseTip = (await shGitOk(`git rev-parse '${values.base}'`, cwd).catch(() => {
159
161
  throw new Error(`--base ${values.base} is not a resolvable ref — pass --base <ref> naming the branch this diff targets`);
@@ -167,7 +169,7 @@ export async function verify(argv, cwd = process.cwd()) {
167
169
  let files = values.files ?? [];
168
170
  let goal = `independent verification of the ${values.base}..HEAD diff`;
169
171
  if (values.task) {
170
- const t = getTask(loadGraph(stateRoot), values.task);
172
+ const t = getTask(loadGraph(fileRoot("graph.json")), values.task);
171
173
  acceptance = t.acceptance;
172
174
  if (!files.length)
173
175
  files = t.files;
@@ -223,7 +225,7 @@ export async function verify(argv, cwd = process.cwd()) {
223
225
  let author = HUMAN_AUTHOR;
224
226
  const adapters = allAdapters();
225
227
  if (wantAcceptance || wantReview) {
226
- const health = readDoctor(stateRoot) ?? (await probeAll(adapters));
228
+ const health = readDoctor(fileRoot("doctor.json")) ?? (await probeAll(adapters));
227
229
  const pools = rolePools(cfg, adapters, health);
228
230
  judgeChannels = pools.judge;
229
231
  channels = pools.review;
@@ -1,6 +1,6 @@
1
1
  import type { TickmarkrConfig } from "../config/config.js";
2
2
  import type { AcceptanceItem } from "../graph/schema.js";
3
- import { type RunCapacity, type ShResult } from "../run/git.js";
3
+ import { type RunCapacity, type ShellOptions, type ShResult } from "../run/git.js";
4
4
  import type { GateResult } from "./types.js";
5
5
  /**
6
6
  * T7: the capacity a gate's own command ran under rides the RESULT, beside the verdict it explains,
@@ -157,7 +157,12 @@ export declare const CAPTURE_CEILING_MS = 1800000;
157
157
  /** OBS-1044: a cached vitest entry whose count is a stdout sum may be inflated; only a manifest-derived
158
158
  * count is safe to apply as a deficit floor. A null count compares nothing and is safe as-is. */
159
159
  export declare function staleFileCountCommands(baseline: Baseline, commands: Record<string, string>, cwd: string): string[];
160
- export declare function captureBaseline(cwd: string, commands: Record<string, string>): Promise<Baseline>;
160
+ export interface BaselineReceiptOptions {
161
+ onReceipt?: ShellOptions["onReceipt"];
162
+ /** Only comparison's build command may carry task-build identity; capture never does. */
163
+ taskBuildAttribution?: ShellOptions["receiptAttribution"];
164
+ }
165
+ export declare function captureBaseline(cwd: string, commands: Record<string, string>, opts?: BaselineReceiptOptions): Promise<Baseline>;
161
166
  export interface VacuousOracleWarning {
162
167
  kind: "vacuous-oracle";
163
168
  taskId: string;
@@ -171,7 +176,7 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
171
176
  export interface RetryOptions {
172
177
  authorizeRetry?: (cause: "infra" | "host-starved") => boolean;
173
178
  }
174
- export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: RetryOptions & {
179
+ export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: RetryOptions & BaselineReceiptOptions & {
175
180
  rerunOf?: HostStarvedRerun;
176
181
  infraRerun?: HostStarvedRerun;
177
182
  selected?: readonly string[];
@@ -487,13 +487,13 @@ export function staleFileCountCommands(baseline, commands, cwd) {
487
487
  })
488
488
  .map(([name]) => name);
489
489
  }
490
- export async function captureBaseline(cwd, commands) {
490
+ export async function captureBaseline(cwd, commands, opts = {}) {
491
491
  const base = { commands: {} };
492
492
  for (const [name, cmd] of Object.entries(commands)) {
493
493
  if (name === "tipTest" && commands.test !== undefined && cmd === commands.test) {
494
494
  continue;
495
495
  }
496
- const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
496
+ const r = await sh(cmd, cwd, CAPTURE_CEILING_MS, { onReceipt: opts.onReceipt });
497
497
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
498
498
  // ponytail: a capture that was itself killed records the ceiling as its "measurement", which
499
499
  // scales the next ceiling up — the right direction for a suite that never finished once.
@@ -634,7 +634,10 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
634
634
  }
635
635
  const entry = baseline.commands[name];
636
636
  const ceilingMs = effectiveCeilingMs(entry);
637
- const r = await sh(cmd, cwd, ceilingMs);
637
+ const r = await sh(cmd, cwd, ceilingMs, {
638
+ onReceipt: opts.onReceipt,
639
+ ...(name === "build" ? { receiptAttribution: opts.taskBuildAttribution } : {}),
640
+ });
638
641
  // T7: every verdict below carries the capacity ITS OWN command ran under, taken off the shell
639
642
  // result rather than re-derived after the fact. The skip row above ran no command and therefore
640
643
  // states no capacity — a row that never divided the machine must not claim that it did.
@@ -185,7 +185,10 @@ export { modelProvider };
185
185
  export function matchClosureId(candidate, target) {
186
186
  if (typeof candidate !== "string")
187
187
  return typeof target === "string" ? false : undefined;
188
- const normCandidate = candidate.replace(/\s+/g, "");
188
+ // OBS-1068: the brief's copy block printed each id as `Fingerprint: <id>` and told the seat to copy
189
+ // it EXACTLY — three faithful approvals in one run were discarded as closure-mismatch. A leading
190
+ // `Fingerprint:` label is the brief's own wording, never the reviewer's id; strip it before comparing.
191
+ const normCandidate = candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "");
189
192
  if (typeof target === "string") {
190
193
  return normCandidate === target.replace(/\s+/g, "");
191
194
  }
@@ -476,6 +479,11 @@ carriedAuthors = []) {
476
479
  return capFail;
477
480
  const nonce = generateVerdictNonce();
478
481
  const repoRoot = daemonRepoRoot(worktree, artifactDir);
482
+ // OBS-880 add.1: guidance only. Do not expand scope globs into permission to run suites.
483
+ const ownTestFiles = [...new Set(task.files.filter((file) => !/[*?[\]{}()!]/.test(file) && /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/.test(file)))];
484
+ const suiteBudget = ownTestFiles.length
485
+ ? `You may run at most the task's own test files explicitly named in files[]; these are the only suites you may run: ${ownTestFiles.map((file) => `\`${file}\``).join(", ")}.`
486
+ : "No suite may be run: files[] names no explicit test file owned by this task.";
479
487
  const prompt = `TICKMARKR-REVIEW
480
488
  You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
481
489
  Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
@@ -489,6 +497,9 @@ ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
489
497
 
490
498
  ${renderDeclaredWriteScope(task.files)}
491
499
 
500
+ ## Reviewer suite budget
501
+ ${suiteBudget} Never run the whole suite (including an unfiltered npm test or vitest run). The gate suite owns the runner lease; a parallel full suite starves the gate.
502
+
492
503
  ${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
493
504
 
494
505
  ` : ""}## Diff
@@ -1,3 +1,4 @@
1
+ import type { CommandReceiptAttribution } from "../run/protocol.js";
1
2
  import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
2
3
  import { type TickmarkrConfig } from "../config/config.js";
3
4
  import { type GateName, type Task } from "../graph/schema.js";
@@ -42,9 +43,10 @@ export type GateEvent = {
42
43
  gate: GateName;
43
44
  name: string;
44
45
  payload: Record<string, unknown>;
45
- result: GateResult;
46
+ result?: GateResult;
46
47
  };
47
48
  export interface GateContext {
49
+ buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
48
50
  authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
49
51
  verificationScope?: VerificationScope;
50
52
  worktree: string;
@@ -62,6 +64,7 @@ export interface GateContext {
62
64
  excludeReviewers?: string[];
63
65
  demotedReviewers?: Set<string>;
64
66
  reviewNoVerdicts?: Map<string, string[]>;
67
+ recheck?: boolean;
65
68
  carriedAuthors?: readonly string[];
66
69
  reviewHistory?: string[];
67
70
  priorReviewers?: PriorReviewer[];
@@ -1,3 +1,4 @@
1
+ import { randomUUID } from "node:crypto";
1
2
  import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
2
3
  import { loadavg, tmpdir } from "node:os";
3
4
  import { join, posix } from "node:path";
@@ -258,6 +259,40 @@ function classifySignalOnlyTest(g) {
258
259
  }
259
260
  export async function runGates(task, ctx) {
260
261
  const results = [];
262
+ // Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
263
+ // allocates a new invocation, including retries whose local spawn counter starts at one again.
264
+ let currentBuild;
265
+ let buildStarted = false;
266
+ let receiptNotes = Promise.resolve();
267
+ const beginBuild = () => {
268
+ buildStarted = false;
269
+ currentBuild = { ...(ctx.buildReceiptIdentity ?? {
270
+ runId: ctx.artifactDir ?? "standalone", taskId: task.id, attempt: 0, gateRound: 0,
271
+ }), invocation: randomUUID() };
272
+ return { ...currentBuild };
273
+ };
274
+ const buildReceipt = (receipt, reason) => {
275
+ const matches = currentBuild !== undefined && receipt.attribution !== undefined
276
+ && Object.keys(currentBuild)
277
+ .every((key) => receipt.attribution[key] === currentBuild[key]);
278
+ const attributed = matches && (receipt.outcome === "started" || !receipt.confirmedStart || buildStarted);
279
+ if (attributed)
280
+ buildStarted = receipt.outcome === "started";
281
+ const { attribution, ...observation } = receipt;
282
+ const payload = attributed ? { ...receipt, gate: "build", ...(reason ? { reason, freshBuildRan: false } : {}) } : {
283
+ ...observation, gate: "build", attributionStatus: "unattributed",
284
+ reportedAttribution: attribution,
285
+ };
286
+ // Shell observers are synchronous. Serialize asynchronous note sinks and drain before the
287
+ // verdict (also on cancellation), without letting an observer change execution or retry policy.
288
+ receiptNotes = receiptNotes.then(async () => {
289
+ await ctx.onGate?.({ phase: "note", gate: "build", name: "build-receipt", payload });
290
+ }).catch(() => { });
291
+ };
292
+ const noBuild = async (outcome, reason) => {
293
+ buildReceipt({ outcome, confirmedStart: false, attribution: beginBuild() }, reason);
294
+ await receiptNotes;
295
+ };
261
296
  let selectionDecision;
262
297
  let commits = [];
263
298
  const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
@@ -272,6 +307,14 @@ export async function runGates(task, ctx) {
272
307
  // the ledger names the reuse and the identity even where the gate-result row's details must stay
273
308
  // the fresh verdict's (see formatReusedRow).
274
309
  const noteReuse = (gate, r, id) => ctx.onGate?.({ phase: "note", gate, name: "gate-reused-verdict", payload: { gate, pass: r.pass, details: r.meta?.reusedDetails, ...reusedIdentity(id) }, result: r });
310
+ // OBS-1055: on a recheck a cached red is the answer the operator just said was wrongly given; it is
311
+ // journaled as discarded and the gate runs. Returns true when the hit must NOT be reused.
312
+ const discardCachedRed = async (gate, hit) => {
313
+ if (!ctx.recheck || hit.pass)
314
+ return false;
315
+ await ctx.onGate?.({ phase: "note", gate, name: "recheck-rerun", payload: { gate, reason: "cached-red-discarded" }, result: hit });
316
+ return true;
317
+ };
275
318
  const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
276
319
  const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
277
320
  const failed = () => results.some((r) => !r.pass);
@@ -595,20 +638,30 @@ export async function runGates(task, ctx) {
595
638
  const hit = verdictStore.get(identity);
596
639
  if (hit)
597
640
  classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
598
- if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
641
+ if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
642
+ && !(await discardCachedRed(g, hit))) {
599
643
  r = formatReusedRow(hit, identity);
600
644
  cached = true;
645
+ if (g === "build")
646
+ await noBuild("reused-result", "verdict reused; no fresh build ran in this invocation");
601
647
  await noteReuse(g, r, identity);
602
648
  }
603
649
  }
604
650
  if (!r) {
605
- // VL-1: a detected vitest test command is judged by its own invocation-bound report — the
606
- // stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
607
- // other scripted test command keeps today's exit-code contract byte-identically.
608
- const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
609
- r = useManifest
610
- ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
611
- : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "test" && selected ? { selected } : {}) })))[0];
651
+ if (g === "build" && !cmd)
652
+ await noBuild("skipped", "no build command detected");
653
+ try {
654
+ // VL-1: a detected vitest test command is judged by its own invocation-bound report — the
655
+ // stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
656
+ // other scripted test command keeps today's exit-code contract byte-identically.
657
+ const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
658
+ r = useManifest
659
+ ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
660
+ : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
661
+ }
662
+ finally {
663
+ await receiptNotes;
664
+ }
612
665
  }
613
666
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
614
667
  if (g === "test" && selected)
@@ -929,6 +982,8 @@ export async function runGates(task, ctx) {
929
982
  if (entryDirt) {
930
983
  addMeasurement(sequence[0], entryMeasurement);
931
984
  await emitStart(sequence[0]);
985
+ if (enabled("build"))
986
+ await noBuild("refused", "dirty worktree; no build start confirmed");
932
987
  await record(await dirtyRefusal(sequence[0], entryDirt));
933
988
  return done();
934
989
  }
@@ -1037,7 +1092,8 @@ export async function runGates(task, ctx) {
1037
1092
  const hit = verdictStore.get(identity);
1038
1093
  if (hit)
1039
1094
  classifySignalOnlyTest(hit);
1040
- if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
1095
+ if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
1096
+ && !(await discardCachedRed("test", hit))) {
1041
1097
  full = formatReusedRow(hit, identity);
1042
1098
  cached = true;
1043
1099
  await noteReuse("test", full, identity);
@@ -36,6 +36,9 @@ export interface TestReport {
36
36
  certificate?: {
37
37
  at: number;
38
38
  exitCode: number;
39
+ /** Unhandled and module collection errors observed by the reporter; absent on older reports. */
40
+ errors?: number;
41
+ diagnostics?: string[];
39
42
  };
40
43
  }
41
44
  /** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
@@ -448,11 +448,29 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
448
448
  });
449
449
  const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
450
450
  report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
451
+ // Preserve the validator's verdict and classification; runner evidence only explains it.
452
+ const stdoutPath = join(dir, `test-runner-stdout-${nonce}.log`);
453
+ const stderrPath = join(dir, `test-runner-stderr-${nonce}.log`);
454
+ const stdoutTail = Buffer.from(invoked.stdout).subarray(-16 * 1024);
455
+ const stderrTail = Buffer.from(invoked.stderr).subarray(-16 * 1024);
456
+ writeFileSync(stdoutPath, stdoutTail);
457
+ writeFileSync(stderrPath, stderrTail);
458
+ const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
459
+ const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
460
+ const errors = report?.certificate?.errors;
461
+ const reporterErrors = typeof errors === "number" && Number.isInteger(errors) && errors >= 0 ? errors : "unknown";
462
+ const reportedDiagnostics = report?.certificate?.diagnostics;
463
+ const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
464
+ const diagnostics = !verdict.pass
465
+ ? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
466
+ + [...runnerErrors,
467
+ stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
468
+ : "";
451
469
  return { pass: verdict.pass, kind: verdict.kind,
452
- details: verdict.details + (!invoked.report && invoked.stderr ? `\nvitest reporter: ${invoked.stderr}` : ""),
470
+ details: verdict.details + diagnostics,
453
471
  classification: verdict.meta.classification,
454
472
  meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
455
- spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid },
473
+ spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid, stdoutPath, stderrPath },
456
474
  exitCode: invoked.exitCode ?? -1, reportPath };
457
475
  }
458
476
  catch (error) {