tickmarkr 1.86.0 → 1.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/dist/adapters/catalog.d.ts +30 -1
  2. package/dist/adapters/catalog.js +58 -2
  3. package/dist/adapters/fake.d.ts +2 -1
  4. package/dist/adapters/fake.js +7 -0
  5. package/dist/adapters/grok.js +11 -0
  6. package/dist/adapters/kimi.d.ts +2 -1
  7. package/dist/adapters/kimi.js +36 -0
  8. package/dist/adapters/opencode.js +17 -0
  9. package/dist/adapters/pi.js +11 -0
  10. package/dist/adapters/prompt.js +8 -1
  11. package/dist/adapters/registry.d.ts +10 -2
  12. package/dist/adapters/registry.js +126 -67
  13. package/dist/adapters/types.d.ts +34 -3
  14. package/dist/adapters/types.js +99 -1
  15. package/dist/cli/commands/approve.d.ts +2 -0
  16. package/dist/cli/commands/approve.js +104 -84
  17. package/dist/cli/commands/compile.d.ts +1 -1
  18. package/dist/cli/commands/compile.js +29 -12
  19. package/dist/cli/commands/init.js +1 -1
  20. package/dist/cli/commands/plan.d.ts +1 -1
  21. package/dist/cli/commands/plan.js +21 -2
  22. package/dist/cli/commands/report.js +49 -0
  23. package/dist/cli/commands/resume.js +7 -1
  24. package/dist/cli/commands/status.js +298 -96
  25. package/dist/cli/harness.d.ts +13 -0
  26. package/dist/cli/harness.js +50 -0
  27. package/dist/compile/collateral.js +4 -4
  28. package/dist/compile/index.d.ts +14 -3
  29. package/dist/compile/index.js +36 -10
  30. package/dist/compile/native.js +108 -25
  31. package/dist/drivers/subprocess.d.ts +6 -1
  32. package/dist/drivers/subprocess.js +9 -4
  33. package/dist/gates/acceptance.d.ts +21 -1
  34. package/dist/gates/acceptance.js +67 -22
  35. package/dist/gates/artifact-manifest.d.ts +119 -0
  36. package/dist/gates/artifact-manifest.js +357 -0
  37. package/dist/gates/baseline.d.ts +6 -0
  38. package/dist/gates/baseline.js +52 -7
  39. package/dist/gates/llm.js +37 -26
  40. package/dist/gates/review.d.ts +16 -11
  41. package/dist/gates/review.js +44 -150
  42. package/dist/gates/run-gates.d.ts +1 -0
  43. package/dist/gates/run-gates.js +145 -9
  44. package/dist/graph/schema.d.ts +3 -1
  45. package/dist/graph/schema.js +4 -1
  46. package/dist/route/preference.d.ts +1 -1
  47. package/dist/route/preference.js +8 -1
  48. package/dist/run/consult.js +14 -1
  49. package/dist/run/daemon.d.ts +42 -0
  50. package/dist/run/daemon.js +2322 -1963
  51. package/dist/run/git.d.ts +50 -0
  52. package/dist/run/git.js +113 -2
  53. package/dist/run/interactive-seed.d.ts +6 -2
  54. package/dist/run/interactive-seed.js +72 -5
  55. package/dist/run/journal.d.ts +9 -1
  56. package/dist/run/journal.js +99 -9
  57. package/dist/run/lock.d.ts +11 -0
  58. package/dist/run/lock.js +97 -6
  59. package/dist/run/outcome.d.ts +50 -0
  60. package/dist/run/outcome.js +152 -0
  61. package/dist/run/protocol.d.ts +460 -0
  62. package/dist/run/protocol.js +433 -0
  63. package/dist/run/supervision.d.ts +29 -0
  64. package/dist/run/supervision.js +189 -0
  65. package/fixtures/gateway-models.json +1 -0
  66. package/fixtures/wrapped-acceptance.native.md +29 -0
  67. package/package.json +1 -1
  68. package/schema/rungraph.schema.json +21 -2
  69. package/skills/tickmarkr-overseer/SKILL.md +366 -4
  70. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +95 -8
  71. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +77 -0
  72. package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
  73. package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
  74. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +183 -0
@@ -77,11 +77,42 @@ export type TrustVerdict = {
77
77
  status: "action-required";
78
78
  command: string;
79
79
  };
80
- export interface TrustDialog {
80
+ export interface CapturedTrustDialog {
81
+ kind?: "dialog";
81
82
  fingerprint: string;
82
83
  key: string;
83
84
  }
84
- export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): boolean;
85
+ export interface NoTrustDialog {
86
+ kind: "none";
87
+ reason: string;
88
+ }
89
+ export type TrustDialog = CapturedTrustDialog | NoTrustDialog;
90
+ export declare const TRUST_DIALOG_BLANK_MESSAGE = "trust-dialog fingerprint must be verbatim bytes captured from a workspace-trust prompt \u2014 a blank or whitespace-only fingerprint matches every pane";
91
+ export declare const CLAUDE_TRUST_PANE: string;
92
+ export declare const CODEX_TRUST_PANE: string;
93
+ export declare const CURSOR_TRUST_PANE = "Workspace Trust Required\nTrust this folder?";
94
+ export declare const KIMI_TRUST_PANE: string;
95
+ export declare const KIMI_MCP_TRUST_PANE: string;
96
+ export declare const RECORDED_TRUST_PANES: readonly string[];
97
+ export declare const APPROVED_TRUST_FINGERPRINTS: readonly string[];
98
+ export declare const TRUST_DIALOG_UNRECORDED_MESSAGE = "trust-dialog fingerprint is not one of the approved workspace-trust captures \u2014 it must be exactly an entry of APPROVED_TRUST_FINGERPRINTS in src/adapters/types.ts (each the distinctive bytes of a recorded pane), not a prefix of one, a substring of one, or a sentence describing the prompt; declare {kind: none, reason} if the CLI renders none";
99
+ export declare const TRUST_DIALOG_VARIANTS: readonly [z.ZodObject<{
100
+ kind: z.ZodOptional<z.ZodLiteral<"dialog">>;
101
+ fingerprint: z.ZodString;
102
+ key: z.ZodString;
103
+ }, z.core.$strict>, z.ZodObject<{
104
+ kind: z.ZodLiteral<"none">;
105
+ reason: z.ZodString;
106
+ }, z.core.$strict>];
107
+ export declare const TrustDialogSchema: z.ZodUnion<readonly [z.ZodObject<{
108
+ kind: z.ZodOptional<z.ZodLiteral<"dialog">>;
109
+ fingerprint: z.ZodString;
110
+ key: z.ZodString;
111
+ }, z.core.$strict>, z.ZodObject<{
112
+ kind: z.ZodLiteral<"none">;
113
+ reason: z.ZodString;
114
+ }, z.core.$strict>]>;
115
+ export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): dialog is CapturedTrustDialog;
85
116
  export interface InputBox {
86
117
  fingerprint: string;
87
118
  match?(paneText: string): boolean;
@@ -118,7 +149,7 @@ export interface WorkerAdapter {
118
149
  collectUsage?(cwd: string, sinceMs: number): TokenUsage | undefined;
119
150
  contextUsage?(session: SessionRef): ContextUsage | null;
120
151
  trust?(repoRoot: string): TrustVerdict;
121
- trustDialog?: TrustDialog;
152
+ trustDialog: TrustDialog;
122
153
  inputBox?: InputBox;
123
154
  hardcodedFlags?: {
124
155
  binary: string;
@@ -27,8 +27,106 @@ export function modelAuthed(health, model, allowUnverifiedModels = false) {
27
27
  const authed = health?.modelAuth?.[model]?.authed;
28
28
  return authed === true || (authed === undefined && allowUnverifiedModels);
29
29
  }
30
+ export const TRUST_DIALOG_BLANK_MESSAGE = "trust-dialog fingerprint must be verbatim bytes captured from a workspace-trust prompt — a blank or whitespace-only fingerprint matches every pane";
31
+ // v1.89 T1 / OBS-414 round 3 — THE EVIDENCE. The word "trust" is not evidence of a capture: codex's
32
+ // real recorded gate is "Do you trust the contents of this directory?" and the handwritten prose
33
+ // "Do you trust this command?" is one word away, so no regex separates them — only the record does.
34
+ // Round 2 accepted the prose, and the daemon presses Enter on a match, which is auto-approval of an
35
+ // arbitrary tool call. So a fingerprint must be an ENUMERATED capture (APPROVED_TRUST_FINGERPRINTS
36
+ // below), and every pane here is VERBATIM, quoted from the record named beside it — the panes are the
37
+ // evidence each approved entry is checked against, never the acceptance rule themselves.
38
+ //
39
+ // Cost of the posture, stated plainly: an operator adding a CLI through the YAML drive contract
40
+ // cannot declare a trust dialog whose capture is not enumerated here — they declare {kind:"none",
41
+ // reason} and page a human, or contribute the capture. That is deliberate. The alternative is a stall
42
+ // protection that any plausible sentence can turn into a permission auto-approver.
43
+ // v1.75 T2 / OBS-137, claude-code 2.1.218 startup gate.
44
+ export const CLAUDE_TRUST_PANE = [
45
+ "Accessing workspace:",
46
+ "/tmp/untrusted-project",
47
+ "Quick safety check: Is this a project you created or one you trust? (Like your own code, a well-known open source project, or work from your team).",
48
+ "Yes, I trust this folder",
49
+ ].join("\n");
50
+ // v1.75 T2 / OBS-137, codex 0.144.6 startup gate.
51
+ export const CODEX_TRUST_PANE = [
52
+ "Do you trust the contents of this directory?",
53
+ "Working with untrusted contents comes with higher risk of prompt injection.",
54
+ "Trusting the directory allows project-local config, hooks, and exec policies to load.",
55
+ "› 1. Yes, continue",
56
+ "Press enter to continue",
57
+ ].join("\n");
58
+ // v1.22 T5 / OBS-19, cursor-agent's per-worktree dialog.
59
+ export const CURSOR_TRUST_PANE = "Workspace Trust Required\nTrust this folder?";
60
+ // OBS-358, live pane wW:p2TB of run-20260805-121252 — the dialog that cost 30 minutes — quoted from
61
+ // the record holding all five lines, .overseer/REPAIR-v186/GATE-T28-1.md (OBSERVATIONS.md abridges
62
+ // it to the first three). "← highlighted" is the observer's annotation of the selected row, kept
63
+ // exactly as recorded rather than tidied away.
64
+ export const KIMI_TRUST_PANE = [
65
+ "Trust this folder?",
66
+ " /Users/…/.tickmarkr/worktrees.noindex/tickmarkr-run-20260805-121252--T29",
67
+ "❯ Trust this folder ← highlighted",
68
+ " Enable project MCP servers. Remembered for this folder.",
69
+ " Don't trust",
70
+ ].join("\n");
71
+ // OBS-406, live pane wW:p32A of run-20260806-121758-…214 T5 — the same gate in its MCP-trust
72
+ // wording, which is why kimi's fingerprint is the cursor+option row both headings share.
73
+ export const KIMI_MCP_TRUST_PANE = [
74
+ "Kimi Code loads project-level MCP servers (.mcp.json, .kimi-code/mcp.json) only in trusted folders.",
75
+ " ❯ Trust this folder / Don't trust",
76
+ ].join("\n");
77
+ export const RECORDED_TRUST_PANES = [
78
+ CLAUDE_TRUST_PANE, CODEX_TRUST_PANE, CURSOR_TRUST_PANE, KIMI_TRUST_PANE, KIMI_MCP_TRUST_PANE,
79
+ ];
80
+ // v1.89 T1 / OBS-414 round 4 — EXACT, not "leading bytes of a recorded line". Round 3 accepted any
81
+ // prefix of a captured line, and every prefix of a trust prompt is also a substring of unrelated
82
+ // panes: `{fingerprint: "Trust"}` passed, then matched a tool-permission pane reading
83
+ // "Trust this command?" and handed the daemon an Enter to press on it — the auto-approval defect,
84
+ // reopened by the check meant to close it. `"Trust this folder"` (kimi's row minus its cursor glyph)
85
+ // passed the same way, and that exact string is the one OBS-406 measured producing 258 false wakes
86
+ // in 25 minutes on supervisor panes with the words on screen as prose.
87
+ //
88
+ // So the approved fingerprints are ENUMERATED, one per shipped capture, each the distinctive bytes
89
+ // of its own pane and nothing shorter. Kimi's carries the selection cursor because a live modal
90
+ // renders one and prose never does. Membership is exact; the corpus check below it keeps a list
91
+ // entry from drifting away from the pane it claims to quote.
92
+ export const APPROVED_TRUST_FINGERPRINTS = [
93
+ "Quick safety check: Is this a project you created or one you trust?", // claude-code 2.1.218, OBS-137
94
+ "Do you trust the contents of this directory?", // codex 0.144.6, OBS-137
95
+ "Workspace Trust Required", // cursor-agent, OBS-19
96
+ "❯ Trust this folder", // kimi 0.29.0, OBS-358 + OBS-406 — the cursor glyph is load-bearing
97
+ ];
98
+ const isApprovedFingerprint = (fingerprint) => APPROVED_TRUST_FINGERPRINTS.includes(fingerprint)
99
+ && RECORDED_TRUST_PANES.some((pane) => pane.includes(fingerprint));
100
+ export const TRUST_DIALOG_UNRECORDED_MESSAGE = "trust-dialog fingerprint is not one of the approved workspace-trust captures — it must be exactly an entry of APPROVED_TRUST_FINGERPRINTS in src/adapters/types.ts (each the distinctive bytes of a recorded pane), not a prefix of one, a substring of one, or a sentence describing the prompt; declare {kind: none, reason} if the CLI renders none";
101
+ // The two variants, exported as one tuple so every schema that embeds a trust declaration (the
102
+ // drive contract in catalog.ts) is built from these bytes rather than restating them.
103
+ export const TRUST_DIALOG_VARIANTS = [
104
+ z.object({
105
+ kind: z.literal("dialog").optional(),
106
+ // superRefine, not chained refine(): one failure yields ONE diagnosis, so the operator reads the
107
+ // reason their declaration was refused instead of every rule it happened to trip.
108
+ fingerprint: z.string().superRefine((fingerprint, ctx) => {
109
+ const message = fingerprint.trim().length === 0 ? TRUST_DIALOG_BLANK_MESSAGE
110
+ : isApprovedFingerprint(fingerprint) ? undefined
111
+ : TRUST_DIALOG_UNRECORDED_MESSAGE;
112
+ if (message)
113
+ ctx.addIssue({ code: "custom", message });
114
+ }),
115
+ key: z.string().min(1),
116
+ }).strict(),
117
+ z.object({
118
+ kind: z.literal("none"),
119
+ reason: z.string().refine((r) => r.trim().length > 0, "a no-dialog declaration must state a falsifiable reason"),
120
+ }).strict(),
121
+ ];
122
+ export const TrustDialogSchema = z.union(TRUST_DIALOG_VARIANTS);
123
+ // The discrimination lives HERE, at the keypress boundary every caller crosses: a no-dialog
124
+ // declaration never matches, so sendKey is unreachable for it without editing this function. The
125
+ // blank guard is the same fail-closed posture one layer below the schema.
30
126
  export function matchesTrustDialog(paneText, dialog) {
31
- return paneText.includes(dialog.fingerprint);
127
+ if (dialog.kind === "none")
128
+ return false;
129
+ return dialog.fingerprint.trim().length > 0 && paneText.includes(dialog.fingerprint);
32
130
  }
33
131
  const inputBoxes = new Map();
34
132
  export function declareInputBox(adapterId, inputBox) {
@@ -1 +1,3 @@
1
+ export type ApprovalStatus = "deferred-live" | "recorded-no-owner";
2
+ /** The production command registered in COMMANDS; its returned bytes are what the CLI prints. */
1
3
  export declare function approve(argv: string[], cwd?: string): Promise<string>;
@@ -1,109 +1,120 @@
1
1
  import { userInfo } from "node:os";
2
2
  import { GATE_NAMES } from "../../graph/schema.js";
3
3
  import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE } from "../../run/journal.js";
4
- // GATE-08 (v1.12): approve a parked human gate so the next `tickmarkr resume <runId>` dispatches it.
5
- //
6
- // The approval is a JOURNAL EVENT (task-approved) carrying who and when — it touches ONLY the
7
- // append-only journal. Writing it into tickmarkr's compiled graph artifact would be silently erased by
8
- // the next recompile (which re-emits humanGate:true from the plan frontmatter) — Phase 42 D-02.
9
- //
10
- // v1.24 OBS-18: when the park kind is attempt-cap (not a humanGate pre-dispatch park), the event
11
- // also carries `release: "attempt-cap"`. replayResumeState zeros the attempt budget on that marker so
12
- // resume dispatches instead of re-parking in the same tick; tried-list is preserved. Unknown kinds
13
- // receive no release and remain fail-closed to a human rather than being inferred from prose.
14
- //
15
- // Fail-closed (D-05): unknown runId, unknown taskId, a not-parked task, and a double-approve are all
16
- // LOUD refusals that name the reason and append NO event — never a silent no-op. A handler throw
17
- // becomes `tickmarkr approve: <message>` at exit 1 (src/cli/index.ts dispatch).
18
- //
19
- // Who/when is truthful, not dressed-up auth (D-03): default actor os.userInfo().username; --by overrides
20
- // for delegated approval; optional --reason; the event's ts (stamped by Journal.append) is the when.
21
- // OBS-189: `--uphold` is the second decision a review park offers. Plain approve accepts the diff the
22
- // reviewer rejected (gate-satisfied); --uphold sides WITH the reviewer and funds ONE fixed worker
23
- // attempt carrying the findings — the park costs an attempt, never the run.
4
+ import { acquireApprovalSerialization, runLockOwner } from "../../run/lock.js";
5
+ /** The production command registered in COMMANDS; its returned bytes are what the CLI prints. */
24
6
  export async function approve(argv, cwd = process.cwd()) {
25
- const { runId, taskId, by, reason, uphold, recheck } = parseArgs(argv);
7
+ const { runId, taskId, by, reason, uphold, recheck, reviewRoundCeiling } = parseArgs(argv);
26
8
  if (uphold && recheck)
27
9
  throw new Error("--uphold and --recheck are different decisions — pass one");
28
- // Journal.open throws `no journal for <runId> at <dir>` on an unknown run — that IS the refusal.
29
- const journal = Journal.open(cwd, runId);
30
- const status = journal.replayStatuses().get(taskId);
31
- if (status === undefined) {
32
- throw new Error(`task ${taskId} has no events in run ${runId} — unknown task or never dispatched`);
33
- }
34
- if (status !== "human") {
35
- // a silent no-op would be worse than a loud refusal — name the actual status (D-05)
36
- throw new Error(`task ${taskId} is ${status}, not a parked human gate — refusing (a silent no-op would be worse)`);
37
- }
38
- // OBS-18: only the most recent task-human for this task decides whether this approval grants a
39
- // fresh attempt budget. The closed daemon-issued kind, never a human prose string, controls release.
40
- const events = journal.read();
41
- let lastHumanIndex = -1;
42
- for (let i = events.length - 1; i >= 0; i--) {
43
- if (events[i].event === "task-human" && events[i].taskId === taskId) {
44
- lastHumanIndex = i;
45
- break;
10
+ const serialization = await acquireApprovalSerialization(cwd, runId);
11
+ try {
12
+ // Journal.open throws `no journal for <runId> at <dir>` on an unknown run — that IS the refusal.
13
+ const journal = Journal.open(cwd, runId);
14
+ const status = journal.replayStatuses().get(taskId);
15
+ if (status === undefined) {
16
+ throw new Error(`task ${taskId} has no events in run ${runId} — unknown task or never dispatched`);
46
17
  }
47
- }
48
- const lastHuman = events[lastHumanIndex];
49
- if (uphold) {
50
- // Fail-closed on the DATA: uphold applies only when the newest failed gate is the review gate —
51
- // any other gate has no reviewer to uphold. Never inferred from the park's prose.
52
- const lastFailed = events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
53
- && typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate;
54
- if (lastFailed !== "review") {
55
- throw new Error(`--uphold applies to a review rejection; ${taskId}'s last failed gate is ${lastFailed ?? "none"} — refusing`);
18
+ if (status !== "human") {
19
+ // a silent no-op would be worse than a loud refusal — name the actual status (D-05)
20
+ throw new Error(`task ${taskId} is ${status}, not a parked human gate — refusing (a silent no-op would be worse)`);
56
21
  }
57
- journal.append("task-approved", taskId, {
58
- by,
59
- ...(reason ? { reason } : {}),
60
- via: "cli",
61
- release: REVIEW_UPHELD_RELEASE,
62
- gate: "review",
63
- });
64
- return `upheld the reviewer for ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to dispatch a fixed attempt carrying the findings`;
65
- }
66
- const capPark = lastHuman?.data.kind === ATTEMPT_CAP_RELEASE;
67
- const gateFailPark = lastHuman?.data.kind === "gate-fail";
68
- if (recheck) {
69
- // OBS-203: fail-closed on the PARK KIND — only a gate-fail park has a gate to re-run. Refusing
70
- // elsewhere keeps --recheck from becoming a silent budget reset on a pre-dispatch human gate.
71
- if (!gateFailPark) {
72
- throw new Error(`--recheck applies to a gate-fail park; ${taskId}'s park kind is ${lastHuman?.data.kind ?? "none"} — refusing`);
22
+ // OBS-18: only the most recent task-human for this task decides whether this approval grants a
23
+ // fresh attempt budget. The closed daemon-issued kind, never a human prose string, controls release.
24
+ const events = journal.read();
25
+ let lastHumanIndex = -1;
26
+ for (let i = events.length - 1; i >= 0; i--) {
27
+ if (events[i].event === "task-human" && events[i].taskId === taskId) {
28
+ lastHumanIndex = i;
29
+ break;
30
+ }
31
+ }
32
+ const lastHuman = events[lastHumanIndex];
33
+ if (uphold) {
34
+ // Fail-closed on the DATA: uphold applies only when the newest failed gate is the review gate —
35
+ // any other gate has no reviewer to uphold. Never inferred from the park's prose.
36
+ const lastFailed = events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
37
+ && typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate;
38
+ if (lastFailed !== "review") {
39
+ throw new Error(`--uphold applies to a review rejection; ${taskId}'s last failed gate is ${lastFailed ?? "none"} — refusing`);
40
+ }
41
+ journal.append("task-approved", taskId, {
42
+ by,
43
+ ...(reason ? { reason } : {}),
44
+ via: "cli",
45
+ release: REVIEW_UPHELD_RELEASE,
46
+ gate: "review",
47
+ ...(reviewRoundCeiling === undefined ? {} : { reviewRoundCeiling }),
48
+ });
49
+ return disposition(cwd, runId, `upheld the reviewer for ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to dispatch a fixed attempt carrying the findings`, serialization.contended);
50
+ }
51
+ const capPark = lastHuman?.data.kind === ATTEMPT_CAP_RELEASE;
52
+ const gateFailPark = lastHuman?.data.kind === "gate-fail";
53
+ if (recheck) {
54
+ // OBS-203: fail-closed on the PARK KIND — only a gate-fail park has a gate to re-run. Refusing
55
+ // elsewhere keeps --recheck from becoming a silent budget reset on a pre-dispatch human gate.
56
+ if (!gateFailPark) {
57
+ throw new Error(`--recheck applies to a gate-fail park; ${taskId}'s park kind is ${lastHuman?.data.kind ?? "none"} — refusing`);
58
+ }
59
+ journal.append("task-approved", taskId, {
60
+ by,
61
+ ...(reason ? { reason } : {}),
62
+ via: "cli",
63
+ release: RECHECK_RELEASE,
64
+ ...(reviewRoundCeiling === undefined ? {} : { reviewRoundCeiling }),
65
+ });
66
+ return disposition(cwd, runId, `re-checking ${taskId} in ${runId} — by ${by}; no gate marked satisfied, run \`tickmarkr resume ${runId}\` to re-dispatch against the full gate suite`, serialization.contended);
67
+ }
68
+ const failedGate = gateFailPark
69
+ ? events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
70
+ && typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate
71
+ : undefined;
72
+ if (gateFailPark && !failedGate) {
73
+ throw new Error(`task ${taskId} is parked on gate-fail but has no failed gate result — refusing to infer one`);
73
74
  }
74
75
  journal.append("task-approved", taskId, {
75
76
  by,
76
77
  ...(reason ? { reason } : {}),
77
78
  via: "cli",
78
- release: RECHECK_RELEASE,
79
+ ...(reviewRoundCeiling === undefined ? {} : { reviewRoundCeiling }),
80
+ ...(capPark ? { release: ATTEMPT_CAP_RELEASE } : {}),
81
+ ...(failedGate ? { release: GATE_SATISFIED_RELEASE, gate: failedGate } : {}),
79
82
  });
80
- return `re-checking ${taskId} in ${runId} — by ${by}; no gate marked satisfied, run \`tickmarkr resume ${runId}\` to re-dispatch against the full gate suite`;
83
+ return disposition(cwd, runId, `approved ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to ${failedGate ? "continue past the approved gate" : "dispatch it"}`, serialization.contended);
81
84
  }
82
- const failedGate = gateFailPark
83
- ? events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
84
- && typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate
85
- : undefined;
86
- if (gateFailPark && !failedGate) {
87
- throw new Error(`task ${taskId} is parked on gate-fail but has no failed gate result — refusing to infer one`);
85
+ finally {
86
+ serialization.release();
88
87
  }
89
- journal.append("task-approved", taskId, {
90
- by,
91
- ...(reason ? { reason } : {}),
92
- via: "cli",
93
- ...(capPark ? { release: ATTEMPT_CAP_RELEASE } : {}),
94
- ...(failedGate ? { release: GATE_SATISFIED_RELEASE, gate: failedGate } : {}),
95
- });
96
- return `approved ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to ${failedGate ? "continue past the approved gate" : "dispatch it"}`;
97
88
  }
98
- const USAGE = "usage: tickmarkr approve <run-id> <task-id> [--uphold|--recheck] [--by <name>] [--reason <text>]";
99
- // hand-parsed argv — no CLI framework (house style). Flags --uphold, --by <name>, --reason <text>;
100
- // positionals are runId then taskId. Throws usage on missing positionals (mirrors resume.ts/unlock.ts).
89
+ // The status is command OUTPUT, not a typed sibling result: the registered approve function returns
90
+ // one primitive string and the dispatcher prints those bytes unchanged. A compact sentinel followed
91
+ // by JSON makes the contract unambiguous to machines without widening the shared command result type.
92
+ // No-lock approvals keep their historical one-line result for the cockpit; a dead recorded owner and
93
+ // a command delayed behind terminalization both emit recorded-no-owner.
94
+ function disposition(cwd, runId, message, contended) {
95
+ const owner = runLockOwner(cwd);
96
+ const resume = `tickmarkr resume ${runId}`;
97
+ if (!owner && !contended)
98
+ return message;
99
+ const status = owner?.live ? "deferred-live" : "recorded-no-owner";
100
+ const record = {
101
+ status,
102
+ resume,
103
+ ...(owner?.pid === undefined ? {} : { ownerPid: owner.pid }),
104
+ ...(owner?.runId === undefined ? {} : { ownerRunId: owner.runId }),
105
+ };
106
+ return `${message}\nTICKMARKR_APPROVAL ${JSON.stringify(record)}`;
107
+ }
108
+ const USAGE = "usage: tickmarkr approve <run-id> <task-id> [--uphold|--recheck] [--review-rounds <positive-integer>] [--by <name>] [--reason <text>]";
109
+ // hand-parsed argv — no CLI framework (house style). Positionals are runId then taskId; decision,
110
+ // ceiling, actor and reason are flags. Throws usage on missing positionals (mirrors resume.ts/unlock.ts).
101
111
  function parseArgs(argv) {
102
112
  const positionals = [];
103
113
  let by;
104
114
  let reason;
105
115
  let uphold = false;
106
116
  let recheck = false;
117
+ let reviewRoundCeiling;
107
118
  for (let i = 0; i < argv.length; i++) {
108
119
  const a = argv[i];
109
120
  if (a === "--by") {
@@ -122,6 +133,15 @@ function parseArgs(argv) {
122
133
  else if (a === "--recheck") {
123
134
  recheck = true;
124
135
  }
136
+ else if (a === "--review-rounds") {
137
+ const value = argv[++i];
138
+ if (value === undefined)
139
+ throw new Error(USAGE);
140
+ if (!/^[1-9]\d*$/.test(value) || !Number.isSafeInteger(Number(value))) {
141
+ throw new Error("--review-rounds must be a positive integer");
142
+ }
143
+ reviewRoundCeiling = Number(value);
144
+ }
125
145
  else {
126
146
  positionals.push(a);
127
147
  }
@@ -130,5 +150,5 @@ function parseArgs(argv) {
130
150
  if (!runId || !taskId) {
131
151
  throw new Error(USAGE);
132
152
  }
133
- return { runId, taskId, by: by ?? userInfo().username, reason, uphold, recheck };
153
+ return { runId, taskId, by: by ?? userInfo().username, reason, uphold, recheck, reviewRoundCeiling };
134
154
  }
@@ -1 +1 @@
1
- export declare function compile(argv: string[], cwd?: string): Promise<string>;
1
+ export declare function compile(argv: string[], cwd?: string, harnessFrom?: string | undefined): Promise<string>;
@@ -1,28 +1,45 @@
1
1
  import { isAbsolute, join } from "node:path";
2
2
  import { parseArgs } from "node:util";
3
+ import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
3
4
  import { compileSource } from "../../compile/index.js";
4
5
  import { saveGraph, stateDirName } from "../../graph/graph.js";
5
6
  import { acquireRunLock, releaseRunLock } from "../../run/lock.js";
6
- export async function compile(argv, cwd = process.cwd()) {
7
+ import { harnessLine, resolveHarness } from "../harness.js";
8
+ // v1.89 T4: harnessFrom is the resolver's INPUT (see plan.ts); the default is the INVOKED entrypoint
9
+ // (`process.argv[1]`, the bin symlink), never this module's own url — that names an internal module.
10
+ export async function compile(argv, cwd = process.cwd(), harnessFrom = process.argv[1]) {
7
11
  const { values, positionals } = parseArgs({
8
12
  args: argv,
9
- options: { type: { type: "string" } },
13
+ options: {
14
+ type: { type: "string" },
15
+ "dry-run": { type: "boolean" },
16
+ },
10
17
  allowPositionals: true,
11
18
  });
12
19
  const src = positionals[0];
13
20
  if (!src)
14
- throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native]");
21
+ throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native] [--dry-run]");
15
22
  // resolve against the target repo, not the process cwd (the CLI test passes a tmp repo)
23
+ // Both modes reach the same pure compiler; --dry-run only removes the lock/write side effect below.
16
24
  const g = compileSource(isAbsolute(src) ? src : join(cwd, src), values.type, cwd);
17
- // HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
18
- // cannot swap graph.json under an active run between the daemon's read and act.
19
25
  const stateDir = stateDirName(cwd);
20
- acquireRunLock(cwd, "compile");
21
- try {
22
- saveGraph(cwd, g);
26
+ if (!values["dry-run"]) {
27
+ // HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
28
+ // cannot swap graph.json under an active run between the daemon's read and act.
29
+ acquireRunLock(cwd, "compile");
30
+ try {
31
+ saveGraph(cwd, g);
32
+ }
33
+ finally {
34
+ releaseRunLock(cwd);
35
+ }
23
36
  }
24
- finally {
25
- releaseRunLock(cwd);
26
- }
27
- return `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
37
+ const summary = values["dry-run"]
38
+ ? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
39
+ : `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
40
+ const scopeLints = [...collateralLints(g.tasks, cwd), ...sourceScopeLints(g.tasks, cwd)];
41
+ const diagnostics = scopeLints.length
42
+ ? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
43
+ : "";
44
+ return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}`;
28
45
  }
@@ -100,7 +100,7 @@ Outside multi-agent environments, run the loop directly.
100
100
 
101
101
  ### Version preflight
102
102
 
103
- Before \`tickmarkr compile\` or \`tickmarkr run\`: run \`tickmarkr version\`, read \`package.json\` version, and if the binary is older on major.minor, stop and tell the operator to update. Never proceed on hope — stale binaries silently skip daemon gates.
103
+ Before \`tickmarkr compile\` or \`tickmarkr run\`: run \`tickmarkr version\`, read \`package.json\` version, and if the binary is older on major.minor, stop and tell the operator to update. Never proceed on hope — stale binaries silently skip daemon gates. Also verify no run is live before starting one: \`pgrep -f "tickmarkr (run|resume)"\` must be empty — match the process, not one install path (\`dist/cli/index.js\` alone misses global and homebrew installs), and treat a held \`.tickmarkr/graph.lock\` as a live run until its holder pid is proven dead.
104
104
 
105
105
  ### Tip-verify-before-green
106
106
 
@@ -1,2 +1,2 @@
1
1
  import type { WorkerAdapter } from "../../adapters/types.js";
2
- export declare function plan(argv: string[], cwd?: string, adapters?: WorkerAdapter[]): Promise<string>;
2
+ export declare function plan(argv: string[], cwd?: string, adapters?: WorkerAdapter[], harnessFrom?: string | undefined): Promise<string>;
@@ -11,6 +11,7 @@ import { staffLedEvidence } from "../../route/profile.js";
11
11
  import { route, RoutingError } from "../../route/router.js";
12
12
  import { modelId } from "../../gates/review.js";
13
13
  import { loadRoutingProfile } from "../../run/journal.js";
14
+ import { harnessLine, resolveHarness } from "../harness.js";
14
15
  // T4 (v1.50): TTY-only brand pass — the title helper frames the routing table, lint/unroutable
15
16
  // markers carry the attention glyph, section labels dim to chrome (the doctor/status system).
16
17
  // Gated on ttyVisual(): the non-TTY surface returns untouched (byte-pinned, machine-consumable).
@@ -34,7 +35,12 @@ const fleetCanCrossVendorReview = (channels) => {
34
35
  return true;
35
36
  return false;
36
37
  };
37
- export async function plan(argv, cwd = process.cwd(), adapters = allAdapters()) {
38
+ // v1.89 T4: harnessFrom is the resolver's INPUT — a caller (the byte-pinned goldens) fixes the location
39
+ // and keeps this machine's absolute paths out of a fixture. The default is the INVOKED entrypoint,
40
+ // `process.argv[1]`: the bin symlink a global install puts on PATH, which resolves to dist/cli/index.js.
41
+ // It is NOT `import.meta.url` — that names dist/cli/commands/plan.js, an internal module of the harness
42
+ // rather than the harness that was invoked, so the banner would identify the wrong file entirely.
43
+ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(), harnessFrom = process.argv[1]) {
38
44
  // ponytail: hardcoded 24h TTL — promote to config when an operator asks. mtime is the signal because
39
45
  // doctor.json has no probe timestamp and a schema field would break the existing-files compat invariant.
40
46
  const DOCTOR_STALE_MS = 24 * 60 * 60 * 1000;
@@ -62,9 +68,12 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters())
62
68
  let deviations = 0;
63
69
  // v1.51 T4: the mode is never invisible — the header names the resolved mode, its winning
64
70
  // source, and the explore posture; each task row carries a floor-derivation line below.
71
+ // v1.89 T4: the harness names itself ABOVE the routing table — the table is only as trustworthy as the
72
+ // binary that produced it, and version equality cannot tell an installed package from a checkout.
65
73
  const lines = [
66
74
  `tickmarkr plan — dry run (${channels.length} channels available)`,
67
75
  `mode: ${mode.mode} (${source}) · explore ${cfg.routing.explore?.mode ?? "on"}`,
76
+ harnessLine(resolveHarness(harnessFrom)),
68
77
  "",
69
78
  ];
70
79
  const derivation = (shape) => {
@@ -127,7 +136,17 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters())
127
136
  lints.push(...modelLints(cfg, health, adapters, { tty: ttyVisual() })); // health may be pre-v1.5/probeAll-fallback — no-detection branch covers both
128
137
  // v1.54 T3: dead-steering sweep — advisory only, renders with the routing lints, never alters routing.
129
138
  lints.push(...preferEntryLints(cfg, health, overlayPreferShapes(cwd)));
130
- if (cfg.review.required && channels.length && !fleetCanCrossVendorReview(channels)) {
139
+ // The guard here was `channels.length && !fleetCanCrossVendorReview(channels)`, so the review lint went
140
+ // SILENT at zero channels — the first-run state, where review.required is set and nothing can route at all.
141
+ // Deleting that clause outright is worse than the silence: it fires "no cross-vendor reviewer pair in fleet"
142
+ // when there IS no fleet. An empty fleet and a single-vendor fleet are different faults with different
143
+ // repairs — install or auth ANY adapter vs. add a second VENDOR — so they are two lints, not one clause.
144
+ // The pair text stays byte-identical: tests/fixtures/brand-surfaces/plan-lints.txt pins it and this task
145
+ // does not own that fixture, so the pair lint keeps naming the waiver as the repair it can offer inline.
146
+ if (cfg.review.required && channels.length === 0) {
147
+ lints.push("review: review.required is set but no channel can route — the fleet is empty; install or authenticate an adapter");
148
+ }
149
+ else if (cfg.review.required && !fleetCanCrossVendorReview(channels)) {
131
150
  lints.push("review: no cross-vendor reviewer pair in fleet — set review.required: false to waive");
132
151
  }
133
152
  let cost = 0;
@@ -9,6 +9,7 @@ import { compareRuns } from "../../report/compare.js";
9
9
  import { estimateCosts } from "../../report/cost.js";
10
10
  import { cellsOf, cellSummary } from "../../route/profile.js";
11
11
  import { Journal, loadRoutingProfile } from "../../run/journal.js";
12
+ import { deriveRunCockpitData } from "../../tui/cockpit/derive.js";
12
13
  const n = (x) => x.toLocaleString("en-US"); // explicit locale — CI/darwin flake guard
13
14
  const EM = "—";
14
15
  // TokenUsage fields that are actually present — filtered, never coalesced to zero (absent ⇒ unmetered).
@@ -144,6 +145,53 @@ const outcomeFor = (events, taskId, runEnd) => {
144
145
  }
145
146
  return "not recorded";
146
147
  };
148
+ const tipStatusItem = (runId, events) => {
149
+ try {
150
+ return deriveRunCockpitData({ fileName: runId, raw: events.map((e) => JSON.stringify(e)).join("\n") }, "", // binaryVersion — unread here; the tip-verify status item is all this asks for
151
+ { isDaemonAlive: () => false }).statusItems.find((item) => item.text.startsWith("tip-verify "));
152
+ }
153
+ catch {
154
+ // A capture the cockpit refuses (empty, or no run-start) verified nothing. Absent, never a pass.
155
+ return undefined;
156
+ }
157
+ };
158
+ /**
159
+ * The events the cockpit's verdict speaks for: a verification cycle ends at the `run-end` that
160
+ * closes it (derive.ts tipVerificationPassed slices to `lastRunEnd + 1`), so verify events a later
161
+ * resume appended belong to a cycle nothing has closed. Both readings below take THIS one slice, so
162
+ * the record can never name a cache that belongs to a cycle the state never judged.
163
+ */
164
+ const closedCycle = (events) => {
165
+ for (let i = events.length - 1; i >= 0; i--) {
166
+ if (events[i].event === "run-end")
167
+ return events.slice(0, i + 1);
168
+ }
169
+ return events;
170
+ };
171
+ const verificationOf = (runId, events) => {
172
+ const cycle = closedCycle(events);
173
+ const tip = tipStatusItem(runId, cycle);
174
+ if (tip?.state === "fail")
175
+ return "failed";
176
+ if (tip?.state !== "pass")
177
+ return "absent";
178
+ // A cached green is a verified green of an EARLIER tip (daemon.ts verifyIntegrationTipCached
179
+ // stamps `cached` on every gate it replays), so the record names it instead of folding it into a
180
+ // plain pass. Still no second window: the cockpit reports "pass" only when the closed cycle
181
+ // carried verdicts, so the last `tip-verify` IN THAT CYCLE is the one it passed on.
182
+ const last = [...cycle].reverse().find((e) => e.event === "tip-verify");
183
+ return last?.data.cached === true ? "cached" : "passed";
184
+ };
185
+ // Absent is a state to render, never a zero to invent — and no reading here calls a run green.
186
+ // NOTE: every reading below is byte-pinned by tests/fixtures/brand-surfaces/report-md.md — the
187
+ // golden freezes the WHOLE markdown record, so changing a reading (or the line that renders it)
188
+ // means regenerating that fixture in the same commit.
189
+ const VERIFICATION_READING = {
190
+ passed: "passed — tickmarkr verified this run's integration tip",
191
+ cached: "cached — carried forward from an earlier verified tip, not re-run for this one",
192
+ failed: "FAILED — the run did not verify its own tip",
193
+ absent: "absent — no tip verification recorded: neither passed nor failed",
194
+ };
147
195
  // VIS-07 / REC-01: derived only from the run journal, telemetry, and local configuration.
148
196
  export function renderMarkdownRecord(runId, events, prices = [], rows = []) {
149
197
  const runStart = events.find((e) => e.event === "run-start");
@@ -180,6 +228,7 @@ export function renderMarkdownRecord(runId, events, prices = [], rows = []) {
180
228
  `- **done:** ${count("done")}`,
181
229
  `- **failed:** ${count("failed")}`,
182
230
  `- **human:** ${count("human")}`,
231
+ `- **verification:** ${VERIFICATION_READING[verificationOf(runId, events)]}`,
183
232
  "",
184
233
  "## Usage & efficiency",
185
234
  "",
@@ -1,5 +1,6 @@
1
1
  import { loadConfig } from "../../config/config.js";
2
2
  import { pickDriver } from "../../drivers/index.js";
3
+ import { loadGraph } from "../../graph/graph.js";
3
4
  import { formatSummary, runDaemon } from "../../run/daemon.js";
4
5
  import { formatJournalNarration } from "../../run/journal.js";
5
6
  import { denyPreferCollisionLine, denyPreferCollisions } from "../../route/preference.js";
@@ -15,7 +16,12 @@ export async function resume(argv, cwd = process.cwd()) {
15
16
  const graphChanged = argv.includes("--graph-changed");
16
17
  const retryFailed = argv.includes("--retry-failed");
17
18
  const cfg = loadConfig(cwd);
18
- const collisions = denyPreferCollisions(cfg);
19
+ // v1.87 T3 (OBS-162, twice-carried workaround): the preflight runs AFTER the graph is read and
20
+ // sees only the shapes the resumed graph carries. A deny∩prefer collision on a shape no resumed
21
+ // task uses is a config fact the run would never resolve — it must not refuse the only
22
+ // crash-recovery path. doctor still walks the whole map.
23
+ const graph = loadGraph(cwd);
24
+ const collisions = denyPreferCollisions(cfg, graph.tasks.map((t) => t.shape));
19
25
  if (collisions.length) {
20
26
  throw new Error(collisions.map(denyPreferCollisionLine).join("; "));
21
27
  }