tickmarkr 2.6.1 → 2.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/README.md +16 -3
  2. package/dist/adapters/catalog-remote.js +89 -47
  3. package/dist/adapters/claude-code.js +9 -6
  4. package/dist/adapters/codex.js +7 -4
  5. package/dist/adapters/prompt.d.ts +1 -0
  6. package/dist/adapters/prompt.js +14 -6
  7. package/dist/adapters/registry.js +3 -3
  8. package/dist/adapters/types.d.ts +12 -4
  9. package/dist/adapters/types.js +6 -0
  10. package/dist/cli/commands/approve.d.ts +11 -4
  11. package/dist/cli/commands/approve.js +82 -27
  12. package/dist/cli/commands/compile.js +13 -3
  13. package/dist/cli/commands/doctor.d.ts +8 -2
  14. package/dist/cli/commands/doctor.js +11 -3
  15. package/dist/cli/commands/fleet.js +87 -11
  16. package/dist/cli/commands/plan.js +13 -8
  17. package/dist/cli/commands/report.d.ts +2 -1
  18. package/dist/cli/commands/report.js +74 -8
  19. package/dist/cli/commands/resume.js +4 -2
  20. package/dist/cli/commands/status.js +43 -20
  21. package/dist/cli/help.d.ts +2 -0
  22. package/dist/cli/help.js +9 -2
  23. package/dist/compile/native.js +7 -0
  24. package/dist/config/config.d.ts +35 -2
  25. package/dist/config/config.js +86 -10
  26. package/dist/config/fleet-overlay.d.ts +13 -2
  27. package/dist/config/fleet-overlay.js +60 -0
  28. package/dist/drivers/herdr.d.ts +12 -0
  29. package/dist/drivers/herdr.js +51 -0
  30. package/dist/drivers/orca.d.ts +35 -2
  31. package/dist/drivers/orca.js +222 -67
  32. package/dist/drivers/types.d.ts +2 -0
  33. package/dist/drivers/types.js +2 -2
  34. package/dist/eval/canary.d.ts +2 -1
  35. package/dist/eval/canary.js +2 -2
  36. package/dist/eval/dispatch.js +1 -0
  37. package/dist/gates/acceptance.d.ts +9 -1
  38. package/dist/gates/acceptance.js +31 -4
  39. package/dist/gates/baseline.d.ts +32 -2
  40. package/dist/gates/baseline.js +111 -24
  41. package/dist/gates/cache.d.ts +8 -0
  42. package/dist/gates/cache.js +12 -2
  43. package/dist/gates/llm.d.ts +11 -4
  44. package/dist/gates/llm.js +40 -21
  45. package/dist/gates/review.d.ts +14 -1
  46. package/dist/gates/review.js +160 -34
  47. package/dist/gates/run-gates.d.ts +56 -4
  48. package/dist/gates/run-gates.js +358 -58
  49. package/dist/gates/test-manifest.d.ts +45 -1
  50. package/dist/gates/test-manifest.js +78 -12
  51. package/dist/graph/schema.d.ts +2 -0
  52. package/dist/graph/schema.js +2 -0
  53. package/dist/plan/scope.js +2 -2
  54. package/dist/route/preference.d.ts +20 -2
  55. package/dist/route/preference.js +48 -13
  56. package/dist/route/router.d.ts +12 -1
  57. package/dist/route/router.js +56 -24
  58. package/dist/run/consult.d.ts +15 -1
  59. package/dist/run/consult.js +18 -7
  60. package/dist/run/daemon.d.ts +38 -2
  61. package/dist/run/daemon.js +895 -192
  62. package/dist/run/git.d.ts +8 -0
  63. package/dist/run/git.js +14 -0
  64. package/dist/run/interactive-seed.d.ts +4 -0
  65. package/dist/run/interactive-seed.js +35 -9
  66. package/dist/run/journal.d.ts +152 -3
  67. package/dist/run/journal.js +551 -50
  68. package/dist/run/lease.d.ts +13 -0
  69. package/dist/run/lease.js +45 -0
  70. package/dist/run/merge.d.ts +3 -1
  71. package/dist/run/merge.js +3 -2
  72. package/dist/run/operator-summary.d.ts +3 -0
  73. package/dist/run/operator-summary.js +3 -1
  74. package/dist/run/protocol.d.ts +46 -1
  75. package/dist/run/protocol.js +14 -2
  76. package/dist/run/receipt-resolver.d.ts +22 -0
  77. package/dist/run/receipt-resolver.js +40 -1
  78. package/dist/run/repair-selection.d.ts +11 -1
  79. package/dist/run/repair-selection.js +17 -9
  80. package/dist/run/supervision.d.ts +7 -1
  81. package/dist/run/supervision.js +5 -2
  82. package/dist/run/wall-budget.d.ts +48 -0
  83. package/dist/run/wall-budget.js +280 -0
  84. package/dist/tui/cockpit/board.js +3 -3
  85. package/dist/tui/cockpit/decision-actions.d.ts +8 -5
  86. package/dist/tui/cockpit/decision-actions.js +55 -32
  87. package/dist/tui/cockpit/derive.js +13 -2
  88. package/dist/tui/cockpit/live-runtime.d.ts +10 -0
  89. package/dist/tui/cockpit/live-runtime.js +50 -3
  90. package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
  91. package/dist/tui/cockpit/run-cockpit.js +27 -2
  92. package/dist/tui/cockpit/run-view.d.ts +9 -2
  93. package/dist/tui/cockpit/run-view.js +66 -9
  94. package/dist/tui/cockpit/setup-cockpit.d.ts +6 -0
  95. package/dist/tui/cockpit/setup-cockpit.js +10 -3
  96. package/dist/tui/ink/fleet-app.d.ts +15 -3
  97. package/dist/tui/ink/fleet-app.js +91 -22
  98. package/package.json +3 -1
  99. package/schema/config.schema.json +825 -0
  100. package/skills/tickmarkr-loop/SKILL.md +15 -3
  101. package/skills/tickmarkr-overseer/SKILL.md +42 -0
  102. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +91 -0
  103. package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
  104. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
  105. package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
@@ -1,5 +1,5 @@
1
1
  import { type EvidenceArtifact, type GateEvidenceReceipt, type ShellReceipt } from "../run/protocol.js";
2
- import type { TickmarkrConfig } from "../config/config.js";
2
+ import { type TickmarkrConfig } from "../config/config.js";
3
3
  import type { AcceptanceItem } from "../graph/schema.js";
4
4
  import { type RunCapacity, type ShellOptions, type ShResult } from "../run/git.js";
5
5
  import type { GateResult } from "./types.js";
@@ -31,8 +31,21 @@ export interface GateEvidenceOptions {
31
31
  write?: (path: string, bytes: Buffer) => void;
32
32
  }
33
33
  export declare const EVIDENCE_TAIL_BYTES: number;
34
+ /** The default per-run quota; `gates.evidenceQuotaBytes` in config overrides it (OBS-1140). */
34
35
  export declare const EVIDENCE_RUN_QUOTA_BYTES: number;
36
+ export type RedactionCounts = NonNullable<GateEvidenceReceipt["redaction"]["counts"]>;
37
+ /**
38
+ * OBS-1139: every category scans the ORIGINAL text on its own; overlapping spans then merge into one
39
+ * span counted once under its highest-priority member — token, assignment, secret environment, benign
40
+ * HOME/TMPDIR — so precedence never depends on which match starts first. Counts only: never a value.
41
+ */
42
+ export declare function classifyGateOutput(text: string, env: NodeJS.ProcessEnv): {
43
+ text: string;
44
+ counts: RedactionCounts;
45
+ };
35
46
  export declare function redactGateOutput(text: string, env: NodeJS.ProcessEnv): string;
47
+ /** Material means bytes were withheld; a benign location substitution withholds nothing. */
48
+ export declare const redactionMaterial: (c: RedactionCounts) => boolean;
36
49
  /** Resolve a snapshot reference against actual bytes: eviction never fabricates a live artifact. */
37
50
  export declare function resolveEvidenceArtifact(root: string, ref: EvidenceArtifact): EvidenceArtifact;
38
51
  export declare function beginGateEvidence(cwd: string, gate: string, command: string, opts?: GateEvidenceOptions, nonce?: string): {
@@ -47,7 +60,7 @@ export declare function beginGateEvidence(cwd: string, gate: string, command: st
47
60
  subjectCommit: string | null;
48
61
  };
49
62
  termination: {
50
- kind: "unknown" | "signal" | "not-started" | "exit" | "timeout";
63
+ kind: "unknown" | "timeout" | "signal" | "not-started" | "exit";
51
64
  exitCode: number | null;
52
65
  signal: string | null;
53
66
  timedOut: boolean | null;
@@ -55,6 +68,12 @@ export declare function beginGateEvidence(cwd: string, gate: string, command: st
55
68
  availability: "available" | "not-started" | "capture-failed" | "killed" | "expired" | "missing";
56
69
  redaction: {
57
70
  material: boolean;
71
+ counts?: {
72
+ token: number;
73
+ assignment: number;
74
+ secretEnv: number;
75
+ benignEnv: number;
76
+ } | undefined;
58
77
  };
59
78
  stdout: {
60
79
  path: string;
@@ -127,9 +146,19 @@ export interface BaselineFileDuration {
127
146
  file: string;
128
147
  durationMs: number;
129
148
  }
149
+ /**
150
+ * OBS-1123: WHICH capture a forgiveness rests on — the commit it measured and when it was published.
151
+ * Stamped by the daemon beside the measurement; absent on every baseline.json written before it, which
152
+ * readers must render as unknown provenance and never date from any other clock.
153
+ */
154
+ export interface BaselineProvenance {
155
+ baseRef: string;
156
+ capturedAt: string;
157
+ }
130
158
  export interface Baseline {
131
159
  /** No command ran because the dependency inventory could not establish isolation. */
132
160
  refusal?: string;
161
+ provenance?: BaselineProvenance;
133
162
  /** Observational capture receipts stay outside the forgiveness-bearing command entries. */
134
163
  evidenceReceipts?: Record<string, GateEvidenceReceipt>;
135
164
  commands: Record<string, BaselineCommand>;
@@ -156,6 +185,7 @@ export declare function fingerprint(output: string): string[];
156
185
  export declare function freshFailures(entry: BaselineCommand | undefined, raw: string): {
157
186
  failing: string[];
158
187
  unreadable: boolean;
188
+ forgiven: string[];
159
189
  };
160
190
  export type PackageManager = "npm" | "pnpm" | "yarn" | "bun";
161
191
  /**
@@ -1,25 +1,81 @@
1
1
  import { createHash, randomUUID } from "node:crypto";
2
2
  import { execFileSync } from "node:child_process";
3
- import { existsSync, readFileSync, mkdirSync, readdirSync, statSync, unlinkSync, writeFileSync, rmdirSync } from "node:fs";
3
+ import { EVICTION_TOMBSTONE_LIMIT, EVICTION_TOMBSTONES_FILE, parseEvictionTombstones } from "../run/receipt-resolver.js";
4
+ import { existsSync, readFileSync, mkdirSync, readdirSync, renameSync, statSync, unlinkSync, writeFileSync, rmdirSync } from "node:fs";
4
5
  import { availableParallelism, loadavg, tmpdir } from "node:os";
5
- import { join } from "node:path";
6
+ import { basename, join } from "node:path";
7
+ import { DEFAULT_EVIDENCE_QUOTA_BYTES } from "../config/config.js";
6
8
  import { dependencyLinkRefusal, DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
7
9
  import { executionSignal } from "../run/execution-budget.js";
8
10
  import { isVitestTestCommand, manifestFileCount } from "./test-manifest.js";
9
11
  export const EVIDENCE_TAIL_BYTES = 16 * 1024;
10
- export const EVIDENCE_RUN_QUOTA_BYTES = 8 * 1024 * 1024;
11
- export function redactGateOutput(text, env) {
12
- // One pass over the original bytes: replacing an environment value must not break a token
13
- // recognizer, and a short environment value must not rewrite the redaction marker itself.
12
+ /** The default per-run quota; `gates.evidenceQuotaBytes` in config overrides it (OBS-1140). */
13
+ export const EVIDENCE_RUN_QUOTA_BYTES = DEFAULT_EVIDENCE_QUOTA_BYTES;
14
+ /** Benign environment locations substituted by name; every other redaction is material. */
15
+ const BENIGN_ENV_KEYS = ["HOME", "TMPDIR"];
16
+ const REDACTION_CATEGORIES = ["token", "assignment", "secretEnv", "benignEnv"];
17
+ /**
18
+ * OBS-1139: every category scans the ORIGINAL text on its own; overlapping spans then merge into one
19
+ * span counted once under its highest-priority member — token, assignment, secret environment, benign
20
+ * HOME/TMPDIR — so precedence never depends on which match starts first. Counts only: never a value.
21
+ */
22
+ export function classifyGateOutput(text, env) {
23
+ // HOME/TMPDIR substitute by name at any length; "/" alone would rewrite every path separator.
24
+ const benign = BENIGN_ENV_KEYS.flatMap(key => { const value = env[key]; return value && value.length > 1 ? [{ value, label: `$${key}` }] : []; });
14
25
  // Short operational values (0, true, vi, test) occur throughout ordinary diagnostics.
15
- // Only secret-named keys justify redacting short values.
16
- const values = [...new Set(Object.entries(env)
17
- .filter(([key, value]) => !!value && (value.length >= 8 || /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL|AUTH/i.test(key)))
18
- .map(([, value]) => value))]
19
- .sort((a, b) => b.length - a.length).map(v => v.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
26
+ // Only secret-named keys justify redacting short values. A secret-named value that happens to equal
27
+ // HOME/TMPDIR stays a secret candidate: precedence decides the span, never candidate pruning.
28
+ const secrets = [...new Set(Object.entries(env)
29
+ .filter(([key, value]) => !!value && !BENIGN_ENV_KEYS.includes(key)
30
+ && (value.length >= 8 || /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL|AUTH/i.test(key)))
31
+ .map(([, value]) => value))];
32
+ const literal = (v) => v.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
20
33
  const tokens = String.raw `\b(?:sk-[A-Za-z0-9_-]{12,}|gh[pousr]_[A-Za-z0-9_]{16,}|github_pat_[A-Za-z0-9_]{16,}|AKIA[A-Z0-9]{16}|eyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+)\b`;
21
34
  const assignments = String.raw `(?:authorization\s*:\s*bearer|(?:api[_-]?key|token|password|secret)\s*[:=])\s*[^\s"']+`;
22
- return text.replace(new RegExp([tokens, assignments, ...values].join("|"), "gi"), "[REDACTED]");
35
+ // Every value scans the original text on its own: an alternation would consume the first match and
36
+ // skip a longer overlapping value's tail, so precedence is resolved on the collected spans instead.
37
+ const scan = (source, rank, label) => [...text.matchAll(new RegExp(source, "gi"))].map(m => ({ start: m.index, end: m.index + m[0].length, rank, label }));
38
+ const spans = [
39
+ ...scan(tokens, 0), ...scan(assignments, 1),
40
+ ...secrets.flatMap(v => scan(literal(v), 2)),
41
+ ...benign.flatMap(b => scan(literal(b.value), 3, b.label)),
42
+ ].sort((a, b) => a.start - b.start || b.end - a.end);
43
+ const merged = [];
44
+ for (const span of spans) {
45
+ const last = merged.at(-1);
46
+ if (!last || span.start >= last.end) {
47
+ merged.push({ ...span });
48
+ continue;
49
+ }
50
+ // C-11 (D-673): a benign value nested in (or identical to) another is one location, named by the containing
51
+ // span, which sorts first. Only a PARTIAL overlap of two different benign values is ambiguous: withhold it.
52
+ if (last.rank === 3 && span.rank === 3 && last.label !== span.label && span.end > last.end)
53
+ last.rank = 2;
54
+ else
55
+ last.rank = Math.min(last.rank, span.rank);
56
+ last.end = Math.max(last.end, span.end);
57
+ }
58
+ const counts = { token: 0, assignment: 0, secretEnv: 0, benignEnv: 0 };
59
+ let out = "", at = 0;
60
+ for (const { start, end, rank, label } of merged) {
61
+ counts[REDACTION_CATEGORIES[rank]]++;
62
+ out += text.slice(at, start) + (rank === 3 ? label : "[REDACTED]");
63
+ at = end;
64
+ }
65
+ return { text: out + text.slice(at), counts };
66
+ }
67
+ export function redactGateOutput(text, env) {
68
+ return classifyGateOutput(text, env).text;
69
+ }
70
+ /** Material means bytes were withheld; a benign location substitution withholds nothing. */
71
+ export const redactionMaterial = (c) => c.token + c.assignment + c.secretEnv > 0;
72
+ function readEvictionTombstones(root) {
73
+ try {
74
+ return parseEvictionTombstones(readFileSync(join(root, EVICTION_TOMBSTONES_FILE)));
75
+ }
76
+ catch {
77
+ return [];
78
+ }
23
79
  }
24
80
  /** Resolve a snapshot reference against actual bytes: eviction never fabricates a live artifact. */
25
81
  export function resolveEvidenceArtifact(root, ref) {
@@ -75,7 +131,12 @@ export function beginGateEvidence(cwd, gate, command, opts = {}, nonce) {
75
131
  exitCode: observed?.exitCode ?? null, signal: observed?.signal ?? null,
76
132
  timedOut: observed ? observed.outcome === "timed-out" : null,
77
133
  };
78
- const clean = [redactGateOutput(stdout, env), redactGateOutput(stderr, env)];
134
+ const classified = [classifyGateOutput(stdout, env), classifyGateOutput(stderr, env)];
135
+ const clean = classified.map(c => c.text);
136
+ const counts = { token: 0, assignment: 0, secretEnv: 0, benignEnv: 0 };
137
+ for (const c of classified)
138
+ for (const k of Object.keys(counts))
139
+ counts[k] += c.counts[k];
79
140
  const refs = clean.map((text, i) => {
80
141
  const bytes = Buffer.from(text);
81
142
  const tail = bytes.subarray(-EVIDENCE_TAIL_BYTES);
@@ -94,19 +155,38 @@ export function beginGateEvidence(cwd, gate, command, opts = {}, nonce) {
94
155
  lock = join(dir, ".write-lock");
95
156
  const quota = opts.quotaBytes !== undefined && Number.isFinite(opts.quotaBytes)
96
157
  ? Math.max(0, Math.floor(opts.quotaBytes)) : EVIDENCE_RUN_QUOTA_BYTES;
158
+ // An empty artifact costs no quota, so it is never evicted: every eviction frees bytes.
97
159
  const files = readdirSync(dir).filter(f => /^[a-zA-Z0-9-]+-(stdout|stderr)\.log$/.test(f)).map(f => ({ path: join(dir, f), stat: statSync(join(dir, f)) }))
98
- .sort((a, b) => a.stat.mtimeMs - b.stat.mtimeMs || a.path.localeCompare(b.path));
160
+ .filter(f => f.stat.size > 0).sort((a, b) => a.stat.mtimeMs - b.stat.mtimeMs || a.path.localeCompare(b.path));
99
161
  let total = files.reduce((sum, f) => sum + f.stat.size, 0);
100
162
  const incoming = refs.reduce((sum, ref) => sum + ref.retainedBytes, 0);
101
- if (incoming > quota)
102
- throw new Error("evidence quota cannot retain this invocation");
103
- while (files.length && total + incoming > quota) {
163
+ // OBS-1140: the quota binds what is already retained too, so a resumed run at zero quota keeps
164
+ // zero bytes. An invocation larger than the whole quota is expired at birth; older artifacts
165
+ // are evicted only as far as the quota itself demands, never for bytes that could not fit.
166
+ const fits = incoming <= quota;
167
+ const evicted = [];
168
+ while (files.length && total + (fits ? incoming : 0) > quota) {
104
169
  const oldest = files.shift();
170
+ const bytes = readFileSync(oldest.path);
105
171
  unlinkSync(oldest.path);
106
172
  total -= oldest.stat.size;
173
+ evicted.push({ path: `gate-evidence/${basename(oldest.path)}`, sha256: createHash("sha256").update(bytes).digest("hex"), retainedBytes: bytes.length });
174
+ }
175
+ if (evicted.length) {
176
+ // Identity-bound (path + hash + length) and bounded: the newest 256 survive a restart.
177
+ const tombstones = join(root, EVICTION_TOMBSTONES_FILE);
178
+ writeFileSync(`${tombstones}.tmp`, JSON.stringify([...readEvictionTombstones(root), ...evicted].slice(-EVICTION_TOMBSTONE_LIMIT)));
179
+ renameSync(`${tombstones}.tmp`, tombstones);
107
180
  }
108
- for (const [i, ref] of refs.entries()) {
109
- (opts.write ?? writeFileSync)(join(root, ref.path), Buffer.from(clean[i]).subarray(-EVIDENCE_TAIL_BYTES));
181
+ if (!fits) {
182
+ availability = "expired";
183
+ for (const ref of refs)
184
+ ref.availability = "expired";
185
+ }
186
+ else {
187
+ for (const [i, ref] of refs.entries()) {
188
+ (opts.write ?? writeFileSync)(join(root, ref.path), Buffer.from(clean[i]).subarray(-EVIDENCE_TAIL_BYTES));
189
+ }
110
190
  }
111
191
  }
112
192
  catch {
@@ -131,7 +211,7 @@ export function beginGateEvidence(cwd, gate, command, opts = {}, nonce) {
131
211
  }
132
212
  return { invocationId, ...(nonce ? { nonce } : {}),
133
213
  subject: { runId: opts.runId ?? "standalone", taskId: opts.taskId ?? null, attempt: opts.attempt ?? null, gate, subjectCommit },
134
- termination, availability, redaction: { material: clean[0] !== stdout || clean[1] !== stderr }, stdout: refs[0], stderr: refs[1] };
214
+ termination, availability, redaction: { material: redactionMaterial(counts), counts }, stdout: refs[0], stderr: refs[1] };
135
215
  },
136
216
  };
137
217
  }
@@ -428,7 +508,9 @@ export function freshFailures(entry, raw) {
428
508
  // OBS-42: diagnostic headings enrich fingerprints but cannot invalidate legacy baselines.
429
509
  const current = fingerprint(raw);
430
510
  const fresh = current.filter((f) => !known.has(f) && (!FAIL_ANCHOR_RE.test(f) || f.startsWith("FAIL ")));
431
- return { failing: fresh.filter((f) => f !== UNRECOGNIZED_FAILURE), unreadable: current.includes(UNRECOGNIZED_FAILURE) };
511
+ // OBS-1123: the baseline-recorded half, named so a reader never has to infer it from the fresh half.
512
+ return { failing: fresh.filter((f) => f !== UNRECOGNIZED_FAILURE), unreadable: current.includes(UNRECOGNIZED_FAILURE),
513
+ forgiven: current.filter((f) => known.has(f)) };
432
514
  }
433
515
  /**
434
516
  * Operator directive 2026-08-12 (D-OBS-10): tickmarkr is package-manager agnostic.
@@ -880,7 +962,10 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
880
962
  // below rather than reading as a verified green. Raise the ceiling by teaching isFailureShaped
881
963
  // that runner's position rule (leading verdict + identifier, or identifier + separator + trailing
882
964
  // verdict); loosening back to vocabulary re-opens OBS-278.
883
- const { failing, unreadable } = freshFailures(entry, raw);
965
+ const { failing, unreadable, forgiven } = freshFailures(entry, raw);
966
+ // OBS-1123: which baseline-recorded fingerprints this verdict carried, and the capture that
967
+ // recorded them. Observational only — nothing below reads it to decide a verdict.
968
+ const provenance = baseline.provenance ? { baselineProvenance: baseline.provenance } : {};
884
969
  // T9: classify the FRESH diff before charging it. The complete runner output can legitimately
885
970
  // contain a baseline-recorded assertion beside a newly introduced infrastructure death; letting
886
971
  // that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
@@ -957,7 +1042,9 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
957
1042
  }
958
1043
  if (failing.length) {
959
1044
  const headlined = headlineDetails(raw, failing);
960
- const meta = { ...headlined.meta, ...(classification ? { classification } : {}) };
1045
+ // A fresh red beside baseline-recorded ones names both halves, so no reader shows the new one as forgiven.
1046
+ const meta = { ...headlined.meta, ...(classification ? { classification } : {}),
1047
+ ...(forgiven.length ? { freshFingerprints: failing, forgivenFingerprints: forgiven, ...provenance } : {}) };
961
1048
  record({ gate: name, pass: false, details: headlined.details, ...(Object.keys(meta).length ? { meta } : {}) });
962
1049
  continue;
963
1050
  }
@@ -981,7 +1068,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
981
1068
  gate: name,
982
1069
  pass: true,
983
1070
  details: `exit ${r.code} but only pre-existing failures (forgiven)${unreadable ? " — no failure shape recognized in this output, so a new failure from this runner is invisible to the baseline gate" : ""}`,
984
- ...(classification ? { meta: { classification } } : {}),
1071
+ meta: { ...(classification ? { classification } : {}), forgivenFingerprints: forgiven, ...provenance },
985
1072
  });
986
1073
  }
987
1074
  return results;
@@ -9,8 +9,15 @@ export declare function lockfileHash(worktree: string): string;
9
9
  export declare function canonicalJson(obj: unknown): string;
10
10
  export declare function baselineIdentity(baseline?: Baseline): string;
11
11
  export declare function getWorktreeTree(worktree: string): Promise<string>;
12
+ /** OBS-635: environment inputs a runner child reads that capacity and lifecycle do not already bind.
13
+ * ponytail: a named list, not the whole env (pids and terminal vars would defeat every reuse); extend
14
+ * it when another variable is shown to change what a runner executes. */
15
+ export declare const RUNNER_ENV_KEYS: readonly ["PATH", "NODE_OPTIONS", "NODE_PATH", "NODE_ENV", "TZ"];
16
+ export declare function runnerInputsHash(env?: NodeJS.ProcessEnv): string;
12
17
  export interface GateEnvironmentInput {
13
18
  nodeRuntime?: string;
19
+ /** the runner inputs hash; defaults to this process's (runnerInputsHash) */
20
+ runner?: string;
14
21
  lockfile?: string;
15
22
  worktree?: string;
16
23
  capacity?: RunCapacity;
@@ -21,6 +28,7 @@ export interface GateEnvironmentInput {
21
28
  }
22
29
  export interface EnvironmentParts {
23
30
  nodeRuntime: string;
31
+ runner: string;
24
32
  lockfile: string;
25
33
  capacity: RunCapacity;
26
34
  selectedSet?: readonly string[];
@@ -77,8 +77,17 @@ export async function getWorktreeTree(worktree) {
77
77
  rmSync(scratch, { recursive: true, force: true });
78
78
  }
79
79
  }
80
+ /** OBS-635: environment inputs a runner child reads that capacity and lifecycle do not already bind.
81
+ * ponytail: a named list, not the whole env (pids and terminal vars would defeat every reuse); extend
82
+ * it when another variable is shown to change what a runner executes. */
83
+ export const RUNNER_ENV_KEYS = ["PATH", "NODE_OPTIONS", "NODE_PATH", "NODE_ENV", "TZ"];
84
+ export function runnerInputsHash(env = process.env) {
85
+ return createHash("sha256").update(canonicalJson(Object.fromEntries(RUNNER_ENV_KEYS.map((k) => [k, env[k] ?? null]))))
86
+ .digest("hex").slice(0, 16);
87
+ }
80
88
  export function environmentFingerprint(env) {
81
89
  const nodeRuntime = env.nodeRuntime ?? process.version;
90
+ const runner = env.runner ?? runnerInputsHash();
82
91
  const lockfile = env.lockfile ?? (env.worktree ? lockfileHash(env.worktree) : "no-lockfile");
83
92
  const cap = env.capacity ?? resolvedCapacity();
84
93
  const capacity = { forkCap: cap.forkCap, cores: cap.cores };
@@ -97,6 +106,7 @@ export function environmentFingerprint(env) {
97
106
  // explicit `false` and an npmrc `false` are the same policy for the child that ran.
98
107
  const payload = canonicalJson({
99
108
  nodeRuntime,
109
+ runner,
100
110
  lockfile,
101
111
  capacity,
102
112
  resolution,
@@ -107,7 +117,7 @@ export function environmentFingerprint(env) {
107
117
  const fingerprint = createHash("sha256").update(payload).digest("hex").slice(0, 16);
108
118
  return {
109
119
  fingerprint,
110
- parts: { nodeRuntime, lockfile, capacity, selectedSet, verification },
120
+ parts: { nodeRuntime, runner, lockfile, capacity, selectedSet, verification },
111
121
  };
112
122
  }
113
123
  export async function computeVerificationIdentity(params) {
@@ -151,7 +161,7 @@ export function verificationIdentityKey(id) {
151
161
  export function formatReusedDetails(originalDetails, id) {
152
162
  const unadorned = originalDetails.replace(/^reused verdict \(identity: [^)]+\):\s*/, "");
153
163
  const envDesc = id.envParts
154
- ? ` [node=${id.envParts.nodeRuntime}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}, protocol=${id.envParts.verification.protocol}, lifecycle=${id.envParts.verification.lifecycle} (${id.envParts.verification.source})]`
164
+ ? ` [node=${id.envParts.nodeRuntime}, runner=${id.envParts.runner}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}, protocol=${id.envParts.verification.protocol}, lifecycle=${id.envParts.verification.lifecycle} (${id.envParts.verification.source})]`
155
165
  : "";
156
166
  const gate = id.gate ?? "gate";
157
167
  const prefix = `reused ${id.scope === "tip" ? "tip " : ""}verdict (identity: gate=${gate} tree=${id.tree}${id.worktree ? ` worktree=${id.worktree}` : ""} command=${id.command} baseline=${id.baseline} env=${id.environment}${envDesc})`;
@@ -1,4 +1,5 @@
1
1
  import { type WorkerAdapter } from "../adapters/types.js";
2
+ import type { Effort } from "../graph/schema.js";
2
3
  import { type ExecutorDriver, type Slot } from "../drivers/types.js";
3
4
  export declare const GATE_PANE_SEP = " \u00B7 ";
4
5
  export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
@@ -68,13 +69,19 @@ export interface LlmRunResult {
68
69
  }
69
70
  export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
70
71
  export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
72
+ /** OBS-1168(b): a judge/review seat whose pane could not be created or launched. Typed so the gates can
73
+ * contain it as seat recovery (another seat) or an infra park — never as a worker failure. */
74
+ export declare class SeatLaunchError extends Error {
75
+ readonly seat: string;
76
+ constructor(seat: string, cause: unknown);
77
+ }
71
78
  export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
72
79
  export declare const REVIEW_SILENT_BYTE_FLOOR = 64;
73
- export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
74
- export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
80
+ export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number, effort?: Effort): Promise<string>;
81
+ export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number, effort?: Effort): Promise<string>;
75
82
  export declare function dewrapPaneVerdict(out: string, nonce: string): string;
76
- export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<LlmRunResult>;
77
- export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
83
+ export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number, effort?: Effort): Promise<LlmRunResult>;
84
+ export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number, effort?: Effort): Promise<string>;
78
85
  export declare function extractJson<T>(raw: string): T | null;
79
86
  /** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
80
87
  export declare function extractVerdictJson<T>(raw: string, nonce: string): T | null;
package/dist/gates/llm.js CHANGED
@@ -81,19 +81,22 @@ export function gatePaneName(role, taskId, suffix = "") {
81
81
  // T2 ownership contract: a canonical owned fallback (the daemon's nameFor now emits one) passes
82
82
  // through untouched; run-gates' "-r1" judge-retry suffix becomes attempt+1 so the retry pane's name
83
83
  // stays contract-parseable (tickmarkr:judge:<task>:1:<runId>) instead of a corrupted-runId shape.
84
+ // C-3 (D-669): the hop suffix is -r<N> — the judge-flake retry is -r1 and the adjudicator -r2, so the two
85
+ // never share an owned name (a retained retry pane must not shadow the adjudicator's slot).
84
86
  export function rolePaneNameFromPrompt(prompt, fallback) {
85
- const retry = fallback.endsWith("-r1");
86
- const base = retry ? fallback.slice(0, -3) : fallback;
87
+ const hop = /-r([1-9])$/.exec(fallback);
88
+ const suffix = hop ? hop[0] : "";
89
+ const base = hop ? fallback.slice(0, -suffix.length) : fallback;
87
90
  const owned = parseOwnedName(base);
88
91
  if (owned)
89
- return retry ? formatOwnedName({ ...owned, attempt: owned.attempt + 1 }) : base;
92
+ return hop ? formatOwnedName({ ...owned, attempt: owned.attempt + Number(hop[1]) }) : base;
90
93
  const id = prompt.match(/## Task ([^\n:]+):/)?.[1];
91
94
  if (!id)
92
95
  return fallback;
93
96
  if (prompt.startsWith("TICKMARKR-JUDGE"))
94
- return gatePaneName("judge", id, retry ? "-r1" : "");
97
+ return gatePaneName("judge", id, suffix);
95
98
  if (prompt.startsWith("TICKMARKR-REVIEW"))
96
- return gatePaneName("review", id, retry ? "-r1" : "");
99
+ return gatePaneName("review", id, suffix);
97
100
  return fallback;
98
101
  }
99
102
  const llmOutputCapture = new AsyncLocalStorage();
@@ -299,17 +302,27 @@ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
299
302
  // often the seat's first byte than the harness's exit marker, and the next read completes either.
300
303
  return trailer ? seat.slice(0, trailer.index) : seat;
301
304
  }
305
+ /** OBS-1168(b): a judge/review seat whose pane could not be created or launched. Typed so the gates can
306
+ * contain it as seat recovery (another seat) or an infra park — never as a worker failure. */
307
+ export class SeatLaunchError extends Error {
308
+ seat;
309
+ constructor(seat, cause) {
310
+ super(`seat ${seat} failed to launch: ${cause instanceof Error ? cause.message : String(cause)}`);
311
+ this.seat = seat;
312
+ this.name = "SeatLaunchError";
313
+ }
314
+ }
302
315
  export const REVIEW_FIRST_LIVENESS_MS = 30_000;
303
316
  // OBS-1039: a seat that wrote ten bytes and went quiet escaped the zero-byte beat and sat to the
304
317
  // ceiling. Below this many seat-authored bytes at the first beat the seat is `silent` — demoted and
305
318
  // re-routed then, not at the ceiling. Pane path only; a headless runner buffers and keeps its ceiling.
306
319
  export const REVIEW_SILENT_BYTE_FLOOR = 64;
307
- async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
320
+ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000, effort) {
308
321
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
309
322
  try {
310
323
  const pf = join(dir, "prompt.md");
311
324
  writeFileSync(pf, prompt);
312
- const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
325
+ const r = await sh(adapter.headlessCommand(pf, model, effort), cwd, timeoutMs);
313
326
  const output = r.stdout + "\n" + r.stderr;
314
327
  const nonce = extractPromptNonce(prompt) ?? "";
315
328
  return { output, exitCode: r.code, timedOut: r.timedOut === true,
@@ -319,12 +332,12 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
319
332
  rmSync(dir, { recursive: true, force: true });
320
333
  }
321
334
  }
322
- export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
323
- return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs)).output;
335
+ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000, effort) {
336
+ return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs, effort)).output;
324
337
  }
325
338
  // v1.1 default path: the same headless CLI call, but dispatched through the driver
326
339
  // as a visible named agent (herdr pane), with the quote-split completion wrapper.
327
- async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
340
+ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
328
341
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
329
342
  let slot;
330
343
  let accountant;
@@ -338,12 +351,18 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
338
351
  writeFileSync(scriptPath, [
339
352
  "export BASH_SILENCE_DEPRECATION_WARNING=1",
340
353
  bannerShell(),
341
- adapter.headlessCommand(pf, model),
354
+ adapter.headlessCommand(pf, model, effort),
342
355
  gateExitTrailer(nonce),
343
356
  ].join("\n"));
344
- slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
345
- via.onSlot?.(slot);
346
- await via.driver.run(slot, paneDispatchCommand(scriptPath));
357
+ try {
358
+ slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
359
+ via.onSlot?.(slot);
360
+ await via.driver.run(slot, paneDispatchCommand(scriptPath));
361
+ }
362
+ catch (error) {
363
+ forceClose = true; // a half-launched pane is closed, never kept
364
+ throw new SeatLaunchError(`${adapter.id}:${model}`, error);
365
+ }
347
366
  if (via.driver.sendKey) {
348
367
  try {
349
368
  if (matchesTrustDialog(await via.driver.read(slot, 400), adapter.trustDialog)) {
@@ -487,8 +506,8 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
487
506
  }
488
507
  }
489
508
  }
490
- export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
491
- return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
509
+ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
510
+ return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)).output;
492
511
  }
493
512
  // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
494
513
  // continuation indent, splitting words mid-token — so literal newlines land inside JSON string
@@ -549,15 +568,15 @@ export function dewrapPaneVerdict(out, nonce) {
549
568
  }
550
569
  return out;
551
570
  }
552
- export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
571
+ export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
553
572
  const result = await (via
554
- ? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)
555
- : runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs));
573
+ ? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)
574
+ : runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs, effort));
556
575
  llmOutputCapture.getStore()?.push(result.output);
557
576
  return result;
558
577
  }
559
- export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
560
- return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
578
+ export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
579
+ return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)).output;
561
580
  }
562
581
  export function extractJson(raw) {
563
582
  const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
@@ -51,6 +51,11 @@ export declare function fetchTaskDiff(worktree: string, baseRef: string, files?:
51
51
  export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
52
52
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
53
53
  export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
54
+ export declare function isGarbageReview(result: GateResult): result is GateResult & {
55
+ meta: {
56
+ reviewer: string;
57
+ };
58
+ };
54
59
  export declare function isDiffCapPark(result: GateResult): boolean;
55
60
  export declare function diffCapParkReason(results: GateResult[]): string | null;
56
61
  export declare function modelId(model: string): string;
@@ -111,7 +116,7 @@ prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, ne
111
116
  floor?: Tier, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
112
117
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
113
118
  onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>, excludeVendors?: ReadonlySet<string>, authors?: readonly string[]): BillingChannel | null;
114
- export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch";
119
+ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch" | "seat-launch-failed";
115
120
  /**
116
121
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
117
122
  * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
@@ -135,4 +140,12 @@ export declare function renderGoalSection(goal: string, repoRoot?: string): stri
135
140
  * closure on a typo and that read as malformed — the block is what a closure list is copied from.
136
141
  */
137
142
  export declare function renderPriorMaterials(priorMaterials: readonly StructuredFinding[]): string;
143
+ /**
144
+ * OBS-1052(2): a seat that lost the top of a long brief, or believed it had already filed its review,
145
+ * answered in prose — and prose is no verdict. So the requirement, naming THIS call's nonce with a
146
+ * valid example, is both the first and the last instruction of the brief. It is best-effort wording:
147
+ * the parser stays the authority, and nothing here reads approval out of prose.
148
+ */
149
+ export declare function reviewResponseExample(nonce: string): string;
150
+ export declare function reviewResponseRequirement(nonce: string): string;
138
151
  export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[], carriedAuthors?: readonly string[], operatorContext?: string): Promise<GateResult>;