tickmarkr 1.62.0 → 1.64.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -103,14 +103,27 @@ const daemonPid = (events) => {
103
103
  }
104
104
  return undefined;
105
105
  };
106
+ const terminalFailureCause = (events) => {
107
+ for (let i = events.length - 1; i >= 0; i--) {
108
+ const e = events[i];
109
+ if (e.event !== "run-end" || typeof e.data.error !== "string")
110
+ continue;
111
+ const phase = typeof e.data.phase === "string" && e.data.phase.trim()
112
+ ? `${e.data.phase.trim()} failed`
113
+ : "run failed";
114
+ return `${phase}: ${e.data.error.replace(/\s+/g, " ").trim()}`;
115
+ }
116
+ return undefined;
117
+ };
106
118
  const liveness = (events) => {
107
119
  const last = events.at(-1);
108
120
  if (!last)
109
121
  return "last event unknown · daemon pid unknown";
110
122
  const age = fmtAge(Date.now() - Date.parse(last.ts));
111
123
  const pid = daemonPid(events);
124
+ const cause = terminalFailureCause(events);
112
125
  if (pid === undefined)
113
- return `last event ${age} ago · daemon pid unknown`;
126
+ return `last event ${age} ago · daemon pid unknown${cause ? ` · ${cause}` : ""}`;
114
127
  // a dead pid after run-end is a clean exit, not a crash — "dead" is only alarming (red) while
115
128
  // the run is still incomplete (operator's crash indicator)
116
129
  const ended = events.some((e) => e.event === "run-end");
@@ -122,7 +135,7 @@ const liveness = (events) => {
122
135
  catch (k) {
123
136
  state = k.code === "ESRCH" ? (ended ? "finished" : "dead") : "alive";
124
137
  } // EPERM ⇒ alive
125
- return `last event ${age} ago · daemon pid ${pid} ${state}`;
138
+ return `last event ${age} ago · daemon pid ${pid} ${state}${cause ? ` · ${cause}` : ""}`;
126
139
  };
127
140
  const renderFrame = (cwd) => {
128
141
  const g = loadGraph(cwd);
@@ -21,6 +21,10 @@ export function compileNative(file) {
21
21
  if (!existsSync(file))
22
22
  throw new CompileError(`no such native spec file: ${file}`);
23
23
  const content = readFileSync(file, "utf8");
24
+ // Exact shared-template check: init writes specTemplate(), and any edit changes the body.
25
+ if (content === specTemplate()) {
26
+ throw new CompileError(`${file} is the unedited tickmarkr init scaffold. Edit ${file} before compiling.`);
27
+ }
24
28
  const drafts = [];
25
29
  let plainCount = 0; // v1.19: plain-string acceptance items compiled as judge oracles (compat) — warn once
26
30
  let specMode;
@@ -8,6 +8,7 @@ export interface JudgeVerdict {
8
8
  criterion: string;
9
9
  met: boolean;
10
10
  reason: string;
11
+ evidence: string;
11
12
  }>;
12
13
  }
13
14
  export declare function judgeCriterionId(index: number): string;
@@ -4,13 +4,16 @@ import { DEFAULT_DIFF_CAP } from "../config/config.js";
4
4
  import { renderAcceptanceItem } from "../graph/schema.js";
5
5
  import { sh } from "../run/git.js";
6
6
  import { checkDiffCap, fetchTaskDiff } from "./review.js";
7
- import { extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
7
+ import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
8
8
  // Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
9
9
  const JUDGE_TIMEOUT_MS = 900_000;
10
10
  const JudgeVerdictRowSchema = z.object({
11
11
  criterion: z.string(),
12
12
  met: z.boolean(),
13
13
  reason: z.string(),
14
+ // v1.64: a verbatim quote from the judged diff grounding the ruling — required; a verdict
15
+ // omitting it is malformed and fails closed like any other shape violation.
16
+ evidence: z.string(),
14
17
  });
15
18
  const JudgeVerdictSchema = z.object({
16
19
  pass: z.boolean(),
@@ -157,6 +160,8 @@ You are a strict acceptance judge. Decide whether the diff satisfies EVERY accep
157
160
  Judge only what the diff proves — plausible-but-wrong must fail. Do not award partial credit.
158
161
  Deterministic command/test oracles have already passed mechanically; judge ONLY the rubric items below.
159
162
 
163
+ ${COMPLETION_FAKING_CHECKLIST}
164
+
160
165
  ## Task ${task.id}: ${task.title}
161
166
  Goal: ${task.goal}
162
167
 
@@ -171,8 +176,9 @@ ${diff}
171
176
  ${verdictNonceLine(nonce)}
172
177
 
173
178
  Respond with ONLY this JSON (no prose before or after):
174
- {"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "..."}]}
179
+ {"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": "..."}]}
175
180
  Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) exactly once.
181
+ Each criteria[].evidence MUST be a short verbatim quote copied from the diff above that grounds the ruling; a quote not found in the diff voids the whole verdict.
176
182
  `;
177
183
  const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
178
184
  const extracted = extractVerdictJson(raw, nonce);
@@ -185,6 +191,15 @@ Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) ex
185
191
  meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
186
192
  }
187
193
  const { verdict: v, inconsistencies } = checkJudgeVerdict(extracted, expectedIds);
194
+ // v1.64: quoted evidence must appear verbatim in `diff` — the exact string embedded in the prompt
195
+ // above, never the worktree or any other artifact. A quote the diff doesn't contain is a
196
+ // hallucinated verdict: treated as unparseable so GATE-09 retries the judge on a failover channel.
197
+ const fabricated = v.criteria.filter((row) => !row.evidence.trim() || !diff.includes(row.evidence));
198
+ if (fabricated.length) {
199
+ return { gate: "acceptance", pass: false,
200
+ details: warn + detBlock + `judge verdict quotes evidence absent from the judged diff (${fabricated.map((row) => row.criterion).join(", ")}) — treating as unparseable, failing closed`,
201
+ meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
202
+ }
188
203
  const pass = v.pass === true && inconsistencies.length === 0 && v.criteria.every((row) => row.met);
189
204
  const lines = v.criteria.map((row) => `${row.met ? "✓" : "✗"} ${row.criterion}: ${row.reason}`);
190
205
  if (!v.pass)
@@ -1,12 +1,30 @@
1
1
  import type { TickmarkrConfig } from "../config/config.js";
2
+ import type { AcceptanceItem } from "../graph/schema.js";
2
3
  import type { GateResult } from "./types.js";
3
4
  export interface Baseline {
4
5
  commands: Record<string, {
5
6
  exitCode: number;
6
7
  fingerprints: string[];
8
+ missingCommand?: boolean;
7
9
  }>;
10
+ warnings?: BaselineWarning[];
11
+ }
12
+ export interface BaselineWarning {
13
+ kind: "wrong-environment";
14
+ commands: string[];
15
+ reason: string;
8
16
  }
9
17
  export declare function fingerprint(output: string): string[];
10
18
  export declare function detectGateCommands(repoRoot: string, cfg: TickmarkrConfig): Record<string, string>;
11
19
  export declare function captureBaseline(cwd: string, commands: Record<string, string>): Promise<Baseline>;
20
+ export interface VacuousOracleWarning {
21
+ kind: "vacuous-oracle";
22
+ taskId: string;
23
+ oracles: string[];
24
+ reason: string;
25
+ }
26
+ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
27
+ id: string;
28
+ acceptance: AcceptanceItem[];
29
+ }>): Promise<VacuousOracleWarning[]>;
12
30
  export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[]): Promise<GateResult[]>;
@@ -41,15 +41,74 @@ export function detectGateCommands(repoRoot, cfg) {
41
41
  }
42
42
  return out;
43
43
  }
44
+ const shellToken = (cmd) => {
45
+ for (const raw of cmd.trim().split(/\s+/)) {
46
+ if (!raw || /^[A-Za-z_][A-Za-z0-9_]*=/.test(raw))
47
+ continue;
48
+ if (raw === "env")
49
+ continue;
50
+ return raw.replace(/^['"]|['"]$/g, "");
51
+ }
52
+ return undefined;
53
+ };
54
+ const reEscape = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
55
+ function missingConfiguredCommand(cmd, result) {
56
+ if (result.code !== 127)
57
+ return false;
58
+ const token = shellToken(cmd);
59
+ if (!token)
60
+ return false;
61
+ const output = `${result.stdout}\n${result.stderr}`;
62
+ return new RegExp(`(?:^|[:\\s])${reEscape(token)}:\\s+(?:command not found|No such file or directory)`, "i").test(output);
63
+ }
44
64
  export async function captureBaseline(cwd, commands) {
45
65
  const base = { commands: {} };
46
66
  for (const [name, cmd] of Object.entries(commands)) {
47
67
  const r = await sh(cmd, cwd);
48
68
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
49
- base.commands[name] = { exitCode: r.code, fingerprints: fingerprint((r.stdout + "\n" + r.stderr).split(cwd).join("")) };
69
+ base.commands[name] = {
70
+ exitCode: r.code,
71
+ fingerprints: fingerprint((r.stdout + "\n" + r.stderr).split(cwd).join("")),
72
+ missingCommand: missingConfiguredCommand(cmd, r),
73
+ };
74
+ }
75
+ const names = Object.keys(commands);
76
+ const missing = names.filter((name) => base.commands[name]?.missingCommand === true);
77
+ if (names.length > 0 && missing.length === names.length) {
78
+ base.warnings = [{
79
+ kind: "wrong-environment",
80
+ commands: missing,
81
+ reason: `wrong environment: every configured baseline command was missing (${missing.join(", ")})`,
82
+ }];
50
83
  }
51
84
  return base;
52
85
  }
86
+ // Tier A #3 (2026-07-21 repo-scan reconciliation): a command oracle that already exits 0 before any
87
+ // work exists cannot falsify the work — surface it at baseline capture. Observational only: journaled
88
+ // warning, never a gate input, and an oracle that fails at baseline changes nothing. Judge oracles
89
+ // (including plain-string compat judges) are never executed; test oracles stay gate-only (they need
90
+ // the detected runner and the worker's diff to mean anything).
91
+ export async function detectVacuousOracles(cwd, tasks) {
92
+ const out = [];
93
+ for (const t of tasks) {
94
+ const vacuous = [];
95
+ for (const a of t.acceptance) {
96
+ if (typeof a !== "object" || a.oracle !== "command")
97
+ continue;
98
+ if ((await sh(a.command, cwd)).code === 0)
99
+ vacuous.push(a.command);
100
+ }
101
+ if (vacuous.length) {
102
+ out.push({
103
+ kind: "vacuous-oracle",
104
+ taskId: t.id,
105
+ oracles: vacuous,
106
+ reason: `vacuous acceptance oracle on ${t.id}: already passes before any work exists — ${vacuous.map((c) => `$ ${c}`).join("; ")}`,
107
+ });
108
+ }
109
+ }
110
+ return out;
111
+ }
53
112
  // HYG-08 (D-01): headline the runner's own failure naming; demote the fingerprint diff to a secondary
54
113
  // section. Extracts from `raw` — the SAME cwd-stripped, per-line ANSI_RE-stripped string that was
55
114
  // fingerprinted, digits UN-normalized (Pitfall 2: normalization mangles test names, and the diff set could
@@ -1,12 +1,13 @@
1
1
  import type { WorkerAdapter } from "../adapters/types.js";
2
2
  import { type ExecutorDriver, type Slot } from "../drivers/types.js";
3
3
  export declare const GATE_PANE_SEP = " \u00B7 ";
4
+ export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
4
5
  /** Fable F3: per-call nonce echoed in verdict JSON and gate exit markers. */
5
6
  export declare function generateVerdictNonce(): string;
6
7
  export declare function verdictNonceLine(nonce: string): string;
7
8
  export declare function extractPromptNonce(prompt: string): string | null;
8
9
  export declare function gateExitTrailer(nonce: string): string;
9
- export declare function augmentFakeVerdictOutput(adapter: WorkerAdapter, out: string, nonce: string): string;
10
+ export declare function augmentFakeVerdictOutput(adapter: WorkerAdapter, out: string, nonce: string, prompt?: string): string;
10
11
  export type GatePaneRole = "judge" | "review" | "consult";
11
12
  /** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
12
13
  export declare function gatePaneName(role: GatePaneRole, taskId: string, suffix?: string): string;
package/dist/gates/llm.js CHANGED
@@ -6,6 +6,22 @@ import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
6
6
  import { bannerShell, paneDispatchCommand } from "../brand.js";
7
7
  import { sh } from "../run/git.js";
8
8
  export const GATE_PANE_SEP = " · ";
9
+ // v1.64 gate-integrity (repo-scan Tier A·1): the concrete completion-faking shortcuts every
10
+ // judge/review verdict must hunt for. Shared verbatim by the acceptance judge and review prompts.
11
+ export const COMPLETION_FAKING_CHECKLIST = `## Completion-faking checklist
12
+ Hunt for these concrete completion-faking shortcuts before ruling on any criterion:
13
+ - hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic
14
+ - test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green
15
+ - vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)
16
+ - fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior
17
+ - echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself
18
+ - stub-left-behind: TODO, throw, or no-op stub where the real implementation should be
19
+ - error-swallowing: catch or fallback that hides failures instead of handling them
20
+ - self-mocking: the code under test mocked or faked so the test exercises the mock
21
+ - check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green
22
+ - rename-as-work: code moved or renamed and presented as the requested change
23
+ - scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched
24
+ When a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.`;
9
25
  /** Fable F3: per-call nonce echoed in verdict JSON and gate exit markers. */
10
26
  export function generateVerdictNonce() {
11
27
  return randomBytes(4).toString("hex");
@@ -19,8 +35,20 @@ export function extractPromptNonce(prompt) {
19
35
  export function gateExitTrailer(nonce) {
20
36
  return `printf '\\nTICKMARKR_''EXIT_${nonce}:%s\\n' $?`;
21
37
  }
38
+ // v1.64: scripted fake judge verdicts predate the required per-criterion evidence field — quote the
39
+ // first line of the prompt's own diff block into rows lacking one so zero-token fixtures keep their
40
+ // outcomes. Rows scripting an explicit evidence value pass through verbatim (tests exercise both paths).
41
+ function injectFakeEvidence(obj, prompt) {
42
+ if (!prompt.startsWith("TICKMARKR-JUDGE") || !Array.isArray(obj.criteria))
43
+ return obj;
44
+ const line = /```diff\n([\s\S]*?)```/.exec(prompt)?.[1].split("\n").find((l) => l.trim());
45
+ if (!line)
46
+ return obj;
47
+ const criteria = obj.criteria.map((row) => row && typeof row === "object" && !("evidence" in row) ? { ...row, evidence: line } : row);
48
+ return { ...obj, criteria };
49
+ }
22
50
  // ponytail: fake adapter serves static verdict JSON without nonce; append a bound copy for zero-token tests.
23
- export function augmentFakeVerdictOutput(adapter, out, nonce) {
51
+ export function augmentFakeVerdictOutput(adapter, out, nonce, prompt = "") {
24
52
  if (adapter.id !== "fake")
25
53
  return out;
26
54
  const obj = extractJson(out);
@@ -28,7 +56,7 @@ export function augmentFakeVerdictOutput(adapter, out, nonce) {
28
56
  return out;
29
57
  if (typeof obj.nonce === "string")
30
58
  return out;
31
- return `${out}\n${JSON.stringify({ ...obj, nonce })}`;
59
+ return `${out}\n${JSON.stringify(injectFakeEvidence({ ...obj, nonce }, prompt))}`;
32
60
  }
33
61
  /** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
34
62
  export function gatePaneName(role, taskId, suffix = "") {
@@ -61,7 +89,7 @@ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 30000
61
89
  const nonce = extractPromptNonce(prompt);
62
90
  let out = r.stdout + "\n" + r.stderr;
63
91
  if (nonce)
64
- out = augmentFakeVerdictOutput(adapter, out, nonce);
92
+ out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
65
93
  return out;
66
94
  }
67
95
  // v1.1 default path: the same headless CLI call, but dispatched through the driver
@@ -88,7 +116,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
88
116
  let out = await via.driver.read(slot, 400);
89
117
  if (!via.keep)
90
118
  await via.driver.close(slot);
91
- out = augmentFakeVerdictOutput(adapter, out, nonce);
119
+ out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
92
120
  return out;
93
121
  }
94
122
  export function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
@@ -4,7 +4,7 @@ import { renderAcceptanceItem } from "../graph/schema.js";
4
4
  import { getAdapter } from "../adapters/registry.js";
5
5
  import { shOk } from "../run/git.js";
6
6
  import { marginalCostRank } from "../route/router.js";
7
- import { extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
7
+ import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
8
8
  // OBS-48: cap on zero-context diff bytes (git diff -U0), not context-padded full diff — scattered
9
9
  // one-line hunks no longer trip at ~370 diff-bytes per changed line. Full diff still goes to the judge.
10
10
  const DIFF_CAP_REMEDY = "split the task, or raise gates.diffCap";
@@ -82,6 +82,8 @@ export async function reviewGate(task, worktree, baseRef, author, channels, adap
82
82
  You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
83
83
  Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
84
84
 
85
+ ${COMPLETION_FAKING_CHECKLIST}
86
+
85
87
  ## Task ${task.id}: ${task.title} (complexity ${task.complexity})
86
88
  ## Acceptance criteria
87
89
  ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
@@ -4,6 +4,7 @@ import { getAdapter } from "../adapters/registry.js";
4
4
  import { bannerShell, paneDispatchCommand } from "../brand.js";
5
5
  import { augmentFakeVerdictOutput, extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
6
6
  import { sh } from "./git.js";
7
+ import { redactSecrets } from "./redact.js";
7
8
  const MAX_RETRY_GUIDANCE_LINES = 10;
8
9
  function guidanceParts(text) {
9
10
  return text.split(/\n+/).flatMap((line) => line.split(/(?<=[.!?])\s+/)).map((s) => s.trim()).filter(Boolean);
@@ -100,7 +101,9 @@ opts = {}) {
100
101
  const dir = join(runDir, "consults");
101
102
  mkdirSync(dir, { recursive: true });
102
103
  const promptFile = join(dir, `${d.taskId}-${n}.md`);
103
- writeFileSync(promptFile, buildDossierPrompt(d, nonce));
104
+ // T3 secret redaction: the persisted dossier artifact (transcript/diff/journal tail — also what the
105
+ // consult model reads) is masked at this seam; the in-memory Dossier stays untouched.
106
+ writeFileSync(promptFile, redactSecrets(buildDossierPrompt(d, nonce)));
104
107
  // One seat = the WHOLE invoke-and-parse unit, both visibility branches (OBS-69 class: a headless-only
105
108
  // failover would leave the production pane path hard-failing on seat one). null = no parseable verdict.
106
109
  const invokeSeat = async (seatAdapter, seatModel, seatIdx) => {
@@ -11,7 +11,7 @@ import { bannerShell, paneDispatchCommand } from "../brand.js";
11
11
  import { globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
12
12
  import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
13
13
  import { formatOwnedName } from "../drivers/types.js";
14
- import { captureBaseline, detectGateCommands } from "../gates/baseline.js";
14
+ import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
15
15
  import { runGates } from "../gates/run-gates.js";
16
16
  import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus } from "../graph/graph.js";
17
17
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
@@ -109,6 +109,10 @@ export async function runDaemon(repoRoot, opts = {}) {
109
109
  // v1.54 T2: declared before the try so the finally can always deregister (a throw before
110
110
  // registration leaves it undefined — the guard below covers that path).
111
111
  let onTermination;
112
+ let journal;
113
+ let runStarted = false;
114
+ let taskLoopStarted = false;
115
+ let branch = "";
112
116
  try {
113
117
  let graph = loadGraph(repoRoot);
114
118
  // v1.51 T2: the routing mode resolves BEFORE any routing input is built — run flag > spec front-matter
@@ -184,12 +188,12 @@ export async function runDaemon(repoRoot, opts = {}) {
184
188
  };
185
189
  process.on("SIGINT", onTermination);
186
190
  process.on("SIGTERM", onTermination);
187
- const journal = opts.resume ? Journal.open(repoRoot, runId, opts.narrate) : Journal.create(repoRoot, runId, opts.narrate);
191
+ journal = opts.resume ? Journal.open(repoRoot, runId, opts.narrate) : Journal.create(repoRoot, runId, opts.narrate);
188
192
  const branchEvent = opts.resume
189
193
  ? [...journal.read()].reverse().find((e) => (e.event === "run-start" || e.event === "run-end" || e.event === "merge") && typeof e.data.branch === "string")
190
194
  : undefined;
191
195
  const recordedBranch = typeof branchEvent?.data.branch === "string" ? branchEvent.data.branch : undefined;
192
- const branch = recordedBranch
196
+ branch = recordedBranch
193
197
  ? branchEvent.event === "merge" ? recordedBranch.slice(0, recordedBranch.lastIndexOf("--")) : recordedBranch
194
198
  : integrationBranch(cfg, runId);
195
199
  if (lock.reclaimed)
@@ -249,15 +253,27 @@ export async function runDaemon(repoRoot, opts = {}) {
249
253
  baseline = await captureBaseline(repoRoot, commands);
250
254
  writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(baseline, null, 2));
251
255
  journal.append("run-start", undefined, { pid: process.pid, baseRef, commands, channels: channels.map(channelKey), branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source, ...(prior ? { supersedes: prior.runId } : {}) }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
256
+ runStarted = true;
252
257
  // v1.53 T5: mark the prior run AFTER this run's run-start exists, so the prior journal never
253
258
  // names a successor that has no journal. Append-only — the prior journal is never rewritten.
254
259
  prior?.append("superseded", undefined, { by: runId });
260
+ for (const warning of baseline.warnings ?? [])
261
+ journal.append("baseline-warning", undefined, { ...warning });
262
+ // Tier A #3: run each task's command-typed acceptance oracles against the pristine baseline —
263
+ // one that already exits 0 verifies nothing. Warning only, taskId-stamped; never a gate input.
264
+ for (const w of await detectVacuousOracles(repoRoot, graph.tasks))
265
+ journal.append("baseline-warning", w.taskId, { ...w });
255
266
  }
256
267
  // T6: open the narrator AFTER run-start/run-resume is journaled so the watch surface has a run to
257
268
  // show. driver.narrator is undefined on subprocess → no-op (subprocess spawns nothing). Swallowed:
258
269
  // a failed-to-open or later-dead watch pane never affects the run.
270
+ // OBS-103: hold the returned slot — narrator() adopts an already-open watch by its owned name
271
+ // (a prior daemon instance's, after a stop→resume cycle), so the run-end sweep below can retire
272
+ // it regardless of which instance split the pane.
273
+ const watchName = formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId });
274
+ let watchSlot;
259
275
  try {
260
- await driver.narrator?.(repoRoot, "tickmarkr status --watch", runId);
276
+ watchSlot = await driver.narrator?.(repoRoot, "tickmarkr status --watch", runId);
261
277
  }
262
278
  catch {
263
279
  /* cosmetic-only — the run proceeds without a live surface */
@@ -286,7 +302,22 @@ export async function runDaemon(repoRoot, opts = {}) {
286
302
  if (keepForever)
287
303
  return;
288
304
  try {
289
- await driver.reconcile?.(desiredPanes(journal.read(), runId), runId, opts);
305
+ const desired = desiredPanes(journal.read(), runId);
306
+ // The watch pane is never the DRIVER sweep's candidate (panesToClose spares role "watch":
307
+ // herdr's watches bookkeeping lives in close(), and a raw pane-close in the sweep would
308
+ // leave narrator() a stale cache) — the driver always sees it as desired; its lifecycle is
309
+ // decided here from the fold alone.
310
+ await driver.reconcile?.(new Set([...desired, watchName]), runId, opts);
311
+ // OBS-103: when the fold retires the watch (run-end boundary), close the narrator. The
312
+ // decision keys on the run identity in the pane name — narrator() adopts a prior daemon
313
+ // instance's pane under the same owned name, so a stop→resume cycle's leftover narrator
314
+ // closes exactly like one this instance opened. A narrator carrying a non-canonical name
315
+ // (no run identity) is never this run's to sweep.
316
+ if (watchSlot && watchSlot.name === watchName && !desired.has(watchName)) {
317
+ const w = watchSlot;
318
+ watchSlot = undefined;
319
+ await driver.close(w);
320
+ }
290
321
  }
291
322
  catch {
292
323
  /* cosmetic — visibility is never a gate */
@@ -979,6 +1010,7 @@ export async function runDaemon(repoRoot, opts = {}) {
979
1010
  return;
980
1011
  }
981
1012
  };
1013
+ taskLoopStarted = true;
982
1014
  const inflight = new Map();
983
1015
  while (true) {
984
1016
  // v1.54 T2: a signal that landed while nothing was racing `aborted` (empty inflight window)
@@ -1069,6 +1101,23 @@ export async function runDaemon(repoRoot, opts = {}) {
1069
1101
  await driver.notify(`tickmarkr ${runId}: ${summary.done.length} done, ${summary.failed.length} failed, ${summary.human.length} awaiting human, ${summary.blocked.length} blocked, ${summary.pending.length} pending${attribution ? ` (${attribution})` : ""}${tipFail} — integration branch ${branch} (merge to main is yours)`, { tier: summary.tipVerify === "failed" ? "attention" : "routine" });
1070
1102
  return summary;
1071
1103
  }
1104
+ catch (err) {
1105
+ if (runStarted && !taskLoopStarted && !journal.read().some((e) => e.event === "run-end")) {
1106
+ journal.append("run-end", undefined, {
1107
+ runId,
1108
+ branch,
1109
+ done: [],
1110
+ failed: [],
1111
+ human: [],
1112
+ blocked: [],
1113
+ pending: [],
1114
+ phase: "setup",
1115
+ fatal: true,
1116
+ error: err instanceof Error ? err.message : String(err),
1117
+ });
1118
+ }
1119
+ throw err;
1120
+ }
1072
1121
  finally {
1073
1122
  // v1.54 T2: deregister on EVERY exit (normal run end, throw, termination unwind) — the daemon
1074
1123
  // test suite runs runDaemon dozens of times in one process; a leaked handler would close a
@@ -5,6 +5,7 @@ import { channelKey, TokenUsageSchema } from "../adapters/types.js";
5
5
  import { stateDirName, tickmarkrDir } from "../graph/graph.js";
6
6
  import { TIERS } from "../graph/schema.js";
7
7
  import { buildProfile } from "../route/profile.js";
8
+ import { redactSecrets } from "./redact.js";
8
9
  export function formatJournalNarration({ event, taskId, data }) {
9
10
  const assignment = data.assignment;
10
11
  const direct = [data.summary, data.reason, data.error, data.step, data.action, data.lint, data.branch, data.from]
@@ -311,9 +312,12 @@ export class Journal {
311
312
  }
312
313
  append(event, taskId, data = {}) {
313
314
  const row = { ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data };
314
- appendFileSync(this.journalPath, JSON.stringify(row) + "\n");
315
+ // T3 secret redaction: only the persisted bytes are masked — the caller's data stays untouched in
316
+ // memory. The narrator receives the persisted (masked) row so a pane sink never shows a credential.
317
+ const line = redactSecrets(JSON.stringify(row));
318
+ appendFileSync(this.journalPath, line + "\n");
315
319
  try {
316
- this.narrate?.(row);
320
+ this.narrate?.(JSON.parse(line));
317
321
  }
318
322
  catch {
319
323
  // narration is observational; a broken sink must not affect the journal or run
@@ -421,7 +425,8 @@ export class Journal {
421
425
  return m;
422
426
  }
423
427
  telemetry(row) {
424
- appendFileSync(join(this.dir, "telemetry.jsonl"), JSON.stringify(row) + "\n");
428
+ // T3 secret redaction: same persistence seam as append — credential-free rows are byte-identical.
429
+ appendFileSync(join(this.dir, "telemetry.jsonl"), redactSecrets(JSON.stringify(row)) + "\n");
425
430
  }
426
431
  // Per-run, raw (NOT schema-validated) — report.ts reads v1.5 core fields; stays byte-compatible.
427
432
  readTelemetry() {
@@ -87,8 +87,10 @@ export function desiredPanes(rows, runId) {
87
87
  clearTask(row.taskId);
88
88
  break;
89
89
  case "run-end":
90
+ // OBS-103: run-end retires EVERY run-tagged pane, the watch narrator included. The fold
91
+ // keys on the run identity in the pane name, so a narrator a prior daemon instance opened
92
+ // (stop→resume cycle) retires all the same; a later run-resume re-desires it above.
90
93
  desired.clear();
91
- desired.add(formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId }));
92
94
  break;
93
95
  }
94
96
  }
@@ -0,0 +1,2 @@
1
+ export declare const MASK = "[REDACTED]";
2
+ export declare function redactSecrets(text: string): string;
@@ -0,0 +1,53 @@
1
+ // T3 (repo-scan Tier A #2): the shared secret-redaction pass for captured agent text at the
2
+ // persistence seams. Every seam that persists captured agent text routes its bytes through
3
+ // redactSecrets before hitting disk — journal event payloads and telemetry rows (journal.ts) and
4
+ // consult dossier artifacts (consult.ts). In-memory originals are never touched: redaction applies
5
+ // to the serialized bytes only, at the write site.
6
+ //
7
+ // Two shape families:
8
+ // - vendor keys: recognizable prefix + high-entropy body. The prefix survives masking so the
9
+ // credential class stays identifiable; the body never persists.
10
+ // - secret assignments: KEY=value / "key": "value" where the key NAME announces a secret. The key
11
+ // survives; the value is masked.
12
+ // Text with no credential shapes passes through byte-identical. Masks contain no quote or backslash
13
+ // and value charsets never cross a JSON string boundary, so redacting a serialized JSON line always
14
+ // yields a line that still parses.
15
+ export const MASK = "[REDACTED]";
16
+ // Order matters: more specific prefixes (sk-ant-, sk-proj-) before the generic sk- form.
17
+ const VENDOR_KEY_RES = [
18
+ /\b(sk-ant-)[A-Za-z0-9_-]{8,}/g, // Anthropic
19
+ /\b(sk-proj-)[A-Za-z0-9_-]{8,}/g, // OpenAI project key
20
+ /\b(sk-)[A-Za-z0-9_-]{16,}/g, // OpenAI classic / sk-prefixed secret keys
21
+ /\b(github_pat_)[A-Za-z0-9_]{8,}/g, // GitHub fine-grained PAT
22
+ /\b(gh[pousr]_)[A-Za-z0-9]{16,}/g, // GitHub token family (ghp_/gho_/ghu_/ghs_/ghr_)
23
+ /\b(glpat-)[A-Za-z0-9_-]{8,}/g, // GitLab PAT
24
+ /\b(xox[baprs]-)[A-Za-z0-9-]{8,}/g, // Slack token family
25
+ /\b(npm_)[A-Za-z0-9]{16,}/g, // npm token
26
+ /\b(AKIA)[0-9A-Z]{16}\b/g, // AWS access key id
27
+ /\b(AIza)[0-9A-Za-z_-]{35}\b/g, // Google API key
28
+ ];
29
+ // Key names that announce a secret. The optional quote after the key closes a JSON key ("apiKey": "…").
30
+ // The match starts AT the keyword — chars before it (MY_ in MY_API_KEY=) are simply left in place by
31
+ // replace, so anchoring on the keyword yields identical output while keeping the scan linear (a
32
+ // leading [A-Za-z0-9_.-]* here backtracks quadratically on long tokens — seconds on a 50K blob, and
33
+ // journal payloads carry parked diffs up to the diff cap).
34
+ // The value charset excludes quotes/backslash (never crosses a JSON string boundary) and square
35
+ // brackets (an already-masked value never re-matches ⇒ idempotent, and a vendor prefix kept by the
36
+ // pass above survives). The letter lookahead skips purely numeric values ("maxTokens":30000000 is a
37
+ // count, not a credential — and masking a bare JSON number would break the line's parse).
38
+ // Quotes may arrive backslash-escaped (a transcript already serialized inside a JSON string), so the
39
+ // optional opening/closing quote around key and value accepts \" as well as ".
40
+ const ASSIGNMENT_RE = /((?:api[_-]?key|apikey|secret|token|passwd|password|credential|access[_-]?key)[A-Za-z0-9_.-]*(?:\\?["'])?\s*[=:]\s*(?:\\?["'])?)(?=[^\s"'`\\,;[\]]*[A-Za-z])([^\s"'`\\,;[\]]{8,})/gi;
41
+ // Authorization headers are the most common credential shape in captured transcripts.
42
+ const BEARER_RE = /\b(Bearer\s+)[A-Za-z0-9._~+/=-]{16,}/g;
43
+ // Cheap literal prescan: every pattern above needs one of these substrings, and most persisted text
44
+ // (diffs, prose, telemetry) has none — those payloads skip all 12 regex passes in one linear scan.
45
+ const HINT_RE = /sk-|github_pat_|gh[pousr]_|glpat-|xox[baprs]-|npm_|AKIA|AIza|Bearer|api[_-]?key|apikey|secret|token|passwd|password|credential|access[_-]?key/i;
46
+ export function redactSecrets(text) {
47
+ if (!HINT_RE.test(text))
48
+ return text;
49
+ let out = text;
50
+ for (const re of VENDOR_KEY_RES)
51
+ out = out.replace(re, `$1${MASK}`);
52
+ return out.replace(ASSIGNMENT_RE, `$1${MASK}`).replace(BEARER_RE, `$1${MASK}`);
53
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "1.62.0",
3
+ "version": "1.64.0",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -16,7 +16,8 @@ When working in a multi-agent terminal environment, decide your role before star
16
16
  - **Supervisor with a live orchestrator:** do not start a second run. Relay the mission to the existing orchestrator with a [verified handoff](#verified-handoffs-agent-to-agent-messaging), then supervise it as OVERSEER.
17
17
  - **Primary session without an orchestrator:** rename your own tab `OVERSEER · <version>` and your agent `overseer`, spawn one child orchestration session with your host's launch form, label its tab `ORCH · <version>` and name its agent, give it the mission and these rules verbatim, then supervise it. Do not drive a duplicate single-tier run yourself. Before spawning, confirm any PREVIOUS orchestrator has stood down (monitors stopped, input box empty — dim ghost-text suggestions are UI, not queued input; ANSI-verify before alarming) and close its tab — the journal, records, and ledger hold the story; scrollback is disposable.
18
18
  - **Claude Code:** `herdr agent start orchestrator --cwd <repo> --no-focus -- claude --permission-mode bypassPermissions`
19
- - **Codex:** `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --ask-for-approval never --sandbox workspace-write`
19
+ - **Codex:** `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
20
+ - **Auxiliary agents you spawn (consultants, reviewers, scouts) follow the same forms.** Never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions` (tickmarkr's own adapter uses exactly this for workers, judges, and consults). A read-only codex consultant may use `--sandbox read-only`; any codex session that must touch git needs the unsandboxed flag above.
20
21
 
21
22
  Outside a multi-agent terminal environment, run the loop directly.
22
23
 
@@ -15,7 +15,8 @@ When working in a multi-agent terminal environment, decide your role before star
15
15
  - **Supervisor with a live orchestrator:** do not start a second run. Relay the mission to the existing orchestrator with a [verified handoff](#verified-handoffs-agent-to-agent-messaging), then supervise it as OVERSEER.
16
16
  - **Primary session without an orchestrator:** rename your own tab `OVERSEER · <version>` and your agent `overseer`, spawn one child orchestration session with your host's launch form, label its tab `ORCH · <version>` and name its agent, give it the mission and these rules verbatim, then supervise it. Do not drive a duplicate single-tier run yourself. Before spawning, confirm any PREVIOUS orchestrator has [stood down](#stand-down-mission-end-and-retirement) and close its tab.
17
17
  - **Claude Code:** `herdr agent start orchestrator --cwd <repo> --no-focus -- claude --permission-mode bypassPermissions`
18
- - **Codex:** `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --ask-for-approval never --sandbox workspace-write`
18
+ - **Codex:** `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
19
+ - **Auxiliary agents you spawn (consultants, reviewers, scouts) follow the same forms.** Never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions` (tickmarkr's own adapter uses exactly this for workers, judges, and consults). A read-only codex consultant may use `--sandbox read-only`; any codex session that must touch git needs the unsandboxed flag above.
19
20
 
20
21
  Outside a multi-agent terminal environment, run the loop directly.
21
22
 
@@ -29,7 +29,7 @@ Requires `HERDR_ENV=1`; if unset, say so and stop.
29
29
  main name plus at most ONE hot-state token. Vocabulary: ORCH carries the milestone and progress
30
30
  fraction (`ORCH · v1.19 4/5`, updated on every task-done); WORKERS carries the task token (tickmarkr
31
31
  updates it). Never long context strings or ✓-chains.
32
- 2. **Orchestrator**: Launch the orchestrator with your agent host. For Claude Code, use `herdr agent start orchestrator --cwd <repo> --no-focus -- claude --permission-mode bypassPermissions` (pin a strong model with `--model <m>` if the operator has a policy). For Codex, use `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --ask-for-approval never --sandbox workspace-write` (add `--model <m>` to specify the model). Workers you never spawn — tickmarkr spawns its own visible worker panes.
32
+ 2. **Orchestrator**: Launch the orchestrator with your agent host. For Claude Code, use `herdr agent start orchestrator --cwd <repo> --no-focus -- claude --permission-mode bypassPermissions` (pin a strong model with `--model <m>` if the operator has a policy). For Codex, use `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions`, and a read-only codex consultant may use `--sandbox read-only`.
33
33
  3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
34
34
  truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
35
35
  (inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then send one line: