tickmarkr 1.68.0 → 1.70.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/adapters/kimi.d.ts +25 -1
  2. package/dist/adapters/kimi.js +80 -0
  3. package/dist/adapters/types.d.ts +6 -0
  4. package/dist/cli/commands/eval.d.ts +4 -0
  5. package/dist/cli/commands/eval.js +26 -0
  6. package/dist/cli/commands/report.js +40 -5
  7. package/dist/cli/index.d.ts +1 -1
  8. package/dist/cli/index.js +3 -1
  9. package/dist/eval/canary.d.ts +46 -0
  10. package/dist/eval/canary.js +113 -0
  11. package/dist/eval/dispatch.d.ts +31 -0
  12. package/dist/eval/dispatch.js +207 -0
  13. package/dist/eval/fixtures.d.ts +22 -0
  14. package/dist/eval/fixtures.js +85 -0
  15. package/dist/eval/report.d.ts +35 -0
  16. package/dist/eval/report.js +82 -0
  17. package/dist/eval/selfcheck.d.ts +22 -0
  18. package/dist/eval/selfcheck.js +177 -0
  19. package/dist/gates/acceptance.d.ts +5 -1
  20. package/dist/gates/acceptance.js +72 -10
  21. package/dist/gates/review.d.ts +10 -2
  22. package/dist/gates/review.js +72 -19
  23. package/dist/report/bundle.d.ts +45 -0
  24. package/dist/report/bundle.js +158 -0
  25. package/dist/report/compare.d.ts +64 -0
  26. package/dist/report/compare.js +231 -0
  27. package/dist/run/daemon.js +152 -103
  28. package/dist/run/environment.d.ts +12 -0
  29. package/dist/run/environment.js +41 -0
  30. package/dist/run/interactive-seed.d.ts +15 -0
  31. package/dist/run/interactive-seed.js +25 -0
  32. package/fixtures/eval/canary/solution/a.txt +1 -0
  33. package/fixtures/eval/canary/spec.md +8 -0
  34. package/fixtures/eval/canary/start/a.txt +1 -0
  35. package/fixtures/eval/sample/solution/a.txt +1 -0
  36. package/fixtures/eval/sample/spec.md +8 -0
  37. package/fixtures/eval/sample/start/a.txt +1 -0
  38. package/fixtures/gsd-sample/07-live-check/07-01-PLAN.md +42 -0
  39. package/fixtures/gsd-sample/07-live-check/07-02-PLAN.md +21 -0
  40. package/fixtures/gsd-sample/07-live-check/07-03-PLAN.md +18 -0
  41. package/fixtures/gsd-sample/07-live-check/07-03-SUMMARY.md +1 -0
  42. package/fixtures/missing-mandatory-gate.native.md +10 -0
  43. package/fixtures/sample-pin.prd.md +18 -0
  44. package/fixtures/sample.native.md +35 -0
  45. package/fixtures/sample.prd.md +22 -0
  46. package/fixtures/speckit-sample/tasks.md +20 -0
  47. package/package.json +3 -2
@@ -0,0 +1,231 @@
1
+ import { estimateCosts } from "./cost.js";
2
+ export function hasRunStart(events) {
3
+ return events.some((e) => e.event === "run-start");
4
+ }
5
+ // First run-start's environment, fail-closed on a missing or malformed stamp.
6
+ export function recordedEnvironment(events) {
7
+ for (const e of events) {
8
+ if (e.event !== "run-start")
9
+ continue;
10
+ const env = e.data.environment;
11
+ if (!env || typeof env !== "object" || Array.isArray(env))
12
+ return undefined;
13
+ const o = env;
14
+ if (typeof o.tickmarkrVersion !== "string" || typeof o.configHash !== "string")
15
+ return undefined;
16
+ if (!o.adapterVersions || typeof o.adapterVersions !== "object" || Array.isArray(o.adapterVersions))
17
+ return undefined;
18
+ const adapterVersions = {};
19
+ for (const [k, v] of Object.entries(o.adapterVersions)) {
20
+ if (typeof v !== "string")
21
+ return undefined;
22
+ adapterVersions[k] = v;
23
+ }
24
+ return { tickmarkrVersion: o.tickmarkrVersion, configHash: o.configHash, adapterVersions };
25
+ }
26
+ return undefined;
27
+ }
28
+ // Canonical identity string for the `recorded` field — configHash is the axis the acceptance
29
+ // criteria name; full environment equality still decides .comparable.
30
+ function envFingerprint(env) {
31
+ return env.configHash;
32
+ }
33
+ function envEqual(a, b) {
34
+ if (a.tickmarkrVersion !== b.tickmarkrVersion || a.configHash !== b.configHash)
35
+ return false;
36
+ const ak = Object.keys(a.adapterVersions).sort();
37
+ const bk = Object.keys(b.adapterVersions).sort();
38
+ if (ak.length !== bk.length)
39
+ return false;
40
+ for (let i = 0; i < ak.length; i++) {
41
+ if (ak[i] !== bk[i])
42
+ return false;
43
+ if (a.adapterVersions[ak[i]] !== b.adapterVersions[bk[i]])
44
+ return false;
45
+ }
46
+ return true;
47
+ }
48
+ // THE environment-identity comparator (criterion: reuses engagementComparable's shape).
49
+ // unbound = either side lacks a usable stamp; mismatch = stamps disagree; comparable = equal.
50
+ export function environmentComparable(baselineEnv, currentEnv) {
51
+ if (baselineEnv === undefined || currentEnv === undefined)
52
+ return { comparable: false, reason: "unbound" };
53
+ const recorded = envFingerprint(baselineEnv);
54
+ return envEqual(baselineEnv, currentEnv)
55
+ ? { comparable: true, recorded }
56
+ : { comparable: false, reason: "mismatch", recorded };
57
+ }
58
+ export function runMetrics(events, rows = [], cost = {}) {
59
+ const start = events.find((e) => e.event === "run-start");
60
+ const end = [...events].reverse().find((e) => e.event === "run-end");
61
+ let durationMs;
62
+ if (start && end) {
63
+ const from = Date.parse(start.ts);
64
+ const to = Date.parse(end.ts);
65
+ if (Number.isFinite(from) && Number.isFinite(to) && to >= from)
66
+ durationMs = to - from;
67
+ }
68
+ const gates = events.filter((e) => e.event === "gate-result");
69
+ const gatePass = gates.filter((e) => e.data.pass === true).length;
70
+ const gateFail = gates.filter((e) => e.data.pass === false).length;
71
+ let costUsd;
72
+ let costSum = 0;
73
+ let hasCost = false;
74
+ for (const p of estimateCosts(rows, cost)) {
75
+ if (p.apiUsd !== undefined) {
76
+ costSum += p.apiUsd;
77
+ hasCost = true;
78
+ }
79
+ else if (p.amortizedUsd) {
80
+ costSum += (p.amortizedUsd[0] + p.amortizedUsd[1]) / 2;
81
+ hasCost = true;
82
+ }
83
+ }
84
+ if (hasCost)
85
+ costUsd = Math.round(costSum * 1e6) / 1e6;
86
+ let tokensTotal;
87
+ let tokenSum = 0;
88
+ let hasTokens = false;
89
+ for (const r of rows) {
90
+ if (!r.tokens)
91
+ continue;
92
+ hasTokens = true;
93
+ const t = r.tokens;
94
+ tokenSum += t.input + t.output + (t.cacheRead ?? 0) + (t.cacheWrite ?? 0) + (t.reasoning ?? 0);
95
+ }
96
+ if (hasTokens)
97
+ tokensTotal = tokenSum;
98
+ return { durationMs, gatePass, gateFail, gateTotal: gates.length, costUsd, tokensTotal };
99
+ }
100
+ const n = (x) => x.toLocaleString("en-US");
101
+ const EM = "—";
102
+ function fmtDuration(ms) {
103
+ if (ms === undefined)
104
+ return EM;
105
+ const seconds = Math.round(ms / 1_000);
106
+ const minutes = Math.floor(seconds / 60);
107
+ return minutes ? `${minutes}m ${seconds % 60}s` : `${seconds}s`;
108
+ }
109
+ function fmtSignedDuration(ms) {
110
+ if (ms === undefined)
111
+ return EM;
112
+ if (ms === 0)
113
+ return "0s";
114
+ const sign = ms > 0 ? "+" : "-";
115
+ return `${sign}${fmtDuration(Math.abs(ms))}`;
116
+ }
117
+ function fmtUsd(v) {
118
+ if (v === undefined)
119
+ return EM;
120
+ return `$${v.toFixed(6)}`;
121
+ }
122
+ function fmtSignedUsd(v) {
123
+ if (v === undefined)
124
+ return EM;
125
+ if (v === 0)
126
+ return "$0.000000";
127
+ const sign = v > 0 ? "+" : "-";
128
+ return `${sign}$${Math.abs(v).toFixed(6)}`;
129
+ }
130
+ function fmtInt(v) {
131
+ if (v === undefined)
132
+ return EM;
133
+ return n(v);
134
+ }
135
+ function fmtSignedInt(v) {
136
+ if (v === undefined)
137
+ return EM;
138
+ if (v === 0)
139
+ return "0";
140
+ return v > 0 ? `+${n(v)}` : `-${n(Math.abs(v))}`;
141
+ }
142
+ function comparabilityLine(cmp, baselineEnv, currentEnv) {
143
+ if (cmp.comparable) {
144
+ return `full comparability (environment identity matches; configHash=${cmp.recorded})`;
145
+ }
146
+ if (cmp.reason === "unbound") {
147
+ return "comparability caveat — one or both runs lack a recorded environment identity; not apples-to-apples";
148
+ }
149
+ const base = baselineEnv?.configHash ?? EM;
150
+ const cur = currentEnv?.configHash ?? EM;
151
+ return `comparability caveat — environment identity disagrees (baseline configHash=${base} ≠ current configHash=${cur}; recorded baseline ${cmp.recorded}); not apples-to-apples`;
152
+ }
153
+ export function renderComparison(opts) {
154
+ const { runId, baselineRunId, comparability, baselineEnv, currentEnv, current, baseline, delta } = opts;
155
+ // Both run ids are always named so the reader knows which is the baseline.
156
+ const lines = [
157
+ `## Comparison`,
158
+ "",
159
+ `- **run:** ${runId}`,
160
+ `- **baseline:** ${baselineRunId}`,
161
+ `- **comparability:** ${comparabilityLine(comparability, baselineEnv, currentEnv)}`,
162
+ "",
163
+ "### Delta (current − baseline)",
164
+ "",
165
+ `| metric | baseline (${baselineRunId}) | current (${runId}) | delta |`,
166
+ `| --- | --- | --- | --- |`,
167
+ `| duration | ${fmtDuration(baseline.durationMs)} | ${fmtDuration(current.durationMs)} | ${fmtSignedDuration(delta.durationMs)} |`,
168
+ `| gate failures | ${fmtInt(baseline.gateFail)} | ${fmtInt(current.gateFail)} | ${fmtSignedInt(delta.gateFail)} |`,
169
+ `| gate pass rate | ${baseline.gateTotal ? `${baseline.gatePass}/${baseline.gateTotal}` : EM} | ${current.gateTotal ? `${current.gatePass}/${current.gateTotal}` : EM} | ${EM} |`,
170
+ `| cost | ${fmtUsd(baseline.costUsd)} | ${fmtUsd(current.costUsd)} | ${fmtSignedUsd(delta.costUsd)} |`,
171
+ `| tokens | ${fmtInt(baseline.tokensTotal)} | ${fmtInt(current.tokensTotal)} | ${fmtSignedInt(delta.tokensTotal)} |`,
172
+ "",
173
+ ];
174
+ if (!comparability.comparable) {
175
+ lines.push("_Caveat: deltas are shown for inspection only — environment identity disagrees, so this is not an apples-to-apples table._", "");
176
+ }
177
+ return lines.join("\n").trimEnd() + "\n";
178
+ }
179
+ // Fail closed when either journal has no run-start (no partial / fabricated comparison).
180
+ // Mismatch of environment identity still yields a rendered delta, with an explicit caveat.
181
+ export function compareRuns(opts) {
182
+ if (!hasRunStart(opts.baselineEvents)) {
183
+ return {
184
+ ok: false,
185
+ reason: `baseline run ${opts.baselineRunId} has no recorded run-start event — cannot compare`,
186
+ };
187
+ }
188
+ if (!hasRunStart(opts.events)) {
189
+ return {
190
+ ok: false,
191
+ reason: `run ${opts.runId} has no recorded run-start event — cannot compare`,
192
+ };
193
+ }
194
+ const baselineEnv = recordedEnvironment(opts.baselineEvents);
195
+ const currentEnv = recordedEnvironment(opts.events);
196
+ const comparability = environmentComparable(baselineEnv, currentEnv);
197
+ const baseline = runMetrics(opts.baselineEvents, opts.baselineRows ?? [], opts.cost ?? {});
198
+ const current = runMetrics(opts.events, opts.rows ?? [], opts.cost ?? {});
199
+ const delta = {
200
+ durationMs: current.durationMs !== undefined && baseline.durationMs !== undefined
201
+ ? current.durationMs - baseline.durationMs
202
+ : undefined,
203
+ gateFail: current.gateFail - baseline.gateFail,
204
+ costUsd: current.costUsd !== undefined && baseline.costUsd !== undefined
205
+ ? Math.round((current.costUsd - baseline.costUsd) * 1e6) / 1e6
206
+ : undefined,
207
+ tokensTotal: current.tokensTotal !== undefined && baseline.tokensTotal !== undefined
208
+ ? current.tokensTotal - baseline.tokensTotal
209
+ : undefined,
210
+ };
211
+ const text = renderComparison({
212
+ runId: opts.runId,
213
+ baselineRunId: opts.baselineRunId,
214
+ comparability,
215
+ baselineEnv,
216
+ currentEnv,
217
+ current,
218
+ baseline,
219
+ delta,
220
+ });
221
+ return {
222
+ ok: true,
223
+ runId: opts.runId,
224
+ baselineRunId: opts.baselineRunId,
225
+ comparability,
226
+ current,
227
+ baseline,
228
+ delta,
229
+ text,
230
+ };
231
+ }
@@ -15,7 +15,9 @@ import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../ga
15
15
  import { runGates } from "../gates/run-gates.js";
16
16
  import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus } from "../graph/graph.js";
17
17
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
18
+ import { runEnvironment } from "./environment.js";
18
19
  import { cleanupRunWorktrees, gitHead, linkNodeModules, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
20
+ import { runInteractiveSeed } from "./interactive-seed.js";
19
21
  import { classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId } from "./journal.js";
20
22
  import { acquireRunLock, releaseRunLock } from "./lock.js";
21
23
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
@@ -64,6 +66,10 @@ export function formatSummary(s) {
64
66
  return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}`;
65
67
  }
66
68
  const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can never loop forever
69
+ // v1.70 T5: request-changes review rounds a single task may draw before it parks for a human decision
70
+ // instead of cycling. Well below MAX_ATTEMPTS so review non-convergence is caught long before the
71
+ // global cap. ponytail: literal constant; lift to cfg.review.roundCap only if a second knob-turner appears.
72
+ const REVIEW_ROUND_CAP = 3;
67
73
  const BLOCKED_POLL_MS = 30_000; // between trailer-wait slices, check whether the pane is blocked on a prompt
68
74
  const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
69
75
  const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
@@ -252,7 +258,11 @@ export async function runDaemon(repoRoot, opts = {}) {
252
258
  baseRef = await gitHead(repoRoot);
253
259
  baseline = await captureBaseline(repoRoot, commands);
254
260
  writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(baseline, null, 2));
255
- journal.append("run-start", undefined, { pid: process.pid, baseRef, commands, channels: channels.map(channelKey), branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source, ...(prior ? { supersedes: prior.runId } : {}) }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
261
+ // v1.70 T2: environment identity beside the graph/branch identity — running tickmarkr version,
262
+ // loaded-config hash, and the probed CLI version of each adapter holding a channel in the run,
263
+ // gathered through the existing probe/config-load paths (no second mechanism).
264
+ const environment = runEnvironment(cfg, channels, health);
265
+ journal.append("run-start", undefined, { pid: process.pid, baseRef, commands, channels: channels.map(channelKey), branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source, environment, ...(prior ? { supersedes: prior.runId } : {}) }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
256
266
  runStarted = true;
257
267
  // v1.53 T5: mark the prior run AFTER this run's run-start exists, so the prior journal never
258
268
  // names a successor that has no journal. Append-only — the prior journal is never rewritten.
@@ -409,6 +419,14 @@ export async function runDaemon(repoRoot, opts = {}) {
409
419
  if (rs)
410
420
  journal.append("resume-restore", t.id, { attempts: rs.attempts, tried: [...tried], assignment });
411
421
  const badReviewers = []; // v1.1: reviewer channels that produced unparseable output for this task
422
+ // v1.70 T5 (review-convergence): failed review rounds this task has drawn, counted from the per-task
423
+ // review history already in the journal — the SAME review gate-result stream onGate reads to grow the
424
+ // reviewer-exclusion list (badReviewers), never a second parallel counter. request-changes rounds do
425
+ // not exclude their reviewer (a fix is re-checked by the same seat), so their count lives here in the
426
+ // shared history rather than in badReviewers' garbage-only exclusion subset.
427
+ const reviewRoundsDrawn = () => journal.read().filter((e) => e.taskId === t.id && e.event === "gate-result" &&
428
+ e.data.gate === "review" &&
429
+ e.data.pass === false).length;
412
430
  let feedback = "";
413
431
  let ladderIdx = 0;
414
432
  let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
@@ -509,6 +527,15 @@ export async function runDaemon(repoRoot, opts = {}) {
509
527
  await park(t, `attempt cap (${MAX_ATTEMPTS}) reached`, "attempt-cap", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
510
528
  return;
511
529
  }
530
+ // v1.70 T5: a task that has already drawn REVIEW_ROUND_CAP request-changes review rounds parks for
531
+ // a human decision instead of dispatching another round. Condition on the review history (data),
532
+ // never the code path — and let an operator's approval (the human decision the cap asked for)
533
+ // release it, mirroring the humanGate guard's condition-on-approval precedent. Rounds only accrue
534
+ // after attempt 0, so the guard skips the journal read on the happy path.
535
+ if (attempt > 0 && !approved.has(t.id) && reviewRoundsDrawn() >= REVIEW_ROUND_CAP) {
536
+ await park(t, `review round cap (${REVIEW_ROUND_CAP}) reached — request-changes reviews not converging; a human should decide`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
537
+ return;
538
+ }
512
539
  // OBS-57: a demoted channel must not be re-dispatched on consult retry or provider requeue.
513
540
  if (demotedChannels.has(channelKey(assignment))) {
514
541
  const next = nextChannel(assignment, t, cfg, channels, tried, profile, demotedChannels);
@@ -601,17 +628,22 @@ export async function runDaemon(repoRoot, opts = {}) {
601
628
  : cfg.visibility.worker === "interactive" && driver.interactive
602
629
  ? adapter.interactiveCommand(promptFile, assignment.model)
603
630
  : null;
604
- if (cfg.visibility.worker === "interactive" && icmd === null && !modeFallbackNoted) {
631
+ // v1.69 T6: adapters that declare interactiveSeed launch the real TUI and inject the prompt as a
632
+ // user turn; they do NOT need the argv-seeding surface that interactiveCommand represents.
633
+ const hasSeed = retryMode !== "resume" && cfg.visibility.worker === "interactive" && driver.interactive && !!adapter.interactiveSeed;
634
+ if (cfg.visibility.worker === "interactive" && icmd === null && !hasSeed && !modeFallbackNoted) {
605
635
  modeFallbackNoted = true;
606
636
  journal.append("worker-mode-fallback", t.id, { reason: driver.interactive ? "adapter" : "driver" });
607
637
  }
608
- const interactive = icmd !== null;
638
+ const interactive = icmd !== null || hasSeed;
609
639
  // OBS-85 (v1.62 T1): both dispatch branches deliver ONE short script invocation — banner,
610
640
  // adapter command, and nonce exit marker live in a per-attempt script beside the prompt
611
641
  // artifact (the same paneDispatchCommand pattern judge/review/consult dispatches use). The
612
642
  // delivered pane line carries no command substitution and no trailing shell text, so paste
613
643
  // timing can never interleave a `$(…)` with what follows it (the codex corruption class).
614
- const workerCmd = interactive ? icmd : adapter.invoke(t, wt, assignment, { promptFile }).command;
644
+ const workerCmd = interactive
645
+ ? (hasSeed ? ":" : icmd)
646
+ : adapter.invoke(t, wt, assignment, { promptFile }).command;
615
647
  const dispatchScript = promptFile.replace(/\.md$/, ".sh");
616
648
  writeFileSync(dispatchScript, [
617
649
  "export BASH_SILENCE_DEPRECATION_WARNING=1",
@@ -659,117 +691,134 @@ export async function runDaemon(repoRoot, opts = {}) {
659
691
  let exitCode;
660
692
  let timedOut = false;
661
693
  let settleParsed;
694
+ let seedResult;
662
695
  if (interactive) {
663
696
  // v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
664
697
  // The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
665
- await driver.run(slot, paneDispatchCommand(dispatchScript));
666
- let paged = false;
667
- // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
668
- // Any other blocked/idle dialog still pages the operator (paged latch below).
669
- let trustAnswered = false;
670
698
  finished = false;
671
699
  exitCode = null;
672
- output = await driver.read(slot, 1000);
673
- // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
674
- // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
675
- const stallWindowMs = taskTimeoutMinutes * 60_000;
676
- // OBS-82: the stall clock compares NORMALIZED snapshots so a spinner glyph/elapsed-time
677
- // repaint is silence, not activity. ONLY this inactivity compare sees normalized text —
678
- // trailer detection, harvest, paging, and quota checks all read the raw pane.
679
- let lastStallSnapshot = normalizeStallSnapshot(output);
680
- let lastOutputAt = Date.now();
681
- while (Date.now() - lastOutputAt < stallWindowMs) {
682
- const sliceStart = Date.now();
683
- const remaining = stallWindowMs - (sliceStart - lastOutputAt);
684
- const slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
685
- if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
686
- // verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
687
- // own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
688
- // parseable trailer or a digit-suffixed exit marker in the harvest is completion.
689
- output = await driver.read(slot, 1000); // TUI transcripts carry chrome — read deeper than print's 500
690
- finished = new RegExp(trailerPattern(nonce)).test(output);
691
- const exit = exitRe.exec(output);
692
- if (finished || exit) {
693
- exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
694
- await sampleContext(); // final poll-seam sample before leaving the wait
695
- break;
696
- }
697
- }
698
- const currentStallSnapshot = normalizeStallSnapshot(await driver.read(slot, 1000));
699
- if (currentStallSnapshot !== lastStallSnapshot) {
700
- lastStallSnapshot = currentStallSnapshot;
701
- lastOutputAt = Date.now();
702
- }
703
- // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
704
- await sampleContext();
705
- // page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
706
- // (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
707
- // a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
708
- const st = paged ? "" : await driver.status(slot);
709
- if (!paged && (st === "blocked" || st === "idle")) {
710
- // T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
711
- // text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
712
- if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
713
- try {
714
- const paneText = await driver.read(slot, 80);
715
- if (matchesTrustDialog(paneText, adapter.trustDialog)) {
716
- trustAnswered = true;
717
- // v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
718
- // Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
719
- journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
720
- await driver.sendKey(slot, adapter.trustDialog.key);
721
- const spent = Date.now() - sliceStart;
722
- if (spent < slice)
723
- await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
724
- continue; // do not page — keep waiting for the trailer
725
- }
726
- }
727
- catch {
728
- /* read/send failed — fall through to page the operator */
729
- }
730
- }
731
- paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
732
- const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
733
- await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
734
- }
735
- // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
736
- const spent = Date.now() - sliceStart;
737
- if (spent < slice)
738
- await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
700
+ if (adapter.interactiveSeed) {
701
+ // v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
702
+ // then fall through to the normal trailer harvest. A failed seed is recorded as a finished
703
+ // failure rather than allowed to race the trailer wait.
704
+ seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
705
+ output = seedResult.output;
739
706
  }
740
- if (!finished && exitCode === null) {
741
- // timed out (or only ever saw false positives): harvest whatever the pane holds now
742
- timedOut = Date.now() - lastOutputAt >= stallWindowMs;
707
+ else {
708
+ await driver.run(slot, paneDispatchCommand(dispatchScript));
743
709
  output = await driver.read(slot, 1000);
744
- finished = new RegExp(trailerPattern(nonce)).test(output);
745
- const exit = exitRe.exec(output);
746
- exitCode = exit ? Number(exit[1]) : null;
747
710
  }
748
- if (finished) {
749
- await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
750
- output = await driver.read(slot, 1000);
711
+ if (seedResult?.seedFailed) {
712
+ finished = false;
751
713
  }
752
- // T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
753
- // contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
754
- // the existing pane-read seam once or twice before recording a malformed-trailer cause.
755
- if (interactive) {
714
+ else {
715
+ let paged = false;
716
+ // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
717
+ // Any other blocked/idle dialog still pages the operator (paged latch below).
718
+ let trustAnswered = false;
719
+ finished = false;
720
+ exitCode = null;
721
+ // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
722
+ // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
756
723
  const stallWindowMs = taskTimeoutMinutes * 60_000;
757
- const settleDeadline = attemptStart + stallWindowMs;
758
- const settleDelayMs = 1_000;
759
- const maxSettleRetries = 2;
760
- let settleTries = 0;
761
- settleParsed = adapter.parse(output, nonce);
762
- while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
763
- const remaining = settleDeadline - Date.now();
764
- if (remaining <= 0)
765
- break;
766
- await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
724
+ // OBS-82: the stall clock compares NORMALIZED snapshots so a spinner glyph/elapsed-time
725
+ // repaint is silence, not activity. ONLY this inactivity compare sees normalized text —
726
+ // trailer detection, harvest, paging, and quota checks all read the raw pane.
727
+ let lastStallSnapshot = normalizeStallSnapshot(output);
728
+ let lastOutputAt = Date.now();
729
+ while (Date.now() - lastOutputAt < stallWindowMs) {
730
+ const sliceStart = Date.now();
731
+ const remaining = stallWindowMs - (sliceStart - lastOutputAt);
732
+ const slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
733
+ if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
734
+ // verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
735
+ // own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
736
+ // parseable trailer or a digit-suffixed exit marker in the harvest is completion.
737
+ output = await driver.read(slot, 1000); // TUI transcripts carry chrome — read deeper than print's 500
738
+ finished = new RegExp(trailerPattern(nonce)).test(output);
739
+ const exit = exitRe.exec(output);
740
+ if (finished || exit) {
741
+ exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
742
+ await sampleContext(); // final poll-seam sample before leaving the wait
743
+ break;
744
+ }
745
+ }
746
+ const currentStallSnapshot = normalizeStallSnapshot(await driver.read(slot, 1000));
747
+ if (currentStallSnapshot !== lastStallSnapshot) {
748
+ lastStallSnapshot = currentStallSnapshot;
749
+ lastOutputAt = Date.now();
750
+ }
751
+ // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
752
+ await sampleContext();
753
+ // page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
754
+ // (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
755
+ // a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
756
+ const st = paged ? "" : await driver.status(slot);
757
+ if (!paged && (st === "blocked" || st === "idle")) {
758
+ // T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
759
+ // text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
760
+ if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
761
+ try {
762
+ const paneText = await driver.read(slot, 80);
763
+ if (matchesTrustDialog(paneText, adapter.trustDialog)) {
764
+ trustAnswered = true;
765
+ // v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
766
+ // Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
767
+ journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
768
+ await driver.sendKey(slot, adapter.trustDialog.key);
769
+ const spent = Date.now() - sliceStart;
770
+ if (spent < slice)
771
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
772
+ continue; // do not page — keep waiting for the trailer
773
+ }
774
+ }
775
+ catch {
776
+ /* read/send failed — fall through to page the operator */
777
+ }
778
+ }
779
+ paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
780
+ const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
781
+ await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
782
+ }
783
+ // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
784
+ const spent = Date.now() - sliceStart;
785
+ if (spent < slice)
786
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
787
+ }
788
+ if (!finished && exitCode === null) {
789
+ // timed out (or only ever saw false positives): harvest whatever the pane holds now
790
+ timedOut = Date.now() - lastOutputAt >= stallWindowMs;
767
791
  output = await driver.read(slot, 1000);
768
- settleParsed = adapter.parse(output, nonce);
769
- settleTries++;
792
+ finished = new RegExp(trailerPattern(nonce)).test(output);
793
+ const exit = exitRe.exec(output);
794
+ exitCode = exit ? Number(exit[1]) : null;
770
795
  }
771
- if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
772
- finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
796
+ if (finished) {
797
+ await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
798
+ output = await driver.read(slot, 1000);
799
+ }
800
+ // T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
801
+ // contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
802
+ // the existing pane-read seam once or twice before recording a malformed-trailer cause.
803
+ if (interactive) {
804
+ const stallWindowMs = taskTimeoutMinutes * 60_000;
805
+ const settleDeadline = attemptStart + stallWindowMs;
806
+ const settleDelayMs = 1_000;
807
+ const maxSettleRetries = 2;
808
+ let settleTries = 0;
809
+ settleParsed = adapter.parse(output, nonce);
810
+ while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
811
+ const remaining = settleDeadline - Date.now();
812
+ if (remaining <= 0)
813
+ break;
814
+ await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
815
+ output = await driver.read(slot, 1000);
816
+ settleParsed = adapter.parse(output, nonce);
817
+ settleTries++;
818
+ }
819
+ if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
820
+ finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
821
+ }
773
822
  }
774
823
  }
775
824
  }
@@ -0,0 +1,12 @@
1
+ import type { AuthHealth, BillingChannel } from "../adapters/types.js";
2
+ import type { TickmarkrConfig } from "../config/config.js";
3
+ export interface RunEnvironment {
4
+ tickmarkrVersion: string;
5
+ configHash: string;
6
+ adapterVersions: Record<string, string>;
7
+ }
8
+ export declare const UNKNOWN_ADAPTER_VERSION = "unknown";
9
+ export declare function tickmarkrVersion(): string;
10
+ export declare function configHash(cfg: TickmarkrConfig): string;
11
+ export declare function adapterVersions(channels: BillingChannel[], health: Record<string, AuthHealth>): Record<string, string>;
12
+ export declare function runEnvironment(cfg: TickmarkrConfig, channels: BillingChannel[], health: Record<string, AuthHealth>): RunEnvironment;
@@ -0,0 +1,41 @@
1
+ import { createHash } from "node:crypto";
2
+ import { readFileSync } from "node:fs";
3
+ import { dirname, join } from "node:path";
4
+ import { fileURLToPath } from "node:url";
5
+ // An adapter whose version probe failed is recorded, not dropped — "unknown", never a fabricated string.
6
+ export const UNKNOWN_ADAPTER_VERSION = "unknown";
7
+ // Same package.json read as src/cli/commands/version.ts (one resolution pattern, two consumers).
8
+ const pkgPath = join(dirname(fileURLToPath(import.meta.url)), "../../package.json");
9
+ export function tickmarkrVersion() {
10
+ const { version: v } = JSON.parse(readFileSync(pkgPath, "utf8"));
11
+ return v;
12
+ }
13
+ // Key order in a parsed config is an accident of the schema/merge layers, so the hash canonicalizes
14
+ // first: object keys sorted recursively, undefined dropped (JSON semantics), array order preserved.
15
+ function stableStringify(v) {
16
+ if (v === null || typeof v !== "object")
17
+ return JSON.stringify(v);
18
+ if (Array.isArray(v))
19
+ return `[${v.map(stableStringify).join(",")}]`;
20
+ const entries = Object.entries(v)
21
+ .filter(([, val]) => val !== undefined)
22
+ .sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0));
23
+ return `{${entries.map(([k, val]) => `${JSON.stringify(k)}:${stableStringify(val)}`).join(",")}}`;
24
+ }
25
+ // sha256 truncated to 16 hex — the graphDefinitionHash convention (stable, grep-friendly).
26
+ export function configHash(cfg) {
27
+ return createHash("sha256").update(stableStringify(cfg)).digest("hex").slice(0, 16);
28
+ }
29
+ // One entry per adapter with a channel in the run (not per channel). The version is whatever the
30
+ // adapter's own probe recorded in health; a missing/undefined probe result becomes "unknown".
31
+ export function adapterVersions(channels, health) {
32
+ const out = {};
33
+ for (const c of channels) {
34
+ if (!(c.adapter in out))
35
+ out[c.adapter] = health[c.adapter]?.version ?? UNKNOWN_ADAPTER_VERSION;
36
+ }
37
+ return out;
38
+ }
39
+ export function runEnvironment(cfg, channels, health) {
40
+ return { tickmarkrVersion: tickmarkrVersion(), configHash: configHash(cfg), adapterVersions: adapterVersions(channels, health) };
41
+ }
@@ -0,0 +1,15 @@
1
+ import type { Assignment, WorkerAdapter } from "../adapters/types.js";
2
+ import type { ExecutorDriver, Slot } from "../drivers/types.js";
3
+ export interface InteractiveSeedResult {
4
+ output: string;
5
+ seedFailed: boolean;
6
+ seedError?: string;
7
+ }
8
+ export declare function runInteractiveSeed(opts: {
9
+ driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
10
+ slot: Slot;
11
+ adapter: WorkerAdapter;
12
+ assignment: Assignment;
13
+ promptFile: string;
14
+ taskTimeoutMinutes: number;
15
+ }): Promise<InteractiveSeedResult>;