tickmarkr 1.68.0 → 1.70.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/kimi.d.ts +25 -1
- package/dist/adapters/kimi.js +80 -0
- package/dist/adapters/types.d.ts +6 -0
- package/dist/cli/commands/eval.d.ts +4 -0
- package/dist/cli/commands/eval.js +26 -0
- package/dist/cli/commands/report.js +40 -5
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +3 -1
- package/dist/eval/canary.d.ts +46 -0
- package/dist/eval/canary.js +113 -0
- package/dist/eval/dispatch.d.ts +31 -0
- package/dist/eval/dispatch.js +207 -0
- package/dist/eval/fixtures.d.ts +22 -0
- package/dist/eval/fixtures.js +85 -0
- package/dist/eval/report.d.ts +35 -0
- package/dist/eval/report.js +82 -0
- package/dist/eval/selfcheck.d.ts +22 -0
- package/dist/eval/selfcheck.js +177 -0
- package/dist/gates/acceptance.d.ts +5 -1
- package/dist/gates/acceptance.js +72 -10
- package/dist/gates/review.d.ts +10 -2
- package/dist/gates/review.js +72 -19
- package/dist/report/bundle.d.ts +45 -0
- package/dist/report/bundle.js +158 -0
- package/dist/report/compare.d.ts +64 -0
- package/dist/report/compare.js +231 -0
- package/dist/run/daemon.js +152 -103
- package/dist/run/environment.d.ts +12 -0
- package/dist/run/environment.js +41 -0
- package/dist/run/interactive-seed.d.ts +15 -0
- package/dist/run/interactive-seed.js +25 -0
- package/fixtures/eval/canary/solution/a.txt +1 -0
- package/fixtures/eval/canary/spec.md +8 -0
- package/fixtures/eval/canary/start/a.txt +1 -0
- package/fixtures/eval/sample/solution/a.txt +1 -0
- package/fixtures/eval/sample/spec.md +8 -0
- package/fixtures/eval/sample/start/a.txt +1 -0
- package/fixtures/gsd-sample/07-live-check/07-01-PLAN.md +42 -0
- package/fixtures/gsd-sample/07-live-check/07-02-PLAN.md +21 -0
- package/fixtures/gsd-sample/07-live-check/07-03-PLAN.md +18 -0
- package/fixtures/gsd-sample/07-live-check/07-03-SUMMARY.md +1 -0
- package/fixtures/missing-mandatory-gate.native.md +10 -0
- package/fixtures/sample-pin.prd.md +18 -0
- package/fixtures/sample.native.md +35 -0
- package/fixtures/sample.prd.md +22 -0
- package/fixtures/speckit-sample/tasks.md +20 -0
- package/package.json +3 -2
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
import { estimateCosts } from "./cost.js";
|
|
2
|
+
export function hasRunStart(events) {
|
|
3
|
+
return events.some((e) => e.event === "run-start");
|
|
4
|
+
}
|
|
5
|
+
// First run-start's environment, fail-closed on a missing or malformed stamp.
|
|
6
|
+
export function recordedEnvironment(events) {
|
|
7
|
+
for (const e of events) {
|
|
8
|
+
if (e.event !== "run-start")
|
|
9
|
+
continue;
|
|
10
|
+
const env = e.data.environment;
|
|
11
|
+
if (!env || typeof env !== "object" || Array.isArray(env))
|
|
12
|
+
return undefined;
|
|
13
|
+
const o = env;
|
|
14
|
+
if (typeof o.tickmarkrVersion !== "string" || typeof o.configHash !== "string")
|
|
15
|
+
return undefined;
|
|
16
|
+
if (!o.adapterVersions || typeof o.adapterVersions !== "object" || Array.isArray(o.adapterVersions))
|
|
17
|
+
return undefined;
|
|
18
|
+
const adapterVersions = {};
|
|
19
|
+
for (const [k, v] of Object.entries(o.adapterVersions)) {
|
|
20
|
+
if (typeof v !== "string")
|
|
21
|
+
return undefined;
|
|
22
|
+
adapterVersions[k] = v;
|
|
23
|
+
}
|
|
24
|
+
return { tickmarkrVersion: o.tickmarkrVersion, configHash: o.configHash, adapterVersions };
|
|
25
|
+
}
|
|
26
|
+
return undefined;
|
|
27
|
+
}
|
|
28
|
+
// Canonical identity string for the `recorded` field — configHash is the axis the acceptance
|
|
29
|
+
// criteria name; full environment equality still decides .comparable.
|
|
30
|
+
function envFingerprint(env) {
|
|
31
|
+
return env.configHash;
|
|
32
|
+
}
|
|
33
|
+
function envEqual(a, b) {
|
|
34
|
+
if (a.tickmarkrVersion !== b.tickmarkrVersion || a.configHash !== b.configHash)
|
|
35
|
+
return false;
|
|
36
|
+
const ak = Object.keys(a.adapterVersions).sort();
|
|
37
|
+
const bk = Object.keys(b.adapterVersions).sort();
|
|
38
|
+
if (ak.length !== bk.length)
|
|
39
|
+
return false;
|
|
40
|
+
for (let i = 0; i < ak.length; i++) {
|
|
41
|
+
if (ak[i] !== bk[i])
|
|
42
|
+
return false;
|
|
43
|
+
if (a.adapterVersions[ak[i]] !== b.adapterVersions[bk[i]])
|
|
44
|
+
return false;
|
|
45
|
+
}
|
|
46
|
+
return true;
|
|
47
|
+
}
|
|
48
|
+
// THE environment-identity comparator (criterion: reuses engagementComparable's shape).
|
|
49
|
+
// unbound = either side lacks a usable stamp; mismatch = stamps disagree; comparable = equal.
|
|
50
|
+
export function environmentComparable(baselineEnv, currentEnv) {
|
|
51
|
+
if (baselineEnv === undefined || currentEnv === undefined)
|
|
52
|
+
return { comparable: false, reason: "unbound" };
|
|
53
|
+
const recorded = envFingerprint(baselineEnv);
|
|
54
|
+
return envEqual(baselineEnv, currentEnv)
|
|
55
|
+
? { comparable: true, recorded }
|
|
56
|
+
: { comparable: false, reason: "mismatch", recorded };
|
|
57
|
+
}
|
|
58
|
+
export function runMetrics(events, rows = [], cost = {}) {
|
|
59
|
+
const start = events.find((e) => e.event === "run-start");
|
|
60
|
+
const end = [...events].reverse().find((e) => e.event === "run-end");
|
|
61
|
+
let durationMs;
|
|
62
|
+
if (start && end) {
|
|
63
|
+
const from = Date.parse(start.ts);
|
|
64
|
+
const to = Date.parse(end.ts);
|
|
65
|
+
if (Number.isFinite(from) && Number.isFinite(to) && to >= from)
|
|
66
|
+
durationMs = to - from;
|
|
67
|
+
}
|
|
68
|
+
const gates = events.filter((e) => e.event === "gate-result");
|
|
69
|
+
const gatePass = gates.filter((e) => e.data.pass === true).length;
|
|
70
|
+
const gateFail = gates.filter((e) => e.data.pass === false).length;
|
|
71
|
+
let costUsd;
|
|
72
|
+
let costSum = 0;
|
|
73
|
+
let hasCost = false;
|
|
74
|
+
for (const p of estimateCosts(rows, cost)) {
|
|
75
|
+
if (p.apiUsd !== undefined) {
|
|
76
|
+
costSum += p.apiUsd;
|
|
77
|
+
hasCost = true;
|
|
78
|
+
}
|
|
79
|
+
else if (p.amortizedUsd) {
|
|
80
|
+
costSum += (p.amortizedUsd[0] + p.amortizedUsd[1]) / 2;
|
|
81
|
+
hasCost = true;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
if (hasCost)
|
|
85
|
+
costUsd = Math.round(costSum * 1e6) / 1e6;
|
|
86
|
+
let tokensTotal;
|
|
87
|
+
let tokenSum = 0;
|
|
88
|
+
let hasTokens = false;
|
|
89
|
+
for (const r of rows) {
|
|
90
|
+
if (!r.tokens)
|
|
91
|
+
continue;
|
|
92
|
+
hasTokens = true;
|
|
93
|
+
const t = r.tokens;
|
|
94
|
+
tokenSum += t.input + t.output + (t.cacheRead ?? 0) + (t.cacheWrite ?? 0) + (t.reasoning ?? 0);
|
|
95
|
+
}
|
|
96
|
+
if (hasTokens)
|
|
97
|
+
tokensTotal = tokenSum;
|
|
98
|
+
return { durationMs, gatePass, gateFail, gateTotal: gates.length, costUsd, tokensTotal };
|
|
99
|
+
}
|
|
100
|
+
const n = (x) => x.toLocaleString("en-US");
|
|
101
|
+
const EM = "—";
|
|
102
|
+
function fmtDuration(ms) {
|
|
103
|
+
if (ms === undefined)
|
|
104
|
+
return EM;
|
|
105
|
+
const seconds = Math.round(ms / 1_000);
|
|
106
|
+
const minutes = Math.floor(seconds / 60);
|
|
107
|
+
return minutes ? `${minutes}m ${seconds % 60}s` : `${seconds}s`;
|
|
108
|
+
}
|
|
109
|
+
function fmtSignedDuration(ms) {
|
|
110
|
+
if (ms === undefined)
|
|
111
|
+
return EM;
|
|
112
|
+
if (ms === 0)
|
|
113
|
+
return "0s";
|
|
114
|
+
const sign = ms > 0 ? "+" : "-";
|
|
115
|
+
return `${sign}${fmtDuration(Math.abs(ms))}`;
|
|
116
|
+
}
|
|
117
|
+
function fmtUsd(v) {
|
|
118
|
+
if (v === undefined)
|
|
119
|
+
return EM;
|
|
120
|
+
return `$${v.toFixed(6)}`;
|
|
121
|
+
}
|
|
122
|
+
function fmtSignedUsd(v) {
|
|
123
|
+
if (v === undefined)
|
|
124
|
+
return EM;
|
|
125
|
+
if (v === 0)
|
|
126
|
+
return "$0.000000";
|
|
127
|
+
const sign = v > 0 ? "+" : "-";
|
|
128
|
+
return `${sign}$${Math.abs(v).toFixed(6)}`;
|
|
129
|
+
}
|
|
130
|
+
function fmtInt(v) {
|
|
131
|
+
if (v === undefined)
|
|
132
|
+
return EM;
|
|
133
|
+
return n(v);
|
|
134
|
+
}
|
|
135
|
+
function fmtSignedInt(v) {
|
|
136
|
+
if (v === undefined)
|
|
137
|
+
return EM;
|
|
138
|
+
if (v === 0)
|
|
139
|
+
return "0";
|
|
140
|
+
return v > 0 ? `+${n(v)}` : `-${n(Math.abs(v))}`;
|
|
141
|
+
}
|
|
142
|
+
function comparabilityLine(cmp, baselineEnv, currentEnv) {
|
|
143
|
+
if (cmp.comparable) {
|
|
144
|
+
return `full comparability (environment identity matches; configHash=${cmp.recorded})`;
|
|
145
|
+
}
|
|
146
|
+
if (cmp.reason === "unbound") {
|
|
147
|
+
return "comparability caveat — one or both runs lack a recorded environment identity; not apples-to-apples";
|
|
148
|
+
}
|
|
149
|
+
const base = baselineEnv?.configHash ?? EM;
|
|
150
|
+
const cur = currentEnv?.configHash ?? EM;
|
|
151
|
+
return `comparability caveat — environment identity disagrees (baseline configHash=${base} ≠ current configHash=${cur}; recorded baseline ${cmp.recorded}); not apples-to-apples`;
|
|
152
|
+
}
|
|
153
|
+
export function renderComparison(opts) {
|
|
154
|
+
const { runId, baselineRunId, comparability, baselineEnv, currentEnv, current, baseline, delta } = opts;
|
|
155
|
+
// Both run ids are always named so the reader knows which is the baseline.
|
|
156
|
+
const lines = [
|
|
157
|
+
`## Comparison`,
|
|
158
|
+
"",
|
|
159
|
+
`- **run:** ${runId}`,
|
|
160
|
+
`- **baseline:** ${baselineRunId}`,
|
|
161
|
+
`- **comparability:** ${comparabilityLine(comparability, baselineEnv, currentEnv)}`,
|
|
162
|
+
"",
|
|
163
|
+
"### Delta (current − baseline)",
|
|
164
|
+
"",
|
|
165
|
+
`| metric | baseline (${baselineRunId}) | current (${runId}) | delta |`,
|
|
166
|
+
`| --- | --- | --- | --- |`,
|
|
167
|
+
`| duration | ${fmtDuration(baseline.durationMs)} | ${fmtDuration(current.durationMs)} | ${fmtSignedDuration(delta.durationMs)} |`,
|
|
168
|
+
`| gate failures | ${fmtInt(baseline.gateFail)} | ${fmtInt(current.gateFail)} | ${fmtSignedInt(delta.gateFail)} |`,
|
|
169
|
+
`| gate pass rate | ${baseline.gateTotal ? `${baseline.gatePass}/${baseline.gateTotal}` : EM} | ${current.gateTotal ? `${current.gatePass}/${current.gateTotal}` : EM} | ${EM} |`,
|
|
170
|
+
`| cost | ${fmtUsd(baseline.costUsd)} | ${fmtUsd(current.costUsd)} | ${fmtSignedUsd(delta.costUsd)} |`,
|
|
171
|
+
`| tokens | ${fmtInt(baseline.tokensTotal)} | ${fmtInt(current.tokensTotal)} | ${fmtSignedInt(delta.tokensTotal)} |`,
|
|
172
|
+
"",
|
|
173
|
+
];
|
|
174
|
+
if (!comparability.comparable) {
|
|
175
|
+
lines.push("_Caveat: deltas are shown for inspection only — environment identity disagrees, so this is not an apples-to-apples table._", "");
|
|
176
|
+
}
|
|
177
|
+
return lines.join("\n").trimEnd() + "\n";
|
|
178
|
+
}
|
|
179
|
+
// Fail closed when either journal has no run-start (no partial / fabricated comparison).
|
|
180
|
+
// Mismatch of environment identity still yields a rendered delta, with an explicit caveat.
|
|
181
|
+
export function compareRuns(opts) {
|
|
182
|
+
if (!hasRunStart(opts.baselineEvents)) {
|
|
183
|
+
return {
|
|
184
|
+
ok: false,
|
|
185
|
+
reason: `baseline run ${opts.baselineRunId} has no recorded run-start event — cannot compare`,
|
|
186
|
+
};
|
|
187
|
+
}
|
|
188
|
+
if (!hasRunStart(opts.events)) {
|
|
189
|
+
return {
|
|
190
|
+
ok: false,
|
|
191
|
+
reason: `run ${opts.runId} has no recorded run-start event — cannot compare`,
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
const baselineEnv = recordedEnvironment(opts.baselineEvents);
|
|
195
|
+
const currentEnv = recordedEnvironment(opts.events);
|
|
196
|
+
const comparability = environmentComparable(baselineEnv, currentEnv);
|
|
197
|
+
const baseline = runMetrics(opts.baselineEvents, opts.baselineRows ?? [], opts.cost ?? {});
|
|
198
|
+
const current = runMetrics(opts.events, opts.rows ?? [], opts.cost ?? {});
|
|
199
|
+
const delta = {
|
|
200
|
+
durationMs: current.durationMs !== undefined && baseline.durationMs !== undefined
|
|
201
|
+
? current.durationMs - baseline.durationMs
|
|
202
|
+
: undefined,
|
|
203
|
+
gateFail: current.gateFail - baseline.gateFail,
|
|
204
|
+
costUsd: current.costUsd !== undefined && baseline.costUsd !== undefined
|
|
205
|
+
? Math.round((current.costUsd - baseline.costUsd) * 1e6) / 1e6
|
|
206
|
+
: undefined,
|
|
207
|
+
tokensTotal: current.tokensTotal !== undefined && baseline.tokensTotal !== undefined
|
|
208
|
+
? current.tokensTotal - baseline.tokensTotal
|
|
209
|
+
: undefined,
|
|
210
|
+
};
|
|
211
|
+
const text = renderComparison({
|
|
212
|
+
runId: opts.runId,
|
|
213
|
+
baselineRunId: opts.baselineRunId,
|
|
214
|
+
comparability,
|
|
215
|
+
baselineEnv,
|
|
216
|
+
currentEnv,
|
|
217
|
+
current,
|
|
218
|
+
baseline,
|
|
219
|
+
delta,
|
|
220
|
+
});
|
|
221
|
+
return {
|
|
222
|
+
ok: true,
|
|
223
|
+
runId: opts.runId,
|
|
224
|
+
baselineRunId: opts.baselineRunId,
|
|
225
|
+
comparability,
|
|
226
|
+
current,
|
|
227
|
+
baseline,
|
|
228
|
+
delta,
|
|
229
|
+
text,
|
|
230
|
+
};
|
|
231
|
+
}
|
package/dist/run/daemon.js
CHANGED
|
@@ -15,7 +15,9 @@ import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../ga
|
|
|
15
15
|
import { runGates } from "../gates/run-gates.js";
|
|
16
16
|
import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus } from "../graph/graph.js";
|
|
17
17
|
import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
18
|
+
import { runEnvironment } from "./environment.js";
|
|
18
19
|
import { cleanupRunWorktrees, gitHead, linkNodeModules, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
20
|
+
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
19
21
|
import { classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId } from "./journal.js";
|
|
20
22
|
import { acquireRunLock, releaseRunLock } from "./lock.js";
|
|
21
23
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
@@ -64,6 +66,10 @@ export function formatSummary(s) {
|
|
|
64
66
|
return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}`;
|
|
65
67
|
}
|
|
66
68
|
const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can never loop forever
|
|
69
|
+
// v1.70 T5: request-changes review rounds a single task may draw before it parks for a human decision
|
|
70
|
+
// instead of cycling. Well below MAX_ATTEMPTS so review non-convergence is caught long before the
|
|
71
|
+
// global cap. ponytail: literal constant; lift to cfg.review.roundCap only if a second knob-turner appears.
|
|
72
|
+
const REVIEW_ROUND_CAP = 3;
|
|
67
73
|
const BLOCKED_POLL_MS = 30_000; // between trailer-wait slices, check whether the pane is blocked on a prompt
|
|
68
74
|
const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
|
|
69
75
|
const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
|
|
@@ -252,7 +258,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
252
258
|
baseRef = await gitHead(repoRoot);
|
|
253
259
|
baseline = await captureBaseline(repoRoot, commands);
|
|
254
260
|
writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(baseline, null, 2));
|
|
255
|
-
|
|
261
|
+
// v1.70 T2: environment identity beside the graph/branch identity — running tickmarkr version,
|
|
262
|
+
// loaded-config hash, and the probed CLI version of each adapter holding a channel in the run,
|
|
263
|
+
// gathered through the existing probe/config-load paths (no second mechanism).
|
|
264
|
+
const environment = runEnvironment(cfg, channels, health);
|
|
265
|
+
journal.append("run-start", undefined, { pid: process.pid, baseRef, commands, channels: channels.map(channelKey), branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source, environment, ...(prior ? { supersedes: prior.runId } : {}) }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
|
|
256
266
|
runStarted = true;
|
|
257
267
|
// v1.53 T5: mark the prior run AFTER this run's run-start exists, so the prior journal never
|
|
258
268
|
// names a successor that has no journal. Append-only — the prior journal is never rewritten.
|
|
@@ -409,6 +419,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
409
419
|
if (rs)
|
|
410
420
|
journal.append("resume-restore", t.id, { attempts: rs.attempts, tried: [...tried], assignment });
|
|
411
421
|
const badReviewers = []; // v1.1: reviewer channels that produced unparseable output for this task
|
|
422
|
+
// v1.70 T5 (review-convergence): failed review rounds this task has drawn, counted from the per-task
|
|
423
|
+
// review history already in the journal — the SAME review gate-result stream onGate reads to grow the
|
|
424
|
+
// reviewer-exclusion list (badReviewers), never a second parallel counter. request-changes rounds do
|
|
425
|
+
// not exclude their reviewer (a fix is re-checked by the same seat), so their count lives here in the
|
|
426
|
+
// shared history rather than in badReviewers' garbage-only exclusion subset.
|
|
427
|
+
const reviewRoundsDrawn = () => journal.read().filter((e) => e.taskId === t.id && e.event === "gate-result" &&
|
|
428
|
+
e.data.gate === "review" &&
|
|
429
|
+
e.data.pass === false).length;
|
|
412
430
|
let feedback = "";
|
|
413
431
|
let ladderIdx = 0;
|
|
414
432
|
let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
|
|
@@ -509,6 +527,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
509
527
|
await park(t, `attempt cap (${MAX_ATTEMPTS}) reached`, "attempt-cap", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
510
528
|
return;
|
|
511
529
|
}
|
|
530
|
+
// v1.70 T5: a task that has already drawn REVIEW_ROUND_CAP request-changes review rounds parks for
|
|
531
|
+
// a human decision instead of dispatching another round. Condition on the review history (data),
|
|
532
|
+
// never the code path — and let an operator's approval (the human decision the cap asked for)
|
|
533
|
+
// release it, mirroring the humanGate guard's condition-on-approval precedent. Rounds only accrue
|
|
534
|
+
// after attempt 0, so the guard skips the journal read on the happy path.
|
|
535
|
+
if (attempt > 0 && !approved.has(t.id) && reviewRoundsDrawn() >= REVIEW_ROUND_CAP) {
|
|
536
|
+
await park(t, `review round cap (${REVIEW_ROUND_CAP}) reached — request-changes reviews not converging; a human should decide`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
537
|
+
return;
|
|
538
|
+
}
|
|
512
539
|
// OBS-57: a demoted channel must not be re-dispatched on consult retry or provider requeue.
|
|
513
540
|
if (demotedChannels.has(channelKey(assignment))) {
|
|
514
541
|
const next = nextChannel(assignment, t, cfg, channels, tried, profile, demotedChannels);
|
|
@@ -601,17 +628,22 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
601
628
|
: cfg.visibility.worker === "interactive" && driver.interactive
|
|
602
629
|
? adapter.interactiveCommand(promptFile, assignment.model)
|
|
603
630
|
: null;
|
|
604
|
-
|
|
631
|
+
// v1.69 T6: adapters that declare interactiveSeed launch the real TUI and inject the prompt as a
|
|
632
|
+
// user turn; they do NOT need the argv-seeding surface that interactiveCommand represents.
|
|
633
|
+
const hasSeed = retryMode !== "resume" && cfg.visibility.worker === "interactive" && driver.interactive && !!adapter.interactiveSeed;
|
|
634
|
+
if (cfg.visibility.worker === "interactive" && icmd === null && !hasSeed && !modeFallbackNoted) {
|
|
605
635
|
modeFallbackNoted = true;
|
|
606
636
|
journal.append("worker-mode-fallback", t.id, { reason: driver.interactive ? "adapter" : "driver" });
|
|
607
637
|
}
|
|
608
|
-
const interactive = icmd !== null;
|
|
638
|
+
const interactive = icmd !== null || hasSeed;
|
|
609
639
|
// OBS-85 (v1.62 T1): both dispatch branches deliver ONE short script invocation — banner,
|
|
610
640
|
// adapter command, and nonce exit marker live in a per-attempt script beside the prompt
|
|
611
641
|
// artifact (the same paneDispatchCommand pattern judge/review/consult dispatches use). The
|
|
612
642
|
// delivered pane line carries no command substitution and no trailing shell text, so paste
|
|
613
643
|
// timing can never interleave a `$(…)` with what follows it (the codex corruption class).
|
|
614
|
-
const workerCmd = interactive
|
|
644
|
+
const workerCmd = interactive
|
|
645
|
+
? (hasSeed ? ":" : icmd)
|
|
646
|
+
: adapter.invoke(t, wt, assignment, { promptFile }).command;
|
|
615
647
|
const dispatchScript = promptFile.replace(/\.md$/, ".sh");
|
|
616
648
|
writeFileSync(dispatchScript, [
|
|
617
649
|
"export BASH_SILENCE_DEPRECATION_WARNING=1",
|
|
@@ -659,117 +691,134 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
659
691
|
let exitCode;
|
|
660
692
|
let timedOut = false;
|
|
661
693
|
let settleParsed;
|
|
694
|
+
let seedResult;
|
|
662
695
|
if (interactive) {
|
|
663
696
|
// v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
|
|
664
697
|
// The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
|
|
665
|
-
await driver.run(slot, paneDispatchCommand(dispatchScript));
|
|
666
|
-
let paged = false;
|
|
667
|
-
// v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
|
|
668
|
-
// Any other blocked/idle dialog still pages the operator (paged latch below).
|
|
669
|
-
let trustAnswered = false;
|
|
670
698
|
finished = false;
|
|
671
699
|
exitCode = null;
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
// trailer detection, harvest, paging, and quota checks all read the raw pane.
|
|
679
|
-
let lastStallSnapshot = normalizeStallSnapshot(output);
|
|
680
|
-
let lastOutputAt = Date.now();
|
|
681
|
-
while (Date.now() - lastOutputAt < stallWindowMs) {
|
|
682
|
-
const sliceStart = Date.now();
|
|
683
|
-
const remaining = stallWindowMs - (sliceStart - lastOutputAt);
|
|
684
|
-
const slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
685
|
-
if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
|
|
686
|
-
// verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
|
|
687
|
-
// own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
|
|
688
|
-
// parseable trailer or a digit-suffixed exit marker in the harvest is completion.
|
|
689
|
-
output = await driver.read(slot, 1000); // TUI transcripts carry chrome — read deeper than print's 500
|
|
690
|
-
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
691
|
-
const exit = exitRe.exec(output);
|
|
692
|
-
if (finished || exit) {
|
|
693
|
-
exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
|
|
694
|
-
await sampleContext(); // final poll-seam sample before leaving the wait
|
|
695
|
-
break;
|
|
696
|
-
}
|
|
697
|
-
}
|
|
698
|
-
const currentStallSnapshot = normalizeStallSnapshot(await driver.read(slot, 1000));
|
|
699
|
-
if (currentStallSnapshot !== lastStallSnapshot) {
|
|
700
|
-
lastStallSnapshot = currentStallSnapshot;
|
|
701
|
-
lastOutputAt = Date.now();
|
|
702
|
-
}
|
|
703
|
-
// v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
|
|
704
|
-
await sampleContext();
|
|
705
|
-
// page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
|
|
706
|
-
// (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
|
|
707
|
-
// a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
|
|
708
|
-
const st = paged ? "" : await driver.status(slot);
|
|
709
|
-
if (!paged && (st === "blocked" || st === "idle")) {
|
|
710
|
-
// T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
|
|
711
|
-
// text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
|
|
712
|
-
if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
|
|
713
|
-
try {
|
|
714
|
-
const paneText = await driver.read(slot, 80);
|
|
715
|
-
if (matchesTrustDialog(paneText, adapter.trustDialog)) {
|
|
716
|
-
trustAnswered = true;
|
|
717
|
-
// v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
|
|
718
|
-
// Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
|
|
719
|
-
journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
|
|
720
|
-
await driver.sendKey(slot, adapter.trustDialog.key);
|
|
721
|
-
const spent = Date.now() - sliceStart;
|
|
722
|
-
if (spent < slice)
|
|
723
|
-
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
724
|
-
continue; // do not page — keep waiting for the trailer
|
|
725
|
-
}
|
|
726
|
-
}
|
|
727
|
-
catch {
|
|
728
|
-
/* read/send failed — fall through to page the operator */
|
|
729
|
-
}
|
|
730
|
-
}
|
|
731
|
-
paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
|
|
732
|
-
const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
|
|
733
|
-
await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
|
|
734
|
-
}
|
|
735
|
-
// a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
|
|
736
|
-
const spent = Date.now() - sliceStart;
|
|
737
|
-
if (spent < slice)
|
|
738
|
-
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
700
|
+
if (adapter.interactiveSeed) {
|
|
701
|
+
// v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
|
|
702
|
+
// then fall through to the normal trailer harvest. A failed seed is recorded as a finished
|
|
703
|
+
// failure rather than allowed to race the trailer wait.
|
|
704
|
+
seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
|
|
705
|
+
output = seedResult.output;
|
|
739
706
|
}
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
timedOut = Date.now() - lastOutputAt >= stallWindowMs;
|
|
707
|
+
else {
|
|
708
|
+
await driver.run(slot, paneDispatchCommand(dispatchScript));
|
|
743
709
|
output = await driver.read(slot, 1000);
|
|
744
|
-
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
745
|
-
const exit = exitRe.exec(output);
|
|
746
|
-
exitCode = exit ? Number(exit[1]) : null;
|
|
747
710
|
}
|
|
748
|
-
if (
|
|
749
|
-
|
|
750
|
-
output = await driver.read(slot, 1000);
|
|
711
|
+
if (seedResult?.seedFailed) {
|
|
712
|
+
finished = false;
|
|
751
713
|
}
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
714
|
+
else {
|
|
715
|
+
let paged = false;
|
|
716
|
+
// v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
|
|
717
|
+
// Any other blocked/idle dialog still pages the operator (paged latch below).
|
|
718
|
+
let trustAnswered = false;
|
|
719
|
+
finished = false;
|
|
720
|
+
exitCode = null;
|
|
721
|
+
// OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
|
|
722
|
+
// stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
|
|
756
723
|
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
let
|
|
761
|
-
|
|
762
|
-
while (
|
|
763
|
-
const
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
await
|
|
724
|
+
// OBS-82: the stall clock compares NORMALIZED snapshots so a spinner glyph/elapsed-time
|
|
725
|
+
// repaint is silence, not activity. ONLY this inactivity compare sees normalized text —
|
|
726
|
+
// trailer detection, harvest, paging, and quota checks all read the raw pane.
|
|
727
|
+
let lastStallSnapshot = normalizeStallSnapshot(output);
|
|
728
|
+
let lastOutputAt = Date.now();
|
|
729
|
+
while (Date.now() - lastOutputAt < stallWindowMs) {
|
|
730
|
+
const sliceStart = Date.now();
|
|
731
|
+
const remaining = stallWindowMs - (sliceStart - lastOutputAt);
|
|
732
|
+
const slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
733
|
+
if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
|
|
734
|
+
// verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
|
|
735
|
+
// own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
|
|
736
|
+
// parseable trailer or a digit-suffixed exit marker in the harvest is completion.
|
|
737
|
+
output = await driver.read(slot, 1000); // TUI transcripts carry chrome — read deeper than print's 500
|
|
738
|
+
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
739
|
+
const exit = exitRe.exec(output);
|
|
740
|
+
if (finished || exit) {
|
|
741
|
+
exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
|
|
742
|
+
await sampleContext(); // final poll-seam sample before leaving the wait
|
|
743
|
+
break;
|
|
744
|
+
}
|
|
745
|
+
}
|
|
746
|
+
const currentStallSnapshot = normalizeStallSnapshot(await driver.read(slot, 1000));
|
|
747
|
+
if (currentStallSnapshot !== lastStallSnapshot) {
|
|
748
|
+
lastStallSnapshot = currentStallSnapshot;
|
|
749
|
+
lastOutputAt = Date.now();
|
|
750
|
+
}
|
|
751
|
+
// v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
|
|
752
|
+
await sampleContext();
|
|
753
|
+
// page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
|
|
754
|
+
// (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
|
|
755
|
+
// a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
|
|
756
|
+
const st = paged ? "" : await driver.status(slot);
|
|
757
|
+
if (!paged && (st === "blocked" || st === "idle")) {
|
|
758
|
+
// T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
|
|
759
|
+
// text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
|
|
760
|
+
if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
|
|
761
|
+
try {
|
|
762
|
+
const paneText = await driver.read(slot, 80);
|
|
763
|
+
if (matchesTrustDialog(paneText, adapter.trustDialog)) {
|
|
764
|
+
trustAnswered = true;
|
|
765
|
+
// v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
|
|
766
|
+
// Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
|
|
767
|
+
journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
|
|
768
|
+
await driver.sendKey(slot, adapter.trustDialog.key);
|
|
769
|
+
const spent = Date.now() - sliceStart;
|
|
770
|
+
if (spent < slice)
|
|
771
|
+
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
772
|
+
continue; // do not page — keep waiting for the trailer
|
|
773
|
+
}
|
|
774
|
+
}
|
|
775
|
+
catch {
|
|
776
|
+
/* read/send failed — fall through to page the operator */
|
|
777
|
+
}
|
|
778
|
+
}
|
|
779
|
+
paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
|
|
780
|
+
const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
|
|
781
|
+
await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
|
|
782
|
+
}
|
|
783
|
+
// a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
|
|
784
|
+
const spent = Date.now() - sliceStart;
|
|
785
|
+
if (spent < slice)
|
|
786
|
+
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
787
|
+
}
|
|
788
|
+
if (!finished && exitCode === null) {
|
|
789
|
+
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
790
|
+
timedOut = Date.now() - lastOutputAt >= stallWindowMs;
|
|
767
791
|
output = await driver.read(slot, 1000);
|
|
768
|
-
|
|
769
|
-
|
|
792
|
+
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
793
|
+
const exit = exitRe.exec(output);
|
|
794
|
+
exitCode = exit ? Number(exit[1]) : null;
|
|
770
795
|
}
|
|
771
|
-
if (
|
|
772
|
-
|
|
796
|
+
if (finished) {
|
|
797
|
+
await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
|
|
798
|
+
output = await driver.read(slot, 1000);
|
|
799
|
+
}
|
|
800
|
+
// T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
|
|
801
|
+
// contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
|
|
802
|
+
// the existing pane-read seam once or twice before recording a malformed-trailer cause.
|
|
803
|
+
if (interactive) {
|
|
804
|
+
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
805
|
+
const settleDeadline = attemptStart + stallWindowMs;
|
|
806
|
+
const settleDelayMs = 1_000;
|
|
807
|
+
const maxSettleRetries = 2;
|
|
808
|
+
let settleTries = 0;
|
|
809
|
+
settleParsed = adapter.parse(output, nonce);
|
|
810
|
+
while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
|
|
811
|
+
const remaining = settleDeadline - Date.now();
|
|
812
|
+
if (remaining <= 0)
|
|
813
|
+
break;
|
|
814
|
+
await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
|
|
815
|
+
output = await driver.read(slot, 1000);
|
|
816
|
+
settleParsed = adapter.parse(output, nonce);
|
|
817
|
+
settleTries++;
|
|
818
|
+
}
|
|
819
|
+
if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
|
|
820
|
+
finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
|
|
821
|
+
}
|
|
773
822
|
}
|
|
774
823
|
}
|
|
775
824
|
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { AuthHealth, BillingChannel } from "../adapters/types.js";
|
|
2
|
+
import type { TickmarkrConfig } from "../config/config.js";
|
|
3
|
+
export interface RunEnvironment {
|
|
4
|
+
tickmarkrVersion: string;
|
|
5
|
+
configHash: string;
|
|
6
|
+
adapterVersions: Record<string, string>;
|
|
7
|
+
}
|
|
8
|
+
export declare const UNKNOWN_ADAPTER_VERSION = "unknown";
|
|
9
|
+
export declare function tickmarkrVersion(): string;
|
|
10
|
+
export declare function configHash(cfg: TickmarkrConfig): string;
|
|
11
|
+
export declare function adapterVersions(channels: BillingChannel[], health: Record<string, AuthHealth>): Record<string, string>;
|
|
12
|
+
export declare function runEnvironment(cfg: TickmarkrConfig, channels: BillingChannel[], health: Record<string, AuthHealth>): RunEnvironment;
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { readFileSync } from "node:fs";
|
|
3
|
+
import { dirname, join } from "node:path";
|
|
4
|
+
import { fileURLToPath } from "node:url";
|
|
5
|
+
// An adapter whose version probe failed is recorded, not dropped — "unknown", never a fabricated string.
|
|
6
|
+
export const UNKNOWN_ADAPTER_VERSION = "unknown";
|
|
7
|
+
// Same package.json read as src/cli/commands/version.ts (one resolution pattern, two consumers).
|
|
8
|
+
const pkgPath = join(dirname(fileURLToPath(import.meta.url)), "../../package.json");
|
|
9
|
+
export function tickmarkrVersion() {
|
|
10
|
+
const { version: v } = JSON.parse(readFileSync(pkgPath, "utf8"));
|
|
11
|
+
return v;
|
|
12
|
+
}
|
|
13
|
+
// Key order in a parsed config is an accident of the schema/merge layers, so the hash canonicalizes
|
|
14
|
+
// first: object keys sorted recursively, undefined dropped (JSON semantics), array order preserved.
|
|
15
|
+
function stableStringify(v) {
|
|
16
|
+
if (v === null || typeof v !== "object")
|
|
17
|
+
return JSON.stringify(v);
|
|
18
|
+
if (Array.isArray(v))
|
|
19
|
+
return `[${v.map(stableStringify).join(",")}]`;
|
|
20
|
+
const entries = Object.entries(v)
|
|
21
|
+
.filter(([, val]) => val !== undefined)
|
|
22
|
+
.sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0));
|
|
23
|
+
return `{${entries.map(([k, val]) => `${JSON.stringify(k)}:${stableStringify(val)}`).join(",")}}`;
|
|
24
|
+
}
|
|
25
|
+
// sha256 truncated to 16 hex — the graphDefinitionHash convention (stable, grep-friendly).
|
|
26
|
+
export function configHash(cfg) {
|
|
27
|
+
return createHash("sha256").update(stableStringify(cfg)).digest("hex").slice(0, 16);
|
|
28
|
+
}
|
|
29
|
+
// One entry per adapter with a channel in the run (not per channel). The version is whatever the
|
|
30
|
+
// adapter's own probe recorded in health; a missing/undefined probe result becomes "unknown".
|
|
31
|
+
export function adapterVersions(channels, health) {
|
|
32
|
+
const out = {};
|
|
33
|
+
for (const c of channels) {
|
|
34
|
+
if (!(c.adapter in out))
|
|
35
|
+
out[c.adapter] = health[c.adapter]?.version ?? UNKNOWN_ADAPTER_VERSION;
|
|
36
|
+
}
|
|
37
|
+
return out;
|
|
38
|
+
}
|
|
39
|
+
export function runEnvironment(cfg, channels, health) {
|
|
40
|
+
return { tickmarkrVersion: tickmarkrVersion(), configHash: configHash(cfg), adapterVersions: adapterVersions(channels, health) };
|
|
41
|
+
}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import type { Assignment, WorkerAdapter } from "../adapters/types.js";
|
|
2
|
+
import type { ExecutorDriver, Slot } from "../drivers/types.js";
|
|
3
|
+
export interface InteractiveSeedResult {
|
|
4
|
+
output: string;
|
|
5
|
+
seedFailed: boolean;
|
|
6
|
+
seedError?: string;
|
|
7
|
+
}
|
|
8
|
+
export declare function runInteractiveSeed(opts: {
|
|
9
|
+
driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
|
|
10
|
+
slot: Slot;
|
|
11
|
+
adapter: WorkerAdapter;
|
|
12
|
+
assignment: Assignment;
|
|
13
|
+
promptFile: string;
|
|
14
|
+
taskTimeoutMinutes: number;
|
|
15
|
+
}): Promise<InteractiveSeedResult>;
|