tickmarkr 1.84.0 → 1.86.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/adapters/catalog-remote.d.ts +64 -0
- package/dist/adapters/catalog-remote.js +287 -0
- package/dist/adapters/catalog.d.ts +96 -0
- package/dist/adapters/catalog.js +176 -0
- package/dist/adapters/claude-code.d.ts +1 -0
- package/dist/adapters/claude-code.js +59 -1
- package/dist/adapters/fake.js +42 -4
- package/dist/adapters/model-lints.d.ts +25 -5
- package/dist/adapters/model-lints.js +184 -50
- package/dist/adapters/model-windows.d.ts +31 -0
- package/dist/adapters/model-windows.js +69 -0
- package/dist/adapters/prompt.d.ts +5 -1
- package/dist/adapters/prompt.js +13 -4
- package/dist/adapters/registry.d.ts +25 -26
- package/dist/adapters/registry.js +173 -110
- package/dist/adapters/types.d.ts +3 -0
- package/dist/adapters/types.js +36 -3
- package/dist/brand.d.ts +5 -1
- package/dist/brand.js +18 -2
- package/dist/cli/commands/doctor.d.ts +3 -0
- package/dist/cli/commands/doctor.js +43 -21
- package/dist/cli/commands/fleet.d.ts +7 -0
- package/dist/cli/commands/fleet.js +94 -74
- package/dist/cli/commands/init.js +118 -5
- package/dist/cli/commands/status.js +202 -46
- package/dist/compile/collateral.d.ts +86 -2
- package/dist/compile/collateral.js +294 -3
- package/dist/compile/gsd.d.ts +2 -1
- package/dist/compile/gsd.js +68 -2
- package/dist/compile/native.d.ts +14 -0
- package/dist/compile/native.js +161 -12
- package/dist/config/config.d.ts +82 -5
- package/dist/config/config.js +253 -66
- package/dist/config/fleet-overlay.d.ts +25 -20
- package/dist/config/fleet-overlay.js +195 -77
- package/dist/config/fleet-why.d.ts +23 -0
- package/dist/config/fleet-why.js +42 -0
- package/dist/drivers/herdr.d.ts +21 -3
- package/dist/drivers/herdr.js +344 -110
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/baseline.d.ts +1 -0
- package/dist/gates/baseline.js +91 -13
- package/dist/gates/llm.d.ts +0 -1
- package/dist/gates/llm.js +5 -30
- package/dist/gates/review.d.ts +9 -1
- package/dist/gates/review.js +105 -10
- package/dist/gates/run-gates.d.ts +9 -0
- package/dist/gates/run-gates.js +285 -41
- package/dist/gates/verdict-cause.d.ts +4 -0
- package/dist/gates/verdict-cause.js +63 -0
- package/dist/graph/schema.d.ts +6 -0
- package/dist/graph/schema.js +8 -5
- package/dist/route/router.d.ts +0 -5
- package/dist/route/router.js +16 -20
- package/dist/run/consult.d.ts +6 -0
- package/dist/run/consult.js +35 -25
- package/dist/run/daemon.d.ts +48 -2
- package/dist/run/daemon.js +1488 -330
- package/dist/run/journal.d.ts +56 -3
- package/dist/run/journal.js +358 -4
- package/dist/run/stall.d.ts +35 -1
- package/dist/run/stall.js +118 -8
- package/dist/tui/cockpit/capture.d.ts +12 -0
- package/dist/tui/cockpit/capture.js +37 -1
- package/dist/tui/cockpit/components.d.ts +2 -0
- package/dist/tui/cockpit/components.js +8 -8
- package/dist/tui/cockpit/derive.d.ts +29 -2
- package/dist/tui/cockpit/derive.js +219 -23
- package/dist/tui/cockpit/run-cockpit.js +128 -27
- package/dist/tui/cockpit/theme.d.ts +32 -26
- package/dist/tui/cockpit/theme.js +11 -5
- package/dist/tui/ink/components.d.ts +0 -15
- package/dist/tui/ink/components.js +0 -17
- package/dist/tui/ink/fleet-app.d.ts +4 -1
- package/dist/tui/ink/fleet-app.js +134 -13
- package/fixtures/sample.native.md +1 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +354 -34
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +70 -0
- package/skills/tickmarkr-overseer/scripts/watch-panes.sh +1 -1
- package/dist/tui/ink/studio-app.d.ts +0 -59
- package/dist/tui/ink/studio-app.js +0 -320
- package/dist/tui/save.d.ts +0 -38
- package/dist/tui/save.js +0 -96
- package/dist/tui/staging.d.ts +0 -29
- package/dist/tui/staging.js +0 -78
package/dist/run/daemon.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { randomBytes } from "node:crypto";
|
|
1
|
+
import { createHash, randomBytes } from "node:crypto";
|
|
2
2
|
import { shq } from "../adapters/types.js";
|
|
3
|
-
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
3
|
+
import { appendFileSync, closeSync, existsSync, mkdirSync, mkdtempSync, openSync, readFileSync, readSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
4
4
|
import { tmpdir } from "node:os";
|
|
5
5
|
import { join } from "node:path";
|
|
6
6
|
import { stringify } from "yaml";
|
|
@@ -8,7 +8,7 @@ import { classifyDeadChannel, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TR
|
|
|
8
8
|
import { allAdapters, discoverChannels, getAdapter, probeAll, readDoctor } from "../adapters/registry.js";
|
|
9
9
|
import { addUsage, channelKey, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
10
10
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
11
|
-
import { globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
|
|
11
|
+
import { DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
|
|
12
12
|
import { DeliveryReadinessError } from "../drivers/herdr.js";
|
|
13
13
|
import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
|
|
14
14
|
import { formatOwnedName } from "../drivers/types.js";
|
|
@@ -20,13 +20,13 @@ import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
|
20
20
|
import { runEnvironment } from "./environment.js";
|
|
21
21
|
import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
22
22
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
23
|
-
import { classifyTaskFailure, classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId, phaseForGate, recordedTaskFailureKind, reviewRoundsSinceApproval } from "./journal.js";
|
|
23
|
+
import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, GATE_FINGERPRINT_CAP, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, pendingRepairFindings, phaseForGate, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
24
24
|
import { isDiffCapPark } from "../gates/review.js";
|
|
25
25
|
import { acquireRunLock, releaseRunLock } from "./lock.js";
|
|
26
26
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
27
27
|
import { nextChannel, route } from "../route/router.js";
|
|
28
28
|
import { desiredPanes } from "./reconcile.js";
|
|
29
|
-
import { StallProgressTracker } from "./stall.js";
|
|
29
|
+
import { NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows } from "./stall.js";
|
|
30
30
|
const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
|
|
31
31
|
// An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
|
|
32
32
|
// carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
|
|
@@ -69,6 +69,107 @@ export function formatSummary(s) {
|
|
|
69
69
|
return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}`;
|
|
70
70
|
}
|
|
71
71
|
const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can never loop forever
|
|
72
|
+
// v1.85 T3 (retry economics): two repairs per engagement, then the fresh ladder. A repair re-uses the
|
|
73
|
+
// findings and the landed diff instead of re-buying onboarding; when two of them have not closed the
|
|
74
|
+
// battery, the cheaper next move is the ladder's channel change, not a third fix-only pass.
|
|
75
|
+
const MAX_REPAIRS = 2;
|
|
76
|
+
/** A named oracle decided this acceptance failure — deterministic, unlike an LLM judge verdict. */
|
|
77
|
+
const isOracleFailure = (g) => g.details.startsWith("oracle failed:");
|
|
78
|
+
/**
|
|
79
|
+
* R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
|
|
80
|
+
* no longer forges `pass: true` to buy passage, so the merge decision has to read the same predicate
|
|
81
|
+
* the run surfaces already read (src/run/activity.ts): pass, or an honest declared skip. Without this
|
|
82
|
+
* the honesty change would silently park every judge-only task at merge — an unrun gate blocking work
|
|
83
|
+
* it was never asked to review. `skipped` is set only by a gate that says so about ITSELF; a red
|
|
84
|
+
* verdict from a review that actually ran still fails here, exactly as before.
|
|
85
|
+
*
|
|
86
|
+
* ONE pair of predicates, every fold. `!g.pass` was correct only while the sole `pass:false` producer
|
|
87
|
+
* was a gate that actually failed; the moment a decline can be recorded red, every `!g.pass` in this
|
|
88
|
+
* file — the retry feedback brief, the review-fix eligibility test, the failing-battery list the
|
|
89
|
+
* ladder and the fingerprint cap are scored on, the structured findings attached to a blocking
|
|
90
|
+
* verdict — reads an unrun gate as a defect. `gateFailed` is the seam they now share, and the journal
|
|
91
|
+
* write below is the seam every OUT-of-file fold shares.
|
|
92
|
+
*/
|
|
93
|
+
const gateSatisfied = (g) => g.pass || g.meta?.skipped === true;
|
|
94
|
+
const gateFailed = (g) => !gateSatisfied(g);
|
|
95
|
+
// v1.85 T3: the gates whose failure IS a deterministic measurement — a machine re-ran a command over a
|
|
96
|
+
// tree and printed the same bytes. Those are the failures the fingerprint cap governs (the ruling names
|
|
97
|
+
// it a "deterministic-gate" cap): a third identical answer to a question already answered twice is the
|
|
98
|
+
// ~663m-across-5-runs loop, whatever rung the ladder happens to stand on. An LLM verdict is a different
|
|
99
|
+
// object — two reviewers, or a judge asked twice, can restate one another without the question being
|
|
100
|
+
// closed — and each already carries a tighter bound of its own: REVIEW_ROUND_CAP parks review at the
|
|
101
|
+
// OPERATOR in two rounds, and a judge verdict rides the ladder and the attempt cap. The boundary is a
|
|
102
|
+
// property of the GATE, never of the ladder rung or of the move that would follow the failure.
|
|
103
|
+
const DETERMINISTIC_GATES = new Set(["build", "test", "lint", "evidence", "scope"]);
|
|
104
|
+
const isDeterministicFailure = (g) => DETERMINISTIC_GATES.has(g.gate) || (g.gate === "acceptance" && isOracleFailure(g));
|
|
105
|
+
/**
|
|
106
|
+
* Is the failing battery narrow enough that a fix-only pass can close it? The ruling's three cases:
|
|
107
|
+
* review-only, a single deterministic test/lint gate, or acceptance decided by a named oracle.
|
|
108
|
+
* Unparseable verdicts and diff-cap trips are excluded exactly as they are from the review-fix retry —
|
|
109
|
+
* neither names anything a worker can fix.
|
|
110
|
+
*/
|
|
111
|
+
function narrowRepairBattery(failing) {
|
|
112
|
+
if (failing.length !== 1)
|
|
113
|
+
return false;
|
|
114
|
+
const g = failing[0];
|
|
115
|
+
if (g.gate === "review")
|
|
116
|
+
return g.meta?.unparseable !== true && !isDiffCapPark(g);
|
|
117
|
+
if (g.gate === "test" || g.gate === "lint")
|
|
118
|
+
return true;
|
|
119
|
+
if (g.gate === "acceptance")
|
|
120
|
+
return isOracleFailure(g) && g.meta?.unparseable !== true;
|
|
121
|
+
return false;
|
|
122
|
+
}
|
|
123
|
+
/**
|
|
124
|
+
* T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
|
|
125
|
+
* review are now launched together, so a round can journal a failed review that the serial walk would
|
|
126
|
+
* never have asked for — it returned at the judge. Those verdicts stay on the record (they are real,
|
|
127
|
+
* and the retry brief carries them), but they must not spend the OPERATOR-facing review round budget:
|
|
128
|
+
* otherwise concurrency alone parks a task rounds early for objections the old pipeline never bought.
|
|
129
|
+
* A round is the gate-result span opened by each `gates` phase-start, per task.
|
|
130
|
+
*/
|
|
131
|
+
export function decisiveReviewRounds(events) {
|
|
132
|
+
const open = new Map();
|
|
133
|
+
const spent = new Set();
|
|
134
|
+
const close = (taskId) => {
|
|
135
|
+
const round = open.get(taskId) ?? [];
|
|
136
|
+
if (round.some((e) => e.data.gate !== "review" && e.data.pass === false)) {
|
|
137
|
+
for (const e of round)
|
|
138
|
+
if (e.data.gate === "review" && e.data.pass === false)
|
|
139
|
+
spent.add(e);
|
|
140
|
+
}
|
|
141
|
+
open.delete(taskId);
|
|
142
|
+
};
|
|
143
|
+
for (const e of events) {
|
|
144
|
+
if (!e.taskId)
|
|
145
|
+
continue;
|
|
146
|
+
if (e.event === "phase-start" && e.data.phase === "gates")
|
|
147
|
+
close(e.taskId);
|
|
148
|
+
else if (e.event === "gate-result")
|
|
149
|
+
open.set(e.taskId, [...(open.get(e.taskId) ?? []), e]);
|
|
150
|
+
}
|
|
151
|
+
for (const taskId of open.keys())
|
|
152
|
+
close(taskId); // Map iteration tolerates deleting the current key
|
|
153
|
+
return events.filter((e) => !spent.has(e));
|
|
154
|
+
}
|
|
155
|
+
/** The fix-only contract: the findings verbatim, then the diff content of the work already landed. */
|
|
156
|
+
function repairBrief(findings, diff, baseRef) {
|
|
157
|
+
const repair = { findings, diff };
|
|
158
|
+
return [
|
|
159
|
+
"## Repair attempt — fix ONLY what these findings name",
|
|
160
|
+
"The commits from your prior attempt are already in this worktree and their diff is reproduced"
|
|
161
|
+
+ " below. Do NOT re-implement that work, do not start over, and do not revert it: make the"
|
|
162
|
+
+ " smallest change that resolves every finding, then commit.",
|
|
163
|
+
"",
|
|
164
|
+
"### Failing gate findings (verbatim)",
|
|
165
|
+
repair.findings,
|
|
166
|
+
"",
|
|
167
|
+
`### The work under review (git diff ${baseRef.slice(0, 12)}..HEAD)`,
|
|
168
|
+
"```diff",
|
|
169
|
+
repair.diff,
|
|
170
|
+
"```",
|
|
171
|
+
].join("\n");
|
|
172
|
+
}
|
|
72
173
|
// v1.70 T5: request-changes review rounds a single task may draw before it parks for a human decision
|
|
73
174
|
// instead of cycling. Well below MAX_ATTEMPTS so review non-convergence is caught long before the
|
|
74
175
|
// global cap. ponytail: literal constant; lift to cfg.review.roundCap only if a second knob-turner appears.
|
|
@@ -92,19 +193,23 @@ export function setEarlyLaunchLivenessMsForTests(ms) {
|
|
|
92
193
|
export function resetEarlyLaunchLivenessMsForTests() {
|
|
93
194
|
earlyLaunchLivenessMs = EARLY_LAUNCH_LIVENESS_MS;
|
|
94
195
|
}
|
|
95
|
-
// OBS-201: the liveness nudge — the daemon's ACTIVE response to
|
|
96
|
-
// replacing page-a-human-then-burn-the-window (289 of 692 worker-minutes in one measured
|
|
97
|
-
// Gate:
|
|
98
|
-
//
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
//
|
|
102
|
-
//
|
|
196
|
+
// OBS-201 + T1 (OBS-262): the liveness nudge — the daemon's ACTIVE response to a worker holding no
|
|
197
|
+
// trailer, replacing page-a-human-then-burn-the-window (289 of 692 worker-minutes in one measured
|
|
198
|
+
// day). Gate: NUDGE_AFTER_SILENT_MS of monotonic-tracker silence, regardless of the herdr status
|
|
199
|
+
// reading (unknown/working no longer suppress it — a wedged TUI often scrapes as either); a
|
|
200
|
+
// `blocked` pane still pages instead, since nudging a dialog prompt can't help. One nudge per
|
|
201
|
+
// attempt; if the grace passes with no progress, the wait concludes as a stall NOW and the consult
|
|
202
|
+
// sees the un-answered nudge instead of an hour of silence. Scope allowlist lives in stall.ts
|
|
203
|
+
// (claude-code only; widening is a fixture-capture chore). The message builder takes no nonce — the
|
|
103
204
|
// self-reference guard holds by construction, an echoed bare token can never match the wait regex.
|
|
104
|
-
export
|
|
205
|
+
export { NUDGEABLE_ADAPTERS } from "./stall.js";
|
|
105
206
|
export const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the task is complete, print your TICKMARKR_RESULT completion trailer exactly as specified in your prompt now. If not, state your next concrete action and continue working.";
|
|
106
|
-
const NUDGE_AFTER_SILENT_MS =
|
|
207
|
+
const NUDGE_AFTER_SILENT_MS = 10 * 60_000; // T1 (OBS-262): >=10m tracker silence — was 3m behind an unreachable status gate
|
|
107
208
|
const WORKER_NUDGE_GRACE_MS = 4 * 60_000;
|
|
209
|
+
// T1 review: a false return from driver.nudge is a DELIVERY outcome (missing pin, readiness
|
|
210
|
+
// stable-frame timeout, read-back hiccup), not proof of an unreachable channel — so one failure
|
|
211
|
+
// is retried once in-slice after this settle, and only a failed retry latches nudgeFailed.
|
|
212
|
+
const NUDGE_REDELIVER_MS = 2_000;
|
|
108
213
|
let nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
|
|
109
214
|
let workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
|
|
110
215
|
/** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
|
|
@@ -116,6 +221,375 @@ export function resetNudgeTimingForTests() {
|
|
|
116
221
|
nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
|
|
117
222
|
workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
|
|
118
223
|
}
|
|
224
|
+
// T1 (OBS-263): in-loop quota-banner classification — the banner IS output, so the empty-output
|
|
225
|
+
// rules can never catch it and the post-loop QUOTA_RE check only runs after the full window. Two
|
|
226
|
+
// consecutive matching slices plus this much monotonic-tracker silence classify (a worker whose
|
|
227
|
+
// diff merely quotes "rate limit" keeps working undisturbed).
|
|
228
|
+
const QUOTA_BANNER_SILENT_MS = 3 * 60_000;
|
|
229
|
+
let quotaBannerSilentMs = QUOTA_BANNER_SILENT_MS;
|
|
230
|
+
/** Test seam — shrink the quota-banner silence gate without minute-long sleeps. */
|
|
231
|
+
export function setQuotaBannerSilentMsForTests(ms) {
|
|
232
|
+
quotaBannerSilentMs = ms;
|
|
233
|
+
}
|
|
234
|
+
export function resetQuotaBannerSilentMsForTests() {
|
|
235
|
+
quotaBannerSilentMs = QUOTA_BANNER_SILENT_MS;
|
|
236
|
+
}
|
|
237
|
+
// T1 (OBS-262): the operator page is UNLATCHED — every eligible slice journals a page, and the
|
|
238
|
+
// notification is delivered again on a status change or once this cadence elapses. The cadence is
|
|
239
|
+
// an operator-spam guard only; it can no longer turn a stall into a single page forever. It sits
|
|
240
|
+
// BELOW the dead-channel fast-kill window on purpose (T1 review): at 5m == 5m the second delivery
|
|
241
|
+
// raced the kill on the same slice boundary, so an idle non-nudgeable pane holding no delta — the
|
|
242
|
+
// exact class the repeat exists for — got exactly one page in production.
|
|
243
|
+
const PAGE_REPEAT_MS = 2 * 60_000;
|
|
244
|
+
let pageRepeatMs = PAGE_REPEAT_MS;
|
|
245
|
+
/** Test seam — shrink the repeat-page cadence without minute-long sleeps. */
|
|
246
|
+
export function setPageRepeatMsForTests(ms) {
|
|
247
|
+
pageRepeatMs = ms;
|
|
248
|
+
}
|
|
249
|
+
export function resetPageRepeatMsForTests() {
|
|
250
|
+
pageRepeatMs = PAGE_REPEAT_MS;
|
|
251
|
+
}
|
|
252
|
+
// T1 (R1 dead-channel fast-kill): a worker with no trailer, no worktree delta, and no output
|
|
253
|
+
// growth for this long is dead — conclude immediately instead of burning the rolling window.
|
|
254
|
+
// Per-task timeoutMinutes stays the escape valve for slow-but-live workers.
|
|
255
|
+
const DEAD_CHANNEL_FAST_KILL_MS = 5 * 60_000;
|
|
256
|
+
let deadChannelFastKillMs = DEAD_CHANNEL_FAST_KILL_MS;
|
|
257
|
+
/** Test seam — shrink the fast-kill window without minute-long sleeps. */
|
|
258
|
+
export function setDeadChannelFastKillMsForTests(ms) {
|
|
259
|
+
deadChannelFastKillMs = ms;
|
|
260
|
+
}
|
|
261
|
+
export function resetDeadChannelFastKillMsForTests() {
|
|
262
|
+
deadChannelFastKillMs = DEAD_CHANNEL_FAST_KILL_MS;
|
|
263
|
+
}
|
|
264
|
+
// T2 (OBS-264): finished work is harvested, never redone. 18 of 18 observed stalls carried 2-33
|
|
265
|
+
// commits, and the redispatch then re-bought verification of work that had already landed. The
|
|
266
|
+
// liveness triad — commits made by this attempt, a FLAT worker-tree CPU delta, and this much
|
|
267
|
+
// monotonic-tracker silence — CONCLUDES the wait. Conclude, never kill: the pane is harvested by
|
|
268
|
+
// the same tail a window expiry uses, and the carried worktree goes straight to gates. Set at the
|
|
269
|
+
// fast-kill's window on purpose (the OBS-264 arithmetic is "a ~36m stall + ~15m redo becomes a
|
|
270
|
+
// ~5m gate pass"); a worker that is merely thinking still burns CPU and is never concluded here.
|
|
271
|
+
const HARVEST_SILENT_MS = 5 * 60_000;
|
|
272
|
+
let harvestSilentMs = HARVEST_SILENT_MS;
|
|
273
|
+
/** Test seam — shrink the harvest silence gate without minute-long sleeps. */
|
|
274
|
+
export function setHarvestSilentMsForTests(ms) {
|
|
275
|
+
harvestSilentMs = ms;
|
|
276
|
+
}
|
|
277
|
+
export function resetHarvestSilentMsForTests() {
|
|
278
|
+
harvestSilentMs = HARVEST_SILENT_MS;
|
|
279
|
+
}
|
|
280
|
+
// A CPU delta needs two samples separated in WALL CLOCK, and the CPU clock is QUANTIZED: darwin's
|
|
281
|
+
// `ps` prints hundredths ("0:00.03"), linux's prints whole seconds ("00:00:01"). Equality across a
|
|
282
|
+
// window shorter than the quantum is not evidence of anything — a worker throttled to a low duty
|
|
283
|
+
// cycle accrues less than one tick per sample and reads flat while genuinely working. So the flat
|
|
284
|
+
// observation must span the LARGER of a floor and this many ticks of the clock actually in use:
|
|
285
|
+
// crossing 30 ticks means the tree burned <1 tick in 30, i.e. under ~3% of one core. On a
|
|
286
|
+
// hundredths host that is a 3s window; on a whole-second host it is 30s — still nothing against the
|
|
287
|
+
// ~15m redispatch it replaces. Resolution is read off the sampled rows, never assumed.
|
|
288
|
+
const HARVEST_CPU_FLAT_MS = 3_000;
|
|
289
|
+
const HARVEST_CPU_FLAT_TICKS = 30;
|
|
290
|
+
// Once the flat window opens, retain descendants often enough to observe brief tool processes that
|
|
291
|
+
// can start and exit between the daemon's ordinary wait slices. This sampler exists only during an
|
|
292
|
+
// eligible silence window; it is stopped on progress or as soon as the worker wait concludes.
|
|
293
|
+
const HARVEST_CPU_ACCOUNTING_POLL_MS = 100;
|
|
294
|
+
// T2 review (material): that 100ms cadence forks a shell plus `ps` ten times a second, and on a host
|
|
295
|
+
// where `ps` is unsupported or denied (the managed-sandbox class) EVERY sample fails — tens of
|
|
296
|
+
// thousands of processes per silent attempt, multiplied by daemon concurrency, for a probe that can
|
|
297
|
+
// never conclude anything. Persistent failure is structural, not transient, so the sampler STOPS
|
|
298
|
+
// after this many consecutive unreadable snapshots. It stays stopped for the silence window it was
|
|
299
|
+
// started for: read() then reports no CPU, the triad refuses to conclude and journals the gap, and a
|
|
300
|
+
// later window (after real progress clears the accountant) starts a fresh one that pays the same
|
|
301
|
+
// bounded probe again.
|
|
302
|
+
const HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP = 20;
|
|
303
|
+
let harvestCpuFlatMs;
|
|
304
|
+
export function harvestCpuFlatWindowMs(resolutionMs) {
|
|
305
|
+
return harvestCpuFlatMs ?? Math.max(HARVEST_CPU_FLAT_MS, resolutionMs * HARVEST_CPU_FLAT_TICKS);
|
|
306
|
+
}
|
|
307
|
+
/** Test seam — pin the flat window so a probe case need not sit through a real one. */
|
|
308
|
+
export function setHarvestCpuFlatMsForTests(ms) {
|
|
309
|
+
harvestCpuFlatMs = ms;
|
|
310
|
+
}
|
|
311
|
+
export function resetHarvestCpuFlatMsForTests() {
|
|
312
|
+
harvestCpuFlatMs = undefined;
|
|
313
|
+
}
|
|
314
|
+
// Once the silence gate is met the CPU probe owns the poll cadence: the trailer-wait slice is 30s,
|
|
315
|
+
// so two samples would otherwise cost a minute of wall clock apiece. Below the gate the only rule
|
|
316
|
+
// is not to sleep PAST it — at the shipped 5m gate that changes no slice a worker sees today.
|
|
317
|
+
const HARVEST_POLL_MS = 2_000;
|
|
318
|
+
function harvestSliceMs(silentMs) {
|
|
319
|
+
return Math.max(100, silentMs >= harvestSilentMs ? HARVEST_POLL_MS : harvestSilentMs - silentMs);
|
|
320
|
+
}
|
|
321
|
+
// The synthesized result a carried no-trailer harvest hands to the gates. Distinct from anything a
|
|
322
|
+
// worker can claim: it never comes from adapter.parse, and it is journaled under its own event.
|
|
323
|
+
export const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries committed work; the worker emitted no TICKMARKR_RESULT trailer";
|
|
324
|
+
/** T4 (OBS-266): identity of the command SET a tip verify ran — a changed command is a different verify. */
|
|
325
|
+
export function commandsHash(commands) {
|
|
326
|
+
return createHash("sha256").update(JSON.stringify(Object.entries(commands).sort())).digest("hex").slice(0, 12);
|
|
327
|
+
}
|
|
328
|
+
/**
|
|
329
|
+
* T4 (OBS-266): the journal's LAST verification cycle — the (tip, cmdHash) pair the most recent run
|
|
330
|
+
* of the verify commands spoke for, the gates it got a pass from, and whether anything failed in it.
|
|
331
|
+
*
|
|
332
|
+
* The LAST one, never a history of every pair ever green. "The last GREEN verified SHA" is what the
|
|
333
|
+
* spec licenses a skip against, and only the last cycle is a statement about the state the run is in
|
|
334
|
+
* now: after A→B→A the tip really moved, and after commands A→B→A the last thing that ran on this
|
|
335
|
+
* SHA was command set B — both re-verify. A cycle is the contiguous run of events sharing one pair,
|
|
336
|
+
* so a cycle cut short by a killed process is missing gates and can never satisfy the caller. A
|
|
337
|
+
* legacy event (no tip/cmdHash) is unattributable and breaks the chain outright.
|
|
338
|
+
*/
|
|
339
|
+
function lastVerifyCycle(events) {
|
|
340
|
+
let cur;
|
|
341
|
+
let afterRunEnd = false;
|
|
342
|
+
for (const e of events) {
|
|
343
|
+
// New journals delimit every attempt explicitly. run-end is the legacy delimiter: it starts a
|
|
344
|
+
// new cycle only when another verify event follows, while preserving the just-closed cycle as
|
|
345
|
+
// the cache candidate for an otherwise unmoved next run-end.
|
|
346
|
+
if (e.event === "tip-verify-start") {
|
|
347
|
+
const { tip, cmdHash } = e.data;
|
|
348
|
+
cur = typeof tip === "string" && typeof cmdHash === "string"
|
|
349
|
+
? { tip, cmdHash, gates: new Set(), failed: false }
|
|
350
|
+
: undefined;
|
|
351
|
+
afterRunEnd = false;
|
|
352
|
+
continue;
|
|
353
|
+
}
|
|
354
|
+
if (e.event === "run-end") {
|
|
355
|
+
afterRunEnd = true;
|
|
356
|
+
continue;
|
|
357
|
+
}
|
|
358
|
+
if (e.event !== "tip-verify" && e.event !== "tip-verify-failed")
|
|
359
|
+
continue;
|
|
360
|
+
const { tip, gate, cmdHash } = e.data;
|
|
361
|
+
if (typeof tip !== "string" || typeof gate !== "string" || typeof cmdHash !== "string") {
|
|
362
|
+
cur = undefined;
|
|
363
|
+
afterRunEnd = false;
|
|
364
|
+
continue;
|
|
365
|
+
}
|
|
366
|
+
if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
|
|
367
|
+
cur = { tip, cmdHash, gates: new Set(), failed: false };
|
|
368
|
+
}
|
|
369
|
+
afterRunEnd = false;
|
|
370
|
+
if (e.event === "tip-verify-failed")
|
|
371
|
+
cur.failed = true;
|
|
372
|
+
else
|
|
373
|
+
cur.gates.add(gate);
|
|
374
|
+
}
|
|
375
|
+
return cur;
|
|
376
|
+
}
|
|
377
|
+
/**
|
|
378
|
+
* OBS-34's strict tip verify, but it stops re-paying for an unmoved tip (~334m corpus-wide; 69.5m in
|
|
379
|
+
* one park-heavy run whose 18 resume cycles merged nothing new). The verify journals the SHA it
|
|
380
|
+
* verified and the hash of the command set, so a later run-end can recognize the same verified state:
|
|
381
|
+
* head equals the LAST green verified SHA, commands unchanged, tree clean → journal
|
|
382
|
+
* `tip-verify-cached` and skip. ANY doubt — moved head, changed commands, dirty tree, a gate missing
|
|
383
|
+
* from that cycle's green set, a failure recorded in it, a cycle older than the last one — runs the
|
|
384
|
+
* full verify. The tip-verify-before-green law is untouched: a cached green is a verified green OF
|
|
385
|
+
* THAT EXACT COMMIT, established by the most recent real run of the same commands.
|
|
386
|
+
* Returns whether the tip is failing.
|
|
387
|
+
*/
|
|
388
|
+
export async function verifyIntegrationTipCached(intWt, commands, journal, opts = {}) {
|
|
389
|
+
const cmdHash = commandsHash(commands);
|
|
390
|
+
const tip = await gitHead(intWt);
|
|
391
|
+
const porcelain = await shGit("git status --porcelain", intWt);
|
|
392
|
+
const clean = porcelain.code === 0 && porcelain.stdout.trim() === "";
|
|
393
|
+
const last = lastVerifyCycle(journal.read());
|
|
394
|
+
const cached = last !== undefined && !last.failed && last.tip === tip && last.cmdHash === cmdHash
|
|
395
|
+
&& Object.keys(commands).every((g) => last.gates.has(g));
|
|
396
|
+
// A pair can be verified red and then green without either SHA or command hash changing (for
|
|
397
|
+
// example, an external service or ignored fixture recovers). Delimit attempts explicitly so that
|
|
398
|
+
// the earlier red cannot remain latched into the later complete green cycle.
|
|
399
|
+
journal.append("tip-verify-start", undefined, { tip, cmdHash, gates: Object.keys(commands), cached: clean && cached });
|
|
400
|
+
if (clean && cached) {
|
|
401
|
+
journal.append("tip-verify-cached", undefined, { tip, cmdHash, gates: Object.keys(commands) });
|
|
402
|
+
// The skip must not read as a red. Every surface derives the tip's verdict from this cycle's
|
|
403
|
+
// `tip-verify` events (cockpit derive.ts tipVerificationPassed: a run-end claiming "passed" with
|
|
404
|
+
// ZERO events is fail-closed to FALSE), so a carried-forward green still journals its per-gate
|
|
405
|
+
// pass — `cached: true` keeps it honest about not having re-run the command.
|
|
406
|
+
for (const gate of Object.keys(commands)) {
|
|
407
|
+
journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash });
|
|
408
|
+
}
|
|
409
|
+
return false;
|
|
410
|
+
}
|
|
411
|
+
let tipFailed = false;
|
|
412
|
+
for (const r of await verifyIntegrationTip(intWt, commands, journal.dir)) {
|
|
413
|
+
if (r.pass) {
|
|
414
|
+
journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, tip, cmdHash });
|
|
415
|
+
}
|
|
416
|
+
else {
|
|
417
|
+
journal.append("tip-verify-failed", undefined, {
|
|
418
|
+
gate: r.gate,
|
|
419
|
+
cmd: r.cmd,
|
|
420
|
+
exitCode: r.exitCode,
|
|
421
|
+
fingerprints: r.fingerprints,
|
|
422
|
+
artifact: r.artifact,
|
|
423
|
+
lastMergedTask: opts.lastMergedTask,
|
|
424
|
+
tip,
|
|
425
|
+
cmdHash,
|
|
426
|
+
});
|
|
427
|
+
tipFailed = true;
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
return tipFailed;
|
|
431
|
+
}
|
|
432
|
+
// `ps` CPU time: "[[dd-]hh:]mm:ss[.frac]" (darwin prints "0:00.03", linux "00:00:01", both print
|
|
433
|
+
// "1-02:03:04" past a day). Anything else is a header or a row this parser must not guess at.
|
|
434
|
+
// `frac` reports whether THIS host prints sub-second digits — the quantum the flat window is sized
|
|
435
|
+
// against, measured rather than assumed (a darwin sample is 10ms, a linux one 1000ms).
|
|
436
|
+
function parsePsCpu(raw) {
|
|
437
|
+
const m = /^(?:(\d+)-)?(?:(\d+):)?(\d+):(\d+(?:\.\d+)?)$/.exec(raw);
|
|
438
|
+
if (!m)
|
|
439
|
+
return undefined;
|
|
440
|
+
const ms = ((Number(m[1] ?? 0) * 24 + Number(m[2] ?? 0)) * 60 + Number(m[3])) * 60_000 + Math.round(Number(m[4]) * 1000);
|
|
441
|
+
return { ms, frac: m[4].includes(".") };
|
|
442
|
+
}
|
|
443
|
+
let linuxClockTickMs;
|
|
444
|
+
function linuxProcessCpuMs(pid, cwd) {
|
|
445
|
+
if (!existsSync("/proc/self/stat"))
|
|
446
|
+
return Promise.resolve(undefined);
|
|
447
|
+
// shGit, not sh: the accountant samples this path at a 100ms cadence, and a LOGIN shell would
|
|
448
|
+
// re-run the operator's profile (nvm/pyenv/direnv side effects included) on every sample.
|
|
449
|
+
linuxClockTickMs ??= shGit("getconf CLK_TCK", cwd, 15_000).then((r) => {
|
|
450
|
+
const ticks = r.code === 0 ? Number(r.stdout.trim()) : Number.NaN;
|
|
451
|
+
return Number.isFinite(ticks) && ticks > 0 ? 1_000 / ticks : undefined;
|
|
452
|
+
});
|
|
453
|
+
return linuxClockTickMs.then((resolutionMs) => {
|
|
454
|
+
if (resolutionMs === undefined)
|
|
455
|
+
return undefined;
|
|
456
|
+
try {
|
|
457
|
+
// `/proc/<pid>/stat` fields 14-17 are user/system jiffies for the process and its waited-for
|
|
458
|
+
// children. The child totals retain tools that start and exit wholly between live-tree polls.
|
|
459
|
+
// Split after the LAST ')' because comm may contain spaces or parentheses; field 3 is rest[0].
|
|
460
|
+
const stat = readFileSync(`/proc/${pid}/stat`, "utf8");
|
|
461
|
+
const fields = stat.slice(stat.lastIndexOf(")") + 2).trim().split(/\s+/);
|
|
462
|
+
const ticks = Number(fields[11]) + Number(fields[12]) + Number(fields[13]) + Number(fields[14]);
|
|
463
|
+
return Number.isFinite(ticks) ? { ms: ticks * resolutionMs, resolutionMs } : undefined;
|
|
464
|
+
}
|
|
465
|
+
catch {
|
|
466
|
+
return undefined; // process exited between ps ancestry capture and the precise CPU read
|
|
467
|
+
}
|
|
468
|
+
});
|
|
469
|
+
}
|
|
470
|
+
// T2 (OBS-264): the triad's CPU leg. Every non-seeded process of an attempt descends from that
|
|
471
|
+
// attempt's own dispatch script, whose path is unique — print, argv-interactive and resume launches
|
|
472
|
+
// all start there. (interactive-seed is intentionally fail-open below because its adapter-owned
|
|
473
|
+
// launch bypasses this script.) ONE `ps` snapshot finds the root and all current descendants:
|
|
474
|
+
// the agent CLI is a CHILD of the script's shell, so the root's own TIME never moves while the CLI
|
|
475
|
+
// thinks. `resolutionMs` is the sampled clock's quantum, which sizes the caller's flat window.
|
|
476
|
+
// Returns 0 when nothing matches: a worker whose process tree is gone is the strongest possible
|
|
477
|
+
// "not working". Returns undefined when the snapshot itself failed or parsed to nothing —
|
|
478
|
+
// unmeasurable CPU is never evidence a worker stopped, and the caller refuses to conclude on it.
|
|
479
|
+
async function workerTreeCpuSnapshot(marker, cwd) {
|
|
480
|
+
// shGit, not sh: same login-shell cost as the CLK_TCK probe above — `ps` needs no profile.
|
|
481
|
+
const snapshot = await shGit("ps -Awwo pid=,ppid=,time=,command=", cwd, 15_000);
|
|
482
|
+
if (snapshot.code !== 0)
|
|
483
|
+
return undefined;
|
|
484
|
+
const rows = [];
|
|
485
|
+
for (const line of snapshot.stdout.split("\n")) {
|
|
486
|
+
const m = /^\s*(\d+)\s+(\d+)\s+(\S+)\s+(.*)$/.exec(line);
|
|
487
|
+
if (!m)
|
|
488
|
+
continue;
|
|
489
|
+
const cpu = parsePsCpu(m[3]);
|
|
490
|
+
if (cpu !== undefined)
|
|
491
|
+
rows.push({ pid: m[1], ppid: m[2], cpuMs: cpu.ms, frac: cpu.frac, cmd: m[4] });
|
|
492
|
+
}
|
|
493
|
+
if (rows.length === 0)
|
|
494
|
+
return undefined;
|
|
495
|
+
const tree = new Set(rows.filter((p) => p.cmd.includes(marker)).map((p) => p.pid));
|
|
496
|
+
// `ps` output is not topologically ordered — relax the parent→child closure until it stops growing.
|
|
497
|
+
for (let grew = true; grew;) {
|
|
498
|
+
grew = false;
|
|
499
|
+
for (const p of rows) {
|
|
500
|
+
if (!tree.has(p.pid) && tree.has(p.ppid)) {
|
|
501
|
+
tree.add(p.pid);
|
|
502
|
+
grew = true;
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
const precise = new Map();
|
|
507
|
+
let preciseResolutionMs;
|
|
508
|
+
for (const p of rows) {
|
|
509
|
+
if (!tree.has(p.pid))
|
|
510
|
+
continue;
|
|
511
|
+
const cpu = await linuxProcessCpuMs(p.pid, cwd);
|
|
512
|
+
precise.set(p.pid, cpu?.ms ?? p.cpuMs);
|
|
513
|
+
if (cpu !== undefined)
|
|
514
|
+
preciseResolutionMs = cpu.resolutionMs;
|
|
515
|
+
}
|
|
516
|
+
// Even an empty worker tree needs the host's actual measurement quantum: on Linux the /proc
|
|
517
|
+
// jiffy clock remains available after the worker exits, while `ps time` only prints whole seconds.
|
|
518
|
+
if (preciseResolutionMs === undefined && existsSync("/proc/self/stat")) {
|
|
519
|
+
preciseResolutionMs = (await linuxProcessCpuMs(String(process.pid), cwd))?.resolutionMs;
|
|
520
|
+
}
|
|
521
|
+
return {
|
|
522
|
+
processes: precise,
|
|
523
|
+
resolutionMs: preciseResolutionMs ?? (rows.some((p) => p.frac) ? 10 : 1_000),
|
|
524
|
+
};
|
|
525
|
+
}
|
|
526
|
+
export async function workerTreeCpuMs(marker, cwd) {
|
|
527
|
+
const snapshot = await workerTreeCpuSnapshot(marker, cwd);
|
|
528
|
+
if (snapshot === undefined)
|
|
529
|
+
return undefined;
|
|
530
|
+
return {
|
|
531
|
+
ms: [...snapshot.processes.values()].reduce((sum, cpuMs) => sum + cpuMs, 0),
|
|
532
|
+
resolutionMs: snapshot.resolutionMs,
|
|
533
|
+
};
|
|
534
|
+
}
|
|
535
|
+
// Sparse live-tree totals forget a tool's CPU as soon as that tool exits. This attempt-local
|
|
536
|
+
// accountant instead adds each observed process's CPU DELTA to a monotonic total and replaces only
|
|
537
|
+
// the live-PID cursor on each sample. When a PID disappears, its contribution stays in `totalMs`;
|
|
538
|
+
// if that PID is later reused, its fresh total is added from zero because it left `live` in between.
|
|
539
|
+
class WorkerTreeCpuAccountant {
|
|
540
|
+
marker;
|
|
541
|
+
cwd;
|
|
542
|
+
active = false;
|
|
543
|
+
loop;
|
|
544
|
+
live = new Map();
|
|
545
|
+
totalMs = 0;
|
|
546
|
+
gaps = 0;
|
|
547
|
+
consecutiveGaps = 0;
|
|
548
|
+
latest;
|
|
549
|
+
constructor(marker, cwd) {
|
|
550
|
+
this.marker = marker;
|
|
551
|
+
this.cwd = cwd;
|
|
552
|
+
}
|
|
553
|
+
async sample() {
|
|
554
|
+
const snapshot = await workerTreeCpuSnapshot(this.marker, this.cwd);
|
|
555
|
+
if (snapshot === undefined) {
|
|
556
|
+
this.gaps++;
|
|
557
|
+
this.live.clear();
|
|
558
|
+
this.latest = undefined;
|
|
559
|
+
// Stop forking `ps` at 10Hz once the host has proved it cannot answer — see the cap's comment.
|
|
560
|
+
if (++this.consecutiveGaps >= HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP)
|
|
561
|
+
this.active = false;
|
|
562
|
+
return;
|
|
563
|
+
}
|
|
564
|
+
this.consecutiveGaps = 0;
|
|
565
|
+
for (const [pid, cpuMs] of snapshot.processes) {
|
|
566
|
+
const prior = this.live.get(pid);
|
|
567
|
+
this.totalMs += prior === undefined || cpuMs < prior ? cpuMs : cpuMs - prior;
|
|
568
|
+
}
|
|
569
|
+
this.live = snapshot.processes;
|
|
570
|
+
this.latest = { ms: this.totalMs, resolutionMs: snapshot.resolutionMs };
|
|
571
|
+
}
|
|
572
|
+
async start() {
|
|
573
|
+
if (this.active)
|
|
574
|
+
return;
|
|
575
|
+
this.active = true;
|
|
576
|
+
await this.sample();
|
|
577
|
+
this.loop = (async () => {
|
|
578
|
+
while (this.active) {
|
|
579
|
+
await new Promise((resolve) => setTimeout(resolve, HARVEST_CPU_ACCOUNTING_POLL_MS));
|
|
580
|
+
if (this.active)
|
|
581
|
+
await this.sample();
|
|
582
|
+
}
|
|
583
|
+
})();
|
|
584
|
+
}
|
|
585
|
+
read() {
|
|
586
|
+
return { cpu: this.latest, gaps: this.gaps };
|
|
587
|
+
}
|
|
588
|
+
async stop() {
|
|
589
|
+
this.active = false;
|
|
590
|
+
await this.loop;
|
|
591
|
+
}
|
|
592
|
+
}
|
|
119
593
|
async function commitsAheadOf(base, wt) {
|
|
120
594
|
const head = await gitHead(wt);
|
|
121
595
|
if (head === base)
|
|
@@ -125,6 +599,26 @@ async function commitsAheadOf(base, wt) {
|
|
|
125
599
|
return [];
|
|
126
600
|
return r.stdout.trim().split("\n").filter(Boolean);
|
|
127
601
|
}
|
|
602
|
+
// T1 (R1 fast-kill): worktree delta = commits ahead of the task base OR any uncommitted change.
|
|
603
|
+
// The fast-kill may only run once the ABSENCE of work has been positively established, so every
|
|
604
|
+
// probe fails OPEN toward "delta": a throwing rev-parse, a non-zero `git log`, and a non-zero
|
|
605
|
+
// `git status` all report delta. commitsAheadOf() is deliberately NOT reused here — it converts a
|
|
606
|
+
// failed log into an empty list, which reads as "no commits" and would kill a live worker.
|
|
607
|
+
async function worktreeHasDelta(wt, base) {
|
|
608
|
+
try {
|
|
609
|
+
const head = await gitHead(wt);
|
|
610
|
+
if (head !== base) {
|
|
611
|
+
const log = await shGit(`git log --reverse --format=%H ${shq(base)}..${shq(head)}`, wt);
|
|
612
|
+
if (log.code !== 0 || log.stdout.trim().length > 0)
|
|
613
|
+
return true;
|
|
614
|
+
}
|
|
615
|
+
const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", wt);
|
|
616
|
+
return r.code !== 0 || r.stdout.trim().length > 0;
|
|
617
|
+
}
|
|
618
|
+
catch {
|
|
619
|
+
return true; // an unreadable worktree is never evidence that the worker did nothing
|
|
620
|
+
}
|
|
621
|
+
}
|
|
128
622
|
async function cherryPickCommits(wt, commits) {
|
|
129
623
|
const carried = [];
|
|
130
624
|
for (const hash of commits) {
|
|
@@ -137,6 +631,71 @@ async function cherryPickCommits(wt, commits) {
|
|
|
137
631
|
}
|
|
138
632
|
return carried;
|
|
139
633
|
}
|
|
634
|
+
// T7 (v1.86): a first run-end append that fails AFTER partial bytes landed leaves a torn tail at
|
|
635
|
+
// EOF with no newline; a blind retry would glue the run-end line onto those bytes and readJsonl's
|
|
636
|
+
// torn-line tolerance would drop the retry too — no terminal record despite a successful write.
|
|
637
|
+
// Terminating the torn fragment keeps it on disk (dropped as a malformed line, never truncated) so
|
|
638
|
+
// the retried run-end lands on a line of its own.
|
|
639
|
+
const terminateTornJournalTail = (journalPath) => {
|
|
640
|
+
if (!existsSync(journalPath))
|
|
641
|
+
return;
|
|
642
|
+
const size = statSync(journalPath).size;
|
|
643
|
+
if (size === 0)
|
|
644
|
+
return;
|
|
645
|
+
const fd = openSync(journalPath, "r");
|
|
646
|
+
try {
|
|
647
|
+
const tail = Buffer.alloc(1);
|
|
648
|
+
if (readSync(fd, tail, 0, 1, size - 1) === 1 && tail[0] !== 0x0a)
|
|
649
|
+
appendFileSync(journalPath, "\n");
|
|
650
|
+
}
|
|
651
|
+
finally {
|
|
652
|
+
closeSync(fd);
|
|
653
|
+
}
|
|
654
|
+
};
|
|
655
|
+
// T7 (v1.86): the fatal handler must never eat the error it reports. Both journal calls are guarded,
|
|
656
|
+
// so a sink failure is reported ALONGSIDE the original error (console.error — the dispatcher's
|
|
657
|
+
// operator-visible line stays the one-line form), never instead of it: a read failure degrades the
|
|
658
|
+
// duplicate-run-end check to "unknown" and fails toward recording; the append is retried ONCE; and
|
|
659
|
+
// a persistently unwritable sink reports a crash naming the journal path and carrying NO terminal
|
|
660
|
+
// record, rather than fabricating an ended run on evidence the harness could not write. (OBS-313:
|
|
661
|
+
// with the sink dead the crash CAUSE is unrecordable — recorded as an observation, not papered over.)
|
|
662
|
+
function recordFatalRunEnd(journal, runId, branch, err) {
|
|
663
|
+
const original = err instanceof Error ? err.message : String(err);
|
|
664
|
+
try {
|
|
665
|
+
if (journal.read().some((e) => e.event === "run-end"))
|
|
666
|
+
return; // already terminal — nothing to add
|
|
667
|
+
}
|
|
668
|
+
catch (readErr) {
|
|
669
|
+
console.error(`tickmarkr ${runId}: journal read failed while recording the fatal run-end (${readErr instanceof Error ? readErr.message : String(readErr)}) — original error: ${original}`);
|
|
670
|
+
}
|
|
671
|
+
const record = {
|
|
672
|
+
runId,
|
|
673
|
+
branch,
|
|
674
|
+
done: [],
|
|
675
|
+
failed: [],
|
|
676
|
+
human: [],
|
|
677
|
+
blocked: [],
|
|
678
|
+
pending: [],
|
|
679
|
+
phase: "setup",
|
|
680
|
+
fatal: true,
|
|
681
|
+
error: original,
|
|
682
|
+
};
|
|
683
|
+
try {
|
|
684
|
+
journal.append("run-end", undefined, record);
|
|
685
|
+
return;
|
|
686
|
+
}
|
|
687
|
+
catch {
|
|
688
|
+
// one retry below — the first failure is reported only if the retry also fails
|
|
689
|
+
}
|
|
690
|
+
const journalPath = join(journal.dir, "journal.jsonl");
|
|
691
|
+
try {
|
|
692
|
+
terminateTornJournalTail(journalPath);
|
|
693
|
+
journal.append("run-end", undefined, record);
|
|
694
|
+
}
|
|
695
|
+
catch (retryErr) {
|
|
696
|
+
console.error(`tickmarkr ${runId}: run crashed — no terminal record written; journal sink unwritable at ${journalPath} (${retryErr instanceof Error ? retryErr.message : String(retryErr)}) — original error: ${original}`);
|
|
697
|
+
}
|
|
698
|
+
}
|
|
140
699
|
export async function runDaemon(repoRoot, opts = {}) {
|
|
141
700
|
// v1.51 T2 / OBS-89 (v1.60): retired --quality env seam. Mode resolution owns premium routing;
|
|
142
701
|
// route() no longer reads the retired env at all, so the old entrypoint scrub is gone with it.
|
|
@@ -295,7 +854,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
295
854
|
for (const [id, st] of journal.replayStatuses()) {
|
|
296
855
|
if (opts.retryFailed && st === "failed" && recordedTaskFailureKind(replayEvents, id) === "dispatch") {
|
|
297
856
|
graph = setStatus(graph, id, "pending");
|
|
298
|
-
|
|
857
|
+
// OBS-254: clear ATTEMPT AND CHANNEL STATE ONLY. Deleting the whole entry also deleted
|
|
858
|
+
// upheldFeedback — the operator's funded brief — and the next dispatch advertised an empty
|
|
859
|
+
// "fix these specifically" heading. A dispatch that died before worker-result then re-ran the
|
|
860
|
+
// worker with the uphold's findings silently gone.
|
|
861
|
+
const prior = resume.get(id);
|
|
862
|
+
resume.set(id, { attempts: 0, tried: [], ...(prior?.upheldFeedback ? { upheldFeedback: prior.upheldFeedback } : {}) });
|
|
299
863
|
continue;
|
|
300
864
|
}
|
|
301
865
|
// operator release: a graph.json edit back to "pending" beats a replayed human/failed park (locked decision 12)
|
|
@@ -485,7 +1049,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
485
1049
|
// reviewer-exclusion list (badReviewers), never a second parallel counter. OBS-189: scoped to the
|
|
486
1050
|
// current engagement — an operator approval (uphold or accept) resets the round budget, so an upheld
|
|
487
1051
|
// task can dispatch its funded attempt instead of re-parking against the whole journal's history.
|
|
488
|
-
const reviewRoundsDrawn = () => reviewRoundsSinceApproval(journal.read(), t.id);
|
|
1052
|
+
const reviewRoundsDrawn = () => reviewRoundsSinceApproval(decisiveReviewRounds(journal.read()), t.id);
|
|
489
1053
|
// OBS-193: journal the in-gate review retry (mirrors judge-retry) and exclude the flaked seat from
|
|
490
1054
|
// later attempts' reviewer picks. One helper, called from both onGate sites (satisfied-gate + main).
|
|
491
1055
|
const noteReviewRetry = (g) => {
|
|
@@ -498,9 +1062,85 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
498
1062
|
badReviewers.push(rr.flaked);
|
|
499
1063
|
}
|
|
500
1064
|
};
|
|
1065
|
+
// v1.85 T3 (ruling R4): every BLOCKING review/judge result lands its findings in the journal
|
|
1066
|
+
// structured — class + canonical path + stable symbol — so a retry, a consult or an auto-uphold
|
|
1067
|
+
// decision reads identity instead of re-parsing prose, and line-number churn is not a new finding.
|
|
1068
|
+
// One helper, both onGate sites (satisfied-gate resume + main attempt loop).
|
|
1069
|
+
const journalGateResult = (g) => {
|
|
1070
|
+
const blocking = gateFailed(g) && (g.gate === "review" || g.gate === "acceptance");
|
|
1071
|
+
// R3 (OBS-186): a gate that DECLINED has no verdict to state, and this row is the ONE seam every
|
|
1072
|
+
// fold outside this file shares. Writing `pass: false` for a decline is what turned a skip into
|
|
1073
|
+
// a failure at all of them at once — the engagement round budget (reviewRoundsSinceApproval,
|
|
1074
|
+
// journal.ts), the operator's failed-gate list (cli/commands/approve.ts), the record's
|
|
1075
|
+
// gate-failure total (cli/commands/report.ts), the cockpit's gate rows (tui/cockpit/derive.ts).
|
|
1076
|
+
// Each keys on `pass === false`; none of them is reachable from this task's file scope, and
|
|
1077
|
+
// patching five copies of the same question would be the wrong fix even if they were. So the
|
|
1078
|
+
// ledger simply does not claim a verdict it does not have.
|
|
1079
|
+
// The legacy baseline declines (a build command the repo never configured) have always written
|
|
1080
|
+
// `pass: true` beside `skipped: true` and every consumer already reads them right, so their row
|
|
1081
|
+
// is untouched: only a decline that would otherwise be recorded RED changes shape here.
|
|
1082
|
+
const unverdicted = g.meta?.skipped === true && !g.pass;
|
|
1083
|
+
journal.append("gate-result", t.id, {
|
|
1084
|
+
gate: g.gate, ...(unverdicted ? {} : { pass: g.pass }), details: g.details,
|
|
1085
|
+
...(g.meta?.skipped === true ? { skipped: true } : {}),
|
|
1086
|
+
// R3 (OBS-186): a declined review is journal truth, not an absence. `skipped: true` alone
|
|
1087
|
+
// says a gate did not run; these say WHICH policy declined it and WHY, so a reader of the
|
|
1088
|
+
// ledger never has to infer participation from a details string. The green-skip branch that
|
|
1089
|
+
// made this row indistinguishable from a pass is gone (src/gates/review.ts).
|
|
1090
|
+
...(g.meta?.verdict === "skipped"
|
|
1091
|
+
? { verdict: "skipped", policy: g.meta.policy, reason: g.meta.reason }
|
|
1092
|
+
: {}),
|
|
1093
|
+
// T4 (OBS-265): a test verdict says WHICH suite spoke. A round runs the selected subset as a
|
|
1094
|
+
// screen and the full suite as the verdict, so without these two the journal would carry a
|
|
1095
|
+
// `test` row whose scope no consumer could recover.
|
|
1096
|
+
...(Array.isArray(g.meta?.selectedTests) ? { selectedTests: g.meta.selectedTests } : {}),
|
|
1097
|
+
...(g.meta?.fullSuite === true ? { fullSuite: true } : {}),
|
|
1098
|
+
// A finding's path is its own evidence path. Do not pass task scope here: a declaration says
|
|
1099
|
+
// where work is allowed, not where this verdict found the defect.
|
|
1100
|
+
...(blocking ? { findings: structuredFindings(g.gate, g.details) } : {}),
|
|
1101
|
+
});
|
|
1102
|
+
};
|
|
1103
|
+
// R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
|
|
1104
|
+
// (run-gates.ts) — a race. Three oracles assert the opposite: a scripted run's journal is
|
|
1105
|
+
// byte-identical run to run (tests/run/narration.test.ts, tests/run/notify-identity.test.ts), and
|
|
1106
|
+
// a round's gate-result order matches its phase-start order (tests/run/daemon.test.ts). Retiring
|
|
1107
|
+
// complexityThreshold is what REACHES this, not what introduces it: those fixtures used to skip
|
|
1108
|
+
// review and journal ONE verdict row per round, so the pair's order was never exercised — and the
|
|
1109
|
+
// operator's config has run `complexityThreshold: 0` since 2026-07-31, so production rounds have
|
|
1110
|
+
// journaled both siblings all along.
|
|
1111
|
+
//
|
|
1112
|
+
// Ordering is the LEDGER's job, not the pipeline's. run-gates still reports each completion the
|
|
1113
|
+
// instant it happens; the daemon writes its ledger in GATE_NAMES order. Only the LATER gate is
|
|
1114
|
+
// ever held, and only while an earlier sibling is still in flight — a review that finishes first
|
|
1115
|
+
// waits for acceptance, never the reverse. That keeps T4's durability where it pays (the first
|
|
1116
|
+
// verdict to land is still published immediately) and bounds the exposure to one row for the
|
|
1117
|
+
// remainder of one already-running gate. A gate that THROWS kills the round before merge, so a
|
|
1118
|
+
// row held behind it is lost with the round it belonged to — not a verdict that could have merged.
|
|
1119
|
+
const parallelPending = new Set();
|
|
1120
|
+
let heldParallel;
|
|
1121
|
+
const notePhaseStart = (e) => {
|
|
1122
|
+
if (e.parentAt !== undefined)
|
|
1123
|
+
parallelPending.add(e.gate);
|
|
1124
|
+
};
|
|
1125
|
+
const inParallelOrder = (gate, publish) => {
|
|
1126
|
+
parallelPending.delete(gate);
|
|
1127
|
+
const rank = GATE_NAMES.indexOf(gate);
|
|
1128
|
+
if ([...parallelPending].some((p) => GATE_NAMES.indexOf(p) < rank)) {
|
|
1129
|
+
heldParallel = publish;
|
|
1130
|
+
return;
|
|
1131
|
+
}
|
|
1132
|
+
publish();
|
|
1133
|
+
const held = heldParallel;
|
|
1134
|
+
heldParallel = undefined;
|
|
1135
|
+
held?.();
|
|
1136
|
+
};
|
|
501
1137
|
// OBS-189: the operator upheld the reviewer — the findings ARE the brief for this funded attempt.
|
|
502
|
-
|
|
503
|
-
|
|
1138
|
+
// OBS-254: RE-DERIVED from the journal here, at prompt-build time, rather than trusted to survive
|
|
1139
|
+
// in resume state. The journal already holds the upheld review's bytes; no reset of attempt or
|
|
1140
|
+
// channel state can take them away, on any path, including `resume --retry-failed`.
|
|
1141
|
+
const upheldFeedback = upheldFeedbackByTask(journal.read()).get(t.id) ?? rs?.upheldFeedback;
|
|
1142
|
+
let feedback = upheldFeedback
|
|
1143
|
+
? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${upheldFeedback}`
|
|
504
1144
|
: "";
|
|
505
1145
|
let ladderIdx = 0;
|
|
506
1146
|
let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
|
|
@@ -509,6 +1149,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
509
1149
|
let tokens; // SPEND-02: accumulated across attempts — parked spend is still spend
|
|
510
1150
|
let metered = 0; // SPEND-02: attempts that returned a usage record; distinguishes unmetered from measured-zero
|
|
511
1151
|
let tipMoves = 0; // OBS-15: one re-gate allowance per task, never reset by a worker retry
|
|
1152
|
+
// T4 (OBS-265): a task whose test gate has failed once stops selecting tests for good — the
|
|
1153
|
+
// selector already proved it cannot speak for this diff, so every later round runs full. SEEDED
|
|
1154
|
+
// FROM THE JOURNAL, not from process memory: a park, a `resume`, or a daemon restart must not
|
|
1155
|
+
// hand the selector a clean slate it did not earn, and the journal is the only state that
|
|
1156
|
+
// survives all three (same read as upheldFeedbackByTask above).
|
|
1157
|
+
let testGateFailed = journal.read().some((e) => e.event === "gate-result" && e.taskId === t.id && e.data.gate === "test" && e.data.pass === false);
|
|
512
1158
|
let retryMode = "fresh";
|
|
513
1159
|
let lastContextTokens; // v1.23 reset signal, including stalled/quota attempts
|
|
514
1160
|
// v1.29: only a gate-failed attempt can seed same-session retry. The next attempt consumes this
|
|
@@ -568,7 +1214,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
568
1214
|
});
|
|
569
1215
|
await driver.notify(`tickmarkr ${runId}: ${t.id} consult verdict: ${v.action}`, { tier: "attention" });
|
|
570
1216
|
if (v.action === "retry") {
|
|
571
|
-
|
|
1217
|
+
// The guidance is ADDED to the brief, never swapped for it: the failure bytes the journal
|
|
1218
|
+
// already holds are the one thing the next attempt cannot rediscover for free. The
|
|
1219
|
+
// fingerprint cap's ban on an identical retry is NOT enforced here — a verdict is only one
|
|
1220
|
+
// of several ways this task reaches a re-dispatch, so the ban is enforced at the dispatch
|
|
1221
|
+
// seam every one of them passes through (see enforceRetryBan).
|
|
1222
|
+
const guidance = renderRetryGuidance(v);
|
|
1223
|
+
if (guidance && !feedback.includes(guidance))
|
|
1224
|
+
feedback = feedback ? `${feedback}\n\n${guidance}` : guidance;
|
|
572
1225
|
return true;
|
|
573
1226
|
}
|
|
574
1227
|
if (v.action === "reroute") {
|
|
@@ -657,7 +1310,36 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
657
1310
|
};
|
|
658
1311
|
const gateAuthor = rs?.lastAssignment ?? assignment;
|
|
659
1312
|
const satisfiedIndex = GATE_NAMES.indexOf(satisfiedGate);
|
|
660
|
-
|
|
1313
|
+
// The serial pipeline could have at most one blocking result, so "everything after the
|
|
1314
|
+
// approved gate" was enough. v1.85 can record both verdict siblings red in one round, and a
|
|
1315
|
+
// selected test screen can be green without a complete suite. Approval waives exactly its
|
|
1316
|
+
// named gate: every other red from that round is re-run, and test is forced unless the prior
|
|
1317
|
+
// journal proves a full suite completed.
|
|
1318
|
+
const priorEvents = journal.read();
|
|
1319
|
+
let priorRoundStart = -1;
|
|
1320
|
+
for (let i = priorEvents.length - 1; i >= 0; i--) {
|
|
1321
|
+
const e = priorEvents[i];
|
|
1322
|
+
if (e.event === "phase-start" && e.taskId === t.id && e.data.phase === "gates") {
|
|
1323
|
+
priorRoundStart = i;
|
|
1324
|
+
break;
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1327
|
+
const priorResults = new Map();
|
|
1328
|
+
for (const e of priorEvents.slice(priorRoundStart + 1)) {
|
|
1329
|
+
if (e.event !== "gate-result" || e.taskId !== t.id || typeof e.data.gate !== "string"
|
|
1330
|
+
|| !GATE_NAMES.includes(e.data.gate))
|
|
1331
|
+
continue;
|
|
1332
|
+
priorResults.set(e.data.gate, e);
|
|
1333
|
+
}
|
|
1334
|
+
const remainingGates = t.gates.filter((gate) => {
|
|
1335
|
+
if (gate === satisfiedGate)
|
|
1336
|
+
return false;
|
|
1337
|
+
const prior = priorResults.get(gate)?.data;
|
|
1338
|
+
const followsApproved = GATE_NAMES.indexOf(gate) > satisfiedIndex;
|
|
1339
|
+
const otherRed = prior?.pass === false;
|
|
1340
|
+
const needsFullSuite = gate === "test" && prior?.fullSuite !== true;
|
|
1341
|
+
return followsApproved || otherRed || needsFullSuite;
|
|
1342
|
+
});
|
|
661
1343
|
const resumedTask = { ...t, gates: remainingGates };
|
|
662
1344
|
gateLoop: while (true) {
|
|
663
1345
|
const gated = await gitHead(wt);
|
|
@@ -665,6 +1347,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
665
1347
|
const { results } = await runGates(resumedTask, {
|
|
666
1348
|
worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
|
|
667
1349
|
commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
|
|
1350
|
+
// a recheck re-verifies a human's release: it never selects tests down, it runs the suite.
|
|
1351
|
+
pipeline: "v185",
|
|
668
1352
|
via: cfg.visibility.llm === "pane"
|
|
669
1353
|
? {
|
|
670
1354
|
driver: trackedDriver,
|
|
@@ -677,25 +1361,25 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
677
1361
|
excludeReviewers: badReviewers,
|
|
678
1362
|
onGate: async (e) => {
|
|
679
1363
|
if (e.phase === "start") {
|
|
680
|
-
|
|
1364
|
+
notePhaseStart(e);
|
|
1365
|
+
journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
|
|
681
1366
|
return;
|
|
682
1367
|
}
|
|
683
1368
|
const g = e.result;
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
1369
|
+
inParallelOrder(g.gate, () => {
|
|
1370
|
+
journalGateResult(g);
|
|
1371
|
+
noteReviewRetry(g);
|
|
1372
|
+
if (g.gate === "review" && !g.pass && /unparseable/.test(g.details)
|
|
1373
|
+
&& typeof g.meta?.reviewer === "string") {
|
|
1374
|
+
badReviewers.push(g.meta.reviewer);
|
|
1375
|
+
}
|
|
687
1376
|
});
|
|
688
|
-
noteReviewRetry(g);
|
|
689
|
-
if (g.gate === "review" && !g.pass && /unparseable/.test(g.details)
|
|
690
|
-
&& typeof g.meta?.reviewer === "string") {
|
|
691
|
-
badReviewers.push(g.meta.reviewer);
|
|
692
|
-
}
|
|
693
1377
|
},
|
|
694
1378
|
});
|
|
695
1379
|
const approvedCommits = await commitsAheadOf(taskBase, wt);
|
|
696
1380
|
graph = addEvidence(graph, t.id, { commits: approvedCommits, gateResults: results });
|
|
697
1381
|
saveGraph(repoRoot, graph);
|
|
698
|
-
if (!results.every(
|
|
1382
|
+
if (!results.every(gateSatisfied)) {
|
|
699
1383
|
gateFails++;
|
|
700
1384
|
await park(t, "post-approval gate failed", "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
701
1385
|
return;
|
|
@@ -772,15 +1456,58 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
772
1456
|
// over-threshold context still forces fresh, and the escalation ladder bounds the chain.
|
|
773
1457
|
const priorSession = retrySession;
|
|
774
1458
|
retrySession = undefined;
|
|
1459
|
+
// v1.85 T3: the fingerprint cap banned an IDENTICAL retry — this task+gate already produced the
|
|
1460
|
+
// same failure bytes twice, so re-running the same channel on the same brief is a paid
|
|
1461
|
+
// re-measurement of an answer the journal already holds. Enforced HERE, at the one seam every
|
|
1462
|
+
// re-dispatch passes through, and NOT at the consult verdict that set it: a terminal verdict
|
|
1463
|
+
// falling through to the cap's own `retry` rung, a review-fix round and the ladder itself all
|
|
1464
|
+
// reach a dispatch without ever consulting again, and worker-launch below would then expire a
|
|
1465
|
+
// ban nothing had honoured. Bound to the channel the cap fired on, so a move that already went
|
|
1466
|
+
// elsewhere is not refused for a ban that was never about it; with no untried channel left the
|
|
1467
|
+
// task parks naming the ban rather than buying a third round.
|
|
1468
|
+
const banned = activeRetryBan(journal.read(), t.id, channelKey(assignment));
|
|
1469
|
+
if (banned) {
|
|
1470
|
+
const next = failover("escalate");
|
|
1471
|
+
journal.append("retry-same-banned", t.id, {
|
|
1472
|
+
gate: banned, from: channelKey(assignment), to: next ? channelKey(next) : null,
|
|
1473
|
+
});
|
|
1474
|
+
if (!next) {
|
|
1475
|
+
await park(t, `identical ${banned} failure twice this engagement — an identical retry is banned and no untried channel is left`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
1476
|
+
return;
|
|
1477
|
+
}
|
|
1478
|
+
assignment = next;
|
|
1479
|
+
if (!tried.includes(channelKey(next)))
|
|
1480
|
+
tried.push(channelKey(next));
|
|
1481
|
+
}
|
|
775
1482
|
const retryAdapter = adapters.find((a) => a.id === assignment.adapter);
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
1483
|
+
// v1.85 T3: a repair attempt is dispatched FRESH by construction — the whole point is that the
|
|
1484
|
+
// brief, not a surviving session, carries the findings and the diff. It therefore outranks the
|
|
1485
|
+
// session-resume choice, and the mode is journaled so the ledger can price repairs against
|
|
1486
|
+
// fresh re-dispatches. Read from the JOURNAL, before this dispatch's own event lands: a run that
|
|
1487
|
+
// stopped between funding the repair and sending it resumes still carrying the findings.
|
|
1488
|
+
const journaledSoFar = journal.read(); // read BEFORE this dispatch's own event lands
|
|
1489
|
+
let repairFindings = pendingRepairFindings(journaledSoFar, t.id);
|
|
1490
|
+
// OBS-254, one layer below the upheld brief: the ordinary gate-fail brief was loop-local, so any
|
|
1491
|
+
// path that rebuilt this task's state (a resume, `--retry-failed`, a fresh daemon) dispatched a
|
|
1492
|
+
// retry that had forgotten why it was retrying. Re-derived from the journal here, at prompt-build
|
|
1493
|
+
// time, and MERGED rather than substituted — a retry never discards what the journal already
|
|
1494
|
+
// holds. Row-wise, because the live brief may already quote one of them (a delivery-readiness
|
|
1495
|
+
// failure the loop just wrote) and repeating it helps no worker.
|
|
1496
|
+
const journaledRows = journaledFailureBrief(journaledSoFar, t.id).filter((row) => !feedback.includes(row));
|
|
1497
|
+
if (journaledRows.length > 0) {
|
|
1498
|
+
const brief = journaledRows.join("\n\n");
|
|
1499
|
+
feedback = feedback ? `${brief}\n\n${feedback}` : brief;
|
|
1500
|
+
}
|
|
1501
|
+
retryMode = repairFindings
|
|
1502
|
+
? "repair"
|
|
1503
|
+
: priorSession
|
|
1504
|
+
&& priorSession.channel === channelKey(assignment)
|
|
1505
|
+
&& (priorSession.contextTokens !== undefined
|
|
1506
|
+
? priorSession.contextTokens < cfg.contextWarnTokens
|
|
1507
|
+
: retryAdapter?.resumeUnknownContext === true)
|
|
1508
|
+
&& retryAdapter?.resumeCommand
|
|
1509
|
+
? "resume"
|
|
1510
|
+
: "fresh";
|
|
784
1511
|
// v1.23 T3: over-threshold context still forces fresh at the retry boundary; never interrupt a
|
|
785
1512
|
// running attempt. Unknown/below emits no reset event.
|
|
786
1513
|
if (attempt > 0 && lastContextTokens !== undefined && lastContextTokens >= cfg.contextWarnTokens) {
|
|
@@ -808,6 +1535,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
808
1535
|
carriedCommits = await cherryPickCommits(wt, commitsToCarry);
|
|
809
1536
|
journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
|
|
810
1537
|
}
|
|
1538
|
+
// T2 review (material): harvest eligibility is "does this WORKTREE carry unverified work",
|
|
1539
|
+
// measured against taskBase — the same base the fast-kill's delta probe and the gates
|
|
1540
|
+
// themselves use. It was measured against this attempt's post-carry HEAD, which excluded
|
|
1541
|
+
// every commit cherry-picked forward: attempt 0 commits and walls, attempt 1 receives that
|
|
1542
|
+
// commit and goes silent, and the silent retry — whose worktree already held the whole
|
|
1543
|
+
// deliverable — was NOT harvested, took the stall consult, and could be redispatched to
|
|
1544
|
+
// re-produce it. The routing branches that "starting HEAD" was protecting no longer need it:
|
|
1545
|
+
// quota, dead-channel and provider-death all classify the PRE-HARVEST outcome below and fire
|
|
1546
|
+
// BEFORE the synthesis, so carried-only work reaches gates without bypassing any failover.
|
|
811
1547
|
const priorNamed = [...new Set([...commitsToCarry, ...carriedCommits])];
|
|
812
1548
|
const presentCommits = new Set(carriedCommits);
|
|
813
1549
|
for (const h of commitsToCarry) {
|
|
@@ -836,6 +1572,36 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
836
1572
|
});
|
|
837
1573
|
await driver.notify(`tickmarkr ${runId}: ${t.id} lost ${lostCommits.length} of ${commitsToCarry.length} landed commit(s) recreating its worktree — it will re-do that work`, { tier: "attention" });
|
|
838
1574
|
}
|
|
1575
|
+
// v1.85 T3: "fully carried commits" is a repair PRECONDITION, and it is re-validated here —
|
|
1576
|
+
// the eligibility test ran one attempt ago, but the carry that decides it happens above, on
|
|
1577
|
+
// this dispatch. A tree that lost part of the prior attempt's work cannot be repaired: a
|
|
1578
|
+
// fix-only contract ("do NOT re-implement that work") over a diff whose implementation is
|
|
1579
|
+
// missing would have the worker patch an incomplete tree and forbid it from rebuilding the
|
|
1580
|
+
// rest. The fresh ladder owns this dispatch instead. `retryMode` is corrected before
|
|
1581
|
+
// worker-launch records it, so the ledger's launch event names what the worker actually got.
|
|
1582
|
+
if (repairFindings !== undefined && lostCommits.length > 0) {
|
|
1583
|
+
journal.append("repair-cancelled", t.id, { reason: "carry incomplete", attempted: commitsToCarry, lost: lostCommits });
|
|
1584
|
+
repairFindings = undefined;
|
|
1585
|
+
retryMode = "fresh";
|
|
1586
|
+
}
|
|
1587
|
+
// v1.85 T3: a repair dispatch replaces the bare gate-fail brief with a fix-only contract that
|
|
1588
|
+
// carries the failing findings VERBATIM and the diff CONTENT of the work already in this
|
|
1589
|
+
// worktree. The measured loss it removes: 62 of 68 re-dispatches were fresh, each re-buying
|
|
1590
|
+
// ~20m of onboarding to rediscover a diff and a finding the journal already held.
|
|
1591
|
+
// The diff is measured HERE, from the worktree the worker will actually open, after the carry —
|
|
1592
|
+
// never from the pre-recreation tree, so what the brief quotes is what the worker has.
|
|
1593
|
+
if (repairFindings) {
|
|
1594
|
+
const raw = await shGit(`git diff ${shq(taskBase)}..HEAD`, wt);
|
|
1595
|
+
const cap = cfg.gates.diffCap ?? DEFAULT_DIFF_CAP; // same fallback the measuring gates use
|
|
1596
|
+
const diff = raw.stdout.length > cap
|
|
1597
|
+
? `${raw.stdout.slice(0, cap)}\n… diff truncated at gates.diffCap (${cap} bytes)`
|
|
1598
|
+
: raw.stdout;
|
|
1599
|
+
const brief = repairBrief(repairFindings, diff, taskBase);
|
|
1600
|
+
// anything the live brief holds beyond the journaled findings (a consult's guidance) is kept:
|
|
1601
|
+
// a repair adds the diff and the fix-only contract, it never subtracts what was already known.
|
|
1602
|
+
feedback = feedback && !brief.includes(feedback) ? `${brief}\n\n${feedback}` : brief;
|
|
1603
|
+
journal.append("repair-dispatch", t.id, { diffBytes: diff.length, capped: raw.stdout.length > cap });
|
|
1604
|
+
}
|
|
839
1605
|
if (feedback || priorNamed.length > 0) {
|
|
840
1606
|
feedback = augmentRetryBrief(feedback, { attempted: commitsToCarry, carried: carriedCommits, present: presentCommits });
|
|
841
1607
|
}
|
|
@@ -892,6 +1658,90 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
892
1658
|
workerCmd,
|
|
893
1659
|
exitMarkerCmd,
|
|
894
1660
|
].join("\n"));
|
|
1661
|
+
// T2 (OBS-264): the liveness triad, shared by BOTH wait loops (a headless worker stalls on
|
|
1662
|
+
// finished work exactly as a visible one does — and rode the whole window before this). It
|
|
1663
|
+
// sits ABOVE the fast-kill and the nudge because the population it governs is the opposite
|
|
1664
|
+
// one: the kill condemns a pane holding NOTHING, while every one of the 18 observed stalls
|
|
1665
|
+
// held 2-33 commits that the redispatch then re-bought. Concluding is not killing — the
|
|
1666
|
+
// post-loop tail harvests this attempt exactly as a window expiry does, and the carried
|
|
1667
|
+
// worktree goes to gates. The probe runs only once the tracker is ALREADY silent, so a
|
|
1668
|
+
// working worker never pays for it, and an unreadable snapshot RESETS the observation:
|
|
1669
|
+
// unmeasurable CPU is never evidence a worker stopped.
|
|
1670
|
+
// v1.85 T3: the prompt has now actually reached a worker. The journal-derived retry decisions (a
|
|
1671
|
+
// funded repair's findings, an identical-retry ban) expire HERE and nowhere earlier: everything
|
|
1672
|
+
// between task-dispatch and this line — worktree recreation, setup, prompt write, slot allocation,
|
|
1673
|
+
// the launch itself — can still die with no worker having seen the brief, and `--retry-failed`
|
|
1674
|
+
// must then re-send that same brief rather than a fresh prompt on a possibly banned channel.
|
|
1675
|
+
const noteLaunched = () => journal.append("worker-launch", t.id, { attempt, retryMode });
|
|
1676
|
+
let cpuFlat;
|
|
1677
|
+
let cpuAccountant;
|
|
1678
|
+
let cpuGapCount = 0;
|
|
1679
|
+
let unmeasurableNoted = false;
|
|
1680
|
+
// One line per attempt, whichever way the CPU leg turns out to be unmeasurable. A triad that
|
|
1681
|
+
// can never conclude is this feature silently ABSENT — on a host whose `ps` the probe cannot
|
|
1682
|
+
// read, every stall would ride its whole window out again with nothing saying why. Named once
|
|
1683
|
+
// per attempt, not per slice: the condition is structural, and a per-slice line would bury it.
|
|
1684
|
+
const noteUnmeasurable = (reason) => {
|
|
1685
|
+
if (unmeasurableNoted)
|
|
1686
|
+
return;
|
|
1687
|
+
unmeasurableNoted = true;
|
|
1688
|
+
journal.append("worker-harvest-unmeasurable", t.id, { slot: slot.name, attempt, reason });
|
|
1689
|
+
};
|
|
1690
|
+
const harvestConcludes = async (silentMs) => {
|
|
1691
|
+
if (silentMs < harvestSilentMs) {
|
|
1692
|
+
await cpuAccountant?.stop();
|
|
1693
|
+
cpuAccountant = undefined;
|
|
1694
|
+
cpuFlat = undefined;
|
|
1695
|
+
cpuGapCount = 0;
|
|
1696
|
+
return false;
|
|
1697
|
+
}
|
|
1698
|
+
// The CPU leg needs a marker in the worker's own argv, and every launch path puts this
|
|
1699
|
+
// attempt's dispatch script there EXCEPT interactiveSeed: runInteractiveSeed launches the
|
|
1700
|
+
// TUI directly, by a command the ADAPTER owns (seed.launch(model)) which tickmarkr cannot
|
|
1701
|
+
// make attempt-unique and must deliver verbatim. A marker that matches nothing reads as
|
|
1702
|
+
// zero CPU — precisely the false "flat" that would harvest a worker mid-turn — so a seeded
|
|
1703
|
+
// attempt has no measurable CPU leg and the triad never concludes it. The other half of
|
|
1704
|
+
// OBS-264 is untouched there: when its window does expire with commits on the worktree, the
|
|
1705
|
+
// no-trailer tail gates them instead of buying a fresh worker to re-produce them.
|
|
1706
|
+
if (hasSeed) {
|
|
1707
|
+
noteUnmeasurable("interactive-seed launch is not in the probed process tree");
|
|
1708
|
+
return false;
|
|
1709
|
+
}
|
|
1710
|
+
if (cpuAccountant === undefined) {
|
|
1711
|
+
cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt);
|
|
1712
|
+
await cpuAccountant.start();
|
|
1713
|
+
}
|
|
1714
|
+
const observation = cpuAccountant.read();
|
|
1715
|
+
if (observation.gaps !== cpuGapCount) {
|
|
1716
|
+
cpuGapCount = observation.gaps;
|
|
1717
|
+
cpuFlat = undefined;
|
|
1718
|
+
noteUnmeasurable("one or more worker process snapshots could not be read");
|
|
1719
|
+
}
|
|
1720
|
+
const cpu = observation.cpu;
|
|
1721
|
+
if (cpu === undefined) {
|
|
1722
|
+
// Unmeasurable CPU is never evidence a worker stopped: RESET the observation rather than
|
|
1723
|
+
// conclude on it, and name the gap — a probe whose snapshot never parses is the same
|
|
1724
|
+
// structural hole as the seeded launch, and must not be the one that stays silent.
|
|
1725
|
+
cpuFlat = undefined;
|
|
1726
|
+
noteUnmeasurable("the worker process snapshot could not be read");
|
|
1727
|
+
return false;
|
|
1728
|
+
}
|
|
1729
|
+
const now = Date.now();
|
|
1730
|
+
if (cpu.ms !== cpuFlat?.ms) {
|
|
1731
|
+
cpuFlat = { ms: cpu.ms, since: now };
|
|
1732
|
+
return false;
|
|
1733
|
+
}
|
|
1734
|
+
if (now - cpuFlat.since < harvestCpuFlatWindowMs(cpu.resolutionMs))
|
|
1735
|
+
return false;
|
|
1736
|
+
const carried = await commitsAheadOf(taskBase, wt);
|
|
1737
|
+
if (carried.length === 0)
|
|
1738
|
+
return false; // nothing landed: not this branch's population
|
|
1739
|
+
journal.append("worker-harvest", t.id, {
|
|
1740
|
+
slot: slot.name, attempt, commits: carried.length,
|
|
1741
|
+
silentMs, cpuMs: cpu.ms, cpuResolutionMs: cpu.resolutionMs,
|
|
1742
|
+
});
|
|
1743
|
+
return true;
|
|
1744
|
+
};
|
|
895
1745
|
// SPEND-01: this attempt's dispatch wall-clock — the usage collect cursor. Captured once here, the
|
|
896
1746
|
// single site, so a test can reason about it; keep Date.now() out of profile.ts (still pure) and
|
|
897
1747
|
// out of adapter module scope (the cursor is a parameter, threaded from the daemon).
|
|
@@ -931,6 +1781,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
931
1781
|
let output;
|
|
932
1782
|
let exitCode;
|
|
933
1783
|
let timedOut = false;
|
|
1784
|
+
// T2 review: print mode's "the exit marker appeared". Kept apart from `finished` (the
|
|
1785
|
+
// trailer) but still needed by the keepPanes decision below, whose contract is about a
|
|
1786
|
+
// subprocess tree that REACHED its exit marker, not about what the worker claimed.
|
|
1787
|
+
let processExited = false;
|
|
934
1788
|
let earlyLaunchDead = false;
|
|
935
1789
|
let settleParsed;
|
|
936
1790
|
let seedResult;
|
|
@@ -966,26 +1820,369 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
966
1820
|
await park(t, "escalation ladder exhausted", "ladder-exhausted", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
967
1821
|
return false;
|
|
968
1822
|
};
|
|
969
|
-
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
1823
|
+
try {
|
|
1824
|
+
if (interactive) {
|
|
1825
|
+
// v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
|
|
1826
|
+
// The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
|
|
1827
|
+
finished = false;
|
|
1828
|
+
exitCode = null;
|
|
1829
|
+
if (adapter.interactiveSeed) {
|
|
1830
|
+
// v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
|
|
1831
|
+
// then fall through to the normal trailer harvest. A failed seed is recorded as a finished
|
|
1832
|
+
// failure rather than allowed to race the trailer wait.
|
|
1833
|
+
try {
|
|
1834
|
+
seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
|
|
1835
|
+
}
|
|
1836
|
+
catch (error) {
|
|
1837
|
+
if (!(error instanceof DeliveryReadinessError))
|
|
1838
|
+
throw error;
|
|
1839
|
+
if (await handleDeliveryReadiness(error))
|
|
1840
|
+
continue attempts;
|
|
1841
|
+
return;
|
|
1842
|
+
}
|
|
1843
|
+
noteLaunched();
|
|
1844
|
+
output = seedResult.output;
|
|
980
1845
|
}
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
1846
|
+
else {
|
|
1847
|
+
try {
|
|
1848
|
+
await driver.run(slot, paneDispatchCommand(dispatchScript));
|
|
1849
|
+
}
|
|
1850
|
+
catch (error) {
|
|
1851
|
+
if (!(error instanceof DeliveryReadinessError))
|
|
1852
|
+
throw error;
|
|
1853
|
+
if (await handleDeliveryReadiness(error))
|
|
1854
|
+
continue attempts;
|
|
1855
|
+
return;
|
|
1856
|
+
}
|
|
1857
|
+
noteLaunched();
|
|
1858
|
+
output = await driver.read(slot, PANE_READ_ROWS);
|
|
1859
|
+
}
|
|
1860
|
+
if (seedResult?.seedFailed) {
|
|
1861
|
+
finished = false;
|
|
1862
|
+
}
|
|
1863
|
+
else {
|
|
1864
|
+
// v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
|
|
1865
|
+
// Any other blocked/idle dialog pages the operator (unlatched since T1 — see below).
|
|
1866
|
+
let trustAnswered = false;
|
|
1867
|
+
// OBS-201: one liveness nudge per attempt; the grace deadline is its OWN timer, never the
|
|
1868
|
+
// stall window (the nudge's pane echo is absorbed before it starts, or the echo itself
|
|
1869
|
+
// would reset the window and make the early conclusion unreachable).
|
|
1870
|
+
let nudged = false;
|
|
1871
|
+
let nudgeFailed = false; // T1: an undeliverable nudge — the operator, not the daemon, is the actor
|
|
1872
|
+
let nudgeDeadline;
|
|
1873
|
+
// T1: page cadence, NOT a latch — a second page fires on a status change or once
|
|
1874
|
+
// pageRepeatMs elapses, so an operator who missed the first one is paged again.
|
|
1875
|
+
let lastPagedStatus;
|
|
1876
|
+
let lastPagedAt = 0;
|
|
1877
|
+
// T1 review: worker-status is journaled on CHANGE, not per slice — every journal append
|
|
1878
|
+
// is narrated to the run's live surface (cli/commands/run.ts) and feeds activity.ts's
|
|
1879
|
+
// `now:` cell, so a per-slice append wrote a status line per worker per ~30s slice and
|
|
1880
|
+
// pinned `now` to worker-status. On-change keeps post-hoc analysis at a fraction of the
|
|
1881
|
+
// volume while still recording which gate held.
|
|
1882
|
+
let lastStatus;
|
|
1883
|
+
finished = false;
|
|
1884
|
+
exitCode = null;
|
|
1885
|
+
// OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
|
|
1886
|
+
// stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
|
|
1887
|
+
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
1888
|
+
// v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
|
|
1889
|
+
// the stall clock. Raw pane differences are terminal chrome until proven otherwise.
|
|
1890
|
+
let everHadOutput = output.length > 0;
|
|
1891
|
+
const stallProgress = new StallProgressTracker();
|
|
1892
|
+
stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
|
|
1893
|
+
let lastProgressAt = Date.now();
|
|
1894
|
+
// T1: in-loop detector state — consecutive quota-banner slices. The dead-channel fast-kill
|
|
1895
|
+
// reads the tracker's RAW row-growth clock (lastRowGrowthAt), not the re-arm-suppressed
|
|
1896
|
+
// lastProgressAt — see the kill below.
|
|
1897
|
+
let quotaStreak = 0;
|
|
1898
|
+
let rowSaturationHeld = false; // journaled once per attempt when the kill stands down
|
|
1899
|
+
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
1900
|
+
const sliceStart = Date.now();
|
|
1901
|
+
const remaining = stallWindowMs - (sliceStart - lastProgressAt);
|
|
1902
|
+
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
1903
|
+
if (!everHadOutput) {
|
|
1904
|
+
const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
|
|
1905
|
+
if (earlyLeft > 0)
|
|
1906
|
+
slice = Math.min(slice, earlyLeft);
|
|
1907
|
+
}
|
|
1908
|
+
// T2 (OBS-264): never sleep PAST the instant the harvest probe becomes eligible, and past
|
|
1909
|
+
// it let the probe own the cadence (HARVEST_POLL_MS) — a 30s trailer slice would
|
|
1910
|
+
// otherwise cost a minute per pair of CPU samples. At the shipped 5m gate this leaves
|
|
1911
|
+
// every slice before the gate exactly as long as it already was.
|
|
1912
|
+
slice = Math.min(slice, harvestSliceMs(sliceStart - lastProgressAt));
|
|
1913
|
+
if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
|
|
1914
|
+
// verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
|
|
1915
|
+
// own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
|
|
1916
|
+
// parseable trailer or a digit-suffixed exit marker in the harvest is completion.
|
|
1917
|
+
output = await driver.read(slot, PANE_READ_ROWS); // TUI transcripts carry chrome — read deeper than print's 500
|
|
1918
|
+
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
1919
|
+
const exit = exitRe.exec(output);
|
|
1920
|
+
if (finished || exit) {
|
|
1921
|
+
exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
|
|
1922
|
+
await sampleContext(); // final poll-seam sample before leaving the wait
|
|
1923
|
+
break;
|
|
1924
|
+
}
|
|
1925
|
+
}
|
|
1926
|
+
const paneText = await driver.read(slot, PANE_READ_ROWS);
|
|
1927
|
+
if (paneText.length > 0)
|
|
1928
|
+
everHadOutput = true;
|
|
1929
|
+
// OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
|
|
1930
|
+
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
1931
|
+
earlyLaunchDead = true;
|
|
1932
|
+
output = paneText;
|
|
1933
|
+
break;
|
|
1934
|
+
}
|
|
1935
|
+
// v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
|
|
1936
|
+
await sampleContext();
|
|
1937
|
+
if (stallProgress.observe({ paneText, contextTokens })) {
|
|
1938
|
+
lastProgressAt = Date.now();
|
|
1939
|
+
// T1 review fix: progress AFTER a delivered nudge means the worker answered it —
|
|
1940
|
+
// disarm the grace deadline. Without this the expiry below fires at the next
|
|
1941
|
+
// quiet patch ≥ the grace (4m) measured from the rolling lastProgressAt, so the
|
|
1942
|
+
// exact population the nudge rescued (workers prone to long silences, e.g. a 6m
|
|
1943
|
+
// test run) was force-concluded as if it had ignored the nudge. The answered
|
|
1944
|
+
// worker returns to the full rolling window, and the consult sees the truth.
|
|
1945
|
+
if (nudged && nudgeDeadline !== undefined) {
|
|
1946
|
+
nudgeDeadline = undefined;
|
|
1947
|
+
journal.append("worker-nudge-answered", t.id, { slot: slot.name, attempt });
|
|
1948
|
+
}
|
|
1949
|
+
}
|
|
1950
|
+
const sliceNow = Date.now();
|
|
1951
|
+
// T1 (OBS-263): quota banners are classified IN-LOOP — two consecutive matching slices
|
|
1952
|
+
// plus >=3m tracker silence — then the post-loop quota failover runs NOW, not after the
|
|
1953
|
+
// window. The match reads the chrome-filtered tail: the bottom of a rendered TUI frame
|
|
1954
|
+
// is fixed composer/welcome chrome (codex pins a "usage limit resets available" line
|
|
1955
|
+
// there — it matched every frame of the wedged-MCP fixture), so the known chrome is
|
|
1956
|
+
// filtered by identity, never by novelty — a banner already on screen at launch
|
|
1957
|
+
// classifies exactly like one printed mid-attempt (T1 review: a novelty baseline
|
|
1958
|
+
// exculpated the launch-throttle case forever).
|
|
1959
|
+
if (QUOTA_RE.test(stallSnapshotBannerRows(paneText)))
|
|
1960
|
+
quotaStreak++;
|
|
1961
|
+
else
|
|
1962
|
+
quotaStreak = 0;
|
|
1963
|
+
if (quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
|
|
1964
|
+
// no `output =` here: the post-loop no-trailer tail re-reads the pane anyway, so an
|
|
1965
|
+
// assignment would only split the classification read from the verdict read.
|
|
1966
|
+
journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt });
|
|
1967
|
+
break;
|
|
1968
|
+
}
|
|
1969
|
+
// T1 (OBS-262): the `paged` latch is deleted — status is sampled EVERY slice (and
|
|
1970
|
+
// journaled on change — see lastStatus above), so post-hoc analysis can see which
|
|
1971
|
+
// gate held. page on "idle" too: herdr's blocked-scrape is strict and proved flaky
|
|
1972
|
+
// for TUI dialogs (live check: cursor's trust dialog scraped as idle).
|
|
1973
|
+
// "unknown"/"working" never page.
|
|
1974
|
+
const st = await driver.status(slot);
|
|
1975
|
+
if (st !== lastStatus) {
|
|
1976
|
+
lastStatus = st;
|
|
1977
|
+
journal.append("worker-status", t.id, { slot: slot.name, status: st, attempt });
|
|
1978
|
+
}
|
|
1979
|
+
if (st === "blocked" || st === "idle") {
|
|
1980
|
+
// T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
|
|
1981
|
+
// text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
|
|
1982
|
+
if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
|
|
1983
|
+
try {
|
|
1984
|
+
const paneText = await driver.read(slot, 80);
|
|
1985
|
+
if (matchesTrustDialog(paneText, adapter.trustDialog)) {
|
|
1986
|
+
trustAnswered = true;
|
|
1987
|
+
// v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
|
|
1988
|
+
// Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
|
|
1989
|
+
journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
|
|
1990
|
+
await driver.sendKey(slot, adapter.trustDialog.key);
|
|
1991
|
+
const spent = Date.now() - sliceStart;
|
|
1992
|
+
if (spent < slice)
|
|
1993
|
+
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
1994
|
+
continue; // do not page — keep waiting for the trailer
|
|
1995
|
+
}
|
|
1996
|
+
}
|
|
1997
|
+
catch {
|
|
1998
|
+
/* read/send failed — fall through to page the operator */
|
|
1999
|
+
}
|
|
2000
|
+
}
|
|
2001
|
+
}
|
|
2002
|
+
// T1 (OBS-262): the daemon ACTS on a silent worker before paging anyone. Gate: monotonic
|
|
2003
|
+
// tracker silent ≥ the nudge threshold — the herdr status reading (idle/unknown/working)
|
|
2004
|
+
// no longer holds the gate hostage; only `blocked` stays page-only (nudging a dialog
|
|
2005
|
+
// prompt can't help). One nudge per attempt, then the grace timer owns the conclusion.
|
|
2006
|
+
const nudgeable = st !== "blocked" && !!driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id);
|
|
2007
|
+
// T1 review (answer-then-die): `nudgeable` is a per-slice property of the adapter and
|
|
2008
|
+
// status — it is NOT "a daemon action is pending". An action is pending only while the
|
|
2009
|
+
// nudge can still fire (un-nudged) or its grace window is armed; once the worker
|
|
2010
|
+
// ANSWERS, the disarm above clears nudgeDeadline while `nudged` stays latched, and the
|
|
2011
|
+
// daemon has nothing left to do — the pane falls back under the fast-kill and page
|
|
2012
|
+
// watchdogs like any other, instead of riding the whole rolling window untended.
|
|
2013
|
+
const nudgePending = nudgeable && (!nudged || nudgeDeadline !== undefined);
|
|
2014
|
+
// T2 (OBS-264): the liveness triad CONCLUDES the wait on finished work — commits ahead
|
|
2015
|
+
// of the task base (this attempt's own AND any carried forward — see the eligibility
|
|
2016
|
+
// comment at the dispatch site), a flat worker-tree CPU delta, and >= harvestSilentMs
|
|
2017
|
+
// of monotonic tracker silence (defined once, above, and run identically by the print loop).
|
|
2018
|
+
// T2 review (material): it carries the SAME nudge hold as the fast-kill below, by the
|
|
2019
|
+
// same clause and for the same reason. The CPU leg cannot tell "idle because finished"
|
|
2020
|
+
// from "idle because holding an unsubmitted turn in its input box" — both read flat CPU
|
|
2021
|
+
// under a silent tracker — and the nudge is the one signal that can. Under the shipped
|
|
2022
|
+
// constants (harvest 5m < nudge 10m) an unheld harvest concluded every committed
|
|
2023
|
+
// claude-code worker before the rescue could fire, leaving T1's nudge dead code for
|
|
2024
|
+
// exactly the committed-and-stalled population OBS-264 is about. Holding concludes at
|
|
2025
|
+
// ~14m (nudge + grace) rather than ~36m — nearly all of the OBS-264 win, and a worker
|
|
2026
|
+
// that only needed a submit answers with a full trailer instead of partial work. An
|
|
2027
|
+
// ANSWERED or twice-undeliverable nudge leaves nothing pending, so the triad governs
|
|
2028
|
+
// again; the hold is on a pending daemon ACTION, never on the adapter being nudgeable.
|
|
2029
|
+
if ((!nudgePending || nudgeFailed) && await harvestConcludes(sliceNow - lastProgressAt))
|
|
2030
|
+
break;
|
|
2031
|
+
// T1 (R1 dead-channel fast-kill): no trailer, no worktree delta, and no output growth
|
|
2032
|
+
// for the fast-kill window — the channel is dead, so conclude NOW
|
|
2033
|
+
// (journaled) and let the existing no-trailer tail classify and route the attempt.
|
|
2034
|
+
// The tracker is the growth signal on purpose: raw pane bytes grow on cosmetic repaint
|
|
2035
|
+
// (an elapsed "9s"→"10s" lengthens the read and would hide a frozen pane). But the
|
|
2036
|
+
// tracker SHARES the read window's ceiling: its row signal saturates once a sample
|
|
2037
|
+
// FILLS a PANE_READ_ROWS read on raw lines (blanks/chrome included — see stall.ts),
|
|
2038
|
+
// and past that point a flat tracker means "unmeasurable", not "dead". For unmetered
|
|
2039
|
+
// adapters (codex, cursor-agent, grok, opencode — no contextUsage, so the token signal
|
|
2040
|
+
// never fires) rows are the ONLY
|
|
2041
|
+
// signal, and those are exactly the non-nudgeable adapters this kill governs — a live
|
|
2042
|
+
// worker past the ceiling would be concluded dead mid-work. So the kill STANDS DOWN on
|
|
2043
|
+
// a saturated row signal (journaled once per attempt); the rolling window still owns
|
|
2044
|
+
// that pane, exactly as pre-T1. Token growth counts as life either way, so a metered
|
|
2045
|
+
// worker thinking through a long tool run survives.
|
|
2046
|
+
// The triad has NO status exemption: a pane that herdr reports as blocked, idle, working
|
|
2047
|
+
// or unknown dies alike once it holds no trailer, no delta and no growth — waiting the
|
|
2048
|
+
// rolling window out on a status reading is exactly the blindness T1 removes. A matched
|
|
2049
|
+
// trust dialog is auto-answered above and `continue`s before ever reaching here.
|
|
2050
|
+
// The NUDGE gets first crack at a nudgeable pane: the fast-kill holds while the daemon
|
|
2051
|
+
// still has an action of its own pending (un-nudged, or inside the grace window) —
|
|
2052
|
+
// under the shipped constants (kill 5m < nudge 10m) a delta-less pane would otherwise
|
|
2053
|
+
// die before the rescue could ever fire. An ANSWERED nudge leaves nothing pending, so
|
|
2054
|
+
// the hold lifts and the triad governs again. Once a nudge has failed to deliver TWICE
|
|
2055
|
+
// (one in-slice retry filters a driver flake — see below), the delta clause drops too:
|
|
2056
|
+
// a delta is past work, and a pane no signal can reach must not buy the rest of the
|
|
2057
|
+
// window with it (a `working` pane can't be paged either, so this is the only bound
|
|
2058
|
+
// that path has).
|
|
2059
|
+
if (stallProgress.rowSignalSaturated && !rowSaturationHeld) {
|
|
2060
|
+
rowSaturationHeld = true;
|
|
2061
|
+
journal.append("worker-dead-held", t.id, { slot: slot.name, attempt, reason: "row-signal-saturated" });
|
|
2062
|
+
}
|
|
2063
|
+
// T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
|
|
2064
|
+
// never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
|
|
2065
|
+
// the re-arm report on row growth once tokens stick, and contextTokens is sticky across
|
|
2066
|
+
// read misses, so a metered non-nudgeable adapter (pi) streaming rows under a stale
|
|
2067
|
+
// counter presented a frozen lastProgressAt and was killed mid-work. lastRowGrowthAt is
|
|
2068
|
+
// recorded on every high-water advance, suppressed or not; token growth already rides
|
|
2069
|
+
// lastProgressAt. Either one advancing is output growth.
|
|
2070
|
+
const lastOutputGrowthAt = Math.max(stallProgress.lastRowGrowthAt ?? 0, lastProgressAt);
|
|
2071
|
+
if (!stallProgress.rowSignalSaturated
|
|
2072
|
+
&& (!nudgePending || nudgeFailed)
|
|
2073
|
+
&& sliceNow - lastOutputGrowthAt >= deadChannelFastKillMs
|
|
2074
|
+
&& (nudgeFailed || !(await worktreeHasDelta(wt, taskBase)))) {
|
|
2075
|
+
journal.append("worker-dead", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastOutputGrowthAt });
|
|
2076
|
+
break;
|
|
2077
|
+
}
|
|
2078
|
+
if (nudgeable && !nudged && sliceNow - lastProgressAt >= nudgeAfterSilentMs) {
|
|
2079
|
+
nudged = true;
|
|
2080
|
+
// T1 review: a false return is a driver-delivery outcome (missing pin, readiness
|
|
2081
|
+
// stable-frame timeout, read-back hiccup), not proof of an unreachable channel — so
|
|
2082
|
+
// one failure is a flake class, retried once in-slice after a short settle. Only a
|
|
2083
|
+
// failed RETRY condemns the channel. Both failures happen inside this slice, so the
|
|
2084
|
+
// latch stays immediate and exactly one failure is journaled per attempt.
|
|
2085
|
+
let delivered = await driver.nudge(slot, WORKER_NUDGE_MESSAGE);
|
|
2086
|
+
if (!delivered) {
|
|
2087
|
+
await new Promise((r) => setTimeout(r, NUDGE_REDELIVER_MS));
|
|
2088
|
+
delivered = await driver.nudge(slot, WORKER_NUDGE_MESSAGE);
|
|
2089
|
+
}
|
|
2090
|
+
if (delivered) {
|
|
2091
|
+
// absorb the nudge's own echo BEFORE arming the grace timer — post-nudge progress
|
|
2092
|
+
// is measured against this baseline, not against the echo.
|
|
2093
|
+
const echo = await driver.read(slot, PANE_READ_ROWS);
|
|
2094
|
+
stallProgress.observe({ paneText: echo, contextTokens });
|
|
2095
|
+
nudgeDeadline = Date.now() + workerNudgeGraceMs;
|
|
2096
|
+
journal.append("worker-nudge", t.id, { slot: slot.name, attempt });
|
|
2097
|
+
}
|
|
2098
|
+
else {
|
|
2099
|
+
// T1: an undeliverable nudge is itself a condemnation — the channel is unreachable,
|
|
2100
|
+
// so the fast-kill above stops requiring a clean worktree for THIS pane (see there).
|
|
2101
|
+
// A blocked/idle pane is still paged below; a `working`/`unknown` one cannot be, and
|
|
2102
|
+
// for it the conclusion's own escalation is what notifies. Nothing here waits the
|
|
2103
|
+
// rolling window out.
|
|
2104
|
+
nudgeFailed = true;
|
|
2105
|
+
journal.append("worker-nudge-failed", t.id, { slot: slot.name, attempt });
|
|
2106
|
+
}
|
|
2107
|
+
}
|
|
2108
|
+
if (nudged && nudgeDeadline !== undefined && sliceNow >= nudgeDeadline && sliceNow - lastProgressAt >= workerNudgeGraceMs) {
|
|
2109
|
+
// grace spent, still no post-nudge progress: re-harvest once (the trailer may have
|
|
2110
|
+
// landed between polls), then conclude the wait as a stall NOW — the consult sees the
|
|
2111
|
+
// un-answered nudge instead of the remainder of the window.
|
|
2112
|
+
nudgeDeadline = undefined;
|
|
2113
|
+
output = await driver.read(slot, PANE_READ_ROWS);
|
|
2114
|
+
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
2115
|
+
const exit = exitRe.exec(output);
|
|
2116
|
+
if (finished || exit) {
|
|
2117
|
+
exitCode = exit ? Number(exit[1]) : null;
|
|
2118
|
+
await sampleContext();
|
|
2119
|
+
break;
|
|
2120
|
+
}
|
|
2121
|
+
journal.append("worker-nudge-expired", t.id, { slot: slot.name, attempt, graceMs: workerNudgeGraceMs });
|
|
2122
|
+
lastProgressAt = Date.now() - stallWindowMs; // the existing harvest/classify tail runs unmodified
|
|
2123
|
+
continue; // conclude via the loop condition — never page over an acted-on nudge
|
|
2124
|
+
}
|
|
2125
|
+
// Unlatched page (T1): the page DECISION fires and is journaled every slice the
|
|
2126
|
+
// operator is the right actor — i.e. the nudge path doesn't own it (non-allowlisted
|
|
2127
|
+
// adapter, no nudge surface, or a blocked dialog) or the nudge was attempted and
|
|
2128
|
+
// FAILED. A nudgeable worker below the silence threshold waits for its nudge; a
|
|
2129
|
+
// DELIVERED nudge's pending grace suppresses the page — the daemon already acted. An
|
|
2130
|
+
// ANSWERED nudge has no action pending (the disarm cleared the deadline), so a pane
|
|
2131
|
+
// that then reads blocked/idle is the operator's again.
|
|
2132
|
+
// Delivery is unlatched too: the operator is notified again on a status change or
|
|
2133
|
+
// once pageRepeatMs elapses, so a missed first page is not the last one.
|
|
2134
|
+
const pageable = (st === "blocked" || st === "idle")
|
|
2135
|
+
&& (!nudgePending || nudgeFailed);
|
|
2136
|
+
if (pageable) {
|
|
2137
|
+
journal.append("operator-page", t.id, { slot: slot.name, attempt, status: st });
|
|
2138
|
+
if (st !== lastPagedStatus || sliceNow - lastPagedAt >= pageRepeatMs) {
|
|
2139
|
+
lastPagedStatus = st;
|
|
2140
|
+
lastPagedAt = sliceNow;
|
|
2141
|
+
const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
|
|
2142
|
+
await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
|
|
2143
|
+
}
|
|
2144
|
+
}
|
|
2145
|
+
// a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
|
|
2146
|
+
const spent = Date.now() - sliceStart;
|
|
2147
|
+
if (spent < slice)
|
|
2148
|
+
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
2149
|
+
}
|
|
2150
|
+
if (!finished && exitCode === null) {
|
|
2151
|
+
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
2152
|
+
timedOut = Date.now() - lastProgressAt >= stallWindowMs;
|
|
2153
|
+
output = await driver.read(slot, PANE_READ_ROWS);
|
|
2154
|
+
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
2155
|
+
const exit = exitRe.exec(output);
|
|
2156
|
+
exitCode = exit ? Number(exit[1]) : null;
|
|
2157
|
+
}
|
|
2158
|
+
if (finished) {
|
|
2159
|
+
await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
|
|
2160
|
+
output = await driver.read(slot, PANE_READ_ROWS);
|
|
2161
|
+
}
|
|
2162
|
+
// T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
|
|
2163
|
+
// contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
|
|
2164
|
+
// the existing pane-read seam once or twice before recording a malformed-trailer cause.
|
|
2165
|
+
if (interactive) {
|
|
2166
|
+
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
2167
|
+
const settleDeadline = attemptStart + stallWindowMs;
|
|
2168
|
+
const settleDelayMs = 1_000;
|
|
2169
|
+
const maxSettleRetries = 2;
|
|
2170
|
+
let settleTries = 0;
|
|
2171
|
+
settleParsed = adapter.parse(output, nonce);
|
|
2172
|
+
while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
|
|
2173
|
+
const remaining = settleDeadline - Date.now();
|
|
2174
|
+
if (remaining <= 0)
|
|
2175
|
+
break;
|
|
2176
|
+
await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
|
|
2177
|
+
output = await driver.read(slot, PANE_READ_ROWS);
|
|
2178
|
+
settleParsed = adapter.parse(output, nonce);
|
|
2179
|
+
settleTries++;
|
|
2180
|
+
}
|
|
2181
|
+
if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
|
|
2182
|
+
finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
|
|
2183
|
+
}
|
|
2184
|
+
}
|
|
987
2185
|
}
|
|
988
|
-
output = seedResult.output;
|
|
989
2186
|
}
|
|
990
2187
|
else {
|
|
991
2188
|
try {
|
|
@@ -998,229 +2195,68 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
998
2195
|
continue attempts;
|
|
999
2196
|
return;
|
|
1000
2197
|
}
|
|
1001
|
-
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
finished = false;
|
|
1005
|
-
}
|
|
1006
|
-
else {
|
|
1007
|
-
let paged = false;
|
|
1008
|
-
// v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
|
|
1009
|
-
// Any other blocked/idle dialog still pages the operator (paged latch below).
|
|
1010
|
-
let trustAnswered = false;
|
|
1011
|
-
// OBS-201: one liveness nudge per attempt; the grace deadline is its OWN timer, never the
|
|
1012
|
-
// stall window (the nudge's pane echo is absorbed before it starts, or the echo itself
|
|
1013
|
-
// would reset the window and make the early conclusion unreachable).
|
|
1014
|
-
let nudged = false;
|
|
1015
|
-
let nudgeDeadline;
|
|
1016
|
-
finished = false;
|
|
1017
|
-
exitCode = null;
|
|
1018
|
-
// OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
|
|
1019
|
-
// stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
|
|
2198
|
+
noteLaunched();
|
|
2199
|
+
// OBS-54: headless workers have the same output-inactivity budget as visible panes.
|
|
2200
|
+
// v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
|
|
1020
2201
|
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
let everHadOutput = output.length > 0;
|
|
2202
|
+
const initialPane = await driver.read(slot, 500);
|
|
2203
|
+
let everHadOutput = initialPane.length > 0;
|
|
1024
2204
|
const stallProgress = new StallProgressTracker();
|
|
1025
|
-
stallProgress.observe({ paneText:
|
|
2205
|
+
stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
|
|
1026
2206
|
let lastProgressAt = Date.now();
|
|
2207
|
+
finished = false;
|
|
2208
|
+
// T2 review (material): the exit marker proves the PROCESS EXITED, never that the worker
|
|
2209
|
+
// emitted a trailer — the two were the same flag here, so a headless worker that committed
|
|
2210
|
+
// and exited cleanly without one entered the tail as finished:true, skipping the harvest
|
|
2211
|
+
// synthesis entirely and reaching gates with the worker's own ok:false and no
|
|
2212
|
+
// worker-result-harvested row. The interactive site has always kept them apart (`finished`
|
|
2213
|
+
// there is the trailer regex; the exit marker only sets exitCode), and the cause taxonomy
|
|
2214
|
+
// already names this shape "clean-exit-no-trailer" — unreachable in print mode until now.
|
|
1027
2215
|
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
1028
|
-
const
|
|
1029
|
-
const remaining = stallWindowMs - (sliceStart - lastProgressAt);
|
|
2216
|
+
const remaining = stallWindowMs - (Date.now() - lastProgressAt);
|
|
1030
2217
|
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
1031
2218
|
if (!everHadOutput) {
|
|
1032
|
-
const earlyLeft = earlyLaunchLivenessMs - (
|
|
2219
|
+
const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
|
|
1033
2220
|
if (earlyLeft > 0)
|
|
1034
2221
|
slice = Math.min(slice, earlyLeft);
|
|
1035
2222
|
}
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
1042
|
-
const exit = exitRe.exec(output);
|
|
1043
|
-
if (finished || exit) {
|
|
1044
|
-
exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
|
|
1045
|
-
await sampleContext(); // final poll-seam sample before leaving the wait
|
|
1046
|
-
break;
|
|
1047
|
-
}
|
|
2223
|
+
// T2 (OBS-264): same probe cadence the interactive loop uses — see harvestSliceMs.
|
|
2224
|
+
slice = Math.min(slice, harvestSliceMs(Date.now() - lastProgressAt));
|
|
2225
|
+
if (await driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
|
|
2226
|
+
processExited = true;
|
|
2227
|
+
break;
|
|
1048
2228
|
}
|
|
1049
|
-
const paneText = await driver.read(slot,
|
|
2229
|
+
const paneText = await driver.read(slot, 500);
|
|
1050
2230
|
if (paneText.length > 0)
|
|
1051
2231
|
everHadOutput = true;
|
|
1052
|
-
// OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
|
|
1053
2232
|
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
1054
2233
|
earlyLaunchDead = true;
|
|
1055
|
-
output = paneText;
|
|
1056
2234
|
break;
|
|
1057
2235
|
}
|
|
1058
|
-
// v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
|
|
1059
2236
|
await sampleContext();
|
|
1060
2237
|
if (stallProgress.observe({ paneText, contextTokens }))
|
|
1061
2238
|
lastProgressAt = Date.now();
|
|
1062
|
-
//
|
|
1063
|
-
//
|
|
1064
|
-
//
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
const paneText = await driver.read(slot, 80);
|
|
1072
|
-
if (matchesTrustDialog(paneText, adapter.trustDialog)) {
|
|
1073
|
-
trustAnswered = true;
|
|
1074
|
-
// v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
|
|
1075
|
-
// Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
|
|
1076
|
-
journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
|
|
1077
|
-
await driver.sendKey(slot, adapter.trustDialog.key);
|
|
1078
|
-
const spent = Date.now() - sliceStart;
|
|
1079
|
-
if (spent < slice)
|
|
1080
|
-
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
1081
|
-
continue; // do not page — keep waiting for the trailer
|
|
1082
|
-
}
|
|
1083
|
-
}
|
|
1084
|
-
catch {
|
|
1085
|
-
/* read/send failed — fall through to page the operator */
|
|
1086
|
-
}
|
|
1087
|
-
}
|
|
1088
|
-
// OBS-201: the daemon ACTS on an idle pane before paging anyone. Gate: idle AND the
|
|
1089
|
-
// monotonic tracker silent ≥ the nudge threshold (a worker inside a tool run reads as
|
|
1090
|
-
// `working` and never gets here; a briefly-settling TUI hasn't been silent long enough).
|
|
1091
|
-
const nudgeable = st === "idle" && !!driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id);
|
|
1092
|
-
if (nudgeable && !nudged) {
|
|
1093
|
-
if (Date.now() - lastProgressAt >= nudgeAfterSilentMs) {
|
|
1094
|
-
nudged = true;
|
|
1095
|
-
if (await driver.nudge(slot, WORKER_NUDGE_MESSAGE)) {
|
|
1096
|
-
// absorb the nudge's own echo BEFORE arming the grace timer — post-nudge progress
|
|
1097
|
-
// is measured against this baseline, not against the echo.
|
|
1098
|
-
const echo = await driver.read(slot, 1000);
|
|
1099
|
-
stallProgress.observe({ paneText: echo, contextTokens });
|
|
1100
|
-
nudgeDeadline = Date.now() + workerNudgeGraceMs;
|
|
1101
|
-
journal.append("worker-nudge", t.id, { slot: slot.name, attempt });
|
|
1102
|
-
}
|
|
1103
|
-
else {
|
|
1104
|
-
journal.append("worker-nudge-failed", t.id, { slot: slot.name, attempt });
|
|
1105
|
-
// nudge undeliverable — fall back to today's behavior: page once, window backstop
|
|
1106
|
-
paged = true;
|
|
1107
|
-
await driver.notify(`tickmarkr ${runId}: ${slot.name} looks idle without finishing — check its pane`, { tier: "attention" });
|
|
1108
|
-
}
|
|
1109
|
-
}
|
|
1110
|
-
// idle but not yet silent past the threshold: neither nudge nor page this slice
|
|
1111
|
-
}
|
|
1112
|
-
else if (nudgeable && nudged && nudgeDeadline !== undefined) {
|
|
1113
|
-
if (Date.now() >= nudgeDeadline && Date.now() - lastProgressAt >= workerNudgeGraceMs) {
|
|
1114
|
-
// grace spent, still idle, no post-nudge progress: re-harvest once (the trailer may
|
|
1115
|
-
// have landed between polls), then conclude the wait as a stall NOW — the consult
|
|
1116
|
-
// sees the un-answered nudge instead of the remainder of the window.
|
|
1117
|
-
nudgeDeadline = undefined;
|
|
1118
|
-
output = await driver.read(slot, 1000);
|
|
1119
|
-
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
1120
|
-
const exit = exitRe.exec(output);
|
|
1121
|
-
if (finished || exit) {
|
|
1122
|
-
exitCode = exit ? Number(exit[1]) : null;
|
|
1123
|
-
await sampleContext();
|
|
1124
|
-
break;
|
|
1125
|
-
}
|
|
1126
|
-
journal.append("worker-nudge-expired", t.id, { slot: slot.name, attempt, graceMs: workerNudgeGraceMs });
|
|
1127
|
-
lastProgressAt = Date.now() - stallWindowMs; // the existing harvest/classify tail runs unmodified
|
|
1128
|
-
}
|
|
1129
|
-
}
|
|
1130
|
-
else {
|
|
1131
|
-
paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
|
|
1132
|
-
const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
|
|
1133
|
-
await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
|
|
1134
|
-
}
|
|
1135
|
-
}
|
|
1136
|
-
// a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
|
|
1137
|
-
const spent = Date.now() - sliceStart;
|
|
1138
|
-
if (spent < slice)
|
|
1139
|
-
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
1140
|
-
}
|
|
1141
|
-
if (!finished && exitCode === null) {
|
|
1142
|
-
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
1143
|
-
timedOut = Date.now() - lastProgressAt >= stallWindowMs;
|
|
1144
|
-
output = await driver.read(slot, 1000);
|
|
1145
|
-
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
1146
|
-
const exit = exitRe.exec(output);
|
|
1147
|
-
exitCode = exit ? Number(exit[1]) : null;
|
|
1148
|
-
}
|
|
1149
|
-
if (finished) {
|
|
1150
|
-
await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
|
|
1151
|
-
output = await driver.read(slot, 1000);
|
|
1152
|
-
}
|
|
1153
|
-
// T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
|
|
1154
|
-
// contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
|
|
1155
|
-
// the existing pane-read seam once or twice before recording a malformed-trailer cause.
|
|
1156
|
-
if (interactive) {
|
|
1157
|
-
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
1158
|
-
const settleDeadline = attemptStart + stallWindowMs;
|
|
1159
|
-
const settleDelayMs = 1_000;
|
|
1160
|
-
const maxSettleRetries = 2;
|
|
1161
|
-
let settleTries = 0;
|
|
1162
|
-
settleParsed = adapter.parse(output, nonce);
|
|
1163
|
-
while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
|
|
1164
|
-
const remaining = settleDeadline - Date.now();
|
|
1165
|
-
if (remaining <= 0)
|
|
1166
|
-
break;
|
|
1167
|
-
await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
|
|
1168
|
-
output = await driver.read(slot, 1000);
|
|
1169
|
-
settleParsed = adapter.parse(output, nonce);
|
|
1170
|
-
settleTries++;
|
|
1171
|
-
}
|
|
1172
|
-
if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
|
|
1173
|
-
finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
|
|
1174
|
-
}
|
|
2239
|
+
// T2 (OBS-264): the same liveness triad the interactive loop runs. A headless worker that
|
|
2240
|
+
// committed and went quiet is finished work too, and before this it rode the entire
|
|
2241
|
+
// window out before anything looked at its commits.
|
|
2242
|
+
// No nudge hold here, deliberately: print mode has no nudge surface at all (no pane to
|
|
2243
|
+
// steer, driver.nudge is never consulted on this path), so there is no pending daemon
|
|
2244
|
+
// action for the triad to preempt — the asymmetry with the interactive call site above is
|
|
2245
|
+
// the absence of the thing being held for, not an oversight.
|
|
2246
|
+
if (await harvestConcludes(Date.now() - lastProgressAt))
|
|
2247
|
+
break;
|
|
1175
2248
|
}
|
|
2249
|
+
output = await driver.read(slot, 500);
|
|
2250
|
+
exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
|
|
2251
|
+
// Completion is the trailer, exactly as in the interactive loop. A process that exited
|
|
2252
|
+
// without one is finished:false with a non-null exitCode — the harvest synthesis then owns
|
|
2253
|
+
// it when the worktree carries work, and classifyWorkerResultCause names it otherwise.
|
|
2254
|
+
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
2255
|
+
timedOut = !processExited && !finished && Date.now() - lastProgressAt >= stallWindowMs;
|
|
1176
2256
|
}
|
|
1177
2257
|
}
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
await driver.run(slot, paneDispatchCommand(dispatchScript));
|
|
1181
|
-
}
|
|
1182
|
-
catch (error) {
|
|
1183
|
-
if (!(error instanceof DeliveryReadinessError))
|
|
1184
|
-
throw error;
|
|
1185
|
-
if (await handleDeliveryReadiness(error))
|
|
1186
|
-
continue attempts;
|
|
1187
|
-
return;
|
|
1188
|
-
}
|
|
1189
|
-
// OBS-54: headless workers have the same output-inactivity budget as visible panes.
|
|
1190
|
-
// v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
|
|
1191
|
-
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
1192
|
-
const initialPane = await driver.read(slot, 500);
|
|
1193
|
-
let everHadOutput = initialPane.length > 0;
|
|
1194
|
-
const stallProgress = new StallProgressTracker();
|
|
1195
|
-
stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
|
|
1196
|
-
let lastProgressAt = Date.now();
|
|
1197
|
-
finished = false;
|
|
1198
|
-
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
1199
|
-
const remaining = stallWindowMs - (Date.now() - lastProgressAt);
|
|
1200
|
-
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
1201
|
-
if (!everHadOutput) {
|
|
1202
|
-
const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
|
|
1203
|
-
if (earlyLeft > 0)
|
|
1204
|
-
slice = Math.min(slice, earlyLeft);
|
|
1205
|
-
}
|
|
1206
|
-
if (await driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
|
|
1207
|
-
finished = true;
|
|
1208
|
-
break;
|
|
1209
|
-
}
|
|
1210
|
-
const paneText = await driver.read(slot, 500);
|
|
1211
|
-
if (paneText.length > 0)
|
|
1212
|
-
everHadOutput = true;
|
|
1213
|
-
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
1214
|
-
earlyLaunchDead = true;
|
|
1215
|
-
break;
|
|
1216
|
-
}
|
|
1217
|
-
await sampleContext();
|
|
1218
|
-
if (stallProgress.observe({ paneText, contextTokens }))
|
|
1219
|
-
lastProgressAt = Date.now();
|
|
1220
|
-
}
|
|
1221
|
-
output = await driver.read(slot, 500);
|
|
1222
|
-
exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
|
|
1223
|
-
timedOut = !finished && Date.now() - lastProgressAt >= stallWindowMs;
|
|
2258
|
+
finally {
|
|
2259
|
+
await cpuAccountant?.stop();
|
|
1224
2260
|
}
|
|
1225
2261
|
// SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
|
|
1226
2262
|
// shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
|
|
@@ -1232,7 +2268,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1232
2268
|
}
|
|
1233
2269
|
// keepPanes retains visible context, not a timed-out subprocess tree. Close before consult/retry
|
|
1234
2270
|
// can recreate the worktree; Herdr and subprocesses that reached their exit marker stay unchanged.
|
|
1235
|
-
if (keepOpen && (finished || driver.id !== "subprocess"))
|
|
2271
|
+
if (keepOpen && (finished || processExited || driver.id !== "subprocess"))
|
|
1236
2272
|
keptSlots.push(slot);
|
|
1237
2273
|
else
|
|
1238
2274
|
await closeSlot(slot);
|
|
@@ -1247,15 +2283,51 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1247
2283
|
tokens = addUsage(tokens, attemptUsage);
|
|
1248
2284
|
metered++;
|
|
1249
2285
|
}
|
|
1250
|
-
|
|
1251
|
-
const
|
|
2286
|
+
let result = settleParsed ?? adapter.parse(output, nonce);
|
|
2287
|
+
const workerFinished = finished;
|
|
2288
|
+
const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut });
|
|
1252
2289
|
journal.append("worker-result", t.id, {
|
|
1253
|
-
ok: result.ok, summary: result.summary, deviations: result.deviations, finished, exitCode,
|
|
1254
|
-
mode: interactive ? "interactive" : "print", ...(
|
|
2290
|
+
ok: result.ok, summary: result.summary, deviations: result.deviations, finished: workerFinished, exitCode,
|
|
2291
|
+
mode: interactive ? "interactive" : "print", ...(workerCause ? { cause: workerCause } : {}),
|
|
1255
2292
|
});
|
|
1256
|
-
|
|
2293
|
+
// T2 review (routing precedence): provider-death, quota and dead-channel classification
|
|
2294
|
+
// derive from ONE rule — the worker's OWN outcome, workerFinished and this PRE-HARVEST
|
|
2295
|
+
// parse — never from the synthesized result below. A worker that committed and then walled
|
|
2296
|
+
// (provider outage, quota banner, dead CLI) must still route; the harvest synthesis only
|
|
2297
|
+
// decides whether THIS attempt's worktree goes to gates, and when routing wins instead, the
|
|
2298
|
+
// commits survive via the existing commitsToCarry/cherryPickCommits carry-forward.
|
|
2299
|
+
const preHarvestResult = result;
|
|
2300
|
+
// T2 (OBS-264): recognize committed no-trailer work BEFORE any no-trailer streak, provider,
|
|
2301
|
+
// quota or dead-channel routing. Gates never trusted the trailer, so this successful synthesis
|
|
2302
|
+
// must enter exactly where a worker-claimed ok enters. Preserve the parsed worker truth in the
|
|
2303
|
+
// worker-result row above and name the synthesized gate input in its own harvested event.
|
|
2304
|
+
let harvestedCommits = [];
|
|
2305
|
+
if (!workerFinished) {
|
|
2306
|
+
harvestedCommits = await commitsAheadOf(taskBase, wt);
|
|
2307
|
+
if (harvestedCommits.length > 0) {
|
|
2308
|
+
result = { ok: true, summary: HARVESTED_RESULT_SUMMARY, deviations: [], raw: output };
|
|
2309
|
+
finished = true;
|
|
2310
|
+
journal.append("worker-result-harvested", t.id, {
|
|
2311
|
+
attempt, commits: harvestedCommits, summary: HARVESTED_RESULT_SUMMARY, source: "harvest",
|
|
2312
|
+
});
|
|
2313
|
+
}
|
|
2314
|
+
}
|
|
2315
|
+
// T2 review: provider-death is the THIRD routing branch that must read the pre-harvest
|
|
2316
|
+
// outcome — the synthesis must not null it. A worker that committed and then printed the
|
|
2317
|
+
// outage banner without exiting (workerFinished false, so the harvest fires) still takes
|
|
2318
|
+
// the capped same-channel requeue below; nulling the cause here would skip that branch and
|
|
2319
|
+
// let classifyDeadChannel(preHarvestResult) demote the channel run-wide on a transient blip.
|
|
2320
|
+
const cause = harvestedCommits.length > 0 && workerCause !== "provider-death" ? undefined : workerCause;
|
|
2321
|
+
// T2 review (family): the no-trailer streak is accounted on the SAME ONE rule the routing
|
|
2322
|
+
// branches above use — workerFinished and the PRE-HARVEST parse — never the synthesized
|
|
2323
|
+
// result. A channel that commits but never emits a parseable trailer still burned a
|
|
2324
|
+
// no-trailer window (OBS-57): the synthesis decides whether THIS worktree goes to gates, it
|
|
2325
|
+
// never certifies the channel. Reading the synthesized `finished`/`ok` here reset the streak
|
|
2326
|
+
// on every harvest, so a CLI that produces commits and swallows every trailer was immune to
|
|
2327
|
+
// the two-window demotion and stayed first pick for the rest of the run.
|
|
2328
|
+
if (preHarvestResult.ok && workerFinished)
|
|
1257
2329
|
noTrailerStreak.set(channelKey(assignment), 0);
|
|
1258
|
-
else if (!
|
|
2330
|
+
else if (!workerFinished && cause !== "provider-death") {
|
|
1259
2331
|
const ck = channelKey(assignment);
|
|
1260
2332
|
const streak = (noTrailerStreak.get(ck) ?? 0) + 1;
|
|
1261
2333
|
noTrailerStreak.set(ck, streak);
|
|
@@ -1275,8 +2347,19 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1275
2347
|
}
|
|
1276
2348
|
// quota exhaustion → failover within floor; does NOT consume the ladder (spec §4)
|
|
1277
2349
|
// print: guarded on exit code — exit-0 output that merely MENTIONS "rate limit" must not failover
|
|
1278
|
-
// interactive: a
|
|
1279
|
-
|
|
2350
|
+
// interactive: a worker-CLAIMED trailer beats quota mentions; without one, quota text fails over
|
|
2351
|
+
// (spec v1.2 §2) — matched on the chrome-filtered tail, the exact discrimination the in-loop
|
|
2352
|
+
// classifier makes, so the two can never disagree. A TUI harvest is the whole retained pane:
|
|
2353
|
+
// an unscoped match failed a worker over for quoting "rate limit" in its own diff (tail
|
|
2354
|
+
// scoping kills that — the mention sits ABOVE the tail), and a raw-tail match fires on fixed
|
|
2355
|
+
// chrome (codex's welcome line — filtered by identity, so a launch-time banner this backstop
|
|
2356
|
+
// exists to catch is never exculpated). Print output keeps the exit-code guard.
|
|
2357
|
+
// T2 review: the gate is `workerFinished`, not the harvest-synthesized `finished` — a
|
|
2358
|
+
// committed-but-quota-walled attempt routes here FIRST (its commits ride the carry-forward
|
|
2359
|
+
// into the next attempt's recreated worktree), it never buys a gate run on throttled work.
|
|
2360
|
+
const quotaHit = interactive
|
|
2361
|
+
? !workerFinished && QUOTA_RE.test(stallSnapshotBannerRows(output))
|
|
2362
|
+
: exitCode !== 0 && QUOTA_RE.test(output);
|
|
1280
2363
|
if (quotaHit) {
|
|
1281
2364
|
const next = failover("quota-failover");
|
|
1282
2365
|
journal.append("quota-failover", t.id, { from: channelKey(assignment), to: next ? channelKey(next) : null });
|
|
@@ -1316,7 +2399,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1316
2399
|
// cap above is spent, so a transient blip still recovers in place.
|
|
1317
2400
|
// OBS-117 (v1.71 T6): a silent launch failure has no CLI signature to parse — the same
|
|
1318
2401
|
// setup-required typed dead-channel path a late-harvest "command not found" would take.
|
|
1319
|
-
|
|
2402
|
+
// T2 review: classify the PRE-HARVEST parse — classifyDeadChannel bails on any ok:true
|
|
2403
|
+
// result, so reading the synthesized harvest result would swallow auth-required /
|
|
2404
|
+
// setup-required / provider-outage for every committed-but-walled attempt (in both modes).
|
|
2405
|
+
const dead = classifyDeadChannel(preHarvestResult) ?? (earlyLaunchDead ? "setup-required" : undefined);
|
|
1320
2406
|
if (dead) {
|
|
1321
2407
|
const from = channelKey(assignment);
|
|
1322
2408
|
demotedChannels.add(from); // excluded for later attempts AND later tasks in this run
|
|
@@ -1382,33 +2468,36 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1382
2468
|
}
|
|
1383
2469
|
const onGate = async (e) => {
|
|
1384
2470
|
if (e.phase === "start") {
|
|
1385
|
-
|
|
2471
|
+
notePhaseStart(e);
|
|
2472
|
+
journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
|
|
1386
2473
|
return;
|
|
1387
2474
|
}
|
|
1388
2475
|
const g = e.result;
|
|
1389
|
-
|
|
1390
|
-
|
|
1391
|
-
|
|
1392
|
-
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
2476
|
+
inParallelOrder(g.gate, () => {
|
|
2477
|
+
// GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
|
|
2478
|
+
// which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
|
|
2479
|
+
// flaked, retried" from "worker failed" (run-20260711-185020 P43-03 L70-72 billed a judge flake as
|
|
2480
|
+
// a worker attempt; 47-01 fixed WHO retries, this closes the audit-trail half). The condition is
|
|
2481
|
+
// META-ONLY (D-03): gate === "acceptance" + typeof-shape guards on meta.judgeRetry — never a
|
|
2482
|
+
// details-regex. The v1.1 review regex below is grandfathered, not precedent. Appended BEFORE the
|
|
2483
|
+
// gate-result so attribution precedes the verdict in the stream. secondUnparseable is derived from
|
|
2484
|
+
// the final result's meta.unparseable (set by run-gates when the retry ALSO flaked — double-garbage).
|
|
2485
|
+
if (g.gate === "acceptance" && typeof g.meta?.judgeRetry === "object" && g.meta.judgeRetry !== null) {
|
|
2486
|
+
const jr = g.meta.judgeRetry;
|
|
2487
|
+
if (typeof jr.flaked === "string" && typeof jr.retried === "string") {
|
|
2488
|
+
journal.append("judge-retry", t.id, {
|
|
2489
|
+
gate: "acceptance", flaked: jr.flaked, retried: jr.retried,
|
|
2490
|
+
...(g.meta.unparseable === true ? { secondUnparseable: true } : {}),
|
|
2491
|
+
});
|
|
2492
|
+
}
|
|
1404
2493
|
}
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
}
|
|
2494
|
+
journalGateResult(g);
|
|
2495
|
+
noteReviewRetry(g);
|
|
2496
|
+
// v1.1 failover: never re-ask a reviewer channel that produced garbage for this task
|
|
2497
|
+
if (g.gate === "review" && !g.pass && /unparseable/.test(g.details) && typeof g.meta?.reviewer === "string") {
|
|
2498
|
+
badReviewers.push(g.meta.reviewer);
|
|
2499
|
+
}
|
|
2500
|
+
});
|
|
1412
2501
|
};
|
|
1413
2502
|
let results = [];
|
|
1414
2503
|
let commits = [];
|
|
@@ -1418,6 +2507,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1418
2507
|
({ results, commits } = await runGates(t, {
|
|
1419
2508
|
worktree: wt, baseRef: taskBase, result, author: assignment,
|
|
1420
2509
|
commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
|
|
2510
|
+
pipeline: "v185", selectTests: !testGateFailed,
|
|
1421
2511
|
via: cfg.visibility.llm === "pane"
|
|
1422
2512
|
? {
|
|
1423
2513
|
driver: trackedDriver,
|
|
@@ -1439,7 +2529,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1439
2529
|
}));
|
|
1440
2530
|
graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
|
|
1441
2531
|
saveGraph(repoRoot, graph);
|
|
1442
|
-
if (results.
|
|
2532
|
+
if (results.some((g) => g.gate === "test" && !g.pass))
|
|
2533
|
+
testGateFailed = true;
|
|
2534
|
+
if (results.every(gateSatisfied)) {
|
|
1443
2535
|
const m = await mergeSerial(taskBranch, t, gated);
|
|
1444
2536
|
if (m.tipMoved) {
|
|
1445
2537
|
journal.append("tip-moved", t.id, m.tipMoved);
|
|
@@ -1484,21 +2576,113 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1484
2576
|
// v1.53 T3: prefer the CLI's own session id captured from this attempt's output (kimi's resume
|
|
1485
2577
|
// trailer) over the harness slot name; absent hook or no capture keeps today's slot-name id.
|
|
1486
2578
|
retrySession = { channel: channelKey(assignment), id: adapter.sessionIdFrom?.(output) ?? sessionId, contextTokens };
|
|
1487
|
-
feedback = results.filter(
|
|
2579
|
+
feedback = results.filter(gateFailed).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
|
|
1488
2580
|
// OBS-189/G3 (park-economics patch): a request-changes review is a findings brief, not a worker
|
|
1489
2581
|
// defect — the fix attempt stays on the same channel with the findings as feedback and consumes
|
|
1490
2582
|
// no escalation-ladder rung. Bounded by the engagement round cap at the top of this loop.
|
|
1491
2583
|
// Unparseable verdicts (already retried in-gate, OBS-193) and diff-cap trips (the diff cannot
|
|
1492
2584
|
// shrink by retrying, OBS-48) fall through to the ladder unchanged. Review runs last, so a
|
|
1493
2585
|
// failed review with every other gate green is exactly "the work landed, the reviewer objects".
|
|
1494
|
-
const reviewFail = results.find((g) => g.gate === "review" &&
|
|
2586
|
+
const reviewFail = results.find((g) => g.gate === "review" && gateFailed(g));
|
|
1495
2587
|
const reviewFixRetry = reviewFail !== undefined
|
|
1496
2588
|
&& reviewFail.meta?.unparseable !== true
|
|
1497
2589
|
&& !isDiffCapPark(reviewFail)
|
|
1498
|
-
&& results.every((g) => g
|
|
1499
|
-
const
|
|
1500
|
-
|
|
1501
|
-
|
|
2590
|
+
&& results.every((g) => gateSatisfied(g) || g.gate === "review");
|
|
2591
|
+
const failing = results.filter(gateFailed);
|
|
2592
|
+
// Decided before the cap, because the cap's question is whether the NEXT move would re-buy a
|
|
2593
|
+
// measurement already made — and a funded repair is one of the moves that would.
|
|
2594
|
+
const landed = await commitsAheadOf(taskBase, wt);
|
|
2595
|
+
// T4 (OBS-265): judge and review are now ONE round, so a round can report both failing where the
|
|
2596
|
+
// serial walk returned at the judge and review never ran. Eligibility is scored on the battery
|
|
2597
|
+
// the serial contract would have surfaced — review only speaks for a round nothing else failed —
|
|
2598
|
+
// so removing the waiting does not re-price the ladder. The journal below still names every
|
|
2599
|
+
// failing gate, and `feedback` still carries every one of them to the next attempt.
|
|
2600
|
+
const repairBattery = failing.some((g) => g.gate !== "review") ? failing.filter((g) => g.gate !== "review") : failing;
|
|
2601
|
+
const repairable = narrowRepairBattery(repairBattery) && lostCommits.length === 0 && landed.length > 0;
|
|
2602
|
+
const repairsDrawn = repairsSinceApproval(journal.read(), t.id);
|
|
2603
|
+
const repair = repairable && repairsDrawn < MAX_REPAIRS;
|
|
2604
|
+
// v1.85 T3: the fingerprint cap. Two normalized-identical failures of one DETERMINISTIC gate on
|
|
2605
|
+
// one task (volatile tokens — worktree prefixes, line refs, durations, run ids — are not
|
|
2606
|
+
// information) mean the round about to be bought is a re-measurement: ~663m across 5 runs went to
|
|
2607
|
+
// exactly this loop. The threshold is a property of the FAILURE, never of the move that would
|
|
2608
|
+
// follow it, so it is evaluated on EVERY such gate at EVERY ladder position — including a rung
|
|
2609
|
+
// that would change channel, and including a review-fix round. Conditioning it on the next rung
|
|
2610
|
+
// was the first shape of this and let a second identical failure buy an escalate/consult round
|
|
2611
|
+
// the criterion says it may not buy. What the cap does NOT reach is an LLM verdict, which is a
|
|
2612
|
+
// different object with its own tighter bound (see isDeterministicFailure).
|
|
2613
|
+
//
|
|
2614
|
+
// It fires on the CROSSING, not as a latch: the consult it forces and the ban it sets govern the
|
|
2615
|
+
// next move, so re-firing on the third identical failure would only re-buy the round it just paid
|
|
2616
|
+
// for — and the ladder, whose rung this failure still spends, bounds the rest.
|
|
2617
|
+
const repeated = failing.find((g) => isDeterministicFailure(g)
|
|
2618
|
+
&& identicalGateFailures(journal.read(), t.id, g.gate, normalizeGateFailure(g.details)) === GATE_FINGERPRINT_CAP);
|
|
2619
|
+
// The rung the cap spent, when it fired — the move below executes THIS instead of drawing a
|
|
2620
|
+
// second one, so a cap costs exactly the rung the failure would have cost anyway.
|
|
2621
|
+
let capStep;
|
|
2622
|
+
if (repeated) {
|
|
2623
|
+
const normalized = normalizeGateFailure(repeated.details);
|
|
2624
|
+
journal.append("gate-fingerprint-cap", t.id, {
|
|
2625
|
+
gate: repeated.gate,
|
|
2626
|
+
occurrences: GATE_FINGERPRINT_CAP,
|
|
2627
|
+
fingerprint: normalized.slice(0, 500),
|
|
2628
|
+
retrySameBanned: true,
|
|
2629
|
+
channel: channelKey(assignment), // the ban is bound to the channel that produced the repeat
|
|
2630
|
+
attempt: attempt + 1,
|
|
2631
|
+
});
|
|
2632
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} ${repeated.gate} failed identically twice — consulting, identical retry banned`, { tier: "attention" });
|
|
2633
|
+
// The cap takes the ladder's MOVE, never its accounting: this failure still spends the rung it
|
|
2634
|
+
// would have spent, so a task that cannot converge still reaches ladder exhaustion on exactly
|
|
2635
|
+
// the budget it always had and the cap can never hand a stuck task extra rounds.
|
|
2636
|
+
capStep = r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
|
|
2637
|
+
journal.append("escalation", t.id, { step: capStep, attempt: attempt + 1, fingerprintCap: true });
|
|
2638
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${capStep}`, { tier: "attention" });
|
|
2639
|
+
const v = await runConsult("gate-fail-repeat", output, feedback, results);
|
|
2640
|
+
// Rule: a terminal cap consult vetoes a same-channel retry, but cannot veto an `escalate`
|
|
2641
|
+
// rung that already satisfies retry-same-banned by changing channel. Thus terminal+retry
|
|
2642
|
+
// parks through the shared verdict boundary, while terminal+escalate records the consult as
|
|
2643
|
+
// advisory and executes the already-spent rung below. Recoverable verdicts still control the
|
|
2644
|
+
// move directly. The paired fixture asserts both directions of this boundary.
|
|
2645
|
+
const recoverable = v.action === "retry" || v.action === "reroute";
|
|
2646
|
+
if (recoverable || capStep !== "escalate") {
|
|
2647
|
+
if (await applyVerdict(v, attempt + 1, "gate-fail"))
|
|
2648
|
+
continue;
|
|
2649
|
+
return;
|
|
2650
|
+
}
|
|
2651
|
+
journal.append("consult-verdict", t.id, { action: v.action, notes: v.notes, capAdvisory: true });
|
|
2652
|
+
}
|
|
2653
|
+
// v1.85 T3: a narrow battery over fully carried commits earns a REPAIR (decided above) — the
|
|
2654
|
+
// next dispatch carries the findings verbatim and the diff content instead of re-onboarding a
|
|
2655
|
+
// fresh worker. Budget is engagement-scoped and journal-derived, so a resume inherits it rather
|
|
2656
|
+
// than refunding it; the third repair-eligible failure falls back to the fresh ladder.
|
|
2657
|
+
//
|
|
2658
|
+
// `landed` is measured from the worktree rather than read from runGates: runGates returns at
|
|
2659
|
+
// the first failure, so a red test or lint gate never reaches the evidence stage and its
|
|
2660
|
+
// `commits` come back empty — reading them would make the ruling's test/lint case unreachable.
|
|
2661
|
+
//
|
|
2662
|
+
// Budget spent means the FRESH LADDER owns this failure — including a review-only one, whose
|
|
2663
|
+
// same-channel fix retry is exactly the round the budget just declared too expensive to repeat.
|
|
2664
|
+
const repairExhausted = repairable && !repair;
|
|
2665
|
+
if (repair && !capStep) {
|
|
2666
|
+
journal.append("repair-attempt", t.id, {
|
|
2667
|
+
repair: repairsDrawn + 1, of: MAX_REPAIRS, gates: failing.map((g) => g.gate),
|
|
2668
|
+
commits: landed.length,
|
|
2669
|
+
findings: feedback, // the failure bytes this repair must carry, replayable across a resume
|
|
2670
|
+
});
|
|
2671
|
+
}
|
|
2672
|
+
else if (repairExhausted && !capStep) {
|
|
2673
|
+
journal.append("repair-exhausted", t.id, { repairs: repairsDrawn, of: MAX_REPAIRS, gates: failing.map((g) => g.gate) });
|
|
2674
|
+
}
|
|
2675
|
+
const step = capStep ?? (repair || (reviewFixRetry && !repairExhausted)
|
|
2676
|
+
? "retry"
|
|
2677
|
+
: r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)]);
|
|
2678
|
+
if (!capStep) { // a capped failure already journaled and announced the rung it spent
|
|
2679
|
+
journal.append("escalation", t.id, {
|
|
2680
|
+
step, attempt: attempt + 1,
|
|
2681
|
+
...(reviewFixRetry && !repairExhausted ? { reviewFix: true } : {}),
|
|
2682
|
+
...(repair ? { repair: repairsDrawn + 1 } : {}),
|
|
2683
|
+
});
|
|
2684
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
|
|
2685
|
+
}
|
|
1502
2686
|
if (step === "retry")
|
|
1503
2687
|
continue;
|
|
1504
2688
|
if (step === "escalate") {
|
|
@@ -1574,24 +2758,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1574
2758
|
// OBS-34: post-merge integration-tip verify — strict exit codes, no baseline forgiveness.
|
|
1575
2759
|
const lastMergedTask = [...journal.read()].reverse().find((e) => e.event === "merge" && e.taskId)?.taskId;
|
|
1576
2760
|
if (summary.done.length > 0 && Object.keys(commands).length > 0) {
|
|
1577
|
-
const
|
|
1578
|
-
let tipFailed = false;
|
|
1579
|
-
for (const r of tipResults) {
|
|
1580
|
-
if (r.pass) {
|
|
1581
|
-
journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details });
|
|
1582
|
-
}
|
|
1583
|
-
else {
|
|
1584
|
-
journal.append("tip-verify-failed", undefined, {
|
|
1585
|
-
gate: r.gate,
|
|
1586
|
-
cmd: r.cmd,
|
|
1587
|
-
exitCode: r.exitCode,
|
|
1588
|
-
fingerprints: r.fingerprints,
|
|
1589
|
-
artifact: r.artifact,
|
|
1590
|
-
lastMergedTask,
|
|
1591
|
-
});
|
|
1592
|
-
tipFailed = true;
|
|
1593
|
-
}
|
|
1594
|
-
}
|
|
2761
|
+
const tipFailed = await verifyIntegrationTipCached(intWt, commands, journal, { lastMergedTask });
|
|
1595
2762
|
summary.tipVerify = tipFailed ? "failed" : "passed";
|
|
1596
2763
|
if (tipFailed && lastMergedTask)
|
|
1597
2764
|
summary.lastMergedTask = lastMergedTask;
|
|
@@ -1617,20 +2784,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1617
2784
|
return summary;
|
|
1618
2785
|
}
|
|
1619
2786
|
catch (err) {
|
|
1620
|
-
|
|
1621
|
-
|
|
1622
|
-
|
|
1623
|
-
|
|
1624
|
-
|
|
1625
|
-
failed: [],
|
|
1626
|
-
human: [],
|
|
1627
|
-
blocked: [],
|
|
1628
|
-
pending: [],
|
|
1629
|
-
phase: "setup",
|
|
1630
|
-
fatal: true,
|
|
1631
|
-
error: err instanceof Error ? err.message : String(err),
|
|
1632
|
-
});
|
|
1633
|
-
}
|
|
2787
|
+
// T7 (v1.86): guarded — a journal read/append failure while recording the fatal run-end is
|
|
2788
|
+
// reported alongside err, never instead of it; recordFatalRunEnd never throws, so the original
|
|
2789
|
+
// error (message, stack, cause) always reaches the caller verbatim.
|
|
2790
|
+
if (runStarted && !taskLoopStarted)
|
|
2791
|
+
recordFatalRunEnd(journal, runId, branch, err);
|
|
1634
2792
|
throw err;
|
|
1635
2793
|
}
|
|
1636
2794
|
finally {
|