tickmarkr 2.6.1 → 2.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -3
- package/dist/adapters/catalog-remote.js +89 -47
- package/dist/adapters/claude-code.js +9 -6
- package/dist/adapters/codex.js +7 -4
- package/dist/adapters/prompt.d.ts +1 -0
- package/dist/adapters/prompt.js +14 -6
- package/dist/adapters/registry.js +3 -3
- package/dist/adapters/types.d.ts +12 -4
- package/dist/adapters/types.js +6 -0
- package/dist/cli/commands/approve.d.ts +5 -1
- package/dist/cli/commands/approve.js +66 -23
- package/dist/cli/commands/compile.js +13 -3
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +11 -3
- package/dist/cli/commands/fleet.js +45 -7
- package/dist/cli/commands/report.js +18 -2
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/status.js +24 -19
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +9 -2
- package/dist/config/config.d.ts +20 -0
- package/dist/config/config.js +47 -8
- package/dist/config/fleet-overlay.d.ts +1 -0
- package/dist/config/fleet-overlay.js +56 -0
- package/dist/drivers/orca.d.ts +26 -1
- package/dist/drivers/orca.js +193 -60
- package/dist/eval/canary.d.ts +2 -1
- package/dist/eval/canary.js +2 -2
- package/dist/eval/dispatch.js +1 -0
- package/dist/gates/acceptance.d.ts +2 -1
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/baseline.d.ts +12 -1
- package/dist/gates/baseline.js +11 -4
- package/dist/gates/llm.d.ts +5 -4
- package/dist/gates/llm.js +13 -13
- package/dist/gates/review.d.ts +8 -0
- package/dist/gates/review.js +40 -4
- package/dist/gates/run-gates.d.ts +2 -1
- package/dist/gates/run-gates.js +27 -13
- package/dist/gates/test-manifest.d.ts +3 -1
- package/dist/gates/test-manifest.js +9 -2
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/plan/scope.js +2 -2
- package/dist/route/preference.d.ts +20 -2
- package/dist/route/preference.js +48 -13
- package/dist/route/router.js +30 -15
- package/dist/run/consult.d.ts +13 -1
- package/dist/run/consult.js +14 -5
- package/dist/run/daemon.d.ts +37 -2
- package/dist/run/daemon.js +579 -141
- package/dist/run/git.d.ts +8 -0
- package/dist/run/git.js +14 -0
- package/dist/run/journal.d.ts +126 -3
- package/dist/run/journal.js +410 -37
- package/dist/run/merge.d.ts +3 -1
- package/dist/run/merge.js +3 -2
- package/dist/run/operator-summary.d.ts +3 -0
- package/dist/run/operator-summary.js +3 -1
- package/dist/run/protocol.d.ts +31 -1
- package/dist/run/protocol.js +3 -1
- package/dist/run/supervision.d.ts +7 -1
- package/dist/run/supervision.js +5 -2
- package/dist/tui/cockpit/board.js +3 -3
- package/dist/tui/cockpit/decision-actions.d.ts +8 -5
- package/dist/tui/cockpit/decision-actions.js +55 -32
- package/dist/tui/cockpit/derive.js +13 -2
- package/dist/tui/cockpit/live-runtime.d.ts +10 -0
- package/dist/tui/cockpit/live-runtime.js +50 -3
- package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
- package/dist/tui/cockpit/run-cockpit.js +26 -1
- package/dist/tui/cockpit/run-view.d.ts +9 -3
- package/dist/tui/cockpit/run-view.js +60 -7
- package/dist/tui/cockpit/setup-cockpit.d.ts +4 -0
- package/dist/tui/cockpit/setup-cockpit.js +6 -3
- package/dist/tui/ink/fleet-app.d.ts +15 -3
- package/dist/tui/ink/fleet-app.js +91 -22
- package/package.json +2 -1
- package/schema/config.schema.json +818 -0
- package/skills/tickmarkr-loop/SKILL.md +8 -2
- package/skills/tickmarkr-overseer/SKILL.md +42 -0
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +88 -0
- package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/dist/run/daemon.js
CHANGED
|
@@ -10,7 +10,7 @@ import { tmpdir } from "node:os";
|
|
|
10
10
|
import { basename, dirname, isAbsolute, join, posix, relative, resolve, sep } from "node:path";
|
|
11
11
|
import { fileURLToPath } from "node:url";
|
|
12
12
|
import { stringify } from "yaml";
|
|
13
|
-
import { classifyDeadChannel, classifyTransientCapacity,
|
|
13
|
+
import { BOOTSTRAP_FAILURE_RE, classifyDeadChannel, classifyTransientCapacity, trailerPattern, writePrompt } from "../adapters/prompt.js";
|
|
14
14
|
import { allAdapters, getAdapter, probeAll, readDoctor, rolePools } from "../adapters/registry.js";
|
|
15
15
|
import { SettledTrailerTracker, addUsage, CAPACITY_RE, channelKey, matchesInputBox, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
16
16
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
@@ -32,16 +32,16 @@ import { executionSignal, remainingExecutionMs, withExecutionBudget, withoutExec
|
|
|
32
32
|
import { repairSelectionDecision } from "./repair-selection.js";
|
|
33
33
|
import { failureDisposition, reserveInfrastructureRetry } from "./recovery.js";
|
|
34
34
|
import { runEnvironment } from "./environment.js";
|
|
35
|
-
import { cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, PRESERVE_COMMIT_SUBJECT, PRESERVE_PRODUCER_TRAILER, preserveWorktree, producerFields, resolvedCapacity, runWithForkBudget, runWithVerificationBudget, sameCapacity, sameVerification, sh, shGit, SUITE_PARENT_ENV, verificationProtocol, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
35
|
+
import { changeRepresented, cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, PRESERVE_COMMIT_SUBJECT, PRESERVE_PRODUCER_TRAILER, preserveWorktree, producerFields, resolvedCapacity, runWithForkBudget, runWithVerificationBudget, sameCapacity, sameVerification, sh, shGit, SUITE_PARENT_ENV, verificationProtocol, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
36
36
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
37
37
|
import { classifyRepairDisposition, resolveScopeHints } from "./repair-disposition.js";
|
|
38
|
-
import { applyScopeAmendments, activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingApprovalActions, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
38
|
+
import { applyScopeAmendments, activeRetryBan, interruptedAttempt, RESUME_HARVEST_SOURCE, resumeHarvestAuthor, approvalAction, APPROVAL_REFUSED, bindingToken, effectiveEvents, foldDecisions, physicalLine, staleApprovals, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingApprovalActions, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, standingRulings, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
39
39
|
import { gateReviewerFloor, isDiffCapPark, pickReviewer } from "../gates/review.js";
|
|
40
40
|
import { acquireApprovalSerialization, acquireRunLock, isPidLive, releaseRunLock } from "./lock.js";
|
|
41
41
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, reusedTipEvidence, verifyIntegrationTip } from "./merge.js";
|
|
42
42
|
import { climbChannel, marginalCostRank, nextChannel, route } from "../route/router.js";
|
|
43
43
|
import { desiredPanes } from "./reconcile.js";
|
|
44
|
-
import { readTierLiveness, readWatchBoard,
|
|
44
|
+
import { readTierLiveness, readWatchBoard, supervisionPresencePath } from "./supervision.js";
|
|
45
45
|
import { harvestCpuFlatWindowMs, NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows, WorkerTreeCpuAccountant, reapOwnedProcessGroup, readOwnedProcessGroup, } from "./stall.js";
|
|
46
46
|
// The live set is also the ownership claim shared by task cleanup and termination.
|
|
47
47
|
export async function closeLiveSlot(liveSlots, driver, slot) {
|
|
@@ -58,6 +58,19 @@ export async function closeLiveSlot(liveSlots, driver, slot) {
|
|
|
58
58
|
// A transport timeout says nothing about whether a mutation reached the terminal.
|
|
59
59
|
class HeldProbeExhausted extends Error {
|
|
60
60
|
}
|
|
61
|
+
// OBS-1108: an owned slot whose contact stayed latched unreadable past its deadline. Same terminal
|
|
62
|
+
// path as an exhausted transport probe — worktree preserved, one infra park (recheck-able) — with
|
|
63
|
+
// its own disposition, so the park names what was actually lost.
|
|
64
|
+
class ContactUnreadableExhausted extends HeldProbeExhausted {
|
|
65
|
+
}
|
|
66
|
+
// OBS-1108: how long one owned slot's contact may stay latched unreadable before its attempt parks.
|
|
67
|
+
// Default: the attempt's own stall window — the latch holds the stall kill down, so without this
|
|
68
|
+
// only the 4x hard deadline bounded a blind wait (2.5.8: 22k rows; 2.6.1: ~12k rows/h per worker).
|
|
69
|
+
let contactUnreadableDeadlineMs;
|
|
70
|
+
export function setContactUnreadableDeadlineMsForTests(ms) { contactUnreadableDeadlineMs = ms; }
|
|
71
|
+
export function resetContactUnreadableDeadlineMsForTests() { contactUnreadableDeadlineMs = undefined; }
|
|
72
|
+
const CONTACT_BACKOFF_BASE_MS = 1_000; // the pre-latch retry pause, doubled per latched retry
|
|
73
|
+
const CONTACT_BACKOFF_DOUBLINGS = 5; // ponytail: 32 s ceiling; the poll slice caps it lower anyway
|
|
61
74
|
function isTransportTimeout(error) {
|
|
62
75
|
if (!(error instanceof Error))
|
|
63
76
|
return false;
|
|
@@ -167,26 +180,53 @@ export function resolveRunMode(repoRoot, opts = {}) {
|
|
|
167
180
|
: undefined;
|
|
168
181
|
return { cfg: resolved.cfg, mode: resolved.mode, source, ...(conflict ? { conflict } : {}) };
|
|
169
182
|
}
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
183
|
+
/** Journal rows are untrusted input: a malformed field reads as not recorded, never as a value. */
|
|
184
|
+
export const fingerprintsOf = (v) => Array.isArray(v) && v.every((x) => typeof x === "string") ? v : undefined;
|
|
185
|
+
export const baselineProvenanceOf = (v) => {
|
|
186
|
+
const { baseRef, capturedAt } = (typeof v === "object" && v !== null ? v : {});
|
|
187
|
+
return typeof baseRef === "string" && typeof capturedAt === "string" ? { baseRef, capturedAt } : undefined;
|
|
188
|
+
};
|
|
189
|
+
/** A forgiven battery green: the structured field, or the battery's own "(forgiven)" wording on a legacy row. */
|
|
190
|
+
export const forgivenGateRow = (data) => data.pass === true && (Array.isArray(data.forgivenFingerprints) || /\(forgiven\)/u.test(String(data.details ?? "")));
|
|
191
|
+
/**
|
|
192
|
+
* OBS-1123: the forgiveness standing at this point in the journal — each task gate's LATEST verdict
|
|
193
|
+
* where it is a forgiven green, then the latest verification cycle's forgiven tip rows. A legacy row
|
|
194
|
+
* keeps what it never recorded absent, so the reader says "unknown" instead of borrowing a date.
|
|
195
|
+
*/
|
|
196
|
+
export function forgivenFingerprints(events) {
|
|
197
|
+
const latest = new Map();
|
|
198
|
+
for (const e of events) {
|
|
199
|
+
if (e.event === "gate-result" && e.taskId && typeof e.data.gate === "string")
|
|
200
|
+
latest.set(`${e.taskId}\0${e.data.gate}`, e);
|
|
188
201
|
}
|
|
189
|
-
|
|
202
|
+
const tasks = [...latest.values()].filter((e) => forgivenGateRow(e.data));
|
|
203
|
+
const start = events.map((e) => e.event).lastIndexOf("tip-verify-start");
|
|
204
|
+
const tip = events.slice(start + 1).filter((e) => e.event === "tip-verify" && e.data.pass === true
|
|
205
|
+
&& e.data.forgiven === true && typeof e.data.gate === "string");
|
|
206
|
+
return [...tasks, ...tip].map((e) => {
|
|
207
|
+
const fingerprints = fingerprintsOf(e.event === "tip-verify" ? e.data.fingerprints : e.data.forgivenFingerprints);
|
|
208
|
+
const baseline = baselineProvenanceOf(e.data.baselineProvenance);
|
|
209
|
+
return { ...(e.taskId ? { taskId: e.taskId } : {}), gate: e.data.gate,
|
|
210
|
+
...(fingerprints ? { fingerprints } : {}), ...(baseline ? { baseline } : {}) };
|
|
211
|
+
});
|
|
212
|
+
}
|
|
213
|
+
/** The capture a forgiveness rests on, or an explicit unknown — never a time read off another clock. */
|
|
214
|
+
export function formatBaselineProvenance(p) {
|
|
215
|
+
return p ? `baseline ${p.baseRef.slice(0, 12)} captured ${p.capturedAt}`
|
|
216
|
+
: "baseline provenance unknown (legacy: no capture identity or time recorded)";
|
|
217
|
+
}
|
|
218
|
+
export function formatFingerprints(fingerprints) {
|
|
219
|
+
return fingerprints === undefined ? "fingerprints not recorded"
|
|
220
|
+
: fingerprints.length ? fingerprints.join(" | ") : "no fingerprint recognized";
|
|
221
|
+
}
|
|
222
|
+
export function formatForgiven(f) {
|
|
223
|
+
return `${f.taskId ?? "tip"} ${f.gate}: ${formatFingerprints(f.fingerprints)} — ${formatBaselineProvenance(f.baseline)}`;
|
|
224
|
+
}
|
|
225
|
+
// OBS-1150: the reviewer reads the same standing rulings the worker brief carries, oldest first. A
|
|
226
|
+
// later approval, with or without a reason, adds to them and never supersedes an earlier one.
|
|
227
|
+
function approvalReviewContext(events, taskId) {
|
|
228
|
+
const rulings = standingRulings(events, taskId);
|
|
229
|
+
return rulings.length ? rulings.map((ruling) => `- ${ruling}`).join("\n") : undefined;
|
|
190
230
|
}
|
|
191
231
|
// T14: the events that prove an approval was ENACTED — narrowly causal, never merely subsequent.
|
|
192
232
|
// Ordinary, attempt-cap and review-upheld approvals buy a WORKER, so their proof is a dispatch;
|
|
@@ -225,7 +265,10 @@ export function pendingDaemonApprovalActions(events) {
|
|
|
225
265
|
*/
|
|
226
266
|
export function outstandingApprovals(events) {
|
|
227
267
|
const newest = new Map();
|
|
228
|
-
|
|
268
|
+
// OBS-1178: this reports ANSWERS, not effects — a refused row was answered; an unsound one the daemon
|
|
269
|
+
// has not yet refused (approved, then failed before any dispatch) is still an unanswered decision.
|
|
270
|
+
const { refused } = foldDecisions(events);
|
|
271
|
+
events.forEach((e, i) => { if (e.event === "task-approved" && e.taskId && !refused.has(physicalLine(events, i)))
|
|
229
272
|
newest.set(e.taskId, i); });
|
|
230
273
|
return [...newest]
|
|
231
274
|
.filter(([taskId, i]) => {
|
|
@@ -262,7 +305,9 @@ export function formatSummary(s) {
|
|
|
262
305
|
+ (resumable.length ? `; \`tickmarkr resume ${s.runId}\` enacts ${resumable.join(", ")}` : "")
|
|
263
306
|
+ (stalled.length ? `; ${stalled.join(", ")} failed before any dispatch — neither resume nor \`--retry-failed\` re-dispatches that` : "")
|
|
264
307
|
: "";
|
|
265
|
-
|
|
308
|
+
// OBS-1123: a green that carried baseline reds names each one and the capture that recorded it.
|
|
309
|
+
const forgiven = (s.forgiven ?? []).map((f) => `\nforgiven vs baseline — ${formatForgiven(f)}`).join("");
|
|
310
|
+
return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}${outstanding}${forgiven}`;
|
|
266
311
|
}
|
|
267
312
|
/** The narrator enters the production Run cockpit for this exact run. The
|
|
268
313
|
* completed static/growing four-hour records precede this default cutover;
|
|
@@ -353,6 +398,12 @@ function narrowRepairBattery(failing) {
|
|
|
353
398
|
return isOracleFailure(g) && g.meta?.unparseable !== true;
|
|
354
399
|
return false;
|
|
355
400
|
}
|
|
401
|
+
// OBS-1182: re-seating copies the channel's identity AND its launch effort — the four-field copy
|
|
402
|
+
// this replaced silently reverted a climbed or pinned seat to the CLI's default effort.
|
|
403
|
+
const seatAssignment = (c) => ({
|
|
404
|
+
adapter: c.adapter, model: c.model, channel: c.channel, tier: c.tier,
|
|
405
|
+
...(c.effort ? { effort: c.effort } : {}),
|
|
406
|
+
});
|
|
356
407
|
// v2.0 T2 (OBS-554): the measurement keys run-gates stamps on a GateResult, lifted verbatim onto the
|
|
357
408
|
// gate row. One list, one lift — both onGate sites record through the same helper.
|
|
358
409
|
const GATE_TELEMETRY_KEYS = ["durationMs", "load1Start", "load1End", "load1Max", "load1Mean", "selectedDurationMs", "fullDurationMs", "invocations"];
|
|
@@ -441,7 +492,8 @@ const DEFERRED_FINDINGS_HEADING = "## Deferred review findings — a reviewer AC
|
|
|
441
492
|
// OBS-419: the newest approval starts the current engagement, so it is also the sole authority for
|
|
442
493
|
// that engagement's optional ceiling. Stop at the newest approval even when the field is absent: a
|
|
443
494
|
// later ordinary release restores the module default instead of inheriting an older operator limit.
|
|
444
|
-
function approvedReviewRoundCeiling(
|
|
495
|
+
function approvedReviewRoundCeiling(journaled, taskId) {
|
|
496
|
+
const events = effectiveEvents(journaled); // OBS-1178: a refused or unsound approval sets no ceiling
|
|
445
497
|
for (let i = events.length - 1; i >= 0; i--) {
|
|
446
498
|
const event = events[i];
|
|
447
499
|
if (event.event !== "task-approved" || event.taskId !== taskId)
|
|
@@ -479,6 +531,14 @@ let capacityBackoffMs = CAPACITY_BACKOFF_MS;
|
|
|
479
531
|
/** Test seam — shrink the capacity backoff without minute-long sleeps. */
|
|
480
532
|
export function setCapacityBackoffMsForTests(ms) { capacityBackoffMs = ms; }
|
|
481
533
|
export function resetCapacityBackoffMsForTests() { capacityBackoffMs = CAPACITY_BACKOFF_MS; }
|
|
534
|
+
// OBS-1169: a CLI bootstrap failure buys ONE same-channel retry after a bounded wait — journal-counted
|
|
535
|
+
// per task and seat like capacity — and then the adapter is escalated away from, never retried forever.
|
|
536
|
+
const BOOTSTRAP_RETRY_CAP = 1;
|
|
537
|
+
const BOOTSTRAP_BACKOFF_MS = 5_000;
|
|
538
|
+
let bootstrapBackoffMs = BOOTSTRAP_BACKOFF_MS;
|
|
539
|
+
/** Test seam — shrink the bootstrap backoff. */
|
|
540
|
+
export function setBootstrapBackoffMsForTests(ms) { bootstrapBackoffMs = ms; }
|
|
541
|
+
export function resetBootstrapBackoffMsForTests() { bootstrapBackoffMs = BOOTSTRAP_BACKOFF_MS; }
|
|
482
542
|
const NO_TRAILER_DEMOTION_STREAK = 2; // OBS-57: consecutive no-trailer windows demote a channel for the rest of the run
|
|
483
543
|
// OBS-117 (v1.71 T6): a worker pane that never prints a byte by T+60s after dispatch is a dead
|
|
484
544
|
// channel — don't burn the full stall window waiting for a silent launch failure. Checked on the
|
|
@@ -544,9 +604,16 @@ export function resetWorkerStartupWindowMsForTests() {
|
|
|
544
604
|
/** One seat, one startup prefix. Once execution or its input box is seen, later panes are ineligible. */
|
|
545
605
|
export class StartupFailureDetector {
|
|
546
606
|
adapter;
|
|
607
|
+
pattern;
|
|
608
|
+
wholePrefix;
|
|
547
609
|
closed = false;
|
|
548
|
-
constructor(adapter
|
|
610
|
+
constructor(adapter, pattern = WORKER_STARTUP_FAILURE_RE, // OBS-1169: or known bootstrap text
|
|
611
|
+
// OBS-1169: bootstrap evidence must own the WHOLE bounded prefix — a match followed by a tool frame
|
|
612
|
+
// or an input box anywhere later in the sample is a worker that ran, not a CLI that died starting.
|
|
613
|
+
wholePrefix = false) {
|
|
549
614
|
this.adapter = adapter;
|
|
615
|
+
this.pattern = pattern;
|
|
616
|
+
this.wholePrefix = wholePrefix;
|
|
550
617
|
}
|
|
551
618
|
sample(output, launchedAt) {
|
|
552
619
|
if (this.closed || launchedAt === undefined || Date.now() - launchedAt > workerStartupWindowMs)
|
|
@@ -583,6 +650,7 @@ export class StartupFailureDetector {
|
|
|
583
650
|
}
|
|
584
651
|
}
|
|
585
652
|
let offset = 0;
|
|
653
|
+
let found;
|
|
586
654
|
for (const [index, row] of rows.entries()) {
|
|
587
655
|
const cleanRow = row.replace(/\u001b\[[0-?]*[ -/]*[@-~]/g, "");
|
|
588
656
|
// Structured tool frames and the terminal's tool headings close the prefix BEFORE their body.
|
|
@@ -591,16 +659,20 @@ export class StartupFailureDetector {
|
|
|
591
659
|
this.closed = true;
|
|
592
660
|
return;
|
|
593
661
|
}
|
|
594
|
-
if (!this.adapter.harnessBannerRows?.includes(cleanRow)) {
|
|
595
|
-
const match =
|
|
596
|
-
if (match)
|
|
597
|
-
|
|
662
|
+
if (!found && !this.adapter.harnessBannerRows?.includes(cleanRow)) {
|
|
663
|
+
const match = this.pattern.exec(row);
|
|
664
|
+
if (match) {
|
|
665
|
+
found = {
|
|
598
666
|
matchedBytes: match[0], offset: offset + Buffer.byteLength(row.slice(0, match.index)),
|
|
599
667
|
row, rowNumber: index + 1,
|
|
600
668
|
};
|
|
669
|
+
if (!this.wholePrefix)
|
|
670
|
+
return found;
|
|
671
|
+
}
|
|
601
672
|
}
|
|
602
673
|
offset += Buffer.byteLength(row) + 1;
|
|
603
674
|
}
|
|
675
|
+
return found;
|
|
604
676
|
}
|
|
605
677
|
}
|
|
606
678
|
// OBS-901/906: one daemon-owned writer for every execution surface. Drivers expose one retained
|
|
@@ -987,7 +1059,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
987
1059
|
};
|
|
988
1060
|
if (r.pass) {
|
|
989
1061
|
// Q121s: a forgiven pass journals its fingerprints — honest about what was carried, never a silent green.
|
|
990
|
-
journal.append("tip-verify", undefined, { ...evidence, gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.reused ? { cached: true } : {}), ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity: measuredCapacity });
|
|
1062
|
+
journal.append("tip-verify", undefined, { ...evidence, gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.reused ? { cached: true } : {}), ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints, ...(r.baselineProvenance ? { baselineProvenance: r.baselineProvenance } : {}) } : {}), tip, cmdHash, capacity: measuredCapacity });
|
|
991
1063
|
}
|
|
992
1064
|
else {
|
|
993
1065
|
journal.append("tip-verify-failed", undefined, {
|
|
@@ -1500,17 +1572,24 @@ async function taskWithMaterializedContext(repoRoot, journal, worktree, task, at
|
|
|
1500
1572
|
}
|
|
1501
1573
|
return { ...task, context };
|
|
1502
1574
|
}
|
|
1575
|
+
// OBS-1107: a refused pick is not lost work by that fact alone. A commit whose own change the
|
|
1576
|
+
// destination already holds (content-empty, or a patch represented there under another hash) is
|
|
1577
|
+
// accounted for and skipped; the carry stops only at a nonempty change the destination lacks.
|
|
1503
1578
|
async function cherryPickCommits(wt, commits) {
|
|
1504
1579
|
const carried = [];
|
|
1580
|
+
const accounted = [];
|
|
1505
1581
|
for (const hash of commits) {
|
|
1506
1582
|
const r = await shGit(`git cherry-pick --no-gpg-sign ${shq(hash)}`, wt);
|
|
1507
|
-
if (r.code
|
|
1508
|
-
|
|
1509
|
-
|
|
1583
|
+
if (r.code === 0) {
|
|
1584
|
+
carried.push(hash);
|
|
1585
|
+
continue;
|
|
1510
1586
|
}
|
|
1511
|
-
|
|
1587
|
+
await shGit("git cherry-pick --abort", wt);
|
|
1588
|
+
if (!(await changeRepresented(wt, hash)))
|
|
1589
|
+
break;
|
|
1590
|
+
accounted.push(hash);
|
|
1512
1591
|
}
|
|
1513
|
-
return carried;
|
|
1592
|
+
return { carried, accounted };
|
|
1514
1593
|
}
|
|
1515
1594
|
const PRESERVED_REF_PREFIX = "refs/tickmarkr/preserved/";
|
|
1516
1595
|
const preservedRefOf = (data) => [data.ref, data.preservedRef].find((v) => typeof v === "string" && v.startsWith(PRESERVED_REF_PREFIX));
|
|
@@ -1662,6 +1741,7 @@ export function recordFatalRunEnd(journal, runId, branch, err, graph, phase = "s
|
|
|
1662
1741
|
const original = err instanceof Error ? err.message : String(err);
|
|
1663
1742
|
// A fatal close never finished a cycle: fresh/reused green is withheld, a recorded red still stands.
|
|
1664
1743
|
let tipProof = { kind: "incomplete" };
|
|
1744
|
+
let forgiven = [];
|
|
1665
1745
|
try {
|
|
1666
1746
|
// Only the latest engagement can already own this terminal outcome.
|
|
1667
1747
|
const events = journal.read();
|
|
@@ -1675,6 +1755,8 @@ export function recordFatalRunEnd(journal, runId, branch, err, graph, phase = "s
|
|
|
1675
1755
|
}
|
|
1676
1756
|
const latest = runEndTipProof(events.slice(from));
|
|
1677
1757
|
tipProof = { ...latest, kind: latest.kind === "failed" ? "failed" : "incomplete" };
|
|
1758
|
+
// OBS-1123: the same fold the normal close records, so a crash never drops what a green carried.
|
|
1759
|
+
forgiven = forgivenFingerprints(events);
|
|
1678
1760
|
}
|
|
1679
1761
|
catch (readErr) {
|
|
1680
1762
|
console.error(`tickmarkr ${runId}: journal read failed while recording the fatal run-end (${readErr instanceof Error ? readErr.message : String(readErr)}) — original error: ${original}`);
|
|
@@ -1691,6 +1773,7 @@ export function recordFatalRunEnd(journal, runId, branch, err, graph, phase = "s
|
|
|
1691
1773
|
fatal: true,
|
|
1692
1774
|
error: original,
|
|
1693
1775
|
tipProof,
|
|
1776
|
+
...(forgiven.length ? { forgiven } : {}),
|
|
1694
1777
|
};
|
|
1695
1778
|
try {
|
|
1696
1779
|
journal.append("run-end", undefined, record);
|
|
@@ -2107,11 +2190,22 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2107
2190
|
// GATE-08 (v1.12): the humanGate guard consults this run's journaled approvals, not just the compiled
|
|
2108
2191
|
// flag. Startup approvals seed the first scheduling pass; live approvals are folded at task
|
|
2109
2192
|
// boundaries below so a sibling can release parked work without waiting for run-end + resume.
|
|
2193
|
+
//
|
|
2194
|
+
// OBS-1178: revalidate every open approval before anything can enact it. The decision fold already
|
|
2195
|
+
// gives an unsound one no effect; the approval-refused row records that answer and consumes it, so
|
|
2196
|
+
// every fold and every operator reads the task as still parked — a stale waive never satisfies a newer gate.
|
|
2197
|
+
const refuseStaleApprovals = () => {
|
|
2198
|
+
for (const [taskId, stale] of staleApprovals(journal.read())) {
|
|
2199
|
+
// `lines` names the voided rows, so a sound decision beside a stale one survives the refusal.
|
|
2200
|
+
journal.append(APPROVAL_REFUSED, taskId, { reason: `stale or unbound decision refused before enactment — ${stale.reason}`, lines: stale.lines });
|
|
2201
|
+
}
|
|
2202
|
+
};
|
|
2203
|
+
refuseStaleApprovals();
|
|
2110
2204
|
const approvalStartupEvents = journal.read();
|
|
2111
|
-
// Continuing human-gate permission survives enactment; malformed releases grant none.
|
|
2112
|
-
const validApproval = (e) => e.event === "task-approved" && e.taskId
|
|
2113
|
-
&&
|
|
2114
|
-
const approved = new Set(approvalStartupEvents.filter(validApproval).map((e) => e.taskId));
|
|
2205
|
+
// Continuing human-gate permission survives enactment; malformed, refused or unsound releases grant none.
|
|
2206
|
+
const validApproval = (e) => e.event === "task-approved" && e.taskId !== undefined
|
|
2207
|
+
&& approvalAction(e.taskId, e).authority !== "inert";
|
|
2208
|
+
const approved = new Set(effectiveEvents(approvalStartupEvents).filter(validApproval).map((e) => e.taskId));
|
|
2115
2209
|
const startupActions = pendingDaemonApprovalActions(approvalStartupEvents);
|
|
2116
2210
|
let approvalSweepCursor = approvalStartupEvents.length;
|
|
2117
2211
|
const commands = detectGateCommands(repoRoot, cfg);
|
|
@@ -2150,11 +2244,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2150
2244
|
console.error(`tickmarkr: narrator not opened: ${placed.error}`);
|
|
2151
2245
|
return undefined;
|
|
2152
2246
|
};
|
|
2153
|
-
// WB-1 (OBS-988): the daemon watches its own cockpit. A board is lost when
|
|
2154
|
-
// aged past the supervision stale bound
|
|
2155
|
-
//
|
|
2156
|
-
//
|
|
2157
|
-
//
|
|
2247
|
+
// WB-1 (OBS-988): the daemon watches its own cockpit. A board is lost when its owner pid is dead, the
|
|
2248
|
+
// recorded arm has no presence, or its beat aged past the supervision stale bound with no owner pid
|
|
2249
|
+
// to consult. OBS-1110: a stale beat under a LIVE MATCHING OWNER — its pid answers and the arm it
|
|
2250
|
+
// claimed is present — is a slow board, not a lost one: journaled once per episode as
|
|
2251
|
+
// `watch-board-slow` and never spending a reopen. The beat and presence checks wait for the UI to
|
|
2252
|
+
// have armed (armId recorded) — a freshly reopened board that has not armed yet is not a second
|
|
2253
|
+
// loss. Reopens are bounded per run; past the bound the loss is journaled `boardless` and the
|
|
2254
|
+
// narrator is never called again.
|
|
2158
2255
|
// Leg-2 T7 M2: one `watch-board-lost` row per LOSS. A loss is identified by the pane and pid the
|
|
2159
2256
|
// owner record names; a reopen that fails leaves that record in place, so the next poll sees the
|
|
2160
2257
|
// same loss, journals only its own reopen attempt, and the bound retires the run boardless from the
|
|
@@ -2163,31 +2260,79 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2163
2260
|
let boardReopens = journal.read().filter((e) => e.event === "watch-board-reopened" || e.event === "watch-board-reopen-failed").length;
|
|
2164
2261
|
let boardless = false;
|
|
2165
2262
|
let journaledLoss;
|
|
2166
|
-
|
|
2263
|
+
let journaledSlow;
|
|
2264
|
+
// Per claim: the next adoption offer, backed off (doubling from the poll cadence to a small cap) but
|
|
2265
|
+
// never exhausted; and whether a reservation is unresolved, so the loop must keep waking to watch it.
|
|
2266
|
+
const MAX_HELD_OFFER_GAP_MS = 4 * APPROVAL_POLL_MS;
|
|
2267
|
+
let heldOffer;
|
|
2268
|
+
let heldRetry = false;
|
|
2269
|
+
const boardHealth = () => {
|
|
2167
2270
|
const owner = readWatchBoard(repoRoot, runId);
|
|
2168
2271
|
if (!owner)
|
|
2169
2272
|
return undefined;
|
|
2170
|
-
const
|
|
2171
|
-
|
|
2172
|
-
|
|
2173
|
-
if (beat.state === "STALE")
|
|
2174
|
-
return { ...lost, beatAgeMs: beat.beatAgeMs };
|
|
2175
|
-
// ponytail: presence path math mirrors supervision.ts's private helper (`<tier>.live.<armId>`)
|
|
2176
|
-
if (!existsSync(join(dirname(supervisionBeatPath(repoRoot, "watch")), `watch.live.${owner.armId}`)))
|
|
2177
|
-
return lost;
|
|
2178
|
-
}
|
|
2273
|
+
const named = { pane: owner.pane, ...(owner.pid !== undefined ? { pid: owner.pid } : {}) };
|
|
2274
|
+
const beat = owner.armId !== undefined ? readTierLiveness(repoRoot, "watch") : undefined;
|
|
2275
|
+
const row = beat?.state === "STALE" ? { ...named, beatAgeMs: beat.beatAgeMs } : named;
|
|
2179
2276
|
if (owner.pid !== undefined && !isPidLive(owner.pid))
|
|
2180
|
-
return lost;
|
|
2277
|
+
return { lost: row };
|
|
2278
|
+
if (owner.armId !== undefined && !existsSync(supervisionPresencePath(repoRoot, "watch", owner.armId)))
|
|
2279
|
+
return { lost: row };
|
|
2280
|
+
if (beat?.state === "STALE")
|
|
2281
|
+
return owner.pid !== undefined ? { slow: row } : { lost: row };
|
|
2181
2282
|
return undefined;
|
|
2182
2283
|
};
|
|
2284
|
+
// OBS-1172: a placement the driver HELD after an indeterminate split receipt has no slot here. While
|
|
2285
|
+
// the reservation stays unresolved the loop keeps polling it — even with every worker slot busy — so
|
|
2286
|
+
// a claim landing mid-task is seen. Once a live observer has claimed, the narrator is offered it
|
|
2287
|
+
// again, backed off per claim (doubling from the poll cadence to MAX_HELD_OFFER_GAP_MS) but never
|
|
2288
|
+
// exhausted: a listing that is unavailable or not yet showing the handle is retried until it proves
|
|
2289
|
+
// the board or the run ends; only success latches (the slot). The driver adopts the proven pane —
|
|
2290
|
+
// tracked from then on like any board — or keeps holding it, and never splits a second board over it.
|
|
2291
|
+
// This spends no reopen: nothing was lost.
|
|
2292
|
+
const adoptHeldBoard = async () => {
|
|
2293
|
+
heldRetry = false;
|
|
2294
|
+
const owner = readWatchBoard(repoRoot, runId);
|
|
2295
|
+
if (watchSlot || owner?.pane !== "" || owner.retired)
|
|
2296
|
+
return;
|
|
2297
|
+
heldRetry = true;
|
|
2298
|
+
if (owner.pid === undefined || !isPidLive(owner.pid))
|
|
2299
|
+
return; // unclaimed, or claimant dead: keep watching
|
|
2300
|
+
const id = `${owner.token}:${owner.pid}`;
|
|
2301
|
+
const now = Date.now();
|
|
2302
|
+
const offer = heldOffer?.id === id ? heldOffer : { id, gapMs: APPROVAL_POLL_MS, nextAt: now };
|
|
2303
|
+
if (now < offer.nextAt) {
|
|
2304
|
+
heldOffer = offer;
|
|
2305
|
+
return;
|
|
2306
|
+
}
|
|
2307
|
+
const adopted = await openBoard();
|
|
2308
|
+
if (adopted.ok) {
|
|
2309
|
+
heldOffer = undefined;
|
|
2310
|
+
heldRetry = false;
|
|
2311
|
+
journal.append("watch-board-adopted", undefined, { pane: adopted.slot.id });
|
|
2312
|
+
return;
|
|
2313
|
+
}
|
|
2314
|
+
heldOffer = { id, gapMs: Math.min(offer.gapMs * 2, MAX_HELD_OFFER_GAP_MS), nextAt: Date.now() + offer.gapMs };
|
|
2315
|
+
};
|
|
2183
2316
|
const watchBoard = async () => {
|
|
2184
|
-
if (
|
|
2317
|
+
if (boardless || !trackedDriver.narrator)
|
|
2185
2318
|
return;
|
|
2186
|
-
|
|
2187
|
-
|
|
2188
|
-
|
|
2319
|
+
if (!boardOpened)
|
|
2320
|
+
return adoptHeldBoard();
|
|
2321
|
+
const health = boardHealth();
|
|
2322
|
+
if (health && "slow" in health) {
|
|
2323
|
+
const identity = `${health.slow.pane}:${health.slow.pid}`;
|
|
2324
|
+
if (identity !== journaledSlow) {
|
|
2325
|
+
journaledSlow = identity;
|
|
2326
|
+
journal.append("watch-board-slow", undefined, health.slow);
|
|
2327
|
+
}
|
|
2189
2328
|
return;
|
|
2190
2329
|
}
|
|
2330
|
+
journaledSlow = undefined;
|
|
2331
|
+
if (!health) {
|
|
2332
|
+
journaledLoss = undefined;
|
|
2333
|
+
return adoptHeldBoard();
|
|
2334
|
+
}
|
|
2335
|
+
const loss = health.lost;
|
|
2191
2336
|
const identity = `${loss.pane}:${loss.pid ?? ""}`;
|
|
2192
2337
|
if (identity !== journaledLoss) {
|
|
2193
2338
|
journaledLoss = identity;
|
|
@@ -2435,7 +2580,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2435
2580
|
// journal.replayStatuses predates typed releases and re-pends even inert rows. Keep that
|
|
2436
2581
|
// legacy reader intact; scheduling accepts only the fold's recognised authorities.
|
|
2437
2582
|
const statuses = new Map();
|
|
2438
|
-
for (const e of replayEvents) {
|
|
2583
|
+
for (const e of effectiveEvents(replayEvents)) { // OBS-1178: a refused or unsound decision released nothing
|
|
2439
2584
|
if (!e.taskId)
|
|
2440
2585
|
continue;
|
|
2441
2586
|
if (e.event === "task-dispatch")
|
|
@@ -2499,7 +2644,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2499
2644
|
if (emptyCapture)
|
|
2500
2645
|
writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(emptyCapture, null, 2));
|
|
2501
2646
|
await initializeHost();
|
|
2502
|
-
|
|
2647
|
+
// OBS-1123: the capture's identity and publication time ride the measurement every forgiveness reads.
|
|
2648
|
+
const captured = { ...(emptyCapture ?? await captureBaseline(repoRoot, commands)),
|
|
2649
|
+
provenance: { baseRef, capturedAt: new Date().toISOString() } };
|
|
2503
2650
|
writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(captured, null, 2));
|
|
2504
2651
|
baseline = captured;
|
|
2505
2652
|
for (const warning of captured.warnings ?? [])
|
|
@@ -2563,16 +2710,29 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2563
2710
|
// owns listing/parsing/closing. Cosmetic by contract: failures are swallowed and subprocess has no
|
|
2564
2711
|
// reconcile (optional chain → no-op), so gates and the oracle suite never feel this. keepPanes
|
|
2565
2712
|
// "forever" is the keep-everything debug override — it disables the sweep entirely.
|
|
2566
|
-
const
|
|
2713
|
+
const resuming = !!opts.resume;
|
|
2714
|
+
const reconcile = async (sweep) => {
|
|
2567
2715
|
if (keepForever)
|
|
2568
2716
|
return;
|
|
2569
2717
|
try {
|
|
2570
|
-
const
|
|
2718
|
+
const rows = journal.read();
|
|
2719
|
+
const desired = desiredPanes(rows, runId);
|
|
2720
|
+
// OBS-1109: an interrupted attempt's pane is evidence — its nonce-bound trailer decides whether
|
|
2721
|
+
// the attempt is harvested, declined as foreign, or recovered — so the fold's run-resume clear
|
|
2722
|
+
// must not reach it before harvestInterruptedAttempt has read it and reaped its processes. Once
|
|
2723
|
+
// that harvest (or a superseding dispatch) is journaled the fold no longer names it: swept then.
|
|
2724
|
+
// Only a resumed daemon can hold such an attempt: a fresh run's sweep is the baseline sweep.
|
|
2725
|
+
if (resuming)
|
|
2726
|
+
for (const t of graph.tasks) {
|
|
2727
|
+
const owned = interruptedAttempt(rows, t.id);
|
|
2728
|
+
if (owned?.launch)
|
|
2729
|
+
desired.add(formatOwnedName({ role: "worker", taskId: t.id, attempt: owned.attempt, runId }));
|
|
2730
|
+
}
|
|
2571
2731
|
// The watch pane is never the DRIVER sweep's candidate (panesToClose spares role "watch":
|
|
2572
2732
|
// herdr's watches bookkeeping lives in close(), and a raw pane-close in the sweep would
|
|
2573
2733
|
// leave narrator() a stale cache) — the driver always sees it as desired; its lifecycle is
|
|
2574
2734
|
// decided here from the fold alone.
|
|
2575
|
-
await driver.reconcile?.(new Set([...desired, watchName]), runId, { ...
|
|
2735
|
+
await driver.reconcile?.(new Set([...desired, watchName]), runId, { ...sweep, endedRunIds });
|
|
2576
2736
|
// OBS-103: when the fold retires the watch (run-end boundary), close the narrator. The
|
|
2577
2737
|
// decision keys on the run identity in the pane name — narrator() adopts a prior daemon
|
|
2578
2738
|
// instance's pane under the same owned name, so a stop→resume cycle's leftover narrator
|
|
@@ -2662,6 +2822,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2662
2822
|
graph = setStatus(graph, t.id, "human");
|
|
2663
2823
|
saveGraph(repoRoot, graph);
|
|
2664
2824
|
journal.append("task-human", t.id, { ...details, reason, kind });
|
|
2825
|
+
// OBS-1178: the notice carries the park token an approval binds to (`approve --park <token>`).
|
|
2826
|
+
const parked = journal.newestBinding(t.id);
|
|
2827
|
+
const token = parked && bindingToken(parked);
|
|
2665
2828
|
if (assignment) {
|
|
2666
2829
|
// OBS-547: `metered` counts CHARGEABLE metered attempts, so an unchargeable dispatch passes 0 and
|
|
2667
2830
|
// the count is omitted rather than written as 0 or as `1` beside `attempts: 0` — a row claiming
|
|
@@ -2670,7 +2833,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2670
2833
|
journal.telemetry({ taskId: t.id, shape: t.shape, adapter: assignment.adapter, model: assignment.model, channel: assignment.channel, attempts, outcome: "human", durationMs: Date.now() - startMs, parkKind: kind, gateFails, consults, tokens, meteredAttempts: tokens && metered ? metered : undefined, retryMode });
|
|
2671
2834
|
}
|
|
2672
2835
|
await reconcile({ spareLiveLlm: true }); // task-human is a terminal event — sweep, sparing sibling tasks' live LLM panes
|
|
2673
|
-
await driver.notify(`tickmarkr ${runId}: ${t.id} needs a human — ${reason}`, { tier: "attention" });
|
|
2836
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} needs a human — ${reason}${token ? `\npark ${token} — bind the decision with \`--park ${token}\`` : ""}`, { tier: "attention" });
|
|
2674
2837
|
};
|
|
2675
2838
|
// OBS-547: cross-reference a scope red against the prediction this run already computed. Every hard
|
|
2676
2839
|
// offender predicted ⇒ an AUTHORING defect whose repair is pre-written: journal the classification
|
|
@@ -2749,7 +2912,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2749
2912
|
let admissionPriority = batteryPriority(startupActions.values());
|
|
2750
2913
|
const admissible = () => readyTasks(graph, admissionPriority);
|
|
2751
2914
|
const sweepLiveApprovals = () => {
|
|
2752
|
-
|
|
2915
|
+
let events = journal.read();
|
|
2916
|
+
if (events.slice(approvalSweepCursor).some((e) => e.event === "task-approved")) {
|
|
2917
|
+
refuseStaleApprovals();
|
|
2918
|
+
events = journal.read();
|
|
2919
|
+
}
|
|
2753
2920
|
admissionPriority = batteryPriority(pendingDaemonApprovalActions(events).values());
|
|
2754
2921
|
const approvals = events.slice(approvalSweepCursor)
|
|
2755
2922
|
.filter((e) => e.event === "task-approved" && e.taskId);
|
|
@@ -2861,6 +3028,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2861
3028
|
// fresh-budget release, prefer nextChannel over the surviving tried-list so burned channels are
|
|
2862
3029
|
// not re-tried first (consult bans / prior failovers survive the release).
|
|
2863
3030
|
const rs = resume.get(t.id);
|
|
3031
|
+
let bootstrapExhausted;
|
|
2864
3032
|
const contentDigest = taskContentDigest(t);
|
|
2865
3033
|
const previousDispatch = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "task-dispatch");
|
|
2866
3034
|
const recordedGraphPath = join(journal.dir, "graph.json");
|
|
@@ -2892,14 +3060,19 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2892
3060
|
// EXISTING nextChannel `tried` parameter — zero router changes (D-03).
|
|
2893
3061
|
// OBS-1034: a PIN is exempt — route() already returned it, and a review-upheld repair of the diff
|
|
2894
3062
|
// the pin produced belongs on the pin, not on the next seat its own tried[] entry would pick.
|
|
2895
|
-
|
|
3063
|
+
// OBS-1169: an adapter a bootstrap failover escalated away from stays out, sibling or none.
|
|
3064
|
+
const escalatedOut = channels.filter((c) => rs.escalatedAdapters?.includes(c.adapter)).map(channelKey);
|
|
3065
|
+
const next = nextChannel(assignment, t, cfg, channels, [...rs.tried, ...Object.keys(rs.escalated ?? {}), ...escalatedOut], profile, demotedChannels);
|
|
2896
3066
|
if (next)
|
|
2897
3067
|
assignment = next;
|
|
2898
3068
|
// ponytail: nextChannel null (every channel already tried / none available) — keep the static
|
|
2899
3069
|
// assignment and proceed. Dispatching on a previously-tried channel beats deadlocking a resumed
|
|
2900
3070
|
// run; a park-instead policy can come later if it ever bites.
|
|
3071
|
+
// OBS-1169: …except onto an escalated adapter — that relaunches the broken bootstrap, so it parks infra below.
|
|
3072
|
+
else if (rs.escalatedAdapters?.includes(assignment.adapter))
|
|
3073
|
+
bootstrapExhausted = channelKey(assignment);
|
|
2901
3074
|
}
|
|
2902
|
-
const taskHistory = journal.read().filter((e) => e.taskId === t.id);
|
|
3075
|
+
const taskHistory = effectiveEvents(journal.read()).filter((e) => e.taskId === t.id); // OBS-1178: only an effective decision opens an engagement
|
|
2903
3076
|
// Lifetime identity counts every dispatch, including legacy and unchargeable rows. Releases
|
|
2904
3077
|
// reset the budget below, never this counter; observational annotations do not consume it.
|
|
2905
3078
|
let nextWorkerDispatchOrdinal = taskHistory.filter((e) => e.event === "task-dispatch").length;
|
|
@@ -2908,13 +3081,38 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2908
3081
|
if (!hintsChanged && priorClimb && taskHistory.indexOf(priorClimb) > taskHistory.map((e) => e.event).lastIndexOf("task-dispatch")) {
|
|
2909
3082
|
const target = channels.find((c) => channelKey(c) === priorClimb.data.to && !demotedChannels.has(channelKey(c)));
|
|
2910
3083
|
if (target)
|
|
2911
|
-
assignment =
|
|
3084
|
+
assignment = seatAssignment(target);
|
|
2912
3085
|
}
|
|
2913
3086
|
// pre-kill invariant: tried always contains the current assignment. Spread, never alias the
|
|
2914
3087
|
// journal-derived array (no hidden mutation of replayed state).
|
|
2915
3088
|
const tried = rs?.tried.length ? [...rs.tried] : [channelKey(assignment)];
|
|
2916
3089
|
if (!tried.includes(channelKey(assignment)))
|
|
2917
3090
|
tried.push(channelKey(assignment));
|
|
3091
|
+
// OBS-1169 add.1: channels excluded because their whole adapter was escalated away from ride
|
|
3092
|
+
// `tried` (nextChannel's one exclusion input) but were never launched — keep the true reason so
|
|
3093
|
+
// the dispatch row never labels an unlaunched sibling "already tried".
|
|
3094
|
+
const vendorEscalated = new Map();
|
|
3095
|
+
// …and a resume restores them with their reasons, apart from the channels actually dispatched.
|
|
3096
|
+
for (const [k, why] of Object.entries(rs?.escalated ?? {})) {
|
|
3097
|
+
if (tried.includes(k))
|
|
3098
|
+
continue;
|
|
3099
|
+
tried.push(k);
|
|
3100
|
+
vendorEscalated.set(k, why);
|
|
3101
|
+
}
|
|
3102
|
+
// OBS-1169 add.2: the ADAPTER exclusion itself, apart from any sibling channel — a single-channel
|
|
3103
|
+
// adapter has no untried sibling to carry it, yet its exhausted channel must stay out of the
|
|
3104
|
+
// recycle pool (and a resume's) until the release that clears `tried`.
|
|
3105
|
+
const escalatedAdapters = new Set(rs?.escalatedAdapters ?? []);
|
|
3106
|
+
const escalateAdapter = (adapter, reason) => {
|
|
3107
|
+
escalatedAdapters.add(adapter);
|
|
3108
|
+
for (const c of channels) {
|
|
3109
|
+
const k = channelKey(c);
|
|
3110
|
+
if (c.adapter !== adapter || tried.includes(k))
|
|
3111
|
+
continue;
|
|
3112
|
+
tried.push(k);
|
|
3113
|
+
vendorEscalated.set(k, reason);
|
|
3114
|
+
}
|
|
3115
|
+
};
|
|
2918
3116
|
// VIS-02 convention: absence = no seeding happened. The observable surface for criterion 2's
|
|
2919
3117
|
// exclusion-list-equality oracle. Daemon-side append only — no journal.ts write-path change (Phase 48
|
|
2920
3118
|
// stays unblocked); inert to replayStatuses (unknown events ignored, pinned at journal.test.ts:70-80).
|
|
@@ -2923,6 +3121,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2923
3121
|
attempts: rs.attempts, tried: [...tried], assignment,
|
|
2924
3122
|
workerDispatchOrdinal: previousDispatch?.data.workerDispatchOrdinal ?? null,
|
|
2925
3123
|
});
|
|
3124
|
+
if (bootstrapExhausted === channelKey(assignment)) { // a later tier-climb restore may have moved it off
|
|
3125
|
+
await park(t, `${bootstrapExhausted} failed at bootstrap and no eligible channel remains after resume`, "infra", assignment, rs?.attempts ?? 0, startMs, 0, 0, undefined, 0, "fresh", { cause: "bootstrap", channel: bootstrapExhausted });
|
|
3126
|
+
return;
|
|
3127
|
+
}
|
|
2926
3128
|
// Keep one live list: recovery retries must see exclusions added by onGate during the round.
|
|
2927
3129
|
const badReviewers = [...replayedReviewerExclusions];
|
|
2928
3130
|
const noteReviewEvent = (e) => {
|
|
@@ -2977,7 +3179,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2977
3179
|
// reviewer-exclusion list (badReviewers), never a second parallel counter. OBS-189: scoped to the
|
|
2978
3180
|
// current engagement — an operator approval (uphold or accept) resets the round budget, so an upheld
|
|
2979
3181
|
// task can dispatch its funded attempt instead of re-parking against the whole journal's history.
|
|
2980
|
-
const reviewRoundsDrawn = () => reviewRoundsSinceApproval(
|
|
3182
|
+
const reviewRoundsDrawn = () => reviewRoundsSinceApproval(journal.read(), t.id, decisiveReviewRounds);
|
|
2981
3183
|
// OBS-193: journal the in-gate review retry (mirrors judge-retry) and exclude the flaked seat from
|
|
2982
3184
|
// later attempts' reviewer picks. One helper, called from both onGate sites (satisfied-gate + main).
|
|
2983
3185
|
const noteReviewRetry = (g) => {
|
|
@@ -3046,7 +3248,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3046
3248
|
...(typeof g.meta.producerAttempt === "number" ? { producerAttempt: g.meta.producerAttempt } : {}),
|
|
3047
3249
|
} : {}),
|
|
3048
3250
|
...(cfg.executionPolicy && !g.pass ? { disposition: failureDisposition(g) } : {}),
|
|
3049
|
-
...Object.fromEntries(["runnerInfraRerun", "hostStarvedRerun", "recoveryBlocked", "failingFiles", "selectionDecision", "failureEvidence"
|
|
3251
|
+
...Object.fromEntries(["runnerInfraRerun", "hostStarvedRerun", "recoveryBlocked", "failingFiles", "selectionDecision", "failureEvidence",
|
|
3252
|
+
"forgivenFingerprints", "freshFingerprints", "baselineProvenance"]
|
|
3050
3253
|
.filter((key) => g.meta?.[key] !== undefined).map((key) => [key, g.meta[key]])),
|
|
3051
3254
|
// OBS-540: preserve terminal-vs-retryable infra exactly. normalizeGateOutcome deliberately
|
|
3052
3255
|
// defaults a legacy infra row to retryable, so dropping an explicit false here reverses the
|
|
@@ -3169,7 +3372,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3169
3372
|
// OBS-254: RE-DERIVED from the journal here, at prompt-build time, rather than trusted to survive
|
|
3170
3373
|
// in resume state. The journal already holds the upheld review's bytes; no reset of attempt or
|
|
3171
3374
|
// channel state can take them away, on any path, including `resume --retry-failed`.
|
|
3172
|
-
const upheldFeedback = upheldFeedbackByTask(journal.read()).get(t.id) ?? rs?.upheldFeedback;
|
|
3375
|
+
const upheldFeedback = upheldFeedbackByTask(effectiveEvents(journal.read())).get(t.id) ?? rs?.upheldFeedback; // OBS-1178: a refused uphold funds no brief
|
|
3173
3376
|
const carriedEvidence = priorRunEvidence.findings
|
|
3174
3377
|
.filter((finding) => finding.taskId === t.id)
|
|
3175
3378
|
.map(formatPriorFindingEvidence)
|
|
@@ -3184,7 +3387,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3184
3387
|
: "");
|
|
3185
3388
|
let ladderIdx = 0;
|
|
3186
3389
|
const engagementRows = () => {
|
|
3187
|
-
const rows = journal.read().filter((e) => e.taskId === t.id);
|
|
3390
|
+
const rows = effectiveEvents(journal.read()).filter((e) => e.taskId === t.id);
|
|
3188
3391
|
const approval = rows.map((e) => e.event).lastIndexOf("task-approved");
|
|
3189
3392
|
return rows.slice(approval + 1);
|
|
3190
3393
|
};
|
|
@@ -3216,7 +3419,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3216
3419
|
if (c.tier === assignment.tier && channelKey(c) !== to)
|
|
3217
3420
|
climbSkips.add(channelKey(c));
|
|
3218
3421
|
climbProvenance = `tier-escalated ${from} → ${to} (${cause})`;
|
|
3219
|
-
assignment =
|
|
3422
|
+
assignment = seatAssignment(next);
|
|
3220
3423
|
tried.push(to);
|
|
3221
3424
|
return "climbed";
|
|
3222
3425
|
};
|
|
@@ -3250,12 +3453,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3250
3453
|
}
|
|
3251
3454
|
return next;
|
|
3252
3455
|
};
|
|
3253
|
-
// OBS-1161: capacity requeues spent on a seat for this task since
|
|
3254
|
-
// journal-derived so the budget survives a resume instead of restarting
|
|
3255
|
-
const
|
|
3256
|
-
const rows = journal.read().filter((e) => e.taskId === t.id);
|
|
3456
|
+
// OBS-1161: capacity requeues (OBS-1169: and bootstrap retries) spent on a seat for this task since
|
|
3457
|
+
// its last operator release — journal-derived so the budget survives a resume instead of restarting.
|
|
3458
|
+
const requeuesOn = (event, channel) => {
|
|
3459
|
+
const rows = effectiveEvents(journal.read()).filter((e) => e.taskId === t.id);
|
|
3257
3460
|
const since = rows.map((e) => e.event).lastIndexOf("task-approved");
|
|
3258
|
-
return rows.slice(since + 1).filter((e) => e.event ===
|
|
3461
|
+
return rows.slice(since + 1).filter((e) => e.event === event && e.data.channel === channel).length;
|
|
3259
3462
|
};
|
|
3260
3463
|
// OBS-202 (operator law: "you can spawn as many as you want"): channels are session FACTORIES,
|
|
3261
3464
|
// not consumed seats — a tried channel can always host a fresh worker session, and a fresh
|
|
@@ -3268,15 +3471,20 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3268
3471
|
const next = failover(site);
|
|
3269
3472
|
if (next)
|
|
3270
3473
|
return next;
|
|
3271
|
-
|
|
3474
|
+
// OBS-1169: a vendor escalated away from stays out of the recycle pool until its release —
|
|
3475
|
+
// recycling any of its channels, launched or not, would launch the broken bootstrap again.
|
|
3476
|
+
const recycleExcluded = [channelKey(assignment), ...channels.filter((c) => escalatedAdapters.has(c.adapter)).map(channelKey)];
|
|
3477
|
+
const recycled = nextChannel(assignment, t, cfg, channels, recycleExcluded, profile, demotedChannels)
|
|
3272
3478
|
?? (demotedChannels.has(channelKey(assignment)) ? null : assignment);
|
|
3273
3479
|
if (recycled)
|
|
3274
3480
|
journal.append("channel-recycle", t.id, { site, channel: channelKey(recycled) });
|
|
3275
3481
|
return recycled;
|
|
3276
3482
|
};
|
|
3277
|
-
const runConsult = (trigger, transcript, diffOrFeedback, gates) => {
|
|
3483
|
+
const runConsult = async (trigger, transcript, diffOrFeedback, gates) => {
|
|
3278
3484
|
consults++;
|
|
3279
|
-
|
|
3485
|
+
// OBS-1182: every launched consult seat, failed ones included, rides the verdict into the journal.
|
|
3486
|
+
const invocations = [];
|
|
3487
|
+
const v = await consult({
|
|
3280
3488
|
taskId: t.id, trigger,
|
|
3281
3489
|
journalTail: JSON.stringify(journal.read().slice(-20)),
|
|
3282
3490
|
transcript: transcript.slice(-8000),
|
|
@@ -3285,7 +3493,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3285
3493
|
// D-07: consult panes self-clean when the verdict is read (keepLlm) — only "forever" keeps them.
|
|
3286
3494
|
// v1.54 T1: channels = this run's doctor-filtered live list — consult.prefer seat liveness
|
|
3287
3495
|
// is judged against it, never rebuilt from config (installed-but-unauthed seats would stall).
|
|
3288
|
-
{ keep: keepLlm, onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined, runId, channels: pools.consult });
|
|
3496
|
+
{ keep: keepLlm, onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined, onInvocation: (inv) => invocations.push(inv), runId, channels: pools.consult });
|
|
3497
|
+
// A lone answering seat is already the row's adapter/model/vendor/effort; the list is journaled
|
|
3498
|
+
// only when a seat failed, so the row never repeats itself and a failed seat never vanishes.
|
|
3499
|
+
return invocations.some((inv) => inv.outcome === "failed") ? { ...v, invocations } : v;
|
|
3289
3500
|
};
|
|
3290
3501
|
// returns true → continue attempting, false → task is terminal (parked)
|
|
3291
3502
|
// trigger (why the consult ran) is threaded in so the decompose/human park keeps its cause —
|
|
@@ -3294,9 +3505,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3294
3505
|
journal.append("consult-verdict", t.id, {
|
|
3295
3506
|
action: v.action, notes: v.notes,
|
|
3296
3507
|
adapter: v.adapter ?? "unknown", model: v.model ?? "unknown", vendor: v.vendor ?? "unknown",
|
|
3508
|
+
...(v.effort ? { effort: v.effort } : {}),
|
|
3297
3509
|
...(v.reason ? { reason: v.reason } : {}),
|
|
3298
3510
|
...(v.guidance ? { guidance: v.guidance } : {}),
|
|
3299
3511
|
...(v.excludeAdapter ? { excludeAdapter: v.excludeAdapter } : {}),
|
|
3512
|
+
...(v.invocations?.length ? { invocations: v.invocations } : {}),
|
|
3300
3513
|
});
|
|
3301
3514
|
await driver.notify(`tickmarkr ${runId}: ${t.id} consult verdict: ${v.action}`, { tier: "attention" });
|
|
3302
3515
|
if (v.action === "retry") {
|
|
@@ -3316,15 +3529,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3316
3529
|
// nextChannel's existing tried parameter — zero router changes (D-03). Unknown adapter
|
|
3317
3530
|
// (zero matches) is a no-op expansion ⇒ ordinary channel-level reroute. Task-scoped:
|
|
3318
3531
|
// `tried` lives inside execTask, so a sibling task is unaffected.
|
|
3319
|
-
if (v.excludeAdapter)
|
|
3320
|
-
|
|
3321
|
-
if (c.adapter === v.excludeAdapter) {
|
|
3322
|
-
const k = channelKey(c);
|
|
3323
|
-
if (!tried.includes(k))
|
|
3324
|
-
tried.push(k);
|
|
3325
|
-
}
|
|
3326
|
-
}
|
|
3327
|
-
}
|
|
3532
|
+
if (v.excludeAdapter)
|
|
3533
|
+
escalateAdapter(v.excludeAdapter, `vendor escalated: consult excluded ${v.excludeAdapter}`);
|
|
3328
3534
|
const next = failoverOrRecycle("consult-reroute");
|
|
3329
3535
|
if (next) {
|
|
3330
3536
|
assignment = next;
|
|
@@ -3345,7 +3551,117 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3345
3551
|
let satisfiedGate = satisfiedGates.get(t.id);
|
|
3346
3552
|
const replayedGates = fundedRerun ? undefined : replayedGateResults.get(t.id);
|
|
3347
3553
|
const recheck = approvalAction?.authority === "battery";
|
|
3348
|
-
|
|
3554
|
+
// OBS-1109: finished work is harvested, never redone — on resume too. A daemon that died while its
|
|
3555
|
+
// worker ran leaves an owned attempt between worker-launch and worker-result: its pane may hold the
|
|
3556
|
+
// attempt's own nonce-bound trailer, its branch the committed work. Only after the attempt's owned
|
|
3557
|
+
// processes are reaped is either read as its result — a matching trailer as the worker's claim,
|
|
3558
|
+
// commits without one as the same no-trailer evidence the live harvest synthesizes — and the gate
|
|
3559
|
+
// replay below then judges it under the PRODUCING attempt's author, with no new dispatch. An
|
|
3560
|
+
// unfinished attempt, a pane holding another attempt's trailer, or cleanup that cannot be proven
|
|
3561
|
+
// is declined onto the ordinary recovery path. A harvest an earlier resume recorded is gated again,
|
|
3562
|
+
// never recorded twice.
|
|
3563
|
+
const harvestInterruptedAttempt = async () => {
|
|
3564
|
+
const found = interruptedAttempt(journal.read(), t.id);
|
|
3565
|
+
if (!found || (!found.launch && !found.result))
|
|
3566
|
+
return found?.assignment;
|
|
3567
|
+
const { attempt } = found;
|
|
3568
|
+
const wt = worktreePath(repoRoot, `${branch}--${t.id}`);
|
|
3569
|
+
const recordHarvest = (commits, finished, summary) => journal.append("worker-result-harvested", t.id, {
|
|
3570
|
+
attempt, commits, summary: finished ? summary : HARVESTED_RESULT_SUMMARY, source: RESUME_HARVEST_SOURCE, trailer: finished,
|
|
3571
|
+
});
|
|
3572
|
+
// A declined attempt leaves the fold's spare (interruptedAttempt closes on this row), so its pane
|
|
3573
|
+
// is swept here, before the ordinary recovery path dispatches beside it — the baseline sweep, late.
|
|
3574
|
+
const decline = async (reason) => {
|
|
3575
|
+
journal.append("resume-harvest-declined", t.id, { attempt, reason });
|
|
3576
|
+
await reconcile({ spareLiveLlm: true });
|
|
3577
|
+
return undefined;
|
|
3578
|
+
};
|
|
3579
|
+
// Owned-process cleanup before any result is read as final: the attempt's own process group and
|
|
3580
|
+
// dispatch marker, reaped through the same reapWorker a live harvest uses. Uncertain is declined.
|
|
3581
|
+
const reapOwned = async (key, launch) => {
|
|
3582
|
+
workerOwners.set(key, { taskId: t.id, attempt, groupFile: `${launch.dispatchScript}.${launch.nonce}.pgid`,
|
|
3583
|
+
marker: launch.dispatchScript, identities: new Map(), descendants: new Map() });
|
|
3584
|
+
try {
|
|
3585
|
+
await reapWorker(key);
|
|
3586
|
+
return undefined;
|
|
3587
|
+
}
|
|
3588
|
+
catch (error) {
|
|
3589
|
+
return `owned-process cleanup uncertain: ${error instanceof Error ? error.message : String(error)}`;
|
|
3590
|
+
}
|
|
3591
|
+
};
|
|
3592
|
+
if (found.result) {
|
|
3593
|
+
// A daemon recorded this attempt's result, then died before its harvest row (a live daemon
|
|
3594
|
+
// before its no-trailer synthesis, or a resume between its two rows): finish that record from
|
|
3595
|
+
// the result it wrote — the pane is never re-read, the result never rewritten. A resume wrote
|
|
3596
|
+
// its result only after reaping; a live daemon writes it BEFORE handling a failed reap, so its
|
|
3597
|
+
// owned processes are reaped again here. A result that neither finished nor left commits has
|
|
3598
|
+
// nothing to gate: the ordinary recovery path owns it.
|
|
3599
|
+
if (!found.result.reaped) {
|
|
3600
|
+
if (!found.launch)
|
|
3601
|
+
return decline("owned-process cleanup unproven: the launch recorded no ownership evidence");
|
|
3602
|
+
const uncertain = await reapOwned(found.launch.slot, found.launch);
|
|
3603
|
+
if (uncertain)
|
|
3604
|
+
return decline(uncertain);
|
|
3605
|
+
}
|
|
3606
|
+
const commits = existsSync(wt) ? await commitsAheadOf(await integrationHead(intWt), wt) : [];
|
|
3607
|
+
if (!found.result.finished && commits.length === 0)
|
|
3608
|
+
return decline("unfinished");
|
|
3609
|
+
recordHarvest(commits, found.result.finished, found.result.summary);
|
|
3610
|
+
return found.assignment;
|
|
3611
|
+
}
|
|
3612
|
+
const launch = found.launch;
|
|
3613
|
+
const { nonce, slot } = launch;
|
|
3614
|
+
if (!existsSync(wt))
|
|
3615
|
+
return decline("task worktree missing");
|
|
3616
|
+
const adapter = adapters.find((a) => a.id === found.assignment.adapter);
|
|
3617
|
+
if (!adapter)
|
|
3618
|
+
return decline(`adapter ${found.assignment.adapter} unavailable to read the trailer`);
|
|
3619
|
+
// The journaled slot is read through THIS daemon's driver, a fresh instance since the last one
|
|
3620
|
+
// died. A pane the driver cannot read is DECLINED, never classified: an unreadable pane proves
|
|
3621
|
+
// neither a matching trailer nor the absence of a foreign one, so it is neither harvested nor
|
|
3622
|
+
// gated as trailerless evidence. A driver whose terminals outlive it (Orca) adopts the owned
|
|
3623
|
+
// terminal: bound only on ownership evidence — the journaled owned title AND the task checkout —
|
|
3624
|
+
// never a title alone. The journaled slot.id is NEVER trusted across daemon instances: a fresh
|
|
3625
|
+
// driver numbers its slots from one again, so an interrupted task's old id can name whatever
|
|
3626
|
+
// terminal THIS instance bound under that id (another task's adoption or allocation).
|
|
3627
|
+
// A driver with no read-only adoption is declined BEFORE anything is bound: slot() allocates —
|
|
3628
|
+
// herdr's reclaims the same-named pane (closing the evidence unread) and holds a dispatch lease
|
|
3629
|
+
// only run() releases — so it is never an adoption fallback; that driver keeps ordinary recovery.
|
|
3630
|
+
const adopt = driver.adopt;
|
|
3631
|
+
if (!adopt)
|
|
3632
|
+
return decline("no read-only adoption on this driver");
|
|
3633
|
+
let ownedSlot;
|
|
3634
|
+
let pane;
|
|
3635
|
+
try {
|
|
3636
|
+
ownedSlot = await adopt.call(driver, slot);
|
|
3637
|
+
pane = await driver.read(ownedSlot, PANE_READ_ROWS);
|
|
3638
|
+
}
|
|
3639
|
+
catch (error) {
|
|
3640
|
+
return decline(`pane unreadable: ${error instanceof Error ? error.message : String(error)}`);
|
|
3641
|
+
}
|
|
3642
|
+
const parsed = adapter.parse(pane, nonce);
|
|
3643
|
+
const finished = parsed.cause === undefined;
|
|
3644
|
+
const foreign = !finished && [...pane.matchAll(/TICKMARKR_RESULT_([0-9a-z]+)/g)]
|
|
3645
|
+
.some(([, other]) => other !== nonce && adapter.parse(pane, other).cause === undefined);
|
|
3646
|
+
if (foreign)
|
|
3647
|
+
return decline("foreign-nonce");
|
|
3648
|
+
const taskBase = await integrationHead(intWt);
|
|
3649
|
+
if (!finished && (await commitsAheadOf(taskBase, wt)).length === 0)
|
|
3650
|
+
return decline("unfinished");
|
|
3651
|
+
const uncertain = await reapOwned(ownedSlot, launch);
|
|
3652
|
+
if (uncertain)
|
|
3653
|
+
return decline(uncertain);
|
|
3654
|
+
const commits = await commitsAheadOf(taskBase, wt); // re-measured: the reaped writer is gone now
|
|
3655
|
+
// The source tag makes a death between these two rows recoverable: the next resume finishes the record.
|
|
3656
|
+
journal.append("worker-result", t.id, {
|
|
3657
|
+
ok: parsed.ok, summary: parsed.summary, deviations: parsed.deviations, finished, exitCode: null, attempt,
|
|
3658
|
+
source: RESUME_HARVEST_SOURCE, ...(parsed.cause ? { cause: parsed.cause } : {}),
|
|
3659
|
+
});
|
|
3660
|
+
recordHarvest(commits, finished, parsed.summary);
|
|
3661
|
+
return found.assignment;
|
|
3662
|
+
};
|
|
3663
|
+
const harvestAuthor = rs && !satisfiedGate && !replayedGates && !recheck ? await harvestInterruptedAttempt() : undefined;
|
|
3664
|
+
resumeGateReplay: if (satisfiedGate || replayedGates || recheck || harvestAuthor) {
|
|
3349
3665
|
const taskBase = await integrationHead(intWt);
|
|
3350
3666
|
taskBases.set(t.id, taskBase);
|
|
3351
3667
|
const taskBranch = `${branch}--${t.id}`;
|
|
@@ -3354,7 +3670,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3354
3670
|
// battery gates — read from the pending approval row, never re-selected here, so the tree
|
|
3355
3671
|
// approve counted and the tree the daemon gates are one and the same, checkout present or not.
|
|
3356
3672
|
const preservedRef = recheck
|
|
3357
|
-
?
|
|
3673
|
+
? effectiveEvents(journal.read()).reverse().find((e) => e.taskId === t.id && e.event === "task-approved" && e.data.release === RECHECK_RELEASE)?.data.recheckedRef
|
|
3358
3674
|
: undefined;
|
|
3359
3675
|
if (!existsSync(priorWt) && !preservedRef) {
|
|
3360
3676
|
if (satisfiedGate || recheck)
|
|
@@ -3366,14 +3682,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3366
3682
|
}
|
|
3367
3683
|
const resumeReason = recheck
|
|
3368
3684
|
? "operator recheck"
|
|
3369
|
-
:
|
|
3370
|
-
?
|
|
3371
|
-
:
|
|
3685
|
+
: harvestAuthor
|
|
3686
|
+
? "harvested interrupted attempt"
|
|
3687
|
+
: satisfiedGate
|
|
3688
|
+
? `approved gate ${satisfiedGate}`
|
|
3689
|
+
: `recorded gates on ${replayedGates.commit.slice(0, 10)}`;
|
|
3372
3690
|
const priorTaskTip = preservedRef ? (await shGit(`git rev-parse ${shq(preservedRef)}`, repoRoot)).stdout.trim() : await gitHead(priorWt);
|
|
3373
3691
|
const priorTaskSubject = await gateCommitSubject(taskBase, priorTaskTip, preservedRef ? repoRoot : priorWt);
|
|
3374
3692
|
const commitsToCarry = preservedRef ? await commitsAheadOfRef(taskBase, priorTaskTip, repoRoot) : await commitsAheadOf(taskBase, priorWt);
|
|
3375
3693
|
const wt = await recreateTaskWorktree(taskBranch, taskBase, priorWt);
|
|
3376
|
-
const carriedCommits = await cherryPickCommits(wt, commitsToCarry);
|
|
3694
|
+
const { carried: carriedCommits, accounted: accountedCommits } = await cherryPickCommits(wt, commitsToCarry);
|
|
3377
3695
|
// Reuse is about the tree the gates will actually inspect. The integration tip may have moved
|
|
3378
3696
|
// while the daemon was down, so compare after recreating the task on today's taskBase rather
|
|
3379
3697
|
// than against the stale worktree whose task-only history cannot see newly merged dependencies.
|
|
@@ -3383,13 +3701,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3383
3701
|
if (recheck) {
|
|
3384
3702
|
satisfiedGate = journal.replaySatisfiedGates(new Map([[t.id, currentTaskSubject]])).get(t.id);
|
|
3385
3703
|
}
|
|
3386
|
-
journal.append("worktree-recreation", t.id, {
|
|
3704
|
+
journal.append("worktree-recreation", t.id, {
|
|
3705
|
+
attempted: commitsToCarry, carried: carriedCommits, ...(accountedCommits.length > 0 ? { accounted: accountedCommits } : {}),
|
|
3706
|
+
});
|
|
3387
3707
|
// OBS-212: same fail-closed rule as the dispatch path — but this path is worse, because it runs
|
|
3388
3708
|
// ONLY the gates after the approved one and then MERGES. T3 took it on run-20260728-110135:
|
|
3389
3709
|
// approved past review at 11:22, recreated at 12:50, and phase-start{gates} / phase-start{merge}
|
|
3390
3710
|
// landed in the same second with zero gate-result events. Work missing here is merged unverified.
|
|
3391
3711
|
{
|
|
3392
|
-
const present = new Set(carriedCommits);
|
|
3712
|
+
const present = new Set([...carriedCommits, ...accountedCommits]);
|
|
3393
3713
|
for (const h of commitsToCarry) {
|
|
3394
3714
|
if (!present.has(h) && (await shGit(`git merge-base --is-ancestor ${shq(h)} HEAD`, wt)).code === 0) {
|
|
3395
3715
|
present.add(h);
|
|
@@ -3429,7 +3749,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3429
3749
|
const parkedAuthor = recheck
|
|
3430
3750
|
? [...journal.read()].reverse().find((e) => e.event === "task-dispatch" && e.taskId === t.id)?.data.assignment
|
|
3431
3751
|
: undefined;
|
|
3432
|
-
|
|
3752
|
+
// OBS-1109: a harvested attempt's author is read from the journal, not from this resume's path —
|
|
3753
|
+
// a later resume that replays the harvest's recorded gates still gates it under its producer.
|
|
3754
|
+
const gateAuthor = parkedAuthor ?? resumeHarvestAuthor(journal.read(), t.id) ?? rs?.lastAssignment ?? assignment;
|
|
3433
3755
|
const satisfiedIndex = satisfiedGate ? GATE_NAMES.indexOf(satisfiedGate) : -1;
|
|
3434
3756
|
// The serial pipeline could have at most one blocking result, so "everything after the
|
|
3435
3757
|
// approved gate" was enough. v1.85 can record both verdict siblings red in one round, and a
|
|
@@ -3458,6 +3780,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3458
3780
|
if (recheck) {
|
|
3459
3781
|
remainingGates = declaredGates.filter((gate) => gate !== satisfiedGate);
|
|
3460
3782
|
}
|
|
3783
|
+
else if (harvestAuthor) {
|
|
3784
|
+
remainingGates = declaredGates; // the harvested attempt was never gated: every declared gate runs
|
|
3785
|
+
}
|
|
3461
3786
|
else if (satisfiedGate) {
|
|
3462
3787
|
remainingGates = t.gates.filter((gate) => {
|
|
3463
3788
|
if (gate === satisfiedGate)
|
|
@@ -3526,7 +3851,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3526
3851
|
const startedAt = Date.now();
|
|
3527
3852
|
const provisioned = await withCommandContext(t.id, () => sh(commands.build, wt));
|
|
3528
3853
|
provisionedRow = {
|
|
3529
|
-
gate: "build", commit: !satisfiedGate && !recheck ? replayedGates.commit : currentTaskSubject,
|
|
3854
|
+
gate: "build", commit: !satisfiedGate && !recheck && replayedGates ? replayedGates.commit : currentTaskSubject,
|
|
3530
3855
|
exitCode: provisioned.code, durationMs: Date.now() - startedAt,
|
|
3531
3856
|
};
|
|
3532
3857
|
if (provisioned.code !== 0 && satisfiedGate !== "build") {
|
|
@@ -3540,7 +3865,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3540
3865
|
if (provisionedRow !== undefined)
|
|
3541
3866
|
journal.append("gate-provisioned", t.id, provisionedRow);
|
|
3542
3867
|
const resumedTask = { ...t, gates: remainingGates };
|
|
3543
|
-
const operatorContext = approvalReviewContext(journal.read(), t.id
|
|
3868
|
+
const operatorContext = approvalReviewContext(journal.read(), t.id);
|
|
3544
3869
|
gateLoop: while (true) {
|
|
3545
3870
|
fatalStop.signal.throwIfAborted();
|
|
3546
3871
|
executionSignal()?.throwIfAborted();
|
|
@@ -3551,7 +3876,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3551
3876
|
// This suffix is re-measured to decide whether resume may advance, but the interrupted
|
|
3552
3877
|
// attempt already paid for its red result. The next worker-backed round remains the next
|
|
3553
3878
|
// deterministic-fingerprint occurrence/review round for budget accounting.
|
|
3554
|
-
...(!satisfiedGate && !recheck ? { replayMeasurement: true } : {}),
|
|
3879
|
+
...(!satisfiedGate && !recheck && !harvestAuthor ? { replayMeasurement: true } : {}),
|
|
3555
3880
|
};
|
|
3556
3881
|
if (recheck && satisfiedGate === "review" && gateSubject.commit !== currentTaskSubject) {
|
|
3557
3882
|
satisfiedGate = undefined;
|
|
@@ -3667,7 +3992,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3667
3992
|
await park(t, `recheck red: pinned ${pin.via}:${pin.model} is unavailable to host the repair — refusing the ladder`, "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
3668
3993
|
return;
|
|
3669
3994
|
}
|
|
3670
|
-
assignment =
|
|
3995
|
+
assignment = seatAssignment(seat);
|
|
3671
3996
|
}
|
|
3672
3997
|
// Carry completeness was verified before this replay battery. Apply the ordinary
|
|
3673
3998
|
// repair bounds here too; a restored red must fund the same findings-bearing dispatch.
|
|
@@ -3886,7 +4211,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3886
4211
|
feedback = feedback ? `${feedback}\n\n${consultBrief}` : consultBrief;
|
|
3887
4212
|
}
|
|
3888
4213
|
}
|
|
3889
|
-
const scopeApproval =
|
|
4214
|
+
const scopeApproval = effectiveEvents(journaledSoFar).reverse().find((e) => e.taskId === t.id && (e.event === "task-approved" || e.event === "task-dispatch"));
|
|
3890
4215
|
retryMode = scopeApproval?.data.release === "scope-request" ? "fresh" : repairFindings
|
|
3891
4216
|
? "repair"
|
|
3892
4217
|
: priorSession
|
|
@@ -3915,6 +4240,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3915
4240
|
// spends two frontier attempts re-deriving a known defect has to be able to answer afterwards.
|
|
3916
4241
|
fatalStop.signal.throwIfAborted();
|
|
3917
4242
|
const workerDispatchOrdinal = nextWorkerDispatchOrdinal++;
|
|
4243
|
+
vendorEscalated.delete(channelKey(assignment)); // dispatched now: from here on it IS tried
|
|
3918
4244
|
journal.append("task-dispatch", t.id, {
|
|
3919
4245
|
...(scopeApproval?.data.release === "scope-request" ? { files: t.files, graphDefinitionHash: graphDefinitionHash(graph) } : {}),
|
|
3920
4246
|
assignment, attempt, workerDispatchOrdinal, provenance: dispatchProvenance([
|
|
@@ -3927,7 +4253,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3927
4253
|
.filter((key) => key !== channelKey(assignment)).sort(),
|
|
3928
4254
|
exclusionReasons: Object.fromEntries([...new Set([...demotedChannels, ...tried, ...climbSkips])]
|
|
3929
4255
|
.filter((key) => key !== channelKey(assignment))
|
|
3930
|
-
.map((key) => [key, demotedChannels.has(key) ? "demoted"
|
|
4256
|
+
.map((key) => [key, demotedChannels.has(key) ? "demoted"
|
|
4257
|
+
: vendorEscalated.get(key) ?? (tried.includes(key) ? "already tried" : "tier climb skipped")])),
|
|
3931
4258
|
...(outstandingFindings.length > 0 ? { carriedFindings: outstandingFindings } : {}),
|
|
3932
4259
|
...(carriedConsultGuidance ? { carriedConsultGuidance } : {}),
|
|
3933
4260
|
});
|
|
@@ -3947,12 +4274,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3947
4274
|
// OBS-58: quota-failover and every retry recreate the task worktree from the integration tip —
|
|
3948
4275
|
// cherry-pick prior attempts' landed commits forward so a failover dispatch cannot silently
|
|
3949
4276
|
// orphan work a consult already verified as landed.
|
|
3950
|
-
|
|
3951
|
-
|
|
3952
|
-
|
|
4277
|
+
const { carried: carriedCommits, accounted: accountedCommits } = commitsToCarry.length > 0
|
|
4278
|
+
? await cherryPickCommits(wt, commitsToCarry) : { carried: [], accounted: [] };
|
|
4279
|
+
if (recreating) {
|
|
4280
|
+
journal.append("worktree-recreation", t.id, {
|
|
4281
|
+
attempted: commitsToCarry, carried: carriedCommits, ...(accountedCommits.length > 0 ? { accounted: accountedCommits } : {}),
|
|
4282
|
+
});
|
|
3953
4283
|
}
|
|
3954
|
-
if (recreating)
|
|
3955
|
-
journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
|
|
3956
4284
|
// T2 review (material): harvest eligibility is "does this WORKTREE carry unverified work",
|
|
3957
4285
|
// measured against taskBase — the same base the fast-kill's delta probe and the gates
|
|
3958
4286
|
// themselves use. It was measured against this attempt's post-carry HEAD, which excluded
|
|
@@ -3963,7 +4291,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3963
4291
|
// quota, dead-channel and provider-death all classify the PRE-HARVEST outcome below and fire
|
|
3964
4292
|
// BEFORE the synthesis, so carried-only work reaches gates without bypassing any failover.
|
|
3965
4293
|
const priorNamed = [...new Set([...commitsToCarry, ...carriedCommits])];
|
|
3966
|
-
const presentCommits = new Set(carriedCommits);
|
|
4294
|
+
const presentCommits = new Set([...carriedCommits, ...accountedCommits]);
|
|
3967
4295
|
for (const h of commitsToCarry) {
|
|
3968
4296
|
if (!presentCommits.has(h) && (await shGit(`git merge-base --is-ancestor ${shq(h)} HEAD`, wt)).code === 0) {
|
|
3969
4297
|
presentCommits.add(h);
|
|
@@ -4079,9 +4407,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4079
4407
|
sessionId, attempt, workerDispatchOrdinal, baselineBytes: resumeBaseline?.bytes ?? null,
|
|
4080
4408
|
});
|
|
4081
4409
|
const icmd = retryMode === "resume"
|
|
4082
|
-
? adapter.resumeCommand(sessionId, promptFile, assignment.model)
|
|
4410
|
+
? adapter.resumeCommand(sessionId, promptFile, assignment.model, assignment.effort)
|
|
4083
4411
|
: cfg.visibility.worker === "interactive" && driver.interactive
|
|
4084
|
-
? adapter.interactiveCommand(promptFile, assignment.model)
|
|
4412
|
+
? adapter.interactiveCommand(promptFile, assignment.model, assignment.effort)
|
|
4085
4413
|
: null;
|
|
4086
4414
|
// v1.69 T6: adapters that declare interactiveSeed launch the real TUI and inject the prompt as a
|
|
4087
4415
|
// user turn; they do NOT need the argv-seeding surface that interactiveCommand represents.
|
|
@@ -4142,6 +4470,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4142
4470
|
journal.append("worker-launch", t.id, {
|
|
4143
4471
|
attempt,
|
|
4144
4472
|
retryMode,
|
|
4473
|
+
// OBS-1109: the attempt's own ownership evidence, so a resumed daemon can harvest it.
|
|
4474
|
+
nonce, dispatchScript,
|
|
4145
4475
|
...(retryMode === "resume" ? { sessionId, workerDispatchOrdinal } : {}),
|
|
4146
4476
|
driver: trackedDriver.id,
|
|
4147
4477
|
slot: { ...slot },
|
|
@@ -4281,13 +4611,46 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4281
4611
|
let capacityBannerKilled = false; // OBS-1161: same banner filter and gates, transient outcome
|
|
4282
4612
|
let driverProbeFailed = false;
|
|
4283
4613
|
let heldLegs = [];
|
|
4614
|
+
// OBS-1108: contact uncertainty is LATCHED per owned slot and dispatch (this attempt's slot): one
|
|
4615
|
+
// row when contact is lost and one when it returns, never one per retry. Retries back off while
|
|
4616
|
+
// latched, and a latch held past its deadline throws into the transport-uncertain park, which
|
|
4617
|
+
// preserves the worktree for a recheck. A read that recovers inside the deadline charges nothing.
|
|
4618
|
+
let contactLost;
|
|
4619
|
+
const contactDeadlineMs = contactUnreadableDeadlineMs ?? taskTimeoutMinutes * 60_000;
|
|
4284
4620
|
const noteDriverUnreadable = (error) => {
|
|
4285
4621
|
if (error instanceof HeldProbeExhausted)
|
|
4286
4622
|
throw error;
|
|
4287
4623
|
driverProbeFailed = true;
|
|
4288
4624
|
heldLegs = ["quota:driver-unreadable", "stall:driver-unreadable"];
|
|
4289
|
-
|
|
4290
|
-
|
|
4625
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
4626
|
+
if (contactLost) {
|
|
4627
|
+
contactLost.retries++;
|
|
4628
|
+
const unreadableMs = Date.now() - contactLost.since;
|
|
4629
|
+
if (unreadableMs >= contactDeadlineMs) {
|
|
4630
|
+
throw new ContactUnreadableExhausted(`contact with ${slot.name} unreadable for ${unreadableMs}ms, past its ${contactDeadlineMs}ms deadline, after ${contactLost.retries} retries: ${message}`);
|
|
4631
|
+
}
|
|
4632
|
+
return;
|
|
4633
|
+
}
|
|
4634
|
+
contactLost = { since: Date.now(), retries: 0 };
|
|
4635
|
+
journal.append("contact-unreadable", t.id, { slot: slot.name, attempt, source: "driver", state: "unreadable",
|
|
4636
|
+
concludes: false, deadlineMs: contactDeadlineMs, error: message });
|
|
4637
|
+
};
|
|
4638
|
+
const noteContactReadable = () => {
|
|
4639
|
+
if (!contactLost)
|
|
4640
|
+
return;
|
|
4641
|
+
journal.append("contact-recovered", t.id, { slot: slot.name, attempt, source: "driver", state: "readable",
|
|
4642
|
+
unreadableMs: Date.now() - contactLost.since, retries: contactLost.retries });
|
|
4643
|
+
contactLost = undefined;
|
|
4644
|
+
};
|
|
4645
|
+
// The pause after a slice whose contact probe failed. Unlatched: the unspent slice, at most 1 s.
|
|
4646
|
+
// Latched: a next-contact-at schedule of its own — doubling per retry, bounded by the contact and
|
|
4647
|
+
// hard deadlines and NEVER by the poll slice, which clamps to 100 ms once the stall clock expires
|
|
4648
|
+
// and would otherwise collapse the backoff into ~10 orca calls a second for the rest of the window.
|
|
4649
|
+
const contactPauseMs = (sliceStart, slice, hardDeadline) => {
|
|
4650
|
+
const now = Date.now();
|
|
4651
|
+
if (!contactLost)
|
|
4652
|
+
return Math.max(0, Math.min(slice - (now - sliceStart), CONTACT_BACKOFF_BASE_MS));
|
|
4653
|
+
return Math.max(1, Math.min(CONTACT_BACKOFF_BASE_MS * 2 ** Math.min(contactLost.retries, CONTACT_BACKOFF_DOUBLINGS), contactLost.since + contactDeadlineMs - now, hardDeadline - now));
|
|
4291
4654
|
};
|
|
4292
4655
|
// T2 review: print mode's "the exit marker appeared". Kept apart from `finished` (the
|
|
4293
4656
|
// trailer) but still needed by the keepPanes decision below, whose contract is about a
|
|
@@ -4297,7 +4660,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4297
4660
|
let startupFailure = false;
|
|
4298
4661
|
let startupEvidence;
|
|
4299
4662
|
const startupDetector = new StartupFailureDetector(adapter);
|
|
4663
|
+
// OBS-1169: bootstrap text is evidence only inside the same startup prefix — every sample below
|
|
4664
|
+
// feeds it, so a tool frame, an input box or a saturated read seen at ANY poll closes it for good.
|
|
4665
|
+
const bootstrapDetector = new StartupFailureDetector(adapter, BOOTSTRAP_FAILURE_RE, true);
|
|
4300
4666
|
const startupFailureInWindow = (text, launchedAt) => {
|
|
4667
|
+
bootstrapDetector.sample(text, launchedAt);
|
|
4301
4668
|
startupEvidence ??= startupDetector.sample(text, launchedAt);
|
|
4302
4669
|
return startupEvidence !== undefined;
|
|
4303
4670
|
};
|
|
@@ -4307,8 +4674,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4307
4674
|
? new SettledTrailerTracker(adapter.busyFrameMarkers) : undefined;
|
|
4308
4675
|
const sampleTrailer = (frame) => {
|
|
4309
4676
|
const parsed = adapter.parse(frame, nonce);
|
|
4310
|
-
|
|
4311
|
-
|
|
4677
|
+
// OBS-1175: the parser's cause, never the summary — a parsed trailer may carry a sentinel string.
|
|
4678
|
+
const trailer = new RegExp(trailerPattern(nonce)).test(frame) && parsed.cause === undefined;
|
|
4312
4679
|
return trailerFrames ? trailerFrames.sample(frame, trailer) : trailer;
|
|
4313
4680
|
};
|
|
4314
4681
|
let seedResult;
|
|
@@ -4363,6 +4730,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4363
4730
|
await park(t, "escalation ladder exhausted", "ladder-exhausted", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
4364
4731
|
return false;
|
|
4365
4732
|
};
|
|
4733
|
+
// OBS-1169: the tree this worker receives — commits past it are this attempt's own work; the
|
|
4734
|
+
// carried ones beneath it are not, and must not make a bootstrap death look like work.
|
|
4735
|
+
const launchHead = await gitHead(wt);
|
|
4366
4736
|
try {
|
|
4367
4737
|
if (interactive) {
|
|
4368
4738
|
// v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
|
|
@@ -4410,7 +4780,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4410
4780
|
return;
|
|
4411
4781
|
}
|
|
4412
4782
|
await noteLaunched();
|
|
4413
|
-
|
|
4783
|
+
// OBS-1108: the first owned-slot read joins the contact latch — an unreadable launch read is
|
|
4784
|
+
// retried by the wait loop below under the same deadline, never a task failure on its own.
|
|
4785
|
+
output = await workerTransport.read(slot, PANE_READ_ROWS).catch((error) => { noteDriverUnreadable(error); return ""; });
|
|
4414
4786
|
}
|
|
4415
4787
|
// The returning paths report the same fact on the result; both callbacks land on the one
|
|
4416
4788
|
// latch, and the second is a no-op. A seed that answered is never re-answered by the loop.
|
|
@@ -4497,7 +4869,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4497
4869
|
await sampleContext(); // final poll-seam sample before leaving the wait
|
|
4498
4870
|
break;
|
|
4499
4871
|
}
|
|
4500
|
-
if (adapter.parse(output, nonce).
|
|
4872
|
+
if (adapter.parse(output, nonce).cause === "malformed-verdict")
|
|
4501
4873
|
break;
|
|
4502
4874
|
if (trailerFrames && new RegExp(trailerPattern(nonce)).test(output)) {
|
|
4503
4875
|
// A matching busy frame is still a live turn. Preserve the rolling progress budget.
|
|
@@ -4524,9 +4896,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4524
4896
|
});
|
|
4525
4897
|
}
|
|
4526
4898
|
await armCpuLeg(true);
|
|
4527
|
-
|
|
4528
|
-
if (spent < slice)
|
|
4529
|
-
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
4899
|
+
await new Promise((r) => setTimeout(r, contactPauseMs(sliceStart, slice, hardDeadline)));
|
|
4530
4900
|
continue;
|
|
4531
4901
|
}
|
|
4532
4902
|
// waitOutput is only a wake hint; a driver may miss a trailer already painted.
|
|
@@ -4539,14 +4909,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4539
4909
|
await sampleContext();
|
|
4540
4910
|
break;
|
|
4541
4911
|
}
|
|
4542
|
-
if (adapter.parse(paneText, nonce).
|
|
4912
|
+
if (adapter.parse(paneText, nonce).cause === "malformed-verdict") {
|
|
4543
4913
|
output = paneText;
|
|
4544
4914
|
break;
|
|
4545
4915
|
}
|
|
4546
4916
|
if (driverProbeFailed) {
|
|
4547
|
-
|
|
4548
|
-
if (spent < slice)
|
|
4549
|
-
await new Promise((resolve) => setTimeout(resolve, Math.min(slice - spent, 1_000)));
|
|
4917
|
+
await new Promise((resolve) => setTimeout(resolve, contactPauseMs(sliceStart, slice, hardDeadline)));
|
|
4550
4918
|
continue;
|
|
4551
4919
|
}
|
|
4552
4920
|
if (paneText.length > 0)
|
|
@@ -4654,11 +5022,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4654
5022
|
});
|
|
4655
5023
|
}
|
|
4656
5024
|
await armCpuLeg(true);
|
|
4657
|
-
|
|
4658
|
-
if (spent < slice)
|
|
4659
|
-
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
5025
|
+
await new Promise((r) => setTimeout(r, contactPauseMs(sliceStart, slice, hardDeadline)));
|
|
4660
5026
|
continue;
|
|
4661
5027
|
}
|
|
5028
|
+
// OBS-1108: every contact probe of this slice answered (a failed wait or read continues
|
|
5029
|
+
// above), so only here — never after one probe while another stays latched — has contact returned.
|
|
5030
|
+
noteContactReadable();
|
|
4662
5031
|
if (st !== lastStatus) {
|
|
4663
5032
|
lastStatus = st;
|
|
4664
5033
|
journal.append("worker-status", t.id, { slot: slot.name, status: st, attempt });
|
|
@@ -4977,7 +5346,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4977
5346
|
if (spent < slice)
|
|
4978
5347
|
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
4979
5348
|
}
|
|
4980
|
-
|
|
5349
|
+
// A wait that concludes on a clean slice's read (trailer, exit, startup failure) has contact back too.
|
|
5350
|
+
if (!driverProbeFailed)
|
|
5351
|
+
noteContactReadable();
|
|
5352
|
+
if (!finished && exitCode === null && adapter.parse(output, nonce).cause !== "malformed-verdict") {
|
|
5353
|
+
// OBS-1108: the hard deadline ended an attempt whose contact was still latched (its last
|
|
5354
|
+
// probe failed). That is contact loss, not a stall: keep the one infra park and the preserved
|
|
5355
|
+
// worktree instead of the hard-timeout classification and its repair charge.
|
|
5356
|
+
if (contactLost && driverProbeFailed && Date.now() >= hardDeadline) {
|
|
5357
|
+
throw new ContactUnreadableExhausted(`contact with ${slot.name} unreadable for ${Date.now() - contactLost.since}ms, still latched at the attempt's hard deadline after ${contactLost.retries} retries`);
|
|
5358
|
+
}
|
|
4981
5359
|
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
4982
5360
|
hardTimedOut = Date.now() >= hardDeadline;
|
|
4983
5361
|
timedOut ||= hardTimedOut;
|
|
@@ -5013,7 +5391,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5013
5391
|
const maxSettleRetries = 2;
|
|
5014
5392
|
let settleTries = 0;
|
|
5015
5393
|
settleParsed = adapter.parse(output, nonce);
|
|
5016
|
-
while (settleParsed.
|
|
5394
|
+
while (settleParsed.cause === "malformed-verdict" && settleTries < maxSettleRetries) {
|
|
5017
5395
|
const remaining = settleDeadline - Date.now();
|
|
5018
5396
|
if (remaining <= 0)
|
|
5019
5397
|
break;
|
|
@@ -5023,8 +5401,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5023
5401
|
settleParsed = adapter.parse(output, nonce);
|
|
5024
5402
|
settleTries++;
|
|
5025
5403
|
}
|
|
5026
|
-
if (settleParsed.
|
|
5027
|
-
finished = settleParsed.
|
|
5404
|
+
if (settleParsed.cause !== "malformed-verdict") {
|
|
5405
|
+
finished = settleParsed.cause === undefined && (!trailerFrames || trailerFrames.settled);
|
|
5028
5406
|
}
|
|
5029
5407
|
}
|
|
5030
5408
|
}
|
|
@@ -5117,6 +5495,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5117
5495
|
finally {
|
|
5118
5496
|
await cpuAccountant?.stop();
|
|
5119
5497
|
}
|
|
5498
|
+
// OBS-1169: the final pane read decides, at exit, whether the prefix is still the CLI's own startup.
|
|
5499
|
+
const bootstrapEvidence = processExited ? bootstrapDetector.sample(output, workerLaunchedAt) : undefined;
|
|
5120
5500
|
// SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
|
|
5121
5501
|
// shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
|
|
5122
5502
|
// exit wrapper). Print mode already waits for TICKMARKR_EXIT, which follows that tail; drain
|
|
@@ -5246,6 +5626,58 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5246
5626
|
stallReaps = 0;
|
|
5247
5627
|
stallSeat = undefined;
|
|
5248
5628
|
}
|
|
5629
|
+
// OBS-1169: a CLI that died in its own bootstrap never read the brief. Startup-owned evidence
|
|
5630
|
+
// (known bootstrap text in the startup prefix StartupFailureDetector owns: inside the window, before
|
|
5631
|
+
// any tool frame or input box, in an unsaturated read), and only when it exited nonzero with no
|
|
5632
|
+
// parsed trailer and no commit of its own: the carried commits
|
|
5633
|
+
// beneath launchHead are an earlier attempt's, and harvesting them would replay that attempt's
|
|
5634
|
+
// red as a second fingerprint occurrence. One same-channel retry after a bounded backoff charges
|
|
5635
|
+
// no attempt, repair or fingerprint; after that the whole adapter shares the broken bootstrap,
|
|
5636
|
+
// so its channels are escalated away from and the existing failover moves on or parks infra.
|
|
5637
|
+
const bootstrap = !workerFinished && exitCode !== null && exitCode !== 0 && bootstrapEvidence
|
|
5638
|
+
&& (await commitsAheadOf(launchHead, wt)).length === 0 ? bootstrapEvidence.matchedBytes : undefined;
|
|
5639
|
+
if (bootstrap) {
|
|
5640
|
+
const from = channelKey(assignment);
|
|
5641
|
+
const retries = requeuesOn("bootstrap-retry", from);
|
|
5642
|
+
if (retries < BOOTSTRAP_RETRY_CAP) {
|
|
5643
|
+
journal.append("bootstrap-retry", t.id, {
|
|
5644
|
+
attempt, retry: retries + 1, of: BOOTSTRAP_RETRY_CAP, channel: from, assignment,
|
|
5645
|
+
matched: bootstrap, exitCode, backoffMs: bootstrapBackoffMs,
|
|
5646
|
+
});
|
|
5647
|
+
await new Promise((r) => setTimeout(r, bootstrapBackoffMs));
|
|
5648
|
+
attempt--;
|
|
5649
|
+
continue;
|
|
5650
|
+
}
|
|
5651
|
+
const reason = `vendor escalated: ${assignment.adapter} bootstrap failed on ${from}`;
|
|
5652
|
+
escalateAdapter(assignment.adapter, reason);
|
|
5653
|
+
const next = failover("bootstrap-failover");
|
|
5654
|
+
// `toAssignment` and `reason` let a resume replay this routing disposition, not only its refund.
|
|
5655
|
+
journal.append("bootstrap-failover", t.id, {
|
|
5656
|
+
from, to: next ? channelKey(next) : null, ...(next ? { toAssignment: next } : {}), matched: bootstrap, retries,
|
|
5657
|
+
escalated: [...vendorEscalated].filter(([, why]) => why === reason).map(([k]) => k), reason,
|
|
5658
|
+
});
|
|
5659
|
+
if (next) {
|
|
5660
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} bootstrap failover`, { tier: "attention" });
|
|
5661
|
+
if (!keepForever) {
|
|
5662
|
+
const idx = keptSlots.indexOf(slot);
|
|
5663
|
+
if (idx >= 0) {
|
|
5664
|
+
keptSlots.splice(idx, 1);
|
|
5665
|
+
try {
|
|
5666
|
+
await closeSlot(slot);
|
|
5667
|
+
}
|
|
5668
|
+
catch { /* cosmetic — reconcile is the backstop */ }
|
|
5669
|
+
}
|
|
5670
|
+
if (supersededWorkerSlot === slot)
|
|
5671
|
+
supersededWorkerSlot = undefined;
|
|
5672
|
+
}
|
|
5673
|
+
assignment = next;
|
|
5674
|
+
tried.push(channelKey(next));
|
|
5675
|
+
attempt--; // the dead bootstrap bought no worker turn
|
|
5676
|
+
continue;
|
|
5677
|
+
}
|
|
5678
|
+
await park(t, `${from} failed at bootstrap after ${retries} same-channel retry and no eligible channel remains`, "infra", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode, { cause: "bootstrap", channel: from, retries });
|
|
5679
|
+
return;
|
|
5680
|
+
}
|
|
5249
5681
|
const preHarvestResult = result;
|
|
5250
5682
|
// T2 (OBS-264): recognize committed no-trailer work BEFORE any no-trailer streak, provider,
|
|
5251
5683
|
// quota or dead-channel routing. Gates never trusted the trailer, so this successful synthesis
|
|
@@ -5322,7 +5754,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5322
5754
|
// the journal, never a loop-local counter, so a resume continues the count it left off at.
|
|
5323
5755
|
if (capacityMatch) {
|
|
5324
5756
|
const from = channelKey(assignment);
|
|
5325
|
-
const requeues =
|
|
5757
|
+
const requeues = requeuesOn("capacity-requeue", from);
|
|
5326
5758
|
const source = capacityBannerKilled ? "banner" : "exit";
|
|
5327
5759
|
if (requeues < CAPACITY_REQUEUE_CAP) {
|
|
5328
5760
|
journal.append("capacity-requeue", t.id, {
|
|
@@ -5799,6 +6231,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5799
6231
|
journal.append("consult-verdict", t.id, {
|
|
5800
6232
|
action: v.action, notes: v.notes,
|
|
5801
6233
|
adapter: v.adapter ?? "unknown", model: v.model ?? "unknown", vendor: v.vendor ?? "unknown",
|
|
6234
|
+
...(v.effort ? { effort: v.effort } : {}),
|
|
6235
|
+
...(v.invocations?.length ? { invocations: v.invocations } : {}),
|
|
5802
6236
|
capAdvisory: true,
|
|
5803
6237
|
});
|
|
5804
6238
|
}
|
|
@@ -5968,7 +6402,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5968
6402
|
throw new Error(`could not preserve ${head}: ${saved.stderr}`);
|
|
5969
6403
|
}
|
|
5970
6404
|
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
5971
|
-
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "transport-uncertain", ref, ...cleanupEvidence });
|
|
6405
|
+
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: err instanceof ContactUnreadableExhausted ? "contact-unreadable" : "transport-uncertain", ref, ...cleanupEvidence });
|
|
5972
6406
|
return;
|
|
5973
6407
|
}
|
|
5974
6408
|
if (err instanceof ExecutionBudgetExceeded) {
|
|
@@ -6031,8 +6465,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
6031
6465
|
const waiters = [...inflight.values(), aborted];
|
|
6032
6466
|
// A free slot is itself a scheduling boundary: poll the append-only approval stream instead of
|
|
6033
6467
|
// sleeping until an unrelated long-running task settles.
|
|
6034
|
-
// An open board is polled on the same cadence, so a dead cockpit is noticed mid-task
|
|
6035
|
-
|
|
6468
|
+
// An open board is polled on the same cadence, so a dead cockpit is noticed mid-task; so is a
|
|
6469
|
+
// held reservation still awaiting its claim or its adoption.
|
|
6470
|
+
if (inflight.size < concurrency || boardOpened || heldRetry) {
|
|
6036
6471
|
waiters.push(new Promise((wake) => setTimeout(wake, APPROVAL_POLL_MS)));
|
|
6037
6472
|
}
|
|
6038
6473
|
await Promise.race(waiters); // aborted rejects on termination — unwinds the run
|
|
@@ -6139,6 +6574,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
6139
6574
|
summary.approvalDisposition = outstanding.length === 0 ? "complete" : "outstanding";
|
|
6140
6575
|
if (outstanding.length > 0)
|
|
6141
6576
|
summary.outstandingApprovals = outstanding;
|
|
6577
|
+
const forgiven = forgivenFingerprints(journal.read());
|
|
6578
|
+
if (forgiven.length > 0)
|
|
6579
|
+
summary.forgiven = forgiven;
|
|
6142
6580
|
journal.append("run-end", undefined, { ...summary });
|
|
6143
6581
|
await reconcile(); // run-end boundary: nothing in flight — full sweep (empty desired set)
|
|
6144
6582
|
// OBS-28: lingering worktrees starve CLI probes; keepPanes:forever is the debug override.
|