tickmarkr 2.6.0 → 2.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -3
- package/dist/adapters/catalog-remote.js +89 -47
- package/dist/adapters/claude-code.js +9 -6
- package/dist/adapters/codex.js +7 -4
- package/dist/adapters/prompt.d.ts +3 -1
- package/dist/adapters/prompt.js +21 -3
- package/dist/adapters/registry.js +3 -3
- package/dist/adapters/types.d.ts +13 -4
- package/dist/adapters/types.js +16 -0
- package/dist/cli/commands/approve.d.ts +5 -1
- package/dist/cli/commands/approve.js +66 -23
- package/dist/cli/commands/compile.js +13 -3
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +11 -3
- package/dist/cli/commands/fleet.js +71 -8
- package/dist/cli/commands/plan.js +7 -3
- package/dist/cli/commands/report.js +18 -2
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/status.js +37 -22
- package/dist/cli/help.d.ts +4 -0
- package/dist/cli/help.js +11 -2
- package/dist/config/config.d.ts +20 -0
- package/dist/config/config.js +47 -8
- package/dist/config/fleet-overlay.d.ts +1 -0
- package/dist/config/fleet-overlay.js +56 -0
- package/dist/drivers/orca.d.ts +26 -1
- package/dist/drivers/orca.js +199 -61
- package/dist/eval/canary.d.ts +2 -1
- package/dist/eval/canary.js +2 -2
- package/dist/eval/dispatch.js +1 -0
- package/dist/gates/acceptance.d.ts +2 -1
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/baseline.d.ts +12 -1
- package/dist/gates/baseline.js +11 -4
- package/dist/gates/cache.d.ts +3 -1
- package/dist/gates/cache.js +10 -3
- package/dist/gates/llm.d.ts +5 -4
- package/dist/gates/llm.js +17 -14
- package/dist/gates/review.d.ts +8 -0
- package/dist/gates/review.js +47 -15
- package/dist/gates/run-gates.d.ts +6 -1
- package/dist/gates/run-gates.js +34 -17
- package/dist/gates/test-manifest.d.ts +14 -0
- package/dist/gates/test-manifest.js +33 -6
- package/dist/graph/graph.d.ts +6 -2
- package/dist/graph/graph.js +15 -4
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/plan/scope.js +2 -2
- package/dist/route/preference.d.ts +20 -2
- package/dist/route/preference.js +48 -13
- package/dist/route/role-pick.d.ts +16 -0
- package/dist/route/role-pick.js +15 -0
- package/dist/route/router.d.ts +14 -0
- package/dist/route/router.js +39 -16
- package/dist/run/consult.d.ts +13 -1
- package/dist/run/consult.js +19 -14
- package/dist/run/daemon.d.ts +46 -2
- package/dist/run/daemon.js +964 -190
- package/dist/run/git.d.ts +44 -1
- package/dist/run/git.js +103 -5
- package/dist/run/host-health.d.ts +20 -0
- package/dist/run/host-health.js +64 -0
- package/dist/run/journal.d.ts +126 -3
- package/dist/run/journal.js +418 -36
- package/dist/run/merge.d.ts +3 -1
- package/dist/run/merge.js +3 -2
- package/dist/run/operator-state.d.ts +24 -2
- package/dist/run/operator-state.js +41 -5
- package/dist/run/operator-summary.d.ts +3 -0
- package/dist/run/operator-summary.js +3 -1
- package/dist/run/protocol.d.ts +31 -1
- package/dist/run/protocol.js +3 -1
- package/dist/run/stall.d.ts +38 -2
- package/dist/run/stall.js +276 -6
- package/dist/run/supervision.d.ts +7 -1
- package/dist/run/supervision.js +5 -2
- package/dist/tui/cockpit/board.d.ts +1 -1
- package/dist/tui/cockpit/board.js +30 -22
- package/dist/tui/cockpit/decision-actions.d.ts +8 -5
- package/dist/tui/cockpit/decision-actions.js +55 -32
- package/dist/tui/cockpit/derive.d.ts +2 -0
- package/dist/tui/cockpit/derive.js +17 -2
- package/dist/tui/cockpit/live-runtime.d.ts +10 -0
- package/dist/tui/cockpit/live-runtime.js +50 -3
- package/dist/tui/cockpit/live-store.d.ts +1 -0
- package/dist/tui/cockpit/live-store.js +31 -8
- package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
- package/dist/tui/cockpit/run-cockpit.js +28 -2
- package/dist/tui/cockpit/run-view.d.ts +11 -7
- package/dist/tui/cockpit/run-view.js +69 -15
- package/dist/tui/cockpit/setup-cockpit.d.ts +4 -0
- package/dist/tui/cockpit/setup-cockpit.js +6 -3
- package/dist/tui/ink/fleet-app.d.ts +15 -3
- package/dist/tui/ink/fleet-app.js +91 -22
- package/package.json +2 -1
- package/schema/config.schema.json +818 -0
- package/skills/tickmarkr-loop/SKILL.md +8 -2
- package/skills/tickmarkr-overseer/SKILL.md +85 -6
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +88 -0
- package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/dist/run/daemon.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { HOST_PROBE_SAMPLE_MS, HOST_PROBE_SAMPLES, HostDegradedError, hostDegraded, observeHost } from "./host-health.js";
|
|
2
|
+
import { VITEST_CACHE_ENV, worktreeVitestCache } from "../gates/test-manifest.js";
|
|
1
3
|
import { COMMAND_LEASE_TOKEN_ENV, commandLeaseEnvironment, CommandLeases, currentCommandLeaseToken, isRunnerCommand, runWithCommandLease, withCommandLease } from "./lease.js";
|
|
2
4
|
import { execFileSync, spawn } from "node:child_process";
|
|
3
5
|
import { createHash, randomBytes } from "node:crypto";
|
|
@@ -8,9 +10,9 @@ import { tmpdir } from "node:os";
|
|
|
8
10
|
import { basename, dirname, isAbsolute, join, posix, relative, resolve, sep } from "node:path";
|
|
9
11
|
import { fileURLToPath } from "node:url";
|
|
10
12
|
import { stringify } from "yaml";
|
|
11
|
-
import { classifyDeadChannel,
|
|
13
|
+
import { BOOTSTRAP_FAILURE_RE, classifyDeadChannel, classifyTransientCapacity, trailerPattern, writePrompt } from "../adapters/prompt.js";
|
|
12
14
|
import { allAdapters, getAdapter, probeAll, readDoctor, rolePools } from "../adapters/registry.js";
|
|
13
|
-
import { SettledTrailerTracker, addUsage, channelKey, matchesInputBox, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
15
|
+
import { SettledTrailerTracker, addUsage, CAPACITY_RE, channelKey, matchesInputBox, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
14
16
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
15
17
|
import { collateralHits } from "../compile/collateral.js";
|
|
16
18
|
import { ExecutionPolicySchema, DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
|
|
@@ -20,8 +22,9 @@ import { herdrSealShellPrefix, MAX_BUF, SubprocessDriver } from "../drivers/subp
|
|
|
20
22
|
import { formatOwnedName } from "../drivers/types.js";
|
|
21
23
|
import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
|
|
22
24
|
import { runGates } from "../gates/run-gates.js";
|
|
25
|
+
import { isInfraResult } from "../gates/cache.js";
|
|
23
26
|
import { filesGlob } from "../graph/files-glob.js";
|
|
24
|
-
import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
|
|
27
|
+
import { addEvidence, attributeBlocked, batteryPriority, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
|
|
25
28
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
26
29
|
import { distFingerprint } from "../cli/commands/version.js";
|
|
27
30
|
import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
@@ -29,16 +32,16 @@ import { executionSignal, remainingExecutionMs, withExecutionBudget, withoutExec
|
|
|
29
32
|
import { repairSelectionDecision } from "./repair-selection.js";
|
|
30
33
|
import { failureDisposition, reserveInfrastructureRetry } from "./recovery.js";
|
|
31
34
|
import { runEnvironment } from "./environment.js";
|
|
32
|
-
import { cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, runWithVerificationBudget, sameCapacity, sameVerification, sh, shGit, SUITE_PARENT_ENV, verificationProtocol, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
35
|
+
import { changeRepresented, cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, PRESERVE_COMMIT_SUBJECT, PRESERVE_PRODUCER_TRAILER, preserveWorktree, producerFields, resolvedCapacity, runWithForkBudget, runWithVerificationBudget, sameCapacity, sameVerification, sh, shGit, SUITE_PARENT_ENV, verificationProtocol, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
33
36
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
34
37
|
import { classifyRepairDisposition, resolveScopeHints } from "./repair-disposition.js";
|
|
35
|
-
import { applyScopeAmendments, activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingApprovalActions, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
38
|
+
import { applyScopeAmendments, activeRetryBan, interruptedAttempt, RESUME_HARVEST_SOURCE, resumeHarvestAuthor, approvalAction, APPROVAL_REFUSED, bindingToken, effectiveEvents, foldDecisions, physicalLine, staleApprovals, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingApprovalActions, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, standingRulings, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
36
39
|
import { gateReviewerFloor, isDiffCapPark, pickReviewer } from "../gates/review.js";
|
|
37
40
|
import { acquireApprovalSerialization, acquireRunLock, isPidLive, releaseRunLock } from "./lock.js";
|
|
38
41
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, reusedTipEvidence, verifyIntegrationTip } from "./merge.js";
|
|
39
42
|
import { climbChannel, marginalCostRank, nextChannel, route } from "../route/router.js";
|
|
40
43
|
import { desiredPanes } from "./reconcile.js";
|
|
41
|
-
import { readTierLiveness, readWatchBoard,
|
|
44
|
+
import { readTierLiveness, readWatchBoard, supervisionPresencePath } from "./supervision.js";
|
|
42
45
|
import { harvestCpuFlatWindowMs, NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows, WorkerTreeCpuAccountant, reapOwnedProcessGroup, readOwnedProcessGroup, } from "./stall.js";
|
|
43
46
|
// The live set is also the ownership claim shared by task cleanup and termination.
|
|
44
47
|
export async function closeLiveSlot(liveSlots, driver, slot) {
|
|
@@ -55,6 +58,19 @@ export async function closeLiveSlot(liveSlots, driver, slot) {
|
|
|
55
58
|
// A transport timeout says nothing about whether a mutation reached the terminal.
|
|
56
59
|
class HeldProbeExhausted extends Error {
|
|
57
60
|
}
|
|
61
|
+
// OBS-1108: an owned slot whose contact stayed latched unreadable past its deadline. Same terminal
|
|
62
|
+
// path as an exhausted transport probe — worktree preserved, one infra park (recheck-able) — with
|
|
63
|
+
// its own disposition, so the park names what was actually lost.
|
|
64
|
+
class ContactUnreadableExhausted extends HeldProbeExhausted {
|
|
65
|
+
}
|
|
66
|
+
// OBS-1108: how long one owned slot's contact may stay latched unreadable before its attempt parks.
|
|
67
|
+
// Default: the attempt's own stall window — the latch holds the stall kill down, so without this
|
|
68
|
+
// only the 4x hard deadline bounded a blind wait (2.5.8: 22k rows; 2.6.1: ~12k rows/h per worker).
|
|
69
|
+
let contactUnreadableDeadlineMs;
|
|
70
|
+
export function setContactUnreadableDeadlineMsForTests(ms) { contactUnreadableDeadlineMs = ms; }
|
|
71
|
+
export function resetContactUnreadableDeadlineMsForTests() { contactUnreadableDeadlineMs = undefined; }
|
|
72
|
+
const CONTACT_BACKOFF_BASE_MS = 1_000; // the pre-latch retry pause, doubled per latched retry
|
|
73
|
+
const CONTACT_BACKOFF_DOUBLINGS = 5; // ponytail: 32 s ceiling; the poll slice caps it lower anyway
|
|
58
74
|
function isTransportTimeout(error) {
|
|
59
75
|
if (!(error instanceof Error))
|
|
60
76
|
return false;
|
|
@@ -164,26 +180,53 @@ export function resolveRunMode(repoRoot, opts = {}) {
|
|
|
164
180
|
: undefined;
|
|
165
181
|
return { cfg: resolved.cfg, mode: resolved.mode, source, ...(conflict ? { conflict } : {}) };
|
|
166
182
|
}
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
183
|
+
/** Journal rows are untrusted input: a malformed field reads as not recorded, never as a value. */
|
|
184
|
+
export const fingerprintsOf = (v) => Array.isArray(v) && v.every((x) => typeof x === "string") ? v : undefined;
|
|
185
|
+
export const baselineProvenanceOf = (v) => {
|
|
186
|
+
const { baseRef, capturedAt } = (typeof v === "object" && v !== null ? v : {});
|
|
187
|
+
return typeof baseRef === "string" && typeof capturedAt === "string" ? { baseRef, capturedAt } : undefined;
|
|
188
|
+
};
|
|
189
|
+
/** A forgiven battery green: the structured field, or the battery's own "(forgiven)" wording on a legacy row. */
|
|
190
|
+
export const forgivenGateRow = (data) => data.pass === true && (Array.isArray(data.forgivenFingerprints) || /\(forgiven\)/u.test(String(data.details ?? "")));
|
|
191
|
+
/**
|
|
192
|
+
* OBS-1123: the forgiveness standing at this point in the journal — each task gate's LATEST verdict
|
|
193
|
+
* where it is a forgiven green, then the latest verification cycle's forgiven tip rows. A legacy row
|
|
194
|
+
* keeps what it never recorded absent, so the reader says "unknown" instead of borrowing a date.
|
|
195
|
+
*/
|
|
196
|
+
export function forgivenFingerprints(events) {
|
|
197
|
+
const latest = new Map();
|
|
198
|
+
for (const e of events) {
|
|
199
|
+
if (e.event === "gate-result" && e.taskId && typeof e.data.gate === "string")
|
|
200
|
+
latest.set(`${e.taskId}\0${e.data.gate}`, e);
|
|
185
201
|
}
|
|
186
|
-
|
|
202
|
+
const tasks = [...latest.values()].filter((e) => forgivenGateRow(e.data));
|
|
203
|
+
const start = events.map((e) => e.event).lastIndexOf("tip-verify-start");
|
|
204
|
+
const tip = events.slice(start + 1).filter((e) => e.event === "tip-verify" && e.data.pass === true
|
|
205
|
+
&& e.data.forgiven === true && typeof e.data.gate === "string");
|
|
206
|
+
return [...tasks, ...tip].map((e) => {
|
|
207
|
+
const fingerprints = fingerprintsOf(e.event === "tip-verify" ? e.data.fingerprints : e.data.forgivenFingerprints);
|
|
208
|
+
const baseline = baselineProvenanceOf(e.data.baselineProvenance);
|
|
209
|
+
return { ...(e.taskId ? { taskId: e.taskId } : {}), gate: e.data.gate,
|
|
210
|
+
...(fingerprints ? { fingerprints } : {}), ...(baseline ? { baseline } : {}) };
|
|
211
|
+
});
|
|
212
|
+
}
|
|
213
|
+
/** The capture a forgiveness rests on, or an explicit unknown — never a time read off another clock. */
|
|
214
|
+
export function formatBaselineProvenance(p) {
|
|
215
|
+
return p ? `baseline ${p.baseRef.slice(0, 12)} captured ${p.capturedAt}`
|
|
216
|
+
: "baseline provenance unknown (legacy: no capture identity or time recorded)";
|
|
217
|
+
}
|
|
218
|
+
export function formatFingerprints(fingerprints) {
|
|
219
|
+
return fingerprints === undefined ? "fingerprints not recorded"
|
|
220
|
+
: fingerprints.length ? fingerprints.join(" | ") : "no fingerprint recognized";
|
|
221
|
+
}
|
|
222
|
+
export function formatForgiven(f) {
|
|
223
|
+
return `${f.taskId ?? "tip"} ${f.gate}: ${formatFingerprints(f.fingerprints)} — ${formatBaselineProvenance(f.baseline)}`;
|
|
224
|
+
}
|
|
225
|
+
// OBS-1150: the reviewer reads the same standing rulings the worker brief carries, oldest first. A
|
|
226
|
+
// later approval, with or without a reason, adds to them and never supersedes an earlier one.
|
|
227
|
+
function approvalReviewContext(events, taskId) {
|
|
228
|
+
const rulings = standingRulings(events, taskId);
|
|
229
|
+
return rulings.length ? rulings.map((ruling) => `- ${ruling}`).join("\n") : undefined;
|
|
187
230
|
}
|
|
188
231
|
// T14: the events that prove an approval was ENACTED — narrowly causal, never merely subsequent.
|
|
189
232
|
// Ordinary, attempt-cap and review-upheld approvals buy a WORKER, so their proof is a dispatch;
|
|
@@ -199,7 +242,8 @@ const RECHECK_ENACTMENT = "recheck-battery";
|
|
|
199
242
|
const GATE_SATISFIED_ENACTMENT = "worktree-recreation";
|
|
200
243
|
// Older daemons enacted rechecks through worker-launch, before recheck-battery existed.
|
|
201
244
|
// Preserve that consumption at every scheduling read without changing continuing permission.
|
|
202
|
-
|
|
245
|
+
// OBS-1158: exported so plan folds battery priority through the SAME consumption-aware seam.
|
|
246
|
+
export function pendingDaemonApprovalActions(events) {
|
|
203
247
|
const actions = pendingApprovalActions(events);
|
|
204
248
|
const rechecks = pendingRechecks(events);
|
|
205
249
|
for (const [id, action] of actions) {
|
|
@@ -221,7 +265,10 @@ function pendingDaemonApprovalActions(events) {
|
|
|
221
265
|
*/
|
|
222
266
|
export function outstandingApprovals(events) {
|
|
223
267
|
const newest = new Map();
|
|
224
|
-
|
|
268
|
+
// OBS-1178: this reports ANSWERS, not effects — a refused row was answered; an unsound one the daemon
|
|
269
|
+
// has not yet refused (approved, then failed before any dispatch) is still an unanswered decision.
|
|
270
|
+
const { refused } = foldDecisions(events);
|
|
271
|
+
events.forEach((e, i) => { if (e.event === "task-approved" && e.taskId && !refused.has(physicalLine(events, i)))
|
|
225
272
|
newest.set(e.taskId, i); });
|
|
226
273
|
return [...newest]
|
|
227
274
|
.filter(([taskId, i]) => {
|
|
@@ -258,7 +305,9 @@ export function formatSummary(s) {
|
|
|
258
305
|
+ (resumable.length ? `; \`tickmarkr resume ${s.runId}\` enacts ${resumable.join(", ")}` : "")
|
|
259
306
|
+ (stalled.length ? `; ${stalled.join(", ")} failed before any dispatch — neither resume nor \`--retry-failed\` re-dispatches that` : "")
|
|
260
307
|
: "";
|
|
261
|
-
|
|
308
|
+
// OBS-1123: a green that carried baseline reds names each one and the capture that recorded it.
|
|
309
|
+
const forgiven = (s.forgiven ?? []).map((f) => `\nforgiven vs baseline — ${formatForgiven(f)}`).join("");
|
|
310
|
+
return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}${outstanding}${forgiven}`;
|
|
262
311
|
}
|
|
263
312
|
/** The narrator enters the production Run cockpit for this exact run. The
|
|
264
313
|
* completed static/growing four-hour records precede this default cutover;
|
|
@@ -310,6 +359,17 @@ function classifySignalOnlyTest(g) {
|
|
|
310
359
|
return;
|
|
311
360
|
g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
|
|
312
361
|
}
|
|
362
|
+
/** OBS-1106: ONE infrastructure predicate for classification, persistence and repair admission. A red
|
|
363
|
+
* that carries an infra fingerprint or classification without `meta.infra` (a legacy journal row, an
|
|
364
|
+
* oracle that named only its classification) is still a non-verdict about the work: it is normalized
|
|
365
|
+
* here BEFORE its journal row and before any park/repair seam reads it, so those seams can key on the
|
|
366
|
+
* same `isInfraResult` the gate cache keys on and never on one metadata field alone. */
|
|
367
|
+
function classifyInfraResult(g) {
|
|
368
|
+
classifySignalOnlyTest(g);
|
|
369
|
+
if (g.pass || g.meta?.infra === true || !isInfraResult(g))
|
|
370
|
+
return;
|
|
371
|
+
g.meta = { ...g.meta, classification: "infra", infra: true };
|
|
372
|
+
}
|
|
313
373
|
// v1.85 T3: the gates whose failure IS a deterministic measurement — a machine re-ran a command over a
|
|
314
374
|
// tree and printed the same bytes. Those are the failures the fingerprint cap governs (the ruling names
|
|
315
375
|
// it a "deterministic-gate" cap): a third identical answer to a question already answered twice is the
|
|
@@ -338,6 +398,12 @@ function narrowRepairBattery(failing) {
|
|
|
338
398
|
return isOracleFailure(g) && g.meta?.unparseable !== true;
|
|
339
399
|
return false;
|
|
340
400
|
}
|
|
401
|
+
// OBS-1182: re-seating copies the channel's identity AND its launch effort — the four-field copy
|
|
402
|
+
// this replaced silently reverted a climbed or pinned seat to the CLI's default effort.
|
|
403
|
+
const seatAssignment = (c) => ({
|
|
404
|
+
adapter: c.adapter, model: c.model, channel: c.channel, tier: c.tier,
|
|
405
|
+
...(c.effort ? { effort: c.effort } : {}),
|
|
406
|
+
});
|
|
341
407
|
// v2.0 T2 (OBS-554): the measurement keys run-gates stamps on a GateResult, lifted verbatim onto the
|
|
342
408
|
// gate row. One list, one lift — both onGate sites record through the same helper.
|
|
343
409
|
const GATE_TELEMETRY_KEYS = ["durationMs", "load1Start", "load1End", "load1Max", "load1Mean", "selectedDurationMs", "fullDurationMs", "invocations"];
|
|
@@ -426,7 +492,8 @@ const DEFERRED_FINDINGS_HEADING = "## Deferred review findings — a reviewer AC
|
|
|
426
492
|
// OBS-419: the newest approval starts the current engagement, so it is also the sole authority for
|
|
427
493
|
// that engagement's optional ceiling. Stop at the newest approval even when the field is absent: a
|
|
428
494
|
// later ordinary release restores the module default instead of inheriting an older operator limit.
|
|
429
|
-
function approvedReviewRoundCeiling(
|
|
495
|
+
function approvedReviewRoundCeiling(journaled, taskId) {
|
|
496
|
+
const events = effectiveEvents(journaled); // OBS-1178: a refused or unsound approval sets no ceiling
|
|
430
497
|
for (let i = events.length - 1; i >= 0; i--) {
|
|
431
498
|
const event = events[i];
|
|
432
499
|
if (event.event !== "task-approved" || event.taskId !== taskId)
|
|
@@ -455,6 +522,23 @@ export const setApprovalWindowForTests = (ms) => { approvalWindowMs = ms; };
|
|
|
455
522
|
export const resetApprovalWindowForTests = () => { approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS; };
|
|
456
523
|
const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
|
|
457
524
|
const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
|
|
525
|
+
// OBS-1161: transient capacity ("Selected model is at capacity") — bounded same-seat requeues with a
|
|
526
|
+
// real wait between them, THEN a same-floor failover; never a demotion. The budget is per task and
|
|
527
|
+
// seat, counted from the journal's own capacity-requeue rows so a resume continues it, never restarts it.
|
|
528
|
+
const CAPACITY_REQUEUE_CAP = 2;
|
|
529
|
+
const CAPACITY_BACKOFF_MS = 60_000;
|
|
530
|
+
let capacityBackoffMs = CAPACITY_BACKOFF_MS;
|
|
531
|
+
/** Test seam — shrink the capacity backoff without minute-long sleeps. */
|
|
532
|
+
export function setCapacityBackoffMsForTests(ms) { capacityBackoffMs = ms; }
|
|
533
|
+
export function resetCapacityBackoffMsForTests() { capacityBackoffMs = CAPACITY_BACKOFF_MS; }
|
|
534
|
+
// OBS-1169: a CLI bootstrap failure buys ONE same-channel retry after a bounded wait — journal-counted
|
|
535
|
+
// per task and seat like capacity — and then the adapter is escalated away from, never retried forever.
|
|
536
|
+
const BOOTSTRAP_RETRY_CAP = 1;
|
|
537
|
+
const BOOTSTRAP_BACKOFF_MS = 5_000;
|
|
538
|
+
let bootstrapBackoffMs = BOOTSTRAP_BACKOFF_MS;
|
|
539
|
+
/** Test seam — shrink the bootstrap backoff. */
|
|
540
|
+
export function setBootstrapBackoffMsForTests(ms) { bootstrapBackoffMs = ms; }
|
|
541
|
+
export function resetBootstrapBackoffMsForTests() { bootstrapBackoffMs = BOOTSTRAP_BACKOFF_MS; }
|
|
458
542
|
const NO_TRAILER_DEMOTION_STREAK = 2; // OBS-57: consecutive no-trailer windows demote a channel for the rest of the run
|
|
459
543
|
// OBS-117 (v1.71 T6): a worker pane that never prints a byte by T+60s after dispatch is a dead
|
|
460
544
|
// channel — don't burn the full stall window waiting for a silent launch failure. Checked on the
|
|
@@ -520,9 +604,16 @@ export function resetWorkerStartupWindowMsForTests() {
|
|
|
520
604
|
/** One seat, one startup prefix. Once execution or its input box is seen, later panes are ineligible. */
|
|
521
605
|
export class StartupFailureDetector {
|
|
522
606
|
adapter;
|
|
607
|
+
pattern;
|
|
608
|
+
wholePrefix;
|
|
523
609
|
closed = false;
|
|
524
|
-
constructor(adapter
|
|
610
|
+
constructor(adapter, pattern = WORKER_STARTUP_FAILURE_RE, // OBS-1169: or known bootstrap text
|
|
611
|
+
// OBS-1169: bootstrap evidence must own the WHOLE bounded prefix — a match followed by a tool frame
|
|
612
|
+
// or an input box anywhere later in the sample is a worker that ran, not a CLI that died starting.
|
|
613
|
+
wholePrefix = false) {
|
|
525
614
|
this.adapter = adapter;
|
|
615
|
+
this.pattern = pattern;
|
|
616
|
+
this.wholePrefix = wholePrefix;
|
|
526
617
|
}
|
|
527
618
|
sample(output, launchedAt) {
|
|
528
619
|
if (this.closed || launchedAt === undefined || Date.now() - launchedAt > workerStartupWindowMs)
|
|
@@ -559,6 +650,7 @@ export class StartupFailureDetector {
|
|
|
559
650
|
}
|
|
560
651
|
}
|
|
561
652
|
let offset = 0;
|
|
653
|
+
let found;
|
|
562
654
|
for (const [index, row] of rows.entries()) {
|
|
563
655
|
const cleanRow = row.replace(/\u001b\[[0-?]*[ -/]*[@-~]/g, "");
|
|
564
656
|
// Structured tool frames and the terminal's tool headings close the prefix BEFORE their body.
|
|
@@ -567,16 +659,20 @@ export class StartupFailureDetector {
|
|
|
567
659
|
this.closed = true;
|
|
568
660
|
return;
|
|
569
661
|
}
|
|
570
|
-
if (!this.adapter.harnessBannerRows?.includes(cleanRow)) {
|
|
571
|
-
const match =
|
|
572
|
-
if (match)
|
|
573
|
-
|
|
662
|
+
if (!found && !this.adapter.harnessBannerRows?.includes(cleanRow)) {
|
|
663
|
+
const match = this.pattern.exec(row);
|
|
664
|
+
if (match) {
|
|
665
|
+
found = {
|
|
574
666
|
matchedBytes: match[0], offset: offset + Buffer.byteLength(row.slice(0, match.index)),
|
|
575
667
|
row, rowNumber: index + 1,
|
|
576
668
|
};
|
|
669
|
+
if (!this.wholePrefix)
|
|
670
|
+
return found;
|
|
671
|
+
}
|
|
577
672
|
}
|
|
578
673
|
offset += Buffer.byteLength(row) + 1;
|
|
579
674
|
}
|
|
675
|
+
return found;
|
|
580
676
|
}
|
|
581
677
|
}
|
|
582
678
|
// OBS-901/906: one daemon-owned writer for every execution surface. Drivers expose one retained
|
|
@@ -963,7 +1059,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
963
1059
|
};
|
|
964
1060
|
if (r.pass) {
|
|
965
1061
|
// Q121s: a forgiven pass journals its fingerprints — honest about what was carried, never a silent green.
|
|
966
|
-
journal.append("tip-verify", undefined, { ...evidence, gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.reused ? { cached: true } : {}), ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity: measuredCapacity });
|
|
1062
|
+
journal.append("tip-verify", undefined, { ...evidence, gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.reused ? { cached: true } : {}), ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints, ...(r.baselineProvenance ? { baselineProvenance: r.baselineProvenance } : {}) } : {}), tip, cmdHash, capacity: measuredCapacity });
|
|
967
1063
|
}
|
|
968
1064
|
else {
|
|
969
1065
|
journal.append("tip-verify-failed", undefined, {
|
|
@@ -1476,18 +1572,51 @@ async function taskWithMaterializedContext(repoRoot, journal, worktree, task, at
|
|
|
1476
1572
|
}
|
|
1477
1573
|
return { ...task, context };
|
|
1478
1574
|
}
|
|
1575
|
+
// OBS-1107: a refused pick is not lost work by that fact alone. A commit whose own change the
|
|
1576
|
+
// destination already holds (content-empty, or a patch represented there under another hash) is
|
|
1577
|
+
// accounted for and skipped; the carry stops only at a nonempty change the destination lacks.
|
|
1479
1578
|
async function cherryPickCommits(wt, commits) {
|
|
1480
1579
|
const carried = [];
|
|
1580
|
+
const accounted = [];
|
|
1481
1581
|
for (const hash of commits) {
|
|
1482
1582
|
const r = await shGit(`git cherry-pick --no-gpg-sign ${shq(hash)}`, wt);
|
|
1483
|
-
if (r.code
|
|
1484
|
-
|
|
1583
|
+
if (r.code === 0) {
|
|
1584
|
+
carried.push(hash);
|
|
1585
|
+
continue;
|
|
1586
|
+
}
|
|
1587
|
+
await shGit("git cherry-pick --abort", wt);
|
|
1588
|
+
if (!(await changeRepresented(wt, hash)))
|
|
1485
1589
|
break;
|
|
1590
|
+
accounted.push(hash);
|
|
1591
|
+
}
|
|
1592
|
+
return { carried, accounted };
|
|
1593
|
+
}
|
|
1594
|
+
const PRESERVED_REF_PREFIX = "refs/tickmarkr/preserved/";
|
|
1595
|
+
const preservedRefOf = (data) => [data.ref, data.preservedRef].find((v) => typeof v === "string" && v.startsWith(PRESERVED_REF_PREFIX));
|
|
1596
|
+
/** The attempt whose worker last launched into the task's current checkout; "unknown" once a
|
|
1597
|
+
* recreation replaced that tree without a new launch (recheck, gate-only restore) or when no
|
|
1598
|
+
* dispatch assignment is on record. Never the newest dispatch by itself. */
|
|
1599
|
+
export function knownProducer(events, taskId) {
|
|
1600
|
+
let dispatched = "unknown";
|
|
1601
|
+
let producer = "unknown";
|
|
1602
|
+
for (const row of events) {
|
|
1603
|
+
if (row.taskId !== taskId)
|
|
1604
|
+
continue;
|
|
1605
|
+
if (row.event === "task-dispatch") {
|
|
1606
|
+
const a = row.data.assignment;
|
|
1607
|
+
dispatched = typeof a?.adapter === "string" && typeof a?.model === "string"
|
|
1608
|
+
? { channel: `${a.adapter}:${a.model}`, attempt: typeof row.data.attempt === "number" ? row.data.attempt : 0 } : "unknown";
|
|
1486
1609
|
}
|
|
1487
|
-
|
|
1610
|
+
else if (row.event === "worker-launch")
|
|
1611
|
+
producer = dispatched;
|
|
1612
|
+
else if (row.event === "worktree-recreation")
|
|
1613
|
+
producer = "unknown";
|
|
1488
1614
|
}
|
|
1489
|
-
return
|
|
1615
|
+
return producer;
|
|
1490
1616
|
}
|
|
1617
|
+
/** Distinct author channels of the subject for task-done/status/board rows: "unknown" replaces every
|
|
1618
|
+
* unresolvable owner (legacy unattributed preservation, dispatch without assignment). */
|
|
1619
|
+
const mergedAuthors = (authors) => [...new Set(authors.map((a) => a.startsWith("unknown author") ? "unknown" : a))].sort();
|
|
1491
1620
|
/** Fold lifetime dispatch/carry evidence, independent of attempt budgets and routing exclusions.
|
|
1492
1621
|
* Recreation rows name SOURCE hashes, so ownership is joined by stable patch identity. Each
|
|
1493
1622
|
* attempt owns only what the next carry (or current subject) adds beyond its own incoming set.
|
|
@@ -1517,12 +1646,15 @@ async function subjectAuthors(events, taskId, wt, base) {
|
|
|
1517
1646
|
let previous;
|
|
1518
1647
|
let awaitingCarry = false;
|
|
1519
1648
|
const owners = new Map();
|
|
1649
|
+
// Preserved engine commits carry their producing attempt on the row; a row without one is legacy
|
|
1650
|
+
// and stays explicitly unknown rather than inheriting the seat that later carried the patch.
|
|
1651
|
+
const preservedOwner = new Map();
|
|
1520
1652
|
const attribute = (ids, attempt) => {
|
|
1521
1653
|
for (const id of ids) {
|
|
1522
1654
|
if (attempt?.incoming.has(id))
|
|
1523
1655
|
continue;
|
|
1524
1656
|
const authors = owners.get(id) ?? new Set();
|
|
1525
|
-
authors.add(attempt?.author ?? "unknown author (missing task-dispatch assignment)");
|
|
1657
|
+
authors.add(preservedOwner.get(id) ?? attempt?.author ?? "unknown author (missing task-dispatch assignment)");
|
|
1526
1658
|
owners.set(id, authors);
|
|
1527
1659
|
}
|
|
1528
1660
|
};
|
|
@@ -1530,6 +1662,24 @@ async function subjectAuthors(events, taskId, wt, base) {
|
|
|
1530
1662
|
for (const row of events) {
|
|
1531
1663
|
if (row.taskId !== taskId)
|
|
1532
1664
|
continue;
|
|
1665
|
+
const preserved = preservedRefOf(row.data);
|
|
1666
|
+
if (preserved) {
|
|
1667
|
+
// Only an engine preserve commit is owned by its row; a ref naming a worker's own commit keeps
|
|
1668
|
+
// that commit's dispatch attribution.
|
|
1669
|
+
const shown = await shGit(`git show -s ${shq(`--format=%H%n%s%n%(trailers:key=${PRESERVE_PRODUCER_TRAILER},valueonly)`)} ${shq(`${preserved}^{commit}`)}`, wt);
|
|
1670
|
+
const [commit, subject, trailer] = shown.stdout.trim().split("\n");
|
|
1671
|
+
if (shown.code === 0 && commit && subject === PRESERVE_COMMIT_SUBJECT) {
|
|
1672
|
+
// A row that merely mentions the ref (a park naming it) defers to the commit's own trailer;
|
|
1673
|
+
// only a commit with neither is legacy. A known owner is never downgraded by a later mention.
|
|
1674
|
+
const owner = typeof row.data.producer === "string" ? row.data.producer
|
|
1675
|
+
: trailer?.trim().replace(/ attempt \d+$/, "") || "unknown author (legacy unattributed preservation)";
|
|
1676
|
+
for (const id of await patches([commit])) {
|
|
1677
|
+
const prior = preservedOwner.get(id);
|
|
1678
|
+
if (prior === undefined || prior === "unknown" || prior.startsWith("unknown author"))
|
|
1679
|
+
preservedOwner.set(id, owner);
|
|
1680
|
+
}
|
|
1681
|
+
}
|
|
1682
|
+
}
|
|
1533
1683
|
if (row.event === "task-dispatch") {
|
|
1534
1684
|
previous = current;
|
|
1535
1685
|
const a = row.data.assignment;
|
|
@@ -1591,6 +1741,7 @@ export function recordFatalRunEnd(journal, runId, branch, err, graph, phase = "s
|
|
|
1591
1741
|
const original = err instanceof Error ? err.message : String(err);
|
|
1592
1742
|
// A fatal close never finished a cycle: fresh/reused green is withheld, a recorded red still stands.
|
|
1593
1743
|
let tipProof = { kind: "incomplete" };
|
|
1744
|
+
let forgiven = [];
|
|
1594
1745
|
try {
|
|
1595
1746
|
// Only the latest engagement can already own this terminal outcome.
|
|
1596
1747
|
const events = journal.read();
|
|
@@ -1604,6 +1755,8 @@ export function recordFatalRunEnd(journal, runId, branch, err, graph, phase = "s
|
|
|
1604
1755
|
}
|
|
1605
1756
|
const latest = runEndTipProof(events.slice(from));
|
|
1606
1757
|
tipProof = { ...latest, kind: latest.kind === "failed" ? "failed" : "incomplete" };
|
|
1758
|
+
// OBS-1123: the same fold the normal close records, so a crash never drops what a green carried.
|
|
1759
|
+
forgiven = forgivenFingerprints(events);
|
|
1607
1760
|
}
|
|
1608
1761
|
catch (readErr) {
|
|
1609
1762
|
console.error(`tickmarkr ${runId}: journal read failed while recording the fatal run-end (${readErr instanceof Error ? readErr.message : String(readErr)}) — original error: ${original}`);
|
|
@@ -1620,6 +1773,7 @@ export function recordFatalRunEnd(journal, runId, branch, err, graph, phase = "s
|
|
|
1620
1773
|
fatal: true,
|
|
1621
1774
|
error: original,
|
|
1622
1775
|
tipProof,
|
|
1776
|
+
...(forgiven.length ? { forgiven } : {}),
|
|
1623
1777
|
};
|
|
1624
1778
|
try {
|
|
1625
1779
|
journal.append("run-end", undefined, record);
|
|
@@ -1706,6 +1860,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1706
1860
|
let fatalPhase = "setup";
|
|
1707
1861
|
let deliberateTermination = false;
|
|
1708
1862
|
const fatalStop = new AbortController();
|
|
1863
|
+
const hostStop = new AbortController();
|
|
1864
|
+
const hostChecks = new Set();
|
|
1709
1865
|
const inflight = new Map();
|
|
1710
1866
|
let retireFatalSlots;
|
|
1711
1867
|
let branch = "";
|
|
@@ -1802,21 +1958,50 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1802
1958
|
if (owner) {
|
|
1803
1959
|
const processGroup = readOwnedProcessGroup(owner.groupFile);
|
|
1804
1960
|
let survivors;
|
|
1961
|
+
const strays = [];
|
|
1962
|
+
let session;
|
|
1805
1963
|
try {
|
|
1806
1964
|
const shared = processGroup !== undefined && [...workerOwners].some(([other, otherOwner]) => other !== slot && liveSlots.has(other) && readOwnedProcessGroup(otherOwner.groupFile) === processGroup);
|
|
1807
1965
|
if (shared)
|
|
1808
1966
|
throw new Error(`worker group ${processGroup} is shared with another live attempt`);
|
|
1809
|
-
|
|
1967
|
+
try {
|
|
1968
|
+
session = readFileSync(`${owner.groupFile}.session`, "utf8").trim() || undefined;
|
|
1969
|
+
}
|
|
1970
|
+
catch { /* pre-launch or older driver */ }
|
|
1971
|
+
let parent;
|
|
1972
|
+
try {
|
|
1973
|
+
const row = /^\s*(\d+)\s+(.+)$/.exec(readFileSync(`${owner.groupFile}.parent`, "utf8").trim());
|
|
1974
|
+
if (row)
|
|
1975
|
+
parent = { pid: Number(row[1]), startedAt: row[2].trim().replace(/\s+/g, " ") };
|
|
1976
|
+
}
|
|
1977
|
+
catch { /* a live dispatch root can still prove its parent */ }
|
|
1978
|
+
const others = [...workerOwners].filter(([other]) => other !== slot && liveSlots.has(other));
|
|
1979
|
+
survivors = await reapOwnedProcessGroup(processGroup, slot.cwd, {
|
|
1980
|
+
marker: owner.marker, session, parent, identities: owner.identities, descendants: owner.descendants, strays,
|
|
1981
|
+
excludedGroups: others.flatMap(([, other]) => {
|
|
1982
|
+
const group = readOwnedProcessGroup(other.groupFile);
|
|
1983
|
+
return group === undefined ? [] : [group];
|
|
1984
|
+
}),
|
|
1985
|
+
excludedWorktrees: others.map(([other]) => other.cwd).filter((cwd) => cwd !== slot.cwd),
|
|
1986
|
+
});
|
|
1810
1987
|
}
|
|
1811
1988
|
catch (error) {
|
|
1812
1989
|
journal.append("worker-process-reaped", owner.taskId, { slot: slot.name, attempt: owner.attempt,
|
|
1813
|
-
processGroup: processGroup ?? null, survivors: null, error: String(error) });
|
|
1990
|
+
processGroup: processGroup ?? null, strays, survivors: null, error: String(error) });
|
|
1991
|
+
workerOwners.delete(slot); // one reap row per attempt: a later close never re-sweeps
|
|
1814
1992
|
throw error;
|
|
1815
1993
|
}
|
|
1816
|
-
reapReports.set(slot, { processGroup: processGroup ?? null, survivors });
|
|
1994
|
+
reapReports.set(slot, { processGroup: processGroup ?? null, strays, survivors });
|
|
1817
1995
|
journal.append("worker-process-reaped", owner.taskId, {
|
|
1818
|
-
slot: slot.name, attempt: owner.attempt, processGroup: processGroup ?? null, survivors,
|
|
1996
|
+
slot: slot.name, attempt: owner.attempt, processGroup: processGroup ?? null, strays, survivors,
|
|
1819
1997
|
});
|
|
1998
|
+
// Every outcome retires the claim: one reap row per attempt, whatever the verdict.
|
|
1999
|
+
workerOwners.delete(slot);
|
|
2000
|
+
// Pre-launch and in-process drivers have no OS dispatch claim; their close owns retirement.
|
|
2001
|
+
// Once any dispatch ownership exists, an unreadable sweep must block the next gate.
|
|
2002
|
+
if (survivors === null && (processGroup !== undefined || session !== undefined || owner.descendants.size > 0)) {
|
|
2003
|
+
throw new Error(`worker group ${processGroup} cleanup unknown`);
|
|
2004
|
+
}
|
|
1820
2005
|
if (survivors && survivors.length > 0)
|
|
1821
2006
|
throw new Error(`worker group ${processGroup} survivors: ${survivors.join(", ")}`);
|
|
1822
2007
|
}
|
|
@@ -1915,6 +2100,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1915
2100
|
// new attempt while this reaper is still closing the old ones.
|
|
1916
2101
|
const termination = new Error(`terminated by ${sig}`);
|
|
1917
2102
|
abortRun(termination);
|
|
2103
|
+
hostStop.abort(termination);
|
|
2104
|
+
await Promise.allSettled(hostChecks);
|
|
1918
2105
|
if (activeTipVerify) {
|
|
1919
2106
|
activeTipVerify.controller.abort(termination);
|
|
1920
2107
|
await activeTipVerify.settled;
|
|
@@ -2003,11 +2190,22 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2003
2190
|
// GATE-08 (v1.12): the humanGate guard consults this run's journaled approvals, not just the compiled
|
|
2004
2191
|
// flag. Startup approvals seed the first scheduling pass; live approvals are folded at task
|
|
2005
2192
|
// boundaries below so a sibling can release parked work without waiting for run-end + resume.
|
|
2193
|
+
//
|
|
2194
|
+
// OBS-1178: revalidate every open approval before anything can enact it. The decision fold already
|
|
2195
|
+
// gives an unsound one no effect; the approval-refused row records that answer and consumes it, so
|
|
2196
|
+
// every fold and every operator reads the task as still parked — a stale waive never satisfies a newer gate.
|
|
2197
|
+
const refuseStaleApprovals = () => {
|
|
2198
|
+
for (const [taskId, stale] of staleApprovals(journal.read())) {
|
|
2199
|
+
// `lines` names the voided rows, so a sound decision beside a stale one survives the refusal.
|
|
2200
|
+
journal.append(APPROVAL_REFUSED, taskId, { reason: `stale or unbound decision refused before enactment — ${stale.reason}`, lines: stale.lines });
|
|
2201
|
+
}
|
|
2202
|
+
};
|
|
2203
|
+
refuseStaleApprovals();
|
|
2006
2204
|
const approvalStartupEvents = journal.read();
|
|
2007
|
-
// Continuing human-gate permission survives enactment; malformed releases grant none.
|
|
2008
|
-
const validApproval = (e) => e.event === "task-approved" && e.taskId
|
|
2009
|
-
&&
|
|
2010
|
-
const approved = new Set(approvalStartupEvents.filter(validApproval).map((e) => e.taskId));
|
|
2205
|
+
// Continuing human-gate permission survives enactment; malformed, refused or unsound releases grant none.
|
|
2206
|
+
const validApproval = (e) => e.event === "task-approved" && e.taskId !== undefined
|
|
2207
|
+
&& approvalAction(e.taskId, e).authority !== "inert";
|
|
2208
|
+
const approved = new Set(effectiveEvents(approvalStartupEvents).filter(validApproval).map((e) => e.taskId));
|
|
2011
2209
|
const startupActions = pendingDaemonApprovalActions(approvalStartupEvents);
|
|
2012
2210
|
let approvalSweepCursor = approvalStartupEvents.length;
|
|
2013
2211
|
const commands = detectGateCommands(repoRoot, cfg);
|
|
@@ -2046,11 +2244,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2046
2244
|
console.error(`tickmarkr: narrator not opened: ${placed.error}`);
|
|
2047
2245
|
return undefined;
|
|
2048
2246
|
};
|
|
2049
|
-
// WB-1 (OBS-988): the daemon watches its own cockpit. A board is lost when
|
|
2050
|
-
// aged past the supervision stale bound
|
|
2051
|
-
//
|
|
2052
|
-
//
|
|
2053
|
-
//
|
|
2247
|
+
// WB-1 (OBS-988): the daemon watches its own cockpit. A board is lost when its owner pid is dead, the
|
|
2248
|
+
// recorded arm has no presence, or its beat aged past the supervision stale bound with no owner pid
|
|
2249
|
+
// to consult. OBS-1110: a stale beat under a LIVE MATCHING OWNER — its pid answers and the arm it
|
|
2250
|
+
// claimed is present — is a slow board, not a lost one: journaled once per episode as
|
|
2251
|
+
// `watch-board-slow` and never spending a reopen. The beat and presence checks wait for the UI to
|
|
2252
|
+
// have armed (armId recorded) — a freshly reopened board that has not armed yet is not a second
|
|
2253
|
+
// loss. Reopens are bounded per run; past the bound the loss is journaled `boardless` and the
|
|
2254
|
+
// narrator is never called again.
|
|
2054
2255
|
// Leg-2 T7 M2: one `watch-board-lost` row per LOSS. A loss is identified by the pane and pid the
|
|
2055
2256
|
// owner record names; a reopen that fails leaves that record in place, so the next poll sees the
|
|
2056
2257
|
// same loss, journals only its own reopen attempt, and the bound retires the run boardless from the
|
|
@@ -2059,31 +2260,79 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2059
2260
|
let boardReopens = journal.read().filter((e) => e.event === "watch-board-reopened" || e.event === "watch-board-reopen-failed").length;
|
|
2060
2261
|
let boardless = false;
|
|
2061
2262
|
let journaledLoss;
|
|
2062
|
-
|
|
2263
|
+
let journaledSlow;
|
|
2264
|
+
// Per claim: the next adoption offer, backed off (doubling from the poll cadence to a small cap) but
|
|
2265
|
+
// never exhausted; and whether a reservation is unresolved, so the loop must keep waking to watch it.
|
|
2266
|
+
const MAX_HELD_OFFER_GAP_MS = 4 * APPROVAL_POLL_MS;
|
|
2267
|
+
let heldOffer;
|
|
2268
|
+
let heldRetry = false;
|
|
2269
|
+
const boardHealth = () => {
|
|
2063
2270
|
const owner = readWatchBoard(repoRoot, runId);
|
|
2064
2271
|
if (!owner)
|
|
2065
2272
|
return undefined;
|
|
2066
|
-
const
|
|
2067
|
-
|
|
2068
|
-
|
|
2069
|
-
if (beat.state === "STALE")
|
|
2070
|
-
return { ...lost, beatAgeMs: beat.beatAgeMs };
|
|
2071
|
-
// ponytail: presence path math mirrors supervision.ts's private helper (`<tier>.live.<armId>`)
|
|
2072
|
-
if (!existsSync(join(dirname(supervisionBeatPath(repoRoot, "watch")), `watch.live.${owner.armId}`)))
|
|
2073
|
-
return lost;
|
|
2074
|
-
}
|
|
2273
|
+
const named = { pane: owner.pane, ...(owner.pid !== undefined ? { pid: owner.pid } : {}) };
|
|
2274
|
+
const beat = owner.armId !== undefined ? readTierLiveness(repoRoot, "watch") : undefined;
|
|
2275
|
+
const row = beat?.state === "STALE" ? { ...named, beatAgeMs: beat.beatAgeMs } : named;
|
|
2075
2276
|
if (owner.pid !== undefined && !isPidLive(owner.pid))
|
|
2076
|
-
return lost;
|
|
2277
|
+
return { lost: row };
|
|
2278
|
+
if (owner.armId !== undefined && !existsSync(supervisionPresencePath(repoRoot, "watch", owner.armId)))
|
|
2279
|
+
return { lost: row };
|
|
2280
|
+
if (beat?.state === "STALE")
|
|
2281
|
+
return owner.pid !== undefined ? { slow: row } : { lost: row };
|
|
2077
2282
|
return undefined;
|
|
2078
2283
|
};
|
|
2284
|
+
// OBS-1172: a placement the driver HELD after an indeterminate split receipt has no slot here. While
|
|
2285
|
+
// the reservation stays unresolved the loop keeps polling it — even with every worker slot busy — so
|
|
2286
|
+
// a claim landing mid-task is seen. Once a live observer has claimed, the narrator is offered it
|
|
2287
|
+
// again, backed off per claim (doubling from the poll cadence to MAX_HELD_OFFER_GAP_MS) but never
|
|
2288
|
+
// exhausted: a listing that is unavailable or not yet showing the handle is retried until it proves
|
|
2289
|
+
// the board or the run ends; only success latches (the slot). The driver adopts the proven pane —
|
|
2290
|
+
// tracked from then on like any board — or keeps holding it, and never splits a second board over it.
|
|
2291
|
+
// This spends no reopen: nothing was lost.
|
|
2292
|
+
const adoptHeldBoard = async () => {
|
|
2293
|
+
heldRetry = false;
|
|
2294
|
+
const owner = readWatchBoard(repoRoot, runId);
|
|
2295
|
+
if (watchSlot || owner?.pane !== "" || owner.retired)
|
|
2296
|
+
return;
|
|
2297
|
+
heldRetry = true;
|
|
2298
|
+
if (owner.pid === undefined || !isPidLive(owner.pid))
|
|
2299
|
+
return; // unclaimed, or claimant dead: keep watching
|
|
2300
|
+
const id = `${owner.token}:${owner.pid}`;
|
|
2301
|
+
const now = Date.now();
|
|
2302
|
+
const offer = heldOffer?.id === id ? heldOffer : { id, gapMs: APPROVAL_POLL_MS, nextAt: now };
|
|
2303
|
+
if (now < offer.nextAt) {
|
|
2304
|
+
heldOffer = offer;
|
|
2305
|
+
return;
|
|
2306
|
+
}
|
|
2307
|
+
const adopted = await openBoard();
|
|
2308
|
+
if (adopted.ok) {
|
|
2309
|
+
heldOffer = undefined;
|
|
2310
|
+
heldRetry = false;
|
|
2311
|
+
journal.append("watch-board-adopted", undefined, { pane: adopted.slot.id });
|
|
2312
|
+
return;
|
|
2313
|
+
}
|
|
2314
|
+
heldOffer = { id, gapMs: Math.min(offer.gapMs * 2, MAX_HELD_OFFER_GAP_MS), nextAt: Date.now() + offer.gapMs };
|
|
2315
|
+
};
|
|
2079
2316
|
const watchBoard = async () => {
|
|
2080
|
-
if (
|
|
2317
|
+
if (boardless || !trackedDriver.narrator)
|
|
2081
2318
|
return;
|
|
2082
|
-
|
|
2083
|
-
|
|
2084
|
-
|
|
2319
|
+
if (!boardOpened)
|
|
2320
|
+
return adoptHeldBoard();
|
|
2321
|
+
const health = boardHealth();
|
|
2322
|
+
if (health && "slow" in health) {
|
|
2323
|
+
const identity = `${health.slow.pane}:${health.slow.pid}`;
|
|
2324
|
+
if (identity !== journaledSlow) {
|
|
2325
|
+
journaledSlow = identity;
|
|
2326
|
+
journal.append("watch-board-slow", undefined, health.slow);
|
|
2327
|
+
}
|
|
2085
2328
|
return;
|
|
2086
2329
|
}
|
|
2330
|
+
journaledSlow = undefined;
|
|
2331
|
+
if (!health) {
|
|
2332
|
+
journaledLoss = undefined;
|
|
2333
|
+
return adoptHeldBoard();
|
|
2334
|
+
}
|
|
2335
|
+
const loss = health.lost;
|
|
2087
2336
|
const identity = `${loss.pane}:${loss.pid ?? ""}`;
|
|
2088
2337
|
if (identity !== journaledLoss) {
|
|
2089
2338
|
journaledLoss = identity;
|
|
@@ -2107,32 +2356,143 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2107
2356
|
journal.append("watch-board-reopen-failed", undefined, { pane: loss.pane, attempt, error: reopened.error, ...(boardless ? { boardless: true } : {}) });
|
|
2108
2357
|
console.error(`tickmarkr: board not reopened (attempt ${attempt}): ${reopened.error}`);
|
|
2109
2358
|
};
|
|
2110
|
-
//
|
|
2111
|
-
const
|
|
2112
|
-
|
|
2113
|
-
|
|
2359
|
+
// Reference rows are the durable source of truth; old journals establish one on first resume.
|
|
2360
|
+
const priorReference = opts.resume ? [...journal.read()].reverse().find(e => ["host-reference", "host-reference-reset"].includes(e.event)
|
|
2361
|
+
&& typeof e.data.medianMs === "number" && Number.isFinite(e.data.medianMs) && e.data.medianMs > 0) : undefined;
|
|
2362
|
+
let hostReferenceMs = priorReference?.data.medianMs;
|
|
2363
|
+
const hostSignal = (signal) => AbortSignal.any([hostStop.signal, fatalStop.signal, signal, executionSignal()].filter((s) => !!s));
|
|
2364
|
+
const trackHost = (promise) => {
|
|
2365
|
+
hostChecks.add(promise);
|
|
2366
|
+
void promise.finally(() => hostChecks.delete(promise)).catch(() => { });
|
|
2367
|
+
return promise;
|
|
2368
|
+
};
|
|
2369
|
+
const recordReference = (observation) => {
|
|
2370
|
+
if (observation.medianMs === null)
|
|
2371
|
+
return;
|
|
2372
|
+
journal.append("host-reference", undefined, { ...observation });
|
|
2373
|
+
hostReferenceMs = observation.medianMs;
|
|
2374
|
+
};
|
|
2375
|
+
// Health and occupancy share a deadline, but only healthy occupancy gets the bounded fallback.
|
|
2376
|
+
const admitHost = async (taskId, signal, initial, resuming = false, gate, onWait) => {
|
|
2114
2377
|
const startedAt = Date.now();
|
|
2378
|
+
let lastCount = -1;
|
|
2379
|
+
let observation = initial;
|
|
2380
|
+
let first = true;
|
|
2381
|
+
let resetEligible = resuming && hostReferenceMs !== undefined;
|
|
2115
2382
|
for (;;) {
|
|
2116
|
-
signal
|
|
2117
|
-
fatalStop.signal.throwIfAborted();
|
|
2118
|
-
executionSignal()?.throwIfAborted();
|
|
2383
|
+
signal.throwIfAborted();
|
|
2119
2384
|
const count = await liveSuiteCount(repoRoot);
|
|
2120
|
-
|
|
2121
|
-
|
|
2122
|
-
|
|
2123
|
-
|
|
2385
|
+
signal.throwIfAborted();
|
|
2386
|
+
const remaining = suiteWaitCeilingMs - (Date.now() - startedAt);
|
|
2387
|
+
// Reserve a full bounded batch. A partial last batch would manufacture an unreadable
|
|
2388
|
+
// host at an otherwise healthy occupancy deadline. Retain the latest complete observation.
|
|
2389
|
+
const probeBudget = HOST_PROBE_SAMPLE_MS * HOST_PROBE_SAMPLES;
|
|
2390
|
+
if (!observation || (!first && remaining >= probeBudget)) {
|
|
2391
|
+
observation = await observeHost(signal, remaining > 0 ? Math.min(probeBudget, remaining) : undefined);
|
|
2392
|
+
journal.append("host-observation", taskId, { ...observation, referenceMs: hostReferenceMs ?? null, resuming });
|
|
2393
|
+
}
|
|
2394
|
+
first = false;
|
|
2395
|
+
if (hostReferenceMs === undefined && observation.medianMs !== null)
|
|
2396
|
+
recordReference(observation);
|
|
2397
|
+
const degraded = hostDegraded(observation, hostReferenceMs);
|
|
2398
|
+
resetEligible &&= count === 0 && observation.medianMs !== null && degraded;
|
|
2399
|
+
if (!degraded && (count === 0 || resuming))
|
|
2400
|
+
return count;
|
|
2401
|
+
onWait?.();
|
|
2402
|
+
if (degraded)
|
|
2403
|
+
journal.append("host-degraded", taskId, {
|
|
2404
|
+
...observation, referenceMs: hostReferenceMs ?? null, ...(gate ? { gate } : {}), count, waitedMs: Date.now() - startedAt,
|
|
2405
|
+
});
|
|
2406
|
+
else if (count !== lastCount)
|
|
2407
|
+
journal.append("suite-wait", taskId, { count, ...(gate ? { gate } : {}) });
|
|
2124
2408
|
lastCount = count;
|
|
2125
2409
|
if (Date.now() - startedAt >= suiteWaitCeilingMs) {
|
|
2410
|
+
if (degraded) {
|
|
2411
|
+
if (resetEligible) {
|
|
2412
|
+
// Append BOTH medians before adoption. An unreadable sample or live suite vetoes reset.
|
|
2413
|
+
journal.append("host-reference-reset", undefined, {
|
|
2414
|
+
referenceMs: hostReferenceMs, medianMs: observation.medianMs, waitedMs: Date.now() - startedAt,
|
|
2415
|
+
});
|
|
2416
|
+
hostReferenceMs = observation.medianMs;
|
|
2417
|
+
return count;
|
|
2418
|
+
}
|
|
2419
|
+
throw new HostDegradedError("host latency remained degraded or unreadable through suite-wait deadline");
|
|
2420
|
+
}
|
|
2126
2421
|
journal.append("suite-wait-ceiling", taskId, { count, waitedMs: Date.now() - startedAt });
|
|
2127
|
-
|
|
2128
|
-
count, occupancyCap: occupancyCapacity.forkCap, conservativeCap: conservativeCapacity.forkCap,
|
|
2129
|
-
});
|
|
2130
|
-
return await runWithVerificationBudget(conservativeCapacity, execute);
|
|
2422
|
+
return count;
|
|
2131
2423
|
}
|
|
2132
|
-
await new Promise((
|
|
2424
|
+
await new Promise((resolve, reject) => {
|
|
2425
|
+
const abort = () => { clearTimeout(timer); reject(signal.reason); };
|
|
2426
|
+
const timer = setTimeout(() => { signal.removeEventListener("abort", abort); resolve(); }, Math.min(SUITE_POLL_MS, Math.max(0, suiteWaitCeilingMs - (Date.now() - startedAt))));
|
|
2427
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
2428
|
+
if (signal.aborted)
|
|
2429
|
+
abort();
|
|
2430
|
+
});
|
|
2133
2431
|
}
|
|
2134
|
-
|
|
2135
|
-
|
|
2432
|
+
};
|
|
2433
|
+
const initializeHost = () => trackHost((async () => {
|
|
2434
|
+
const signal = hostSignal();
|
|
2435
|
+
const observation = await observeHost(signal);
|
|
2436
|
+
journal.append("host-observation", undefined, { ...observation, referenceMs: hostReferenceMs ?? null, resuming: !!opts.resume });
|
|
2437
|
+
if (hostReferenceMs === undefined)
|
|
2438
|
+
recordReference(observation);
|
|
2439
|
+
if (opts.resume || observation.medianMs === null)
|
|
2440
|
+
await admitHost(undefined, signal, observation, !!opts.resume);
|
|
2441
|
+
})());
|
|
2442
|
+
// Context carries attribution through gates and remote inference; only shell commands acquire.
|
|
2443
|
+
const commandLeases = new CommandLeases();
|
|
2444
|
+
// Only an unambiguous, currently open phase can attribute a command wait. Parallel siblings
|
|
2445
|
+
// deliberately leave gate absent; neither the last phase nor the last red is a safe substitute.
|
|
2446
|
+
const activeGatePhases = new Map();
|
|
2447
|
+
const withCommandContext = (taskId, run, signal = executionSignal()) => {
|
|
2448
|
+
let hostFailure;
|
|
2449
|
+
return runWithCommandLease((_command, execute) => {
|
|
2450
|
+
const active = taskId ? activeGatePhases.get(taskId) : undefined;
|
|
2451
|
+
const gate = active?.size === 1 ? [...active][0] : undefined;
|
|
2452
|
+
let waited = false;
|
|
2453
|
+
return commandLeases.run(async () => {
|
|
2454
|
+
if (hostFailure)
|
|
2455
|
+
throw hostFailure;
|
|
2456
|
+
let count;
|
|
2457
|
+
try {
|
|
2458
|
+
count = await trackHost(admitHost(taskId, hostSignal(signal), undefined, false, gate, () => { waited = true; }));
|
|
2459
|
+
}
|
|
2460
|
+
catch (error) {
|
|
2461
|
+
if (error instanceof HostDegradedError)
|
|
2462
|
+
hostFailure = error;
|
|
2463
|
+
throw error;
|
|
2464
|
+
}
|
|
2465
|
+
if (waited && taskId) {
|
|
2466
|
+
if (gate)
|
|
2467
|
+
journal.phaseStart(taskId, phaseForGate(gate), { gate, admitted: true });
|
|
2468
|
+
else
|
|
2469
|
+
journal.append("suite-admitted", taskId, {});
|
|
2470
|
+
}
|
|
2471
|
+
if (count > 0) {
|
|
2472
|
+
journal.append("suite-budget", taskId, {
|
|
2473
|
+
count, occupancyCap: occupancyCapacity.forkCap, conservativeCap: conservativeCapacity.forkCap,
|
|
2474
|
+
});
|
|
2475
|
+
return await runWithVerificationBudget(conservativeCapacity, execute);
|
|
2476
|
+
}
|
|
2477
|
+
return await execute();
|
|
2478
|
+
}, (count) => {
|
|
2479
|
+
waited = true;
|
|
2480
|
+
journal.append("suite-wait", taskId, { count, ...(gate ? { gate } : {}) });
|
|
2481
|
+
}, SUITE_POLL_MS, signal);
|
|
2482
|
+
}, async () => {
|
|
2483
|
+
try {
|
|
2484
|
+
const result = await run();
|
|
2485
|
+
// Some command oracles turn launch errors into results. Admission failure still parks infra.
|
|
2486
|
+
if (hostFailure)
|
|
2487
|
+
throw hostFailure;
|
|
2488
|
+
return result;
|
|
2489
|
+
}
|
|
2490
|
+
finally {
|
|
2491
|
+
if (taskId)
|
|
2492
|
+
activeGatePhases.delete(taskId);
|
|
2493
|
+
}
|
|
2494
|
+
});
|
|
2495
|
+
};
|
|
2136
2496
|
let baseRef;
|
|
2137
2497
|
let baseline;
|
|
2138
2498
|
let baselinePending = false;
|
|
@@ -2220,7 +2580,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2220
2580
|
// journal.replayStatuses predates typed releases and re-pends even inert rows. Keep that
|
|
2221
2581
|
// legacy reader intact; scheduling accepts only the fold's recognised authorities.
|
|
2222
2582
|
const statuses = new Map();
|
|
2223
|
-
for (const e of replayEvents) {
|
|
2583
|
+
for (const e of effectiveEvents(replayEvents)) { // OBS-1178: a refused or unsound decision released nothing
|
|
2224
2584
|
if (!e.taskId)
|
|
2225
2585
|
continue;
|
|
2226
2586
|
if (e.event === "task-dispatch")
|
|
@@ -2264,6 +2624,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2264
2624
|
});
|
|
2265
2625
|
runStarted = true;
|
|
2266
2626
|
await placeBoard();
|
|
2627
|
+
// A terminal resume with no commands has no execution to admit.
|
|
2628
|
+
if (Object.keys(commands).length > 0 || graph.tasks.some(t => ["pending", "running", "gated"].includes(t.status))) {
|
|
2629
|
+
baselineCapture = initializeHost();
|
|
2630
|
+
void baselineCapture.catch(() => { baselineFailed = true; });
|
|
2631
|
+
}
|
|
2267
2632
|
}
|
|
2268
2633
|
else {
|
|
2269
2634
|
baseRef = await gitHead(repoRoot);
|
|
@@ -2273,7 +2638,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2273
2638
|
// Workers can run beside capture, but no gate may observe an absent or partial baseline.
|
|
2274
2639
|
// Keep publication and warnings inside the same barrier as the suite's final verdict.
|
|
2275
2640
|
baselineCapture = withCommandContext(undefined, async () => {
|
|
2276
|
-
|
|
2641
|
+
// With no commands, capture is already complete: persist it before the probe can wait.
|
|
2642
|
+
// Otherwise a kill during startup can strand a resumable run without baseline.json.
|
|
2643
|
+
const emptyCapture = Object.keys(commands).length === 0 ? await captureBaseline(repoRoot, commands) : undefined;
|
|
2644
|
+
if (emptyCapture)
|
|
2645
|
+
writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(emptyCapture, null, 2));
|
|
2646
|
+
await initializeHost();
|
|
2647
|
+
// OBS-1123: the capture's identity and publication time ride the measurement every forgiveness reads.
|
|
2648
|
+
const captured = { ...(emptyCapture ?? await captureBaseline(repoRoot, commands)),
|
|
2649
|
+
provenance: { baseRef, capturedAt: new Date().toISOString() } };
|
|
2277
2650
|
writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(captured, null, 2));
|
|
2278
2651
|
baseline = captured;
|
|
2279
2652
|
for (const warning of captured.warnings ?? [])
|
|
@@ -2337,16 +2710,29 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2337
2710
|
// owns listing/parsing/closing. Cosmetic by contract: failures are swallowed and subprocess has no
|
|
2338
2711
|
// reconcile (optional chain → no-op), so gates and the oracle suite never feel this. keepPanes
|
|
2339
2712
|
// "forever" is the keep-everything debug override — it disables the sweep entirely.
|
|
2340
|
-
const
|
|
2713
|
+
const resuming = !!opts.resume;
|
|
2714
|
+
const reconcile = async (sweep) => {
|
|
2341
2715
|
if (keepForever)
|
|
2342
2716
|
return;
|
|
2343
2717
|
try {
|
|
2344
|
-
const
|
|
2718
|
+
const rows = journal.read();
|
|
2719
|
+
const desired = desiredPanes(rows, runId);
|
|
2720
|
+
// OBS-1109: an interrupted attempt's pane is evidence — its nonce-bound trailer decides whether
|
|
2721
|
+
// the attempt is harvested, declined as foreign, or recovered — so the fold's run-resume clear
|
|
2722
|
+
// must not reach it before harvestInterruptedAttempt has read it and reaped its processes. Once
|
|
2723
|
+
// that harvest (or a superseding dispatch) is journaled the fold no longer names it: swept then.
|
|
2724
|
+
// Only a resumed daemon can hold such an attempt: a fresh run's sweep is the baseline sweep.
|
|
2725
|
+
if (resuming)
|
|
2726
|
+
for (const t of graph.tasks) {
|
|
2727
|
+
const owned = interruptedAttempt(rows, t.id);
|
|
2728
|
+
if (owned?.launch)
|
|
2729
|
+
desired.add(formatOwnedName({ role: "worker", taskId: t.id, attempt: owned.attempt, runId }));
|
|
2730
|
+
}
|
|
2345
2731
|
// The watch pane is never the DRIVER sweep's candidate (panesToClose spares role "watch":
|
|
2346
2732
|
// herdr's watches bookkeeping lives in close(), and a raw pane-close in the sweep would
|
|
2347
2733
|
// leave narrator() a stale cache) — the driver always sees it as desired; its lifecycle is
|
|
2348
2734
|
// decided here from the fold alone.
|
|
2349
|
-
await driver.reconcile?.(new Set([...desired, watchName]), runId, { ...
|
|
2735
|
+
await driver.reconcile?.(new Set([...desired, watchName]), runId, { ...sweep, endedRunIds });
|
|
2350
2736
|
// OBS-103: when the fold retires the watch (run-end boundary), close the narrator. The
|
|
2351
2737
|
// decision keys on the run identity in the pane name — narrator() adopts a prior daemon
|
|
2352
2738
|
// instance's pane under the same owned name, so a stop→resume cycle's leftover narrator
|
|
@@ -2436,6 +2822,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2436
2822
|
graph = setStatus(graph, t.id, "human");
|
|
2437
2823
|
saveGraph(repoRoot, graph);
|
|
2438
2824
|
journal.append("task-human", t.id, { ...details, reason, kind });
|
|
2825
|
+
// OBS-1178: the notice carries the park token an approval binds to (`approve --park <token>`).
|
|
2826
|
+
const parked = journal.newestBinding(t.id);
|
|
2827
|
+
const token = parked && bindingToken(parked);
|
|
2439
2828
|
if (assignment) {
|
|
2440
2829
|
// OBS-547: `metered` counts CHARGEABLE metered attempts, so an unchargeable dispatch passes 0 and
|
|
2441
2830
|
// the count is omitted rather than written as 0 or as `1` beside `attempts: 0` — a row claiming
|
|
@@ -2444,7 +2833,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2444
2833
|
journal.telemetry({ taskId: t.id, shape: t.shape, adapter: assignment.adapter, model: assignment.model, channel: assignment.channel, attempts, outcome: "human", durationMs: Date.now() - startMs, parkKind: kind, gateFails, consults, tokens, meteredAttempts: tokens && metered ? metered : undefined, retryMode });
|
|
2445
2834
|
}
|
|
2446
2835
|
await reconcile({ spareLiveLlm: true }); // task-human is a terminal event — sweep, sparing sibling tasks' live LLM panes
|
|
2447
|
-
await driver.notify(`tickmarkr ${runId}: ${t.id} needs a human — ${reason}`, { tier: "attention" });
|
|
2836
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} needs a human — ${reason}${token ? `\npark ${token} — bind the decision with \`--park ${token}\`` : ""}`, { tier: "attention" });
|
|
2448
2837
|
};
|
|
2449
2838
|
// OBS-547: cross-reference a scope red against the prediction this run already computed. Every hard
|
|
2450
2839
|
// offender predicted ⇒ an AUTHORING defect whose repair is pre-written: journal the classification
|
|
@@ -2519,8 +2908,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2519
2908
|
}
|
|
2520
2909
|
return false;
|
|
2521
2910
|
};
|
|
2911
|
+
// OBS-1158: admission reads the same journal snapshot the sweep folded, through the shared seam.
|
|
2912
|
+
let admissionPriority = batteryPriority(startupActions.values());
|
|
2913
|
+
const admissible = () => readyTasks(graph, admissionPriority);
|
|
2522
2914
|
const sweepLiveApprovals = () => {
|
|
2523
|
-
|
|
2915
|
+
let events = journal.read();
|
|
2916
|
+
if (events.slice(approvalSweepCursor).some((e) => e.event === "task-approved")) {
|
|
2917
|
+
refuseStaleApprovals();
|
|
2918
|
+
events = journal.read();
|
|
2919
|
+
}
|
|
2920
|
+
admissionPriority = batteryPriority(pendingDaemonApprovalActions(events).values());
|
|
2524
2921
|
const approvals = events.slice(approvalSweepCursor)
|
|
2525
2922
|
.filter((e) => e.event === "task-approved" && e.taskId);
|
|
2526
2923
|
approvalSweepCursor = events.length;
|
|
@@ -2589,10 +2986,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2589
2986
|
// old path. A preservation failure throws and therefore leaves the old checkout in place. The
|
|
2590
2987
|
// row is deliberately written before the later worktree-recreation row so the journal cannot
|
|
2591
2988
|
// describe only the commits it carried while omitting uncommitted work the removal destroyed.
|
|
2989
|
+
const producerNow = () => knownProducer(journal.read(), t.id);
|
|
2592
2990
|
const recreateTaskWorktree = async (taskBranch, taskBase, priorWt) => {
|
|
2593
|
-
const
|
|
2991
|
+
const producer = producerNow();
|
|
2992
|
+
const ref = await preserveWorktree(priorWt, producer);
|
|
2594
2993
|
if (ref)
|
|
2595
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
2994
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
2596
2995
|
return driver.worktree(repoRoot, taskBranch, taskBase);
|
|
2597
2996
|
};
|
|
2598
2997
|
// A dead worker with a clean checkout still needs a durable recovery handle: there may be no
|
|
@@ -2625,10 +3024,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2625
3024
|
// IS the fix; the tried seed and the attempt-loop start close RES-01/RES-02 alongside it.
|
|
2626
3025
|
//
|
|
2627
3026
|
// v1.24 OBS-18: a task-approved{release:attempt-cap} zeros rs.attempts (fresh budget) and clears
|
|
2628
|
-
// lastAssignment while keeping tried
|
|
3027
|
+
// lastAssignment while keeping tried, so the restore below is skipped — after a
|
|
2629
3028
|
// fresh-budget release, prefer nextChannel over the surviving tried-list so burned channels are
|
|
2630
3029
|
// not re-tried first (consult bans / prior failovers survive the release).
|
|
2631
3030
|
const rs = resume.get(t.id);
|
|
3031
|
+
let bootstrapExhausted;
|
|
2632
3032
|
const contentDigest = taskContentDigest(t);
|
|
2633
3033
|
const previousDispatch = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "task-dispatch");
|
|
2634
3034
|
const recordedGraphPath = join(journal.dir, "graph.json");
|
|
@@ -2646,7 +3046,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2646
3046
|
to: channelKey(assignment),
|
|
2647
3047
|
reason: JSON.stringify(previousHints?.pin) !== JSON.stringify(t.routingHints?.pin) ? "pin changed" : "floor changed",
|
|
2648
3048
|
});
|
|
2649
|
-
|
|
3049
|
+
// OBS-1161: no `attempts > 0` guard — every release already clears lastAssignment in the replay,
|
|
3050
|
+
// and a lastAssignment at zero attempts is a first dispatch whose capacity requeue was taken back:
|
|
3051
|
+
// the seat is still in force, so restore it instead of failing over its own tried[] entry early.
|
|
3052
|
+
if (!hintsChanged && rs?.lastAssignment
|
|
2650
3053
|
&& channels.some((c) => channelKey(c) === channelKey(rs.lastAssignment))
|
|
2651
3054
|
&& !demotedChannels.has(channelKey(rs.lastAssignment))) {
|
|
2652
3055
|
assignment = rs.lastAssignment; // restore the consult-chosen assignment (bypasses route()'s static re-pick)
|
|
@@ -2657,14 +3060,19 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2657
3060
|
// EXISTING nextChannel `tried` parameter — zero router changes (D-03).
|
|
2658
3061
|
// OBS-1034: a PIN is exempt — route() already returned it, and a review-upheld repair of the diff
|
|
2659
3062
|
// the pin produced belongs on the pin, not on the next seat its own tried[] entry would pick.
|
|
2660
|
-
|
|
3063
|
+
// OBS-1169: an adapter a bootstrap failover escalated away from stays out, sibling or none.
|
|
3064
|
+
const escalatedOut = channels.filter((c) => rs.escalatedAdapters?.includes(c.adapter)).map(channelKey);
|
|
3065
|
+
const next = nextChannel(assignment, t, cfg, channels, [...rs.tried, ...Object.keys(rs.escalated ?? {}), ...escalatedOut], profile, demotedChannels);
|
|
2661
3066
|
if (next)
|
|
2662
3067
|
assignment = next;
|
|
2663
3068
|
// ponytail: nextChannel null (every channel already tried / none available) — keep the static
|
|
2664
3069
|
// assignment and proceed. Dispatching on a previously-tried channel beats deadlocking a resumed
|
|
2665
3070
|
// run; a park-instead policy can come later if it ever bites.
|
|
3071
|
+
// OBS-1169: …except onto an escalated adapter — that relaunches the broken bootstrap, so it parks infra below.
|
|
3072
|
+
else if (rs.escalatedAdapters?.includes(assignment.adapter))
|
|
3073
|
+
bootstrapExhausted = channelKey(assignment);
|
|
2666
3074
|
}
|
|
2667
|
-
const taskHistory = journal.read().filter((e) => e.taskId === t.id);
|
|
3075
|
+
const taskHistory = effectiveEvents(journal.read()).filter((e) => e.taskId === t.id); // OBS-1178: only an effective decision opens an engagement
|
|
2668
3076
|
// Lifetime identity counts every dispatch, including legacy and unchargeable rows. Releases
|
|
2669
3077
|
// reset the budget below, never this counter; observational annotations do not consume it.
|
|
2670
3078
|
let nextWorkerDispatchOrdinal = taskHistory.filter((e) => e.event === "task-dispatch").length;
|
|
@@ -2673,13 +3081,38 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2673
3081
|
if (!hintsChanged && priorClimb && taskHistory.indexOf(priorClimb) > taskHistory.map((e) => e.event).lastIndexOf("task-dispatch")) {
|
|
2674
3082
|
const target = channels.find((c) => channelKey(c) === priorClimb.data.to && !demotedChannels.has(channelKey(c)));
|
|
2675
3083
|
if (target)
|
|
2676
|
-
assignment =
|
|
3084
|
+
assignment = seatAssignment(target);
|
|
2677
3085
|
}
|
|
2678
3086
|
// pre-kill invariant: tried always contains the current assignment. Spread, never alias the
|
|
2679
3087
|
// journal-derived array (no hidden mutation of replayed state).
|
|
2680
3088
|
const tried = rs?.tried.length ? [...rs.tried] : [channelKey(assignment)];
|
|
2681
3089
|
if (!tried.includes(channelKey(assignment)))
|
|
2682
3090
|
tried.push(channelKey(assignment));
|
|
3091
|
+
// OBS-1169 add.1: channels excluded because their whole adapter was escalated away from ride
|
|
3092
|
+
// `tried` (nextChannel's one exclusion input) but were never launched — keep the true reason so
|
|
3093
|
+
// the dispatch row never labels an unlaunched sibling "already tried".
|
|
3094
|
+
const vendorEscalated = new Map();
|
|
3095
|
+
// …and a resume restores them with their reasons, apart from the channels actually dispatched.
|
|
3096
|
+
for (const [k, why] of Object.entries(rs?.escalated ?? {})) {
|
|
3097
|
+
if (tried.includes(k))
|
|
3098
|
+
continue;
|
|
3099
|
+
tried.push(k);
|
|
3100
|
+
vendorEscalated.set(k, why);
|
|
3101
|
+
}
|
|
3102
|
+
// OBS-1169 add.2: the ADAPTER exclusion itself, apart from any sibling channel — a single-channel
|
|
3103
|
+
// adapter has no untried sibling to carry it, yet its exhausted channel must stay out of the
|
|
3104
|
+
// recycle pool (and a resume's) until the release that clears `tried`.
|
|
3105
|
+
const escalatedAdapters = new Set(rs?.escalatedAdapters ?? []);
|
|
3106
|
+
const escalateAdapter = (adapter, reason) => {
|
|
3107
|
+
escalatedAdapters.add(adapter);
|
|
3108
|
+
for (const c of channels) {
|
|
3109
|
+
const k = channelKey(c);
|
|
3110
|
+
if (c.adapter !== adapter || tried.includes(k))
|
|
3111
|
+
continue;
|
|
3112
|
+
tried.push(k);
|
|
3113
|
+
vendorEscalated.set(k, reason);
|
|
3114
|
+
}
|
|
3115
|
+
};
|
|
2683
3116
|
// VIS-02 convention: absence = no seeding happened. The observable surface for criterion 2's
|
|
2684
3117
|
// exclusion-list-equality oracle. Daemon-side append only — no journal.ts write-path change (Phase 48
|
|
2685
3118
|
// stays unblocked); inert to replayStatuses (unknown events ignored, pinned at journal.test.ts:70-80).
|
|
@@ -2688,6 +3121,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2688
3121
|
attempts: rs.attempts, tried: [...tried], assignment,
|
|
2689
3122
|
workerDispatchOrdinal: previousDispatch?.data.workerDispatchOrdinal ?? null,
|
|
2690
3123
|
});
|
|
3124
|
+
if (bootstrapExhausted === channelKey(assignment)) { // a later tier-climb restore may have moved it off
|
|
3125
|
+
await park(t, `${bootstrapExhausted} failed at bootstrap and no eligible channel remains after resume`, "infra", assignment, rs?.attempts ?? 0, startMs, 0, 0, undefined, 0, "fresh", { cause: "bootstrap", channel: bootstrapExhausted });
|
|
3126
|
+
return;
|
|
3127
|
+
}
|
|
2691
3128
|
// Keep one live list: recovery retries must see exclusions added by onGate during the round.
|
|
2692
3129
|
const badReviewers = [...replayedReviewerExclusions];
|
|
2693
3130
|
const noteReviewEvent = (e) => {
|
|
@@ -2742,7 +3179,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2742
3179
|
// reviewer-exclusion list (badReviewers), never a second parallel counter. OBS-189: scoped to the
|
|
2743
3180
|
// current engagement — an operator approval (uphold or accept) resets the round budget, so an upheld
|
|
2744
3181
|
// task can dispatch its funded attempt instead of re-parking against the whole journal's history.
|
|
2745
|
-
const reviewRoundsDrawn = () => reviewRoundsSinceApproval(
|
|
3182
|
+
const reviewRoundsDrawn = () => reviewRoundsSinceApproval(journal.read(), t.id, decisiveReviewRounds);
|
|
2746
3183
|
// OBS-193: journal the in-gate review retry (mirrors judge-retry) and exclude the flaked seat from
|
|
2747
3184
|
// later attempts' reviewer picks. One helper, called from both onGate sites (satisfied-gate + main).
|
|
2748
3185
|
const noteReviewRetry = (g) => {
|
|
@@ -2807,9 +3244,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2807
3244
|
dirtyWorktree: true, dirtyPaths: g.meta.paths,
|
|
2808
3245
|
...(typeof g.meta.culprit === "string" ? { culprit: g.meta.culprit } : {}),
|
|
2809
3246
|
...(typeof g.meta.preservedRef === "string" ? { preservedRef: g.meta.preservedRef } : {}),
|
|
3247
|
+
...(typeof g.meta.producer === "string" ? { producer: g.meta.producer } : {}),
|
|
3248
|
+
...(typeof g.meta.producerAttempt === "number" ? { producerAttempt: g.meta.producerAttempt } : {}),
|
|
2810
3249
|
} : {}),
|
|
2811
3250
|
...(cfg.executionPolicy && !g.pass ? { disposition: failureDisposition(g) } : {}),
|
|
2812
|
-
...Object.fromEntries(["runnerInfraRerun", "hostStarvedRerun", "recoveryBlocked", "failingFiles", "selectionDecision", "failureEvidence"
|
|
3251
|
+
...Object.fromEntries(["runnerInfraRerun", "hostStarvedRerun", "recoveryBlocked", "failingFiles", "selectionDecision", "failureEvidence",
|
|
3252
|
+
"forgivenFingerprints", "freshFingerprints", "baselineProvenance"]
|
|
2813
3253
|
.filter((key) => g.meta?.[key] !== undefined).map((key) => [key, g.meta[key]])),
|
|
2814
3254
|
// OBS-540: preserve terminal-vs-retryable infra exactly. normalizeGateOutcome deliberately
|
|
2815
3255
|
// defaults a legacy infra row to retryable, so dropping an explicit false here reverses the
|
|
@@ -2910,6 +3350,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2910
3350
|
const parallelPending = new Set();
|
|
2911
3351
|
let heldParallel;
|
|
2912
3352
|
const notePhaseStart = (e) => {
|
|
3353
|
+
const active = activeGatePhases.get(t.id) ?? new Set();
|
|
3354
|
+
active.add(e.gate);
|
|
3355
|
+
activeGatePhases.set(t.id, active);
|
|
2913
3356
|
if (e.parentAt !== undefined)
|
|
2914
3357
|
parallelPending.add(e.gate);
|
|
2915
3358
|
};
|
|
@@ -2929,7 +3372,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2929
3372
|
// OBS-254: RE-DERIVED from the journal here, at prompt-build time, rather than trusted to survive
|
|
2930
3373
|
// in resume state. The journal already holds the upheld review's bytes; no reset of attempt or
|
|
2931
3374
|
// channel state can take them away, on any path, including `resume --retry-failed`.
|
|
2932
|
-
const upheldFeedback = upheldFeedbackByTask(journal.read()).get(t.id) ?? rs?.upheldFeedback;
|
|
3375
|
+
const upheldFeedback = upheldFeedbackByTask(effectiveEvents(journal.read())).get(t.id) ?? rs?.upheldFeedback; // OBS-1178: a refused uphold funds no brief
|
|
2933
3376
|
const carriedEvidence = priorRunEvidence.findings
|
|
2934
3377
|
.filter((finding) => finding.taskId === t.id)
|
|
2935
3378
|
.map(formatPriorFindingEvidence)
|
|
@@ -2944,7 +3387,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2944
3387
|
: "");
|
|
2945
3388
|
let ladderIdx = 0;
|
|
2946
3389
|
const engagementRows = () => {
|
|
2947
|
-
const rows = journal.read().filter((e) => e.taskId === t.id);
|
|
3390
|
+
const rows = effectiveEvents(journal.read()).filter((e) => e.taskId === t.id);
|
|
2948
3391
|
const approval = rows.map((e) => e.event).lastIndexOf("task-approved");
|
|
2949
3392
|
return rows.slice(approval + 1);
|
|
2950
3393
|
};
|
|
@@ -2976,7 +3419,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2976
3419
|
if (c.tier === assignment.tier && channelKey(c) !== to)
|
|
2977
3420
|
climbSkips.add(channelKey(c));
|
|
2978
3421
|
climbProvenance = `tier-escalated ${from} → ${to} (${cause})`;
|
|
2979
|
-
assignment =
|
|
3422
|
+
assignment = seatAssignment(next);
|
|
2980
3423
|
tried.push(to);
|
|
2981
3424
|
return "climbed";
|
|
2982
3425
|
};
|
|
@@ -3010,6 +3453,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3010
3453
|
}
|
|
3011
3454
|
return next;
|
|
3012
3455
|
};
|
|
3456
|
+
// OBS-1161: capacity requeues (OBS-1169: and bootstrap retries) spent on a seat for this task since
|
|
3457
|
+
// its last operator release — journal-derived so the budget survives a resume instead of restarting.
|
|
3458
|
+
const requeuesOn = (event, channel) => {
|
|
3459
|
+
const rows = effectiveEvents(journal.read()).filter((e) => e.taskId === t.id);
|
|
3460
|
+
const since = rows.map((e) => e.event).lastIndexOf("task-approved");
|
|
3461
|
+
return rows.slice(since + 1).filter((e) => e.event === event && e.data.channel === channel).length;
|
|
3462
|
+
};
|
|
3013
3463
|
// OBS-202 (operator law: "you can spawn as many as you want"): channels are session FACTORIES,
|
|
3014
3464
|
// not consumed seats — a tried channel can always host a fresh worker session, and a fresh
|
|
3015
3465
|
// session carries none of the failed attempt's baggage. When the untried pool is empty, recycle
|
|
@@ -3021,15 +3471,20 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3021
3471
|
const next = failover(site);
|
|
3022
3472
|
if (next)
|
|
3023
3473
|
return next;
|
|
3024
|
-
|
|
3474
|
+
// OBS-1169: a vendor escalated away from stays out of the recycle pool until its release —
|
|
3475
|
+
// recycling any of its channels, launched or not, would launch the broken bootstrap again.
|
|
3476
|
+
const recycleExcluded = [channelKey(assignment), ...channels.filter((c) => escalatedAdapters.has(c.adapter)).map(channelKey)];
|
|
3477
|
+
const recycled = nextChannel(assignment, t, cfg, channels, recycleExcluded, profile, demotedChannels)
|
|
3025
3478
|
?? (demotedChannels.has(channelKey(assignment)) ? null : assignment);
|
|
3026
3479
|
if (recycled)
|
|
3027
3480
|
journal.append("channel-recycle", t.id, { site, channel: channelKey(recycled) });
|
|
3028
3481
|
return recycled;
|
|
3029
3482
|
};
|
|
3030
|
-
const runConsult = (trigger, transcript, diffOrFeedback, gates) => {
|
|
3483
|
+
const runConsult = async (trigger, transcript, diffOrFeedback, gates) => {
|
|
3031
3484
|
consults++;
|
|
3032
|
-
|
|
3485
|
+
// OBS-1182: every launched consult seat, failed ones included, rides the verdict into the journal.
|
|
3486
|
+
const invocations = [];
|
|
3487
|
+
const v = await consult({
|
|
3033
3488
|
taskId: t.id, trigger,
|
|
3034
3489
|
journalTail: JSON.stringify(journal.read().slice(-20)),
|
|
3035
3490
|
transcript: transcript.slice(-8000),
|
|
@@ -3038,7 +3493,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3038
3493
|
// D-07: consult panes self-clean when the verdict is read (keepLlm) — only "forever" keeps them.
|
|
3039
3494
|
// v1.54 T1: channels = this run's doctor-filtered live list — consult.prefer seat liveness
|
|
3040
3495
|
// is judged against it, never rebuilt from config (installed-but-unauthed seats would stall).
|
|
3041
|
-
{ keep: keepLlm, onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined, runId, channels: pools.consult });
|
|
3496
|
+
{ keep: keepLlm, onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined, onInvocation: (inv) => invocations.push(inv), runId, channels: pools.consult });
|
|
3497
|
+
// A lone answering seat is already the row's adapter/model/vendor/effort; the list is journaled
|
|
3498
|
+
// only when a seat failed, so the row never repeats itself and a failed seat never vanishes.
|
|
3499
|
+
return invocations.some((inv) => inv.outcome === "failed") ? { ...v, invocations } : v;
|
|
3042
3500
|
};
|
|
3043
3501
|
// returns true → continue attempting, false → task is terminal (parked)
|
|
3044
3502
|
// trigger (why the consult ran) is threaded in so the decompose/human park keeps its cause —
|
|
@@ -3047,9 +3505,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3047
3505
|
journal.append("consult-verdict", t.id, {
|
|
3048
3506
|
action: v.action, notes: v.notes,
|
|
3049
3507
|
adapter: v.adapter ?? "unknown", model: v.model ?? "unknown", vendor: v.vendor ?? "unknown",
|
|
3508
|
+
...(v.effort ? { effort: v.effort } : {}),
|
|
3050
3509
|
...(v.reason ? { reason: v.reason } : {}),
|
|
3051
3510
|
...(v.guidance ? { guidance: v.guidance } : {}),
|
|
3052
3511
|
...(v.excludeAdapter ? { excludeAdapter: v.excludeAdapter } : {}),
|
|
3512
|
+
...(v.invocations?.length ? { invocations: v.invocations } : {}),
|
|
3053
3513
|
});
|
|
3054
3514
|
await driver.notify(`tickmarkr ${runId}: ${t.id} consult verdict: ${v.action}`, { tier: "attention" });
|
|
3055
3515
|
if (v.action === "retry") {
|
|
@@ -3069,15 +3529,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3069
3529
|
// nextChannel's existing tried parameter — zero router changes (D-03). Unknown adapter
|
|
3070
3530
|
// (zero matches) is a no-op expansion ⇒ ordinary channel-level reroute. Task-scoped:
|
|
3071
3531
|
// `tried` lives inside execTask, so a sibling task is unaffected.
|
|
3072
|
-
if (v.excludeAdapter)
|
|
3073
|
-
|
|
3074
|
-
if (c.adapter === v.excludeAdapter) {
|
|
3075
|
-
const k = channelKey(c);
|
|
3076
|
-
if (!tried.includes(k))
|
|
3077
|
-
tried.push(k);
|
|
3078
|
-
}
|
|
3079
|
-
}
|
|
3080
|
-
}
|
|
3532
|
+
if (v.excludeAdapter)
|
|
3533
|
+
escalateAdapter(v.excludeAdapter, `vendor escalated: consult excluded ${v.excludeAdapter}`);
|
|
3081
3534
|
const next = failoverOrRecycle("consult-reroute");
|
|
3082
3535
|
if (next) {
|
|
3083
3536
|
assignment = next;
|
|
@@ -3098,7 +3551,117 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3098
3551
|
let satisfiedGate = satisfiedGates.get(t.id);
|
|
3099
3552
|
const replayedGates = fundedRerun ? undefined : replayedGateResults.get(t.id);
|
|
3100
3553
|
const recheck = approvalAction?.authority === "battery";
|
|
3101
|
-
|
|
3554
|
+
// OBS-1109: finished work is harvested, never redone — on resume too. A daemon that died while its
|
|
3555
|
+
// worker ran leaves an owned attempt between worker-launch and worker-result: its pane may hold the
|
|
3556
|
+
// attempt's own nonce-bound trailer, its branch the committed work. Only after the attempt's owned
|
|
3557
|
+
// processes are reaped is either read as its result — a matching trailer as the worker's claim,
|
|
3558
|
+
// commits without one as the same no-trailer evidence the live harvest synthesizes — and the gate
|
|
3559
|
+
// replay below then judges it under the PRODUCING attempt's author, with no new dispatch. An
|
|
3560
|
+
// unfinished attempt, a pane holding another attempt's trailer, or cleanup that cannot be proven
|
|
3561
|
+
// is declined onto the ordinary recovery path. A harvest an earlier resume recorded is gated again,
|
|
3562
|
+
// never recorded twice.
|
|
3563
|
+
const harvestInterruptedAttempt = async () => {
|
|
3564
|
+
const found = interruptedAttempt(journal.read(), t.id);
|
|
3565
|
+
if (!found || (!found.launch && !found.result))
|
|
3566
|
+
return found?.assignment;
|
|
3567
|
+
const { attempt } = found;
|
|
3568
|
+
const wt = worktreePath(repoRoot, `${branch}--${t.id}`);
|
|
3569
|
+
const recordHarvest = (commits, finished, summary) => journal.append("worker-result-harvested", t.id, {
|
|
3570
|
+
attempt, commits, summary: finished ? summary : HARVESTED_RESULT_SUMMARY, source: RESUME_HARVEST_SOURCE, trailer: finished,
|
|
3571
|
+
});
|
|
3572
|
+
// A declined attempt leaves the fold's spare (interruptedAttempt closes on this row), so its pane
|
|
3573
|
+
// is swept here, before the ordinary recovery path dispatches beside it — the baseline sweep, late.
|
|
3574
|
+
const decline = async (reason) => {
|
|
3575
|
+
journal.append("resume-harvest-declined", t.id, { attempt, reason });
|
|
3576
|
+
await reconcile({ spareLiveLlm: true });
|
|
3577
|
+
return undefined;
|
|
3578
|
+
};
|
|
3579
|
+
// Owned-process cleanup before any result is read as final: the attempt's own process group and
|
|
3580
|
+
// dispatch marker, reaped through the same reapWorker a live harvest uses. Uncertain is declined.
|
|
3581
|
+
const reapOwned = async (key, launch) => {
|
|
3582
|
+
workerOwners.set(key, { taskId: t.id, attempt, groupFile: `${launch.dispatchScript}.${launch.nonce}.pgid`,
|
|
3583
|
+
marker: launch.dispatchScript, identities: new Map(), descendants: new Map() });
|
|
3584
|
+
try {
|
|
3585
|
+
await reapWorker(key);
|
|
3586
|
+
return undefined;
|
|
3587
|
+
}
|
|
3588
|
+
catch (error) {
|
|
3589
|
+
return `owned-process cleanup uncertain: ${error instanceof Error ? error.message : String(error)}`;
|
|
3590
|
+
}
|
|
3591
|
+
};
|
|
3592
|
+
if (found.result) {
|
|
3593
|
+
// A daemon recorded this attempt's result, then died before its harvest row (a live daemon
|
|
3594
|
+
// before its no-trailer synthesis, or a resume between its two rows): finish that record from
|
|
3595
|
+
// the result it wrote — the pane is never re-read, the result never rewritten. A resume wrote
|
|
3596
|
+
// its result only after reaping; a live daemon writes it BEFORE handling a failed reap, so its
|
|
3597
|
+
// owned processes are reaped again here. A result that neither finished nor left commits has
|
|
3598
|
+
// nothing to gate: the ordinary recovery path owns it.
|
|
3599
|
+
if (!found.result.reaped) {
|
|
3600
|
+
if (!found.launch)
|
|
3601
|
+
return decline("owned-process cleanup unproven: the launch recorded no ownership evidence");
|
|
3602
|
+
const uncertain = await reapOwned(found.launch.slot, found.launch);
|
|
3603
|
+
if (uncertain)
|
|
3604
|
+
return decline(uncertain);
|
|
3605
|
+
}
|
|
3606
|
+
const commits = existsSync(wt) ? await commitsAheadOf(await integrationHead(intWt), wt) : [];
|
|
3607
|
+
if (!found.result.finished && commits.length === 0)
|
|
3608
|
+
return decline("unfinished");
|
|
3609
|
+
recordHarvest(commits, found.result.finished, found.result.summary);
|
|
3610
|
+
return found.assignment;
|
|
3611
|
+
}
|
|
3612
|
+
const launch = found.launch;
|
|
3613
|
+
const { nonce, slot } = launch;
|
|
3614
|
+
if (!existsSync(wt))
|
|
3615
|
+
return decline("task worktree missing");
|
|
3616
|
+
const adapter = adapters.find((a) => a.id === found.assignment.adapter);
|
|
3617
|
+
if (!adapter)
|
|
3618
|
+
return decline(`adapter ${found.assignment.adapter} unavailable to read the trailer`);
|
|
3619
|
+
// The journaled slot is read through THIS daemon's driver, a fresh instance since the last one
|
|
3620
|
+
// died. A pane the driver cannot read is DECLINED, never classified: an unreadable pane proves
|
|
3621
|
+
// neither a matching trailer nor the absence of a foreign one, so it is neither harvested nor
|
|
3622
|
+
// gated as trailerless evidence. A driver whose terminals outlive it (Orca) adopts the owned
|
|
3623
|
+
// terminal: bound only on ownership evidence — the journaled owned title AND the task checkout —
|
|
3624
|
+
// never a title alone. The journaled slot.id is NEVER trusted across daemon instances: a fresh
|
|
3625
|
+
// driver numbers its slots from one again, so an interrupted task's old id can name whatever
|
|
3626
|
+
// terminal THIS instance bound under that id (another task's adoption or allocation).
|
|
3627
|
+
// A driver with no read-only adoption is declined BEFORE anything is bound: slot() allocates —
|
|
3628
|
+
// herdr's reclaims the same-named pane (closing the evidence unread) and holds a dispatch lease
|
|
3629
|
+
// only run() releases — so it is never an adoption fallback; that driver keeps ordinary recovery.
|
|
3630
|
+
const adopt = driver.adopt;
|
|
3631
|
+
if (!adopt)
|
|
3632
|
+
return decline("no read-only adoption on this driver");
|
|
3633
|
+
let ownedSlot;
|
|
3634
|
+
let pane;
|
|
3635
|
+
try {
|
|
3636
|
+
ownedSlot = await adopt.call(driver, slot);
|
|
3637
|
+
pane = await driver.read(ownedSlot, PANE_READ_ROWS);
|
|
3638
|
+
}
|
|
3639
|
+
catch (error) {
|
|
3640
|
+
return decline(`pane unreadable: ${error instanceof Error ? error.message : String(error)}`);
|
|
3641
|
+
}
|
|
3642
|
+
const parsed = adapter.parse(pane, nonce);
|
|
3643
|
+
const finished = parsed.cause === undefined;
|
|
3644
|
+
const foreign = !finished && [...pane.matchAll(/TICKMARKR_RESULT_([0-9a-z]+)/g)]
|
|
3645
|
+
.some(([, other]) => other !== nonce && adapter.parse(pane, other).cause === undefined);
|
|
3646
|
+
if (foreign)
|
|
3647
|
+
return decline("foreign-nonce");
|
|
3648
|
+
const taskBase = await integrationHead(intWt);
|
|
3649
|
+
if (!finished && (await commitsAheadOf(taskBase, wt)).length === 0)
|
|
3650
|
+
return decline("unfinished");
|
|
3651
|
+
const uncertain = await reapOwned(ownedSlot, launch);
|
|
3652
|
+
if (uncertain)
|
|
3653
|
+
return decline(uncertain);
|
|
3654
|
+
const commits = await commitsAheadOf(taskBase, wt); // re-measured: the reaped writer is gone now
|
|
3655
|
+
// The source tag makes a death between these two rows recoverable: the next resume finishes the record.
|
|
3656
|
+
journal.append("worker-result", t.id, {
|
|
3657
|
+
ok: parsed.ok, summary: parsed.summary, deviations: parsed.deviations, finished, exitCode: null, attempt,
|
|
3658
|
+
source: RESUME_HARVEST_SOURCE, ...(parsed.cause ? { cause: parsed.cause } : {}),
|
|
3659
|
+
});
|
|
3660
|
+
recordHarvest(commits, finished, parsed.summary);
|
|
3661
|
+
return found.assignment;
|
|
3662
|
+
};
|
|
3663
|
+
const harvestAuthor = rs && !satisfiedGate && !replayedGates && !recheck ? await harvestInterruptedAttempt() : undefined;
|
|
3664
|
+
resumeGateReplay: if (satisfiedGate || replayedGates || recheck || harvestAuthor) {
|
|
3102
3665
|
const taskBase = await integrationHead(intWt);
|
|
3103
3666
|
taskBases.set(t.id, taskBase);
|
|
3104
3667
|
const taskBranch = `${branch}--${t.id}`;
|
|
@@ -3107,7 +3670,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3107
3670
|
// battery gates — read from the pending approval row, never re-selected here, so the tree
|
|
3108
3671
|
// approve counted and the tree the daemon gates are one and the same, checkout present or not.
|
|
3109
3672
|
const preservedRef = recheck
|
|
3110
|
-
?
|
|
3673
|
+
? effectiveEvents(journal.read()).reverse().find((e) => e.taskId === t.id && e.event === "task-approved" && e.data.release === RECHECK_RELEASE)?.data.recheckedRef
|
|
3111
3674
|
: undefined;
|
|
3112
3675
|
if (!existsSync(priorWt) && !preservedRef) {
|
|
3113
3676
|
if (satisfiedGate || recheck)
|
|
@@ -3119,14 +3682,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3119
3682
|
}
|
|
3120
3683
|
const resumeReason = recheck
|
|
3121
3684
|
? "operator recheck"
|
|
3122
|
-
:
|
|
3123
|
-
?
|
|
3124
|
-
:
|
|
3685
|
+
: harvestAuthor
|
|
3686
|
+
? "harvested interrupted attempt"
|
|
3687
|
+
: satisfiedGate
|
|
3688
|
+
? `approved gate ${satisfiedGate}`
|
|
3689
|
+
: `recorded gates on ${replayedGates.commit.slice(0, 10)}`;
|
|
3125
3690
|
const priorTaskTip = preservedRef ? (await shGit(`git rev-parse ${shq(preservedRef)}`, repoRoot)).stdout.trim() : await gitHead(priorWt);
|
|
3126
3691
|
const priorTaskSubject = await gateCommitSubject(taskBase, priorTaskTip, preservedRef ? repoRoot : priorWt);
|
|
3127
3692
|
const commitsToCarry = preservedRef ? await commitsAheadOfRef(taskBase, priorTaskTip, repoRoot) : await commitsAheadOf(taskBase, priorWt);
|
|
3128
3693
|
const wt = await recreateTaskWorktree(taskBranch, taskBase, priorWt);
|
|
3129
|
-
const carriedCommits = await cherryPickCommits(wt, commitsToCarry);
|
|
3694
|
+
const { carried: carriedCommits, accounted: accountedCommits } = await cherryPickCommits(wt, commitsToCarry);
|
|
3130
3695
|
// Reuse is about the tree the gates will actually inspect. The integration tip may have moved
|
|
3131
3696
|
// while the daemon was down, so compare after recreating the task on today's taskBase rather
|
|
3132
3697
|
// than against the stale worktree whose task-only history cannot see newly merged dependencies.
|
|
@@ -3136,13 +3701,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3136
3701
|
if (recheck) {
|
|
3137
3702
|
satisfiedGate = journal.replaySatisfiedGates(new Map([[t.id, currentTaskSubject]])).get(t.id);
|
|
3138
3703
|
}
|
|
3139
|
-
journal.append("worktree-recreation", t.id, {
|
|
3704
|
+
journal.append("worktree-recreation", t.id, {
|
|
3705
|
+
attempted: commitsToCarry, carried: carriedCommits, ...(accountedCommits.length > 0 ? { accounted: accountedCommits } : {}),
|
|
3706
|
+
});
|
|
3140
3707
|
// OBS-212: same fail-closed rule as the dispatch path — but this path is worse, because it runs
|
|
3141
3708
|
// ONLY the gates after the approved one and then MERGES. T3 took it on run-20260728-110135:
|
|
3142
3709
|
// approved past review at 11:22, recreated at 12:50, and phase-start{gates} / phase-start{merge}
|
|
3143
3710
|
// landed in the same second with zero gate-result events. Work missing here is merged unverified.
|
|
3144
3711
|
{
|
|
3145
|
-
const present = new Set(carriedCommits);
|
|
3712
|
+
const present = new Set([...carriedCommits, ...accountedCommits]);
|
|
3146
3713
|
for (const h of commitsToCarry) {
|
|
3147
3714
|
if (!present.has(h) && (await shGit(`git merge-base --is-ancestor ${shq(h)} HEAD`, wt)).code === 0) {
|
|
3148
3715
|
present.add(h);
|
|
@@ -3182,7 +3749,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3182
3749
|
const parkedAuthor = recheck
|
|
3183
3750
|
? [...journal.read()].reverse().find((e) => e.event === "task-dispatch" && e.taskId === t.id)?.data.assignment
|
|
3184
3751
|
: undefined;
|
|
3185
|
-
|
|
3752
|
+
// OBS-1109: a harvested attempt's author is read from the journal, not from this resume's path —
|
|
3753
|
+
// a later resume that replays the harvest's recorded gates still gates it under its producer.
|
|
3754
|
+
const gateAuthor = parkedAuthor ?? resumeHarvestAuthor(journal.read(), t.id) ?? rs?.lastAssignment ?? assignment;
|
|
3186
3755
|
const satisfiedIndex = satisfiedGate ? GATE_NAMES.indexOf(satisfiedGate) : -1;
|
|
3187
3756
|
// The serial pipeline could have at most one blocking result, so "everything after the
|
|
3188
3757
|
// approved gate" was enough. v1.85 can record both verdict siblings red in one round, and a
|
|
@@ -3211,6 +3780,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3211
3780
|
if (recheck) {
|
|
3212
3781
|
remainingGates = declaredGates.filter((gate) => gate !== satisfiedGate);
|
|
3213
3782
|
}
|
|
3783
|
+
else if (harvestAuthor) {
|
|
3784
|
+
remainingGates = declaredGates; // the harvested attempt was never gated: every declared gate runs
|
|
3785
|
+
}
|
|
3214
3786
|
else if (satisfiedGate) {
|
|
3215
3787
|
remainingGates = t.gates.filter((gate) => {
|
|
3216
3788
|
if (gate === satisfiedGate)
|
|
@@ -3279,7 +3851,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3279
3851
|
const startedAt = Date.now();
|
|
3280
3852
|
const provisioned = await withCommandContext(t.id, () => sh(commands.build, wt));
|
|
3281
3853
|
provisionedRow = {
|
|
3282
|
-
gate: "build", commit: !satisfiedGate && !recheck ? replayedGates.commit : currentTaskSubject,
|
|
3854
|
+
gate: "build", commit: !satisfiedGate && !recheck && replayedGates ? replayedGates.commit : currentTaskSubject,
|
|
3283
3855
|
exitCode: provisioned.code, durationMs: Date.now() - startedAt,
|
|
3284
3856
|
};
|
|
3285
3857
|
if (provisioned.code !== 0 && satisfiedGate !== "build") {
|
|
@@ -3293,7 +3865,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3293
3865
|
if (provisionedRow !== undefined)
|
|
3294
3866
|
journal.append("gate-provisioned", t.id, provisionedRow);
|
|
3295
3867
|
const resumedTask = { ...t, gates: remainingGates };
|
|
3296
|
-
const operatorContext = approvalReviewContext(journal.read(), t.id
|
|
3868
|
+
const operatorContext = approvalReviewContext(journal.read(), t.id);
|
|
3297
3869
|
gateLoop: while (true) {
|
|
3298
3870
|
fatalStop.signal.throwIfAborted();
|
|
3299
3871
|
executionSignal()?.throwIfAborted();
|
|
@@ -3304,7 +3876,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3304
3876
|
// This suffix is re-measured to decide whether resume may advance, but the interrupted
|
|
3305
3877
|
// attempt already paid for its red result. The next worker-backed round remains the next
|
|
3306
3878
|
// deterministic-fingerprint occurrence/review round for budget accounting.
|
|
3307
|
-
...(!satisfiedGate && !recheck ? { replayMeasurement: true } : {}),
|
|
3879
|
+
...(!satisfiedGate && !recheck && !harvestAuthor ? { replayMeasurement: true } : {}),
|
|
3308
3880
|
};
|
|
3309
3881
|
if (recheck && satisfiedGate === "review" && gateSubject.commit !== currentTaskSubject) {
|
|
3310
3882
|
satisfiedGate = undefined;
|
|
@@ -3318,6 +3890,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3318
3890
|
journal.phaseStart(t.id, "gates");
|
|
3319
3891
|
const { results } = await withCommandContext(t.id, async () => runReviewRecovery(resumedTask, {
|
|
3320
3892
|
carriedAuthors: await subjectAuthors(journal.read(), t.id, wt, taskBase),
|
|
3893
|
+
producer: producerNow(),
|
|
3321
3894
|
carriedFindings: outstandingReviewFindings(journal.read(), t.id),
|
|
3322
3895
|
operatorContext,
|
|
3323
3896
|
worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
|
|
@@ -3350,7 +3923,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3350
3923
|
return;
|
|
3351
3924
|
}
|
|
3352
3925
|
const g = e.result;
|
|
3353
|
-
|
|
3926
|
+
activeGatePhases.get(t.id)?.delete(e.gate);
|
|
3927
|
+
classifyInfraResult(g);
|
|
3354
3928
|
inParallelOrder(g.gate, () => {
|
|
3355
3929
|
journalGateResult(g);
|
|
3356
3930
|
noteReviewRetry(g);
|
|
@@ -3361,7 +3935,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3361
3935
|
});
|
|
3362
3936
|
},
|
|
3363
3937
|
}, false));
|
|
3364
|
-
results.forEach(
|
|
3938
|
+
results.forEach(classifyInfraResult);
|
|
3365
3939
|
if (pendingDaemonApprovalActions(journal.read()).get(t.id)?.authority === "battery") {
|
|
3366
3940
|
journal.append("recheck-battery", t.id, {
|
|
3367
3941
|
commit: gateSubject.commit,
|
|
@@ -3378,7 +3952,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3378
3952
|
await park(t, gateFailApprovalReason(t.id, unavailableReview.details, true), "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
3379
3953
|
return;
|
|
3380
3954
|
}
|
|
3381
|
-
|
|
3955
|
+
// OBS-1106: same predicate as classification — an infra replay is parked, never repaired.
|
|
3956
|
+
const infra = results.find((g) => gateFailed(g) && isInfraResult(g));
|
|
3382
3957
|
if (infra) {
|
|
3383
3958
|
await park(t, `${infra.gate}: ${infra.details}`, "infra", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
3384
3959
|
return;
|
|
@@ -3417,7 +3992,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3417
3992
|
await park(t, `recheck red: pinned ${pin.via}:${pin.model} is unavailable to host the repair — refusing the ladder`, "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
3418
3993
|
return;
|
|
3419
3994
|
}
|
|
3420
|
-
assignment =
|
|
3995
|
+
assignment = seatAssignment(seat);
|
|
3421
3996
|
}
|
|
3422
3997
|
// Carry completeness was verified before this replay battery. Apply the ordinary
|
|
3423
3998
|
// repair bounds here too; a restored red must fund the same findings-bearing dispatch.
|
|
@@ -3461,6 +4036,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3461
4036
|
saveGraph(repoRoot, graph);
|
|
3462
4037
|
journal.append("task-done", t.id, {
|
|
3463
4038
|
attempts: rs?.attempts ?? 0, assignment: gateAuthor, taskContentDigest: contentDigest,
|
|
4039
|
+
authors: mergedAuthors(await subjectAuthors(journal.read(), t.id, wt, taskBase)),
|
|
3464
4040
|
});
|
|
3465
4041
|
journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
|
|
3466
4042
|
await trackedDriver.project?.(t.id, "completed");
|
|
@@ -3635,7 +4211,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3635
4211
|
feedback = feedback ? `${feedback}\n\n${consultBrief}` : consultBrief;
|
|
3636
4212
|
}
|
|
3637
4213
|
}
|
|
3638
|
-
const scopeApproval =
|
|
4214
|
+
const scopeApproval = effectiveEvents(journaledSoFar).reverse().find((e) => e.taskId === t.id && (e.event === "task-approved" || e.event === "task-dispatch"));
|
|
3639
4215
|
retryMode = scopeApproval?.data.release === "scope-request" ? "fresh" : repairFindings
|
|
3640
4216
|
? "repair"
|
|
3641
4217
|
: priorSession
|
|
@@ -3664,6 +4240,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3664
4240
|
// spends two frontier attempts re-deriving a known defect has to be able to answer afterwards.
|
|
3665
4241
|
fatalStop.signal.throwIfAborted();
|
|
3666
4242
|
const workerDispatchOrdinal = nextWorkerDispatchOrdinal++;
|
|
4243
|
+
vendorEscalated.delete(channelKey(assignment)); // dispatched now: from here on it IS tried
|
|
3667
4244
|
journal.append("task-dispatch", t.id, {
|
|
3668
4245
|
...(scopeApproval?.data.release === "scope-request" ? { files: t.files, graphDefinitionHash: graphDefinitionHash(graph) } : {}),
|
|
3669
4246
|
assignment, attempt, workerDispatchOrdinal, provenance: dispatchProvenance([
|
|
@@ -3676,7 +4253,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3676
4253
|
.filter((key) => key !== channelKey(assignment)).sort(),
|
|
3677
4254
|
exclusionReasons: Object.fromEntries([...new Set([...demotedChannels, ...tried, ...climbSkips])]
|
|
3678
4255
|
.filter((key) => key !== channelKey(assignment))
|
|
3679
|
-
.map((key) => [key, demotedChannels.has(key) ? "demoted"
|
|
4256
|
+
.map((key) => [key, demotedChannels.has(key) ? "demoted"
|
|
4257
|
+
: vendorEscalated.get(key) ?? (tried.includes(key) ? "already tried" : "tier climb skipped")])),
|
|
3680
4258
|
...(outstandingFindings.length > 0 ? { carriedFindings: outstandingFindings } : {}),
|
|
3681
4259
|
...(carriedConsultGuidance ? { carriedConsultGuidance } : {}),
|
|
3682
4260
|
});
|
|
@@ -3696,12 +4274,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3696
4274
|
// OBS-58: quota-failover and every retry recreate the task worktree from the integration tip —
|
|
3697
4275
|
// cherry-pick prior attempts' landed commits forward so a failover dispatch cannot silently
|
|
3698
4276
|
// orphan work a consult already verified as landed.
|
|
3699
|
-
|
|
3700
|
-
|
|
3701
|
-
|
|
4277
|
+
const { carried: carriedCommits, accounted: accountedCommits } = commitsToCarry.length > 0
|
|
4278
|
+
? await cherryPickCommits(wt, commitsToCarry) : { carried: [], accounted: [] };
|
|
4279
|
+
if (recreating) {
|
|
4280
|
+
journal.append("worktree-recreation", t.id, {
|
|
4281
|
+
attempted: commitsToCarry, carried: carriedCommits, ...(accountedCommits.length > 0 ? { accounted: accountedCommits } : {}),
|
|
4282
|
+
});
|
|
3702
4283
|
}
|
|
3703
|
-
if (recreating)
|
|
3704
|
-
journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
|
|
3705
4284
|
// T2 review (material): harvest eligibility is "does this WORKTREE carry unverified work",
|
|
3706
4285
|
// measured against taskBase — the same base the fast-kill's delta probe and the gates
|
|
3707
4286
|
// themselves use. It was measured against this attempt's post-carry HEAD, which excluded
|
|
@@ -3712,7 +4291,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3712
4291
|
// quota, dead-channel and provider-death all classify the PRE-HARVEST outcome below and fire
|
|
3713
4292
|
// BEFORE the synthesis, so carried-only work reaches gates without bypassing any failover.
|
|
3714
4293
|
const priorNamed = [...new Set([...commitsToCarry, ...carriedCommits])];
|
|
3715
|
-
const presentCommits = new Set(carriedCommits);
|
|
4294
|
+
const presentCommits = new Set([...carriedCommits, ...accountedCommits]);
|
|
3716
4295
|
for (const h of commitsToCarry) {
|
|
3717
4296
|
if (!presentCommits.has(h) && (await shGit(`git merge-base --is-ancestor ${shq(h)} HEAD`, wt)).code === 0) {
|
|
3718
4297
|
presentCommits.add(h);
|
|
@@ -3828,9 +4407,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3828
4407
|
sessionId, attempt, workerDispatchOrdinal, baselineBytes: resumeBaseline?.bytes ?? null,
|
|
3829
4408
|
});
|
|
3830
4409
|
const icmd = retryMode === "resume"
|
|
3831
|
-
? adapter.resumeCommand(sessionId, promptFile, assignment.model)
|
|
4410
|
+
? adapter.resumeCommand(sessionId, promptFile, assignment.model, assignment.effort)
|
|
3832
4411
|
: cfg.visibility.worker === "interactive" && driver.interactive
|
|
3833
|
-
? adapter.interactiveCommand(promptFile, assignment.model)
|
|
4412
|
+
? adapter.interactiveCommand(promptFile, assignment.model, assignment.effort)
|
|
3834
4413
|
: null;
|
|
3835
4414
|
// v1.69 T6: adapters that declare interactiveSeed launch the real TUI and inject the prompt as a
|
|
3836
4415
|
// user turn; they do NOT need the argv-seeding surface that interactiveCommand represents.
|
|
@@ -3852,13 +4431,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3852
4431
|
// Approval can reset the attempt counter while the old pane remains retained.
|
|
3853
4432
|
// Its ownership claim must survive a new engagement reusing the script path.
|
|
3854
4433
|
const groupFile = `${dispatchScript}.${nonce}.pgid`;
|
|
3855
|
-
workerOwners.set(slot, { taskId: t.id, attempt, groupFile });
|
|
4434
|
+
workerOwners.set(slot, { taskId: t.id, attempt, groupFile, marker: dispatchScript, identities: new Map(), descendants: new Map() });
|
|
3856
4435
|
writeFileSync(dispatchScript, [
|
|
3857
4436
|
// Shell startup can swallow the driver's leading cd; the payload owns its checkout too.
|
|
3858
4437
|
`cd ${shq(wt)} || exit 1`,
|
|
4438
|
+
`export ${VITEST_CACHE_ENV}=${shq(worktreeVitestCache(wt))}`,
|
|
3859
4439
|
// A driver may launch inside the daemon's group. That group is never worker-owned.
|
|
3860
4440
|
`worker_pgid=$(ps -o pgid= -p $$ 2>/dev/null); daemon_pgid=$(ps -o pgid= -p ${process.pid} 2>/dev/null)`,
|
|
3861
4441
|
`if [ -n "$worker_pgid" ] && [ -n "$daemon_pgid" ] && [ "$worker_pgid" != "$daemon_pgid" ]; then printf '%s\\n' "$worker_pgid" > ${shq(groupFile)}; fi`,
|
|
4442
|
+
`ps -o sess= -p $$ > ${shq(`${groupFile}.session`)} 2>/dev/null`,
|
|
4443
|
+
`ps -o pid=,lstart= -p $PPID > ${shq(`${groupFile}.parent`)} 2>/dev/null`,
|
|
3862
4444
|
"export BASH_SILENCE_DEPRECATION_WARNING=1",
|
|
3863
4445
|
bannerShell(),
|
|
3864
4446
|
`printf '%s\\n' 'TICKMARKR_DISPATCH_${nonce}'`,
|
|
@@ -3888,6 +4470,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3888
4470
|
journal.append("worker-launch", t.id, {
|
|
3889
4471
|
attempt,
|
|
3890
4472
|
retryMode,
|
|
4473
|
+
// OBS-1109: the attempt's own ownership evidence, so a resumed daemon can harvest it.
|
|
4474
|
+
nonce, dispatchScript,
|
|
3891
4475
|
...(retryMode === "resume" ? { sessionId, workerDispatchOrdinal } : {}),
|
|
3892
4476
|
driver: trackedDriver.id,
|
|
3893
4477
|
slot: { ...slot },
|
|
@@ -3928,7 +4512,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3928
4512
|
}
|
|
3929
4513
|
if (hasSeed || cpuAccountant !== undefined)
|
|
3930
4514
|
return;
|
|
3931
|
-
cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt, () => readOwnedProcessGroup(groupFile));
|
|
4515
|
+
cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt, () => readOwnedProcessGroup(groupFile), workerOwners.get(slot).descendants);
|
|
3932
4516
|
await cpuAccountant.start();
|
|
3933
4517
|
};
|
|
3934
4518
|
const readCpuLeg = () => {
|
|
@@ -4024,15 +4608,49 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4024
4608
|
let deadChannelKilled = false;
|
|
4025
4609
|
let hardTimedOut = false;
|
|
4026
4610
|
let quotaBannerKilled = false;
|
|
4611
|
+
let capacityBannerKilled = false; // OBS-1161: same banner filter and gates, transient outcome
|
|
4027
4612
|
let driverProbeFailed = false;
|
|
4028
4613
|
let heldLegs = [];
|
|
4614
|
+
// OBS-1108: contact uncertainty is LATCHED per owned slot and dispatch (this attempt's slot): one
|
|
4615
|
+
// row when contact is lost and one when it returns, never one per retry. Retries back off while
|
|
4616
|
+
// latched, and a latch held past its deadline throws into the transport-uncertain park, which
|
|
4617
|
+
// preserves the worktree for a recheck. A read that recovers inside the deadline charges nothing.
|
|
4618
|
+
let contactLost;
|
|
4619
|
+
const contactDeadlineMs = contactUnreadableDeadlineMs ?? taskTimeoutMinutes * 60_000;
|
|
4029
4620
|
const noteDriverUnreadable = (error) => {
|
|
4030
4621
|
if (error instanceof HeldProbeExhausted)
|
|
4031
4622
|
throw error;
|
|
4032
4623
|
driverProbeFailed = true;
|
|
4033
4624
|
heldLegs = ["quota:driver-unreadable", "stall:driver-unreadable"];
|
|
4034
|
-
|
|
4035
|
-
|
|
4625
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
4626
|
+
if (contactLost) {
|
|
4627
|
+
contactLost.retries++;
|
|
4628
|
+
const unreadableMs = Date.now() - contactLost.since;
|
|
4629
|
+
if (unreadableMs >= contactDeadlineMs) {
|
|
4630
|
+
throw new ContactUnreadableExhausted(`contact with ${slot.name} unreadable for ${unreadableMs}ms, past its ${contactDeadlineMs}ms deadline, after ${contactLost.retries} retries: ${message}`);
|
|
4631
|
+
}
|
|
4632
|
+
return;
|
|
4633
|
+
}
|
|
4634
|
+
contactLost = { since: Date.now(), retries: 0 };
|
|
4635
|
+
journal.append("contact-unreadable", t.id, { slot: slot.name, attempt, source: "driver", state: "unreadable",
|
|
4636
|
+
concludes: false, deadlineMs: contactDeadlineMs, error: message });
|
|
4637
|
+
};
|
|
4638
|
+
const noteContactReadable = () => {
|
|
4639
|
+
if (!contactLost)
|
|
4640
|
+
return;
|
|
4641
|
+
journal.append("contact-recovered", t.id, { slot: slot.name, attempt, source: "driver", state: "readable",
|
|
4642
|
+
unreadableMs: Date.now() - contactLost.since, retries: contactLost.retries });
|
|
4643
|
+
contactLost = undefined;
|
|
4644
|
+
};
|
|
4645
|
+
// The pause after a slice whose contact probe failed. Unlatched: the unspent slice, at most 1 s.
|
|
4646
|
+
// Latched: a next-contact-at schedule of its own — doubling per retry, bounded by the contact and
|
|
4647
|
+
// hard deadlines and NEVER by the poll slice, which clamps to 100 ms once the stall clock expires
|
|
4648
|
+
// and would otherwise collapse the backoff into ~10 orca calls a second for the rest of the window.
|
|
4649
|
+
const contactPauseMs = (sliceStart, slice, hardDeadline) => {
|
|
4650
|
+
const now = Date.now();
|
|
4651
|
+
if (!contactLost)
|
|
4652
|
+
return Math.max(0, Math.min(slice - (now - sliceStart), CONTACT_BACKOFF_BASE_MS));
|
|
4653
|
+
return Math.max(1, Math.min(CONTACT_BACKOFF_BASE_MS * 2 ** Math.min(contactLost.retries, CONTACT_BACKOFF_DOUBLINGS), contactLost.since + contactDeadlineMs - now, hardDeadline - now));
|
|
4036
4654
|
};
|
|
4037
4655
|
// T2 review: print mode's "the exit marker appeared". Kept apart from `finished` (the
|
|
4038
4656
|
// trailer) but still needed by the keepPanes decision below, whose contract is about a
|
|
@@ -4042,7 +4660,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4042
4660
|
let startupFailure = false;
|
|
4043
4661
|
let startupEvidence;
|
|
4044
4662
|
const startupDetector = new StartupFailureDetector(adapter);
|
|
4663
|
+
// OBS-1169: bootstrap text is evidence only inside the same startup prefix — every sample below
|
|
4664
|
+
// feeds it, so a tool frame, an input box or a saturated read seen at ANY poll closes it for good.
|
|
4665
|
+
const bootstrapDetector = new StartupFailureDetector(adapter, BOOTSTRAP_FAILURE_RE, true);
|
|
4045
4666
|
const startupFailureInWindow = (text, launchedAt) => {
|
|
4667
|
+
bootstrapDetector.sample(text, launchedAt);
|
|
4046
4668
|
startupEvidence ??= startupDetector.sample(text, launchedAt);
|
|
4047
4669
|
return startupEvidence !== undefined;
|
|
4048
4670
|
};
|
|
@@ -4052,8 +4674,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4052
4674
|
? new SettledTrailerTracker(adapter.busyFrameMarkers) : undefined;
|
|
4053
4675
|
const sampleTrailer = (frame) => {
|
|
4054
4676
|
const parsed = adapter.parse(frame, nonce);
|
|
4055
|
-
|
|
4056
|
-
|
|
4677
|
+
// OBS-1175: the parser's cause, never the summary — a parsed trailer may carry a sentinel string.
|
|
4678
|
+
const trailer = new RegExp(trailerPattern(nonce)).test(frame) && parsed.cause === undefined;
|
|
4057
4679
|
return trailerFrames ? trailerFrames.sample(frame, trailer) : trailer;
|
|
4058
4680
|
};
|
|
4059
4681
|
let seedResult;
|
|
@@ -4108,6 +4730,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4108
4730
|
await park(t, "escalation ladder exhausted", "ladder-exhausted", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
4109
4731
|
return false;
|
|
4110
4732
|
};
|
|
4733
|
+
// OBS-1169: the tree this worker receives — commits past it are this attempt's own work; the
|
|
4734
|
+
// carried ones beneath it are not, and must not make a bootstrap death look like work.
|
|
4735
|
+
const launchHead = await gitHead(wt);
|
|
4111
4736
|
try {
|
|
4112
4737
|
if (interactive) {
|
|
4113
4738
|
// v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
|
|
@@ -4155,7 +4780,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4155
4780
|
return;
|
|
4156
4781
|
}
|
|
4157
4782
|
await noteLaunched();
|
|
4158
|
-
|
|
4783
|
+
// OBS-1108: the first owned-slot read joins the contact latch — an unreadable launch read is
|
|
4784
|
+
// retried by the wait loop below under the same deadline, never a task failure on its own.
|
|
4785
|
+
output = await workerTransport.read(slot, PANE_READ_ROWS).catch((error) => { noteDriverUnreadable(error); return ""; });
|
|
4159
4786
|
}
|
|
4160
4787
|
// The returning paths report the same fact on the result; both callbacks land on the one
|
|
4161
4788
|
// latch, and the second is a no-op. A seed that answered is never re-answered by the loop.
|
|
@@ -4242,7 +4869,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4242
4869
|
await sampleContext(); // final poll-seam sample before leaving the wait
|
|
4243
4870
|
break;
|
|
4244
4871
|
}
|
|
4245
|
-
if (adapter.parse(output, nonce).
|
|
4872
|
+
if (adapter.parse(output, nonce).cause === "malformed-verdict")
|
|
4246
4873
|
break;
|
|
4247
4874
|
if (trailerFrames && new RegExp(trailerPattern(nonce)).test(output)) {
|
|
4248
4875
|
// A matching busy frame is still a live turn. Preserve the rolling progress budget.
|
|
@@ -4269,9 +4896,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4269
4896
|
});
|
|
4270
4897
|
}
|
|
4271
4898
|
await armCpuLeg(true);
|
|
4272
|
-
|
|
4273
|
-
if (spent < slice)
|
|
4274
|
-
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
4899
|
+
await new Promise((r) => setTimeout(r, contactPauseMs(sliceStart, slice, hardDeadline)));
|
|
4275
4900
|
continue;
|
|
4276
4901
|
}
|
|
4277
4902
|
// waitOutput is only a wake hint; a driver may miss a trailer already painted.
|
|
@@ -4284,14 +4909,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4284
4909
|
await sampleContext();
|
|
4285
4910
|
break;
|
|
4286
4911
|
}
|
|
4287
|
-
if (adapter.parse(paneText, nonce).
|
|
4912
|
+
if (adapter.parse(paneText, nonce).cause === "malformed-verdict") {
|
|
4288
4913
|
output = paneText;
|
|
4289
4914
|
break;
|
|
4290
4915
|
}
|
|
4291
4916
|
if (driverProbeFailed) {
|
|
4292
|
-
|
|
4293
|
-
if (spent < slice)
|
|
4294
|
-
await new Promise((resolve) => setTimeout(resolve, Math.min(slice - spent, 1_000)));
|
|
4917
|
+
await new Promise((resolve) => setTimeout(resolve, contactPauseMs(sliceStart, slice, hardDeadline)));
|
|
4295
4918
|
continue;
|
|
4296
4919
|
}
|
|
4297
4920
|
if (paneText.length > 0)
|
|
@@ -4352,19 +4975,32 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4352
4975
|
// filtered by identity, never by novelty — a banner already on screen at launch
|
|
4353
4976
|
// classifies exactly like one printed mid-attempt (T1 review: a novelty baseline
|
|
4354
4977
|
// exculpated the launch-throttle case forever).
|
|
4355
|
-
|
|
4978
|
+
// OBS-1161: a transient-capacity banner rides the SAME filter, streak and silence gates,
|
|
4979
|
+
// and concludes the attempt the same way — only its post-loop outcome differs (bounded
|
|
4980
|
+
// requeue on this seat, then same-floor failover, never demotion). Quota wins a tie.
|
|
4981
|
+
const bannerRows = stallSnapshotBannerRows(paneText);
|
|
4982
|
+
const quotaBanner = QUOTA_RE.exec(bannerRows);
|
|
4983
|
+
const bannerMatch = quotaBanner ?? CAPACITY_RE.exec(bannerRows);
|
|
4356
4984
|
if (bannerMatch)
|
|
4357
4985
|
quotaStreak++;
|
|
4358
4986
|
else
|
|
4359
4987
|
quotaStreak = 0;
|
|
4360
|
-
if (bannerMatch && !stallProgress.rowSignalSaturated && !nudgeFailed
|
|
4988
|
+
if (bannerMatch && !quotaBanner && !stallProgress.rowSignalSaturated && !nudgeFailed
|
|
4989
|
+
&& !(driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id) && (!nudged || nudgeDeadline !== undefined))
|
|
4990
|
+
&& readCpuLeg().state === "flat"
|
|
4991
|
+
&& quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
|
|
4992
|
+
capacityBannerKilled = true;
|
|
4993
|
+
journal.append("capacity-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched: bannerMatch[0], excerpt: bannerMatch.input, regex: CAPACITY_RE.source });
|
|
4994
|
+
break;
|
|
4995
|
+
}
|
|
4996
|
+
if (quotaBanner && !stallProgress.rowSignalSaturated && !nudgeFailed
|
|
4361
4997
|
&& !(driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id) && (!nudged || nudgeDeadline !== undefined))
|
|
4362
4998
|
&& readCpuLeg().state === "flat"
|
|
4363
4999
|
&& quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
|
|
4364
5000
|
// no `output =` here: the post-loop no-trailer tail re-reads the pane anyway, so an
|
|
4365
5001
|
// assignment would only split the classification read from the verdict read.
|
|
4366
5002
|
quotaBannerKilled = true;
|
|
4367
|
-
journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched:
|
|
5003
|
+
journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched: quotaBanner[0], excerpt: quotaBanner.input, regex: QUOTA_RE.source });
|
|
4368
5004
|
break;
|
|
4369
5005
|
}
|
|
4370
5006
|
// T1 (OBS-262): the `paged` latch is deleted — status is sampled EVERY slice (and
|
|
@@ -4386,11 +5022,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4386
5022
|
});
|
|
4387
5023
|
}
|
|
4388
5024
|
await armCpuLeg(true);
|
|
4389
|
-
|
|
4390
|
-
if (spent < slice)
|
|
4391
|
-
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
5025
|
+
await new Promise((r) => setTimeout(r, contactPauseMs(sliceStart, slice, hardDeadline)));
|
|
4392
5026
|
continue;
|
|
4393
5027
|
}
|
|
5028
|
+
// OBS-1108: every contact probe of this slice answered (a failed wait or read continues
|
|
5029
|
+
// above), so only here — never after one probe while another stays latched — has contact returned.
|
|
5030
|
+
noteContactReadable();
|
|
4394
5031
|
if (st !== lastStatus) {
|
|
4395
5032
|
lastStatus = st;
|
|
4396
5033
|
journal.append("worker-status", t.id, { slot: slot.name, status: st, attempt });
|
|
@@ -4520,7 +5157,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4520
5157
|
const ref = preservation.ref;
|
|
4521
5158
|
const reason = `worker is unambiguously dead: pane absent, process tree empty, and worktree unchanged; preserved at ${ref}`;
|
|
4522
5159
|
deadWorkerPark = { ref, reason };
|
|
4523
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
5160
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producerNow()) });
|
|
4524
5161
|
noteWorkerLiveness("worker-dead-held", {
|
|
4525
5162
|
slot: slot.name, attempt, reason: "unambiguous-worker-death", ref,
|
|
4526
5163
|
});
|
|
@@ -4709,7 +5346,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4709
5346
|
if (spent < slice)
|
|
4710
5347
|
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
4711
5348
|
}
|
|
4712
|
-
|
|
5349
|
+
// A wait that concludes on a clean slice's read (trailer, exit, startup failure) has contact back too.
|
|
5350
|
+
if (!driverProbeFailed)
|
|
5351
|
+
noteContactReadable();
|
|
5352
|
+
if (!finished && exitCode === null && adapter.parse(output, nonce).cause !== "malformed-verdict") {
|
|
5353
|
+
// OBS-1108: the hard deadline ended an attempt whose contact was still latched (its last
|
|
5354
|
+
// probe failed). That is contact loss, not a stall: keep the one infra park and the preserved
|
|
5355
|
+
// worktree instead of the hard-timeout classification and its repair charge.
|
|
5356
|
+
if (contactLost && driverProbeFailed && Date.now() >= hardDeadline) {
|
|
5357
|
+
throw new ContactUnreadableExhausted(`contact with ${slot.name} unreadable for ${Date.now() - contactLost.since}ms, still latched at the attempt's hard deadline after ${contactLost.retries} retries`);
|
|
5358
|
+
}
|
|
4713
5359
|
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
4714
5360
|
hardTimedOut = Date.now() >= hardDeadline;
|
|
4715
5361
|
timedOut ||= hardTimedOut;
|
|
@@ -4745,7 +5391,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4745
5391
|
const maxSettleRetries = 2;
|
|
4746
5392
|
let settleTries = 0;
|
|
4747
5393
|
settleParsed = adapter.parse(output, nonce);
|
|
4748
|
-
while (settleParsed.
|
|
5394
|
+
while (settleParsed.cause === "malformed-verdict" && settleTries < maxSettleRetries) {
|
|
4749
5395
|
const remaining = settleDeadline - Date.now();
|
|
4750
5396
|
if (remaining <= 0)
|
|
4751
5397
|
break;
|
|
@@ -4755,8 +5401,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4755
5401
|
settleParsed = adapter.parse(output, nonce);
|
|
4756
5402
|
settleTries++;
|
|
4757
5403
|
}
|
|
4758
|
-
if (settleParsed.
|
|
4759
|
-
finished = settleParsed.
|
|
5404
|
+
if (settleParsed.cause !== "malformed-verdict") {
|
|
5405
|
+
finished = settleParsed.cause === undefined && (!trailerFrames || trailerFrames.settled);
|
|
4760
5406
|
}
|
|
4761
5407
|
}
|
|
4762
5408
|
}
|
|
@@ -4849,6 +5495,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4849
5495
|
finally {
|
|
4850
5496
|
await cpuAccountant?.stop();
|
|
4851
5497
|
}
|
|
5498
|
+
// OBS-1169: the final pane read decides, at exit, whether the prefix is still the CLI's own startup.
|
|
5499
|
+
const bootstrapEvidence = processExited ? bootstrapDetector.sample(output, workerLaunchedAt) : undefined;
|
|
4852
5500
|
// SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
|
|
4853
5501
|
// shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
|
|
4854
5502
|
// exit wrapper). Print mode already waits for TICKMARKR_EXIT, which follows that tail; drain
|
|
@@ -4892,9 +5540,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4892
5540
|
try {
|
|
4893
5541
|
await closeSlot(slot);
|
|
4894
5542
|
if (!workerFinished) {
|
|
4895
|
-
const
|
|
5543
|
+
const producer = producerNow();
|
|
5544
|
+
const ref = await preserveWorktree(wt, producer);
|
|
4896
5545
|
if (ref) {
|
|
4897
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
5546
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
4898
5547
|
reapedWorktreeRef = ref;
|
|
4899
5548
|
}
|
|
4900
5549
|
journal.append("worker-reaped-before-harvest", t.id, {
|
|
@@ -4910,6 +5559,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4910
5559
|
}
|
|
4911
5560
|
}
|
|
4912
5561
|
else if (keepOpen && (workerFinished || processExited || driver.id !== "subprocess")) {
|
|
5562
|
+
await reapWorker(slot);
|
|
4913
5563
|
keptSlots.push(slot);
|
|
4914
5564
|
supersededWorkerSlot = slot;
|
|
4915
5565
|
}
|
|
@@ -4964,9 +5614,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4964
5614
|
stallReaps = stallSeat === seat ? stallReaps + 1 : 1;
|
|
4965
5615
|
stallSeat = seat;
|
|
4966
5616
|
if (stallReaps >= 2) {
|
|
4967
|
-
const
|
|
5617
|
+
const producer = producerNow();
|
|
5618
|
+
const ref = await preserveWorktree(wt, producer);
|
|
4968
5619
|
if (ref)
|
|
4969
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
5620
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
4970
5621
|
await park(t, `two consecutive stall reaps without a gate on seat ${seat}`, "stall", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { seat, stallReaps });
|
|
4971
5622
|
return;
|
|
4972
5623
|
}
|
|
@@ -4975,6 +5626,58 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4975
5626
|
stallReaps = 0;
|
|
4976
5627
|
stallSeat = undefined;
|
|
4977
5628
|
}
|
|
5629
|
+
// OBS-1169: a CLI that died in its own bootstrap never read the brief. Startup-owned evidence
|
|
5630
|
+
// (known bootstrap text in the startup prefix StartupFailureDetector owns: inside the window, before
|
|
5631
|
+
// any tool frame or input box, in an unsaturated read), and only when it exited nonzero with no
|
|
5632
|
+
// parsed trailer and no commit of its own: the carried commits
|
|
5633
|
+
// beneath launchHead are an earlier attempt's, and harvesting them would replay that attempt's
|
|
5634
|
+
// red as a second fingerprint occurrence. One same-channel retry after a bounded backoff charges
|
|
5635
|
+
// no attempt, repair or fingerprint; after that the whole adapter shares the broken bootstrap,
|
|
5636
|
+
// so its channels are escalated away from and the existing failover moves on or parks infra.
|
|
5637
|
+
const bootstrap = !workerFinished && exitCode !== null && exitCode !== 0 && bootstrapEvidence
|
|
5638
|
+
&& (await commitsAheadOf(launchHead, wt)).length === 0 ? bootstrapEvidence.matchedBytes : undefined;
|
|
5639
|
+
if (bootstrap) {
|
|
5640
|
+
const from = channelKey(assignment);
|
|
5641
|
+
const retries = requeuesOn("bootstrap-retry", from);
|
|
5642
|
+
if (retries < BOOTSTRAP_RETRY_CAP) {
|
|
5643
|
+
journal.append("bootstrap-retry", t.id, {
|
|
5644
|
+
attempt, retry: retries + 1, of: BOOTSTRAP_RETRY_CAP, channel: from, assignment,
|
|
5645
|
+
matched: bootstrap, exitCode, backoffMs: bootstrapBackoffMs,
|
|
5646
|
+
});
|
|
5647
|
+
await new Promise((r) => setTimeout(r, bootstrapBackoffMs));
|
|
5648
|
+
attempt--;
|
|
5649
|
+
continue;
|
|
5650
|
+
}
|
|
5651
|
+
const reason = `vendor escalated: ${assignment.adapter} bootstrap failed on ${from}`;
|
|
5652
|
+
escalateAdapter(assignment.adapter, reason);
|
|
5653
|
+
const next = failover("bootstrap-failover");
|
|
5654
|
+
// `toAssignment` and `reason` let a resume replay this routing disposition, not only its refund.
|
|
5655
|
+
journal.append("bootstrap-failover", t.id, {
|
|
5656
|
+
from, to: next ? channelKey(next) : null, ...(next ? { toAssignment: next } : {}), matched: bootstrap, retries,
|
|
5657
|
+
escalated: [...vendorEscalated].filter(([, why]) => why === reason).map(([k]) => k), reason,
|
|
5658
|
+
});
|
|
5659
|
+
if (next) {
|
|
5660
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} bootstrap failover`, { tier: "attention" });
|
|
5661
|
+
if (!keepForever) {
|
|
5662
|
+
const idx = keptSlots.indexOf(slot);
|
|
5663
|
+
if (idx >= 0) {
|
|
5664
|
+
keptSlots.splice(idx, 1);
|
|
5665
|
+
try {
|
|
5666
|
+
await closeSlot(slot);
|
|
5667
|
+
}
|
|
5668
|
+
catch { /* cosmetic — reconcile is the backstop */ }
|
|
5669
|
+
}
|
|
5670
|
+
if (supersededWorkerSlot === slot)
|
|
5671
|
+
supersededWorkerSlot = undefined;
|
|
5672
|
+
}
|
|
5673
|
+
assignment = next;
|
|
5674
|
+
tried.push(channelKey(next));
|
|
5675
|
+
attempt--; // the dead bootstrap bought no worker turn
|
|
5676
|
+
continue;
|
|
5677
|
+
}
|
|
5678
|
+
await park(t, `${from} failed at bootstrap after ${retries} same-channel retry and no eligible channel remains`, "infra", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode, { cause: "bootstrap", channel: from, retries });
|
|
5679
|
+
return;
|
|
5680
|
+
}
|
|
4978
5681
|
const preHarvestResult = result;
|
|
4979
5682
|
// T2 (OBS-264): recognize committed no-trailer work BEFORE any no-trailer streak, provider,
|
|
4980
5683
|
// quota or dead-channel routing. Gates never trusted the trailer, so this successful synthesis
|
|
@@ -5011,6 +5714,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5011
5714
|
const quotaMatch = (interactive ? !workerFinished : exitCode !== 0)
|
|
5012
5715
|
? QUOTA_RE.exec(stallSnapshotBannerRows(output))
|
|
5013
5716
|
: null;
|
|
5717
|
+
// OBS-1161: transient capacity reads the SAME tail under the same guards — a live idle banner
|
|
5718
|
+
// and a no-trailer capacity exit classify identically — through the parse-boundary rule that a
|
|
5719
|
+
// parsed verdict (either way) is work, so a trailer QUOTING the phrase never lands here.
|
|
5720
|
+
const capacityMatch = !quotaMatch && (interactive ? !workerFinished : exitCode !== 0)
|
|
5721
|
+
? classifyTransientCapacity({ ...preHarvestResult, raw: stallSnapshotBannerRows(output) })
|
|
5722
|
+
: null;
|
|
5014
5723
|
// Q-1: a graph pin is an operator instruction — a quota match ALONE never overrides it; only a
|
|
5015
5724
|
// channel-attributed error (auth/setup/outage/timeout, the typed dead-channel classes) may.
|
|
5016
5725
|
const pin = t.routingHints?.pin;
|
|
@@ -5021,7 +5730,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5021
5730
|
// demote the pin two tails in and the demotion re-dispatch would move the task off it.
|
|
5022
5731
|
if (preHarvestResult.ok && workerFinished)
|
|
5023
5732
|
noTrailerStreak.set(channelKey(assignment), 0);
|
|
5024
|
-
else if (!workerFinished && cause !== "provider-death" && !pinRefusesQuota) {
|
|
5733
|
+
else if (!workerFinished && cause !== "provider-death" && !pinRefusesQuota && !capacityMatch) {
|
|
5025
5734
|
const ck = channelKey(assignment);
|
|
5026
5735
|
const streak = (noTrailerStreak.get(ck) ?? 0) + 1;
|
|
5027
5736
|
noTrailerStreak.set(ck, streak);
|
|
@@ -5039,6 +5748,48 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5039
5748
|
attempt--;
|
|
5040
5749
|
continue;
|
|
5041
5750
|
}
|
|
5751
|
+
// OBS-1161: transient capacity → bounded same-seat requeue with backoff (no attempt burn, no
|
|
5752
|
+
// consult), then a same-floor failover exactly like quota — but the seat is NEVER demoted or
|
|
5753
|
+
// excluded: it is busy, not dead, and a later task may find it free. The budget is read from
|
|
5754
|
+
// the journal, never a loop-local counter, so a resume continues the count it left off at.
|
|
5755
|
+
if (capacityMatch) {
|
|
5756
|
+
const from = channelKey(assignment);
|
|
5757
|
+
const requeues = requeuesOn("capacity-requeue", from);
|
|
5758
|
+
const source = capacityBannerKilled ? "banner" : "exit";
|
|
5759
|
+
if (requeues < CAPACITY_REQUEUE_CAP) {
|
|
5760
|
+
journal.append("capacity-requeue", t.id, {
|
|
5761
|
+
attempt, requeue: requeues + 1, of: CAPACITY_REQUEUE_CAP, channel: from, assignment,
|
|
5762
|
+
matched: capacityMatch[0], source, backoffMs: capacityBackoffMs,
|
|
5763
|
+
});
|
|
5764
|
+
await new Promise((r) => setTimeout(r, capacityBackoffMs));
|
|
5765
|
+
attempt--;
|
|
5766
|
+
continue;
|
|
5767
|
+
}
|
|
5768
|
+
const next = failover("capacity-failover");
|
|
5769
|
+
journal.append("capacity-failover", t.id, {
|
|
5770
|
+
from, to: next ? channelKey(next) : null, matched: capacityMatch[0], source, requeues, cause: "capacity",
|
|
5771
|
+
});
|
|
5772
|
+
if (next) {
|
|
5773
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} capacity failover`, { tier: "attention" });
|
|
5774
|
+
if (!keepForever) {
|
|
5775
|
+
const idx = keptSlots.indexOf(slot);
|
|
5776
|
+
if (idx >= 0) {
|
|
5777
|
+
keptSlots.splice(idx, 1);
|
|
5778
|
+
try {
|
|
5779
|
+
await closeSlot(slot);
|
|
5780
|
+
}
|
|
5781
|
+
catch { /* cosmetic — reconcile is the backstop */ }
|
|
5782
|
+
}
|
|
5783
|
+
if (supersededWorkerSlot === slot)
|
|
5784
|
+
supersededWorkerSlot = undefined;
|
|
5785
|
+
}
|
|
5786
|
+
assignment = next;
|
|
5787
|
+
tried.push(channelKey(next));
|
|
5788
|
+
continue;
|
|
5789
|
+
}
|
|
5790
|
+
await park(t, `capacity exhausted on ${from} after ${requeues} requeues and no eligible channel at floor`, "quota", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { cause: "capacity", channel: from, requeues });
|
|
5791
|
+
return;
|
|
5792
|
+
}
|
|
5042
5793
|
// quota exhaustion → failover within floor; does NOT consume the ladder (spec §4)
|
|
5043
5794
|
// print: guarded on exit code — exit-0 output that merely MENTIONS "rate limit" must not failover
|
|
5044
5795
|
// interactive: a worker-CLAIMED trailer beats quota mentions; without one, quota text fails over
|
|
@@ -5202,7 +5953,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5202
5953
|
return;
|
|
5203
5954
|
}
|
|
5204
5955
|
const g = e.result;
|
|
5205
|
-
|
|
5956
|
+
activeGatePhases.get(t.id)?.delete(e.gate);
|
|
5957
|
+
classifyInfraResult(g);
|
|
5206
5958
|
inParallelOrder(g.gate, () => {
|
|
5207
5959
|
// GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
|
|
5208
5960
|
// which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
|
|
@@ -5258,6 +6010,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5258
6010
|
meta: Object.fromEntries(Object.entries(data).filter(([key]) => !GATE_TELEMETRY_KEYS.includes(key) && key !== "capacity")),
|
|
5259
6011
|
}));
|
|
5260
6012
|
commits = await commitsAheadOf(taskBase, wt);
|
|
6013
|
+
// OBS-1106: a replayed legacy infra row lacking `infra` is re-classified before it is
|
|
6014
|
+
// re-journaled and before the infra park below reads it.
|
|
6015
|
+
results.forEach(classifyInfraResult);
|
|
5261
6016
|
for (const g of results) {
|
|
5262
6017
|
journal.append("gate-replayed", t.id, {
|
|
5263
6018
|
attempt, priorAttempt: attempt - 1, gate: g.gate, commit: gateSubject.commit,
|
|
@@ -5270,6 +6025,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5270
6025
|
else {
|
|
5271
6026
|
({ results, commits } = await withCommandContext(t.id, async () => runReviewRecovery(t, {
|
|
5272
6027
|
carriedAuthors: await subjectAuthors(journal.read(), t.id, wt, taskBase),
|
|
6028
|
+
producer: producerNow(),
|
|
5273
6029
|
carriedFindings: outstandingFindings,
|
|
5274
6030
|
operatorContext,
|
|
5275
6031
|
worktree: wt, baseRef: taskBase, result, author: assignment,
|
|
@@ -5299,7 +6055,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5299
6055
|
onGate,
|
|
5300
6056
|
})));
|
|
5301
6057
|
}
|
|
5302
|
-
results.forEach(
|
|
6058
|
+
results.forEach(classifyInfraResult);
|
|
5303
6059
|
graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
|
|
5304
6060
|
saveGraph(repoRoot, graph);
|
|
5305
6061
|
if (results.some((g) => g.gate === "test" && !g.pass
|
|
@@ -5336,6 +6092,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5336
6092
|
saveGraph(repoRoot, graph);
|
|
5337
6093
|
journal.append("task-done", t.id, {
|
|
5338
6094
|
attempts: attempt + 1, assignment, taskContentDigest: contentDigest,
|
|
6095
|
+
authors: mergedAuthors(await subjectAuthors(journal.read(), t.id, wt, taskBase)),
|
|
5339
6096
|
});
|
|
5340
6097
|
journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
|
|
5341
6098
|
await trackedDriver.project?.(t.id, "completed");
|
|
@@ -5369,7 +6126,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5369
6126
|
await park(t, gateFailApprovalReason(t.id, unavailableReview.details, true), "gate-fail", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
5370
6127
|
return;
|
|
5371
6128
|
}
|
|
5372
|
-
const infraFailure = results.find((g) => gateFailed(g) && g
|
|
6129
|
+
const infraFailure = results.find((g) => gateFailed(g) && isInfraResult(g));
|
|
5373
6130
|
if (infraFailure) {
|
|
5374
6131
|
await park(t, `${infraFailure.gate}: ${infraFailure.details}${infraFailure.meta?.recoveryBlocked ? ` — ${infraFailure.meta.recoveryBlocked}` : ""}`, "infra", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
5375
6132
|
return;
|
|
@@ -5474,6 +6231,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5474
6231
|
journal.append("consult-verdict", t.id, {
|
|
5475
6232
|
action: v.action, notes: v.notes,
|
|
5476
6233
|
adapter: v.adapter ?? "unknown", model: v.model ?? "unknown", vendor: v.vendor ?? "unknown",
|
|
6234
|
+
...(v.effort ? { effort: v.effort } : {}),
|
|
6235
|
+
...(v.invocations?.length ? { invocations: v.invocations } : {}),
|
|
5477
6236
|
capAdvisory: true,
|
|
5478
6237
|
});
|
|
5479
6238
|
}
|
|
@@ -5586,7 +6345,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5586
6345
|
const holdEndCondition = (closing = true) => {
|
|
5587
6346
|
sweepLiveApprovals();
|
|
5588
6347
|
const freeSlots = Math.max(0, concurrency - inflight.size);
|
|
5589
|
-
const dispatchable =
|
|
6348
|
+
const dispatchable = admissible().filter((t) => !inflight.has(t.id));
|
|
5590
6349
|
if (freeSlots === 0 || dispatchable.length === 0)
|
|
5591
6350
|
return false;
|
|
5592
6351
|
if (closing) {
|
|
@@ -5606,7 +6365,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5606
6365
|
throw new Error(`terminated by ${termSignal}`);
|
|
5607
6366
|
await watchBoard();
|
|
5608
6367
|
sweepLiveApprovals();
|
|
5609
|
-
const ready =
|
|
6368
|
+
const ready = admissible()
|
|
5610
6369
|
.filter((t) => !inflight.has(t.id))
|
|
5611
6370
|
.slice(0, Math.max(0, concurrency - inflight.size));
|
|
5612
6371
|
for (const t of ready) {
|
|
@@ -5627,9 +6386,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5627
6386
|
if (fatalStop.signal.aborted)
|
|
5628
6387
|
return;
|
|
5629
6388
|
const cleanupEvidence = cleanupErrors.length ? { cleanupErrors } : {};
|
|
6389
|
+
if (err instanceof HostDegradedError) {
|
|
6390
|
+
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "host-degraded", ...cleanupEvidence });
|
|
6391
|
+
return;
|
|
6392
|
+
}
|
|
5630
6393
|
if (err instanceof HeldProbeExhausted) {
|
|
5631
6394
|
const wt = worktreePath(repoRoot, `${branch}--${t.id}`);
|
|
5632
|
-
|
|
6395
|
+
const producer = knownProducer(journal.read(), t.id);
|
|
6396
|
+
let ref = await withoutExecutionBudget(() => preserveWorktree(wt, producer));
|
|
5633
6397
|
if (!ref) {
|
|
5634
6398
|
const head = await gitHead(wt);
|
|
5635
6399
|
ref = `refs/tickmarkr/preserved/${head}`;
|
|
@@ -5637,15 +6401,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5637
6401
|
if (saved.code !== 0)
|
|
5638
6402
|
throw new Error(`could not preserve ${head}: ${saved.stderr}`);
|
|
5639
6403
|
}
|
|
5640
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
5641
|
-
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "transport-uncertain", ref, ...cleanupEvidence });
|
|
6404
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
6405
|
+
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: err instanceof ContactUnreadableExhausted ? "contact-unreadable" : "transport-uncertain", ref, ...cleanupEvidence });
|
|
5642
6406
|
return;
|
|
5643
6407
|
}
|
|
5644
6408
|
if (err instanceof ExecutionBudgetExceeded) {
|
|
5645
6409
|
const wt = worktreePath(repoRoot, `${branch}--${t.id}`);
|
|
5646
|
-
const
|
|
6410
|
+
const producer = knownProducer(journal.read(), t.id);
|
|
6411
|
+
const ref = await preserveWorktree(wt, producer);
|
|
5647
6412
|
if (ref)
|
|
5648
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
6413
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
5649
6414
|
const dispatch = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "task-dispatch");
|
|
5650
6415
|
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "execution-budget-exhausted", limitMs: cfg.executionPolicy.taskExecutionLimitMs,
|
|
5651
6416
|
...cleanupEvidence,
|
|
@@ -5679,7 +6444,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5679
6444
|
journal.append("approval-window-start", undefined, { windowMs, parked: [...parked] });
|
|
5680
6445
|
// The narrator may itself append a decision at this boundary.
|
|
5681
6446
|
sweepLiveApprovals();
|
|
5682
|
-
if (
|
|
6447
|
+
if (admissible().length) {
|
|
5683
6448
|
approvalDeadline = undefined;
|
|
5684
6449
|
continue;
|
|
5685
6450
|
}
|
|
@@ -5700,8 +6465,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5700
6465
|
const waiters = [...inflight.values(), aborted];
|
|
5701
6466
|
// A free slot is itself a scheduling boundary: poll the append-only approval stream instead of
|
|
5702
6467
|
// sleeping until an unrelated long-running task settles.
|
|
5703
|
-
// An open board is polled on the same cadence, so a dead cockpit is noticed mid-task
|
|
5704
|
-
|
|
6468
|
+
// An open board is polled on the same cadence, so a dead cockpit is noticed mid-task; so is a
|
|
6469
|
+
// held reservation still awaiting its claim or its adoption.
|
|
6470
|
+
if (inflight.size < concurrency || boardOpened || heldRetry) {
|
|
5705
6471
|
waiters.push(new Promise((wake) => setTimeout(wake, APPROVAL_POLL_MS)));
|
|
5706
6472
|
}
|
|
5707
6473
|
await Promise.race(waiters); // aborted rejects on termination — unwinds the run
|
|
@@ -5729,7 +6495,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5729
6495
|
pending: pendingTasks(graph).map((t) => t.id),
|
|
5730
6496
|
};
|
|
5731
6497
|
fatalPhase = "baseline";
|
|
5732
|
-
await baselineCapture
|
|
6498
|
+
await baselineCapture.catch(error => {
|
|
6499
|
+
// A host park is resumable only after baseline publication. Before that, retain the
|
|
6500
|
+
// fatal baseline failure: resume requires baseline.json and cannot recover this capture.
|
|
6501
|
+
if (!(error instanceof HostDegradedError) || !existsSync(join(journal.dir, "baseline.json")))
|
|
6502
|
+
throw error;
|
|
6503
|
+
});
|
|
5733
6504
|
fatalPhase = "tip-verify";
|
|
5734
6505
|
// OBS-34: post-merge integration-tip verify — strict exit codes, no baseline forgiveness.
|
|
5735
6506
|
const lastMergedTask = [...journal.read()].reverse().find((e) => e.event === "merge" && e.taskId)?.taskId;
|
|
@@ -5739,7 +6510,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5739
6510
|
const checkApprovals = () => {
|
|
5740
6511
|
try {
|
|
5741
6512
|
sweepLiveApprovals();
|
|
5742
|
-
if (
|
|
6513
|
+
if (admissible().length)
|
|
5743
6514
|
controller.abort(cancellation);
|
|
5744
6515
|
if (termSignal)
|
|
5745
6516
|
controller.abort(new Error(`terminated by ${termSignal}`));
|
|
@@ -5803,6 +6574,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5803
6574
|
summary.approvalDisposition = outstanding.length === 0 ? "complete" : "outstanding";
|
|
5804
6575
|
if (outstanding.length > 0)
|
|
5805
6576
|
summary.outstandingApprovals = outstanding;
|
|
6577
|
+
const forgiven = forgivenFingerprints(journal.read());
|
|
6578
|
+
if (forgiven.length > 0)
|
|
6579
|
+
summary.forgiven = forgiven;
|
|
5806
6580
|
journal.append("run-end", undefined, { ...summary });
|
|
5807
6581
|
await reconcile(); // run-end boundary: nothing in flight — full sweep (empty desired set)
|
|
5808
6582
|
// OBS-28: lingering worktrees starve CLI probes; keepPanes:forever is the debug override.
|