tickmarkr 2.5.9 → 2.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code.js +19 -6
- package/dist/adapters/prompt.d.ts +2 -1
- package/dist/adapters/prompt.js +11 -1
- package/dist/adapters/types.d.ts +4 -0
- package/dist/adapters/types.js +10 -0
- package/dist/cli/commands/fleet.js +26 -1
- package/dist/cli/commands/init.js +1 -1
- package/dist/cli/commands/plan.js +7 -3
- package/dist/cli/commands/report.js +11 -1
- package/dist/cli/commands/status.js +13 -3
- package/dist/cli/commands/verify.js +2 -0
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +2 -0
- package/dist/compile/native.js +39 -4
- package/dist/drivers/orca.d.ts +1 -1
- package/dist/drivers/orca.js +27 -5
- package/dist/gates/baseline.d.ts +2 -0
- package/dist/gates/baseline.js +9 -1
- package/dist/gates/cache.d.ts +11 -1
- package/dist/gates/cache.js +35 -15
- package/dist/gates/llm.js +4 -1
- package/dist/gates/review.d.ts +2 -3
- package/dist/gates/review.js +28 -35
- package/dist/gates/run-gates.d.ts +4 -0
- package/dist/gates/run-gates.js +17 -4
- package/dist/gates/test-manifest.d.ts +17 -0
- package/dist/gates/test-manifest.js +109 -9
- package/dist/gates/test-reporter.js +4 -0
- package/dist/graph/graph.d.ts +7 -3
- package/dist/graph/graph.js +21 -5
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/route/preference.d.ts +1 -1
- package/dist/route/preference.js +10 -39
- package/dist/route/role-pick.d.ts +16 -0
- package/dist/route/role-pick.js +15 -0
- package/dist/route/router.d.ts +14 -0
- package/dist/route/router.js +9 -1
- package/dist/run/consult.js +5 -9
- package/dist/run/daemon.d.ts +16 -1
- package/dist/run/daemon.js +554 -77
- package/dist/run/git.d.ts +46 -1
- package/dist/run/git.js +138 -6
- package/dist/run/host-health.d.ts +20 -0
- package/dist/run/host-health.js +64 -0
- package/dist/run/journal.d.ts +1 -1
- package/dist/run/journal.js +51 -6
- package/dist/run/merge.d.ts +1 -1
- package/dist/run/merge.js +30 -6
- package/dist/run/operator-state.d.ts +24 -2
- package/dist/run/operator-state.js +41 -5
- package/dist/run/stall.d.ts +38 -2
- package/dist/run/stall.js +276 -6
- package/dist/tui/cockpit/board.d.ts +1 -1
- package/dist/tui/cockpit/board.js +27 -19
- package/dist/tui/cockpit/derive.d.ts +2 -0
- package/dist/tui/cockpit/derive.js +4 -0
- package/dist/tui/cockpit/live-runtime.d.ts +4 -0
- package/dist/tui/cockpit/live-runtime.js +37 -3
- package/dist/tui/cockpit/live-store.d.ts +3 -0
- package/dist/tui/cockpit/live-store.js +31 -8
- package/dist/tui/cockpit/run-cockpit.js +2 -1
- package/dist/tui/cockpit/run-view.d.ts +2 -4
- package/dist/tui/cockpit/run-view.js +9 -8
- package/package.json +1 -1
- package/schema/rungraph.schema.json +7 -0
- package/skills/tickmarkr-overseer/SKILL.md +76 -14
package/dist/run/daemon.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { HOST_PROBE_SAMPLE_MS, HOST_PROBE_SAMPLES, HostDegradedError, hostDegraded, observeHost } from "./host-health.js";
|
|
2
|
+
import { VITEST_CACHE_ENV, worktreeVitestCache } from "../gates/test-manifest.js";
|
|
1
3
|
import { COMMAND_LEASE_TOKEN_ENV, commandLeaseEnvironment, CommandLeases, currentCommandLeaseToken, isRunnerCommand, runWithCommandLease, withCommandLease } from "./lease.js";
|
|
2
4
|
import { execFileSync, spawn } from "node:child_process";
|
|
3
5
|
import { createHash, randomBytes } from "node:crypto";
|
|
@@ -8,9 +10,9 @@ import { tmpdir } from "node:os";
|
|
|
8
10
|
import { basename, dirname, isAbsolute, join, posix, relative, resolve, sep } from "node:path";
|
|
9
11
|
import { fileURLToPath } from "node:url";
|
|
10
12
|
import { stringify } from "yaml";
|
|
11
|
-
import { classifyDeadChannel, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TRAILER_SUMMARY, writePrompt } from "../adapters/prompt.js";
|
|
13
|
+
import { classifyDeadChannel, classifyTransientCapacity, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TRAILER_SUMMARY, writePrompt } from "../adapters/prompt.js";
|
|
12
14
|
import { allAdapters, getAdapter, probeAll, readDoctor, rolePools } from "../adapters/registry.js";
|
|
13
|
-
import { SettledTrailerTracker, addUsage, channelKey, matchesInputBox, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
15
|
+
import { SettledTrailerTracker, addUsage, CAPACITY_RE, channelKey, matchesInputBox, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
14
16
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
15
17
|
import { collateralHits } from "../compile/collateral.js";
|
|
16
18
|
import { ExecutionPolicySchema, DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
|
|
@@ -20,8 +22,9 @@ import { herdrSealShellPrefix, MAX_BUF, SubprocessDriver } from "../drivers/subp
|
|
|
20
22
|
import { formatOwnedName } from "../drivers/types.js";
|
|
21
23
|
import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
|
|
22
24
|
import { runGates } from "../gates/run-gates.js";
|
|
25
|
+
import { isInfraResult } from "../gates/cache.js";
|
|
23
26
|
import { filesGlob } from "../graph/files-glob.js";
|
|
24
|
-
import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
|
|
27
|
+
import { addEvidence, attributeBlocked, batteryPriority, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
|
|
25
28
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
26
29
|
import { distFingerprint } from "../cli/commands/version.js";
|
|
27
30
|
import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
@@ -29,11 +32,11 @@ import { executionSignal, remainingExecutionMs, withExecutionBudget, withoutExec
|
|
|
29
32
|
import { repairSelectionDecision } from "./repair-selection.js";
|
|
30
33
|
import { failureDisposition, reserveInfrastructureRetry } from "./recovery.js";
|
|
31
34
|
import { runEnvironment } from "./environment.js";
|
|
32
|
-
import { cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, runWithVerificationBudget, sameCapacity, sameVerification, sh, shGit, SUITE_PARENT_ENV, verificationProtocol, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
35
|
+
import { cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, PRESERVE_COMMIT_SUBJECT, PRESERVE_PRODUCER_TRAILER, preserveWorktree, producerFields, resolvedCapacity, runWithForkBudget, runWithVerificationBudget, sameCapacity, sameVerification, sh, shGit, SUITE_PARENT_ENV, verificationProtocol, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
33
36
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
34
37
|
import { classifyRepairDisposition, resolveScopeHints } from "./repair-disposition.js";
|
|
35
38
|
import { applyScopeAmendments, activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingApprovalActions, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
36
|
-
import { isDiffCapPark, pickReviewer } from "../gates/review.js";
|
|
39
|
+
import { gateReviewerFloor, isDiffCapPark, pickReviewer } from "../gates/review.js";
|
|
37
40
|
import { acquireApprovalSerialization, acquireRunLock, isPidLive, releaseRunLock } from "./lock.js";
|
|
38
41
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, reusedTipEvidence, verifyIntegrationTip } from "./merge.js";
|
|
39
42
|
import { climbChannel, marginalCostRank, nextChannel, route } from "../route/router.js";
|
|
@@ -199,7 +202,8 @@ const RECHECK_ENACTMENT = "recheck-battery";
|
|
|
199
202
|
const GATE_SATISFIED_ENACTMENT = "worktree-recreation";
|
|
200
203
|
// Older daemons enacted rechecks through worker-launch, before recheck-battery existed.
|
|
201
204
|
// Preserve that consumption at every scheduling read without changing continuing permission.
|
|
202
|
-
|
|
205
|
+
// OBS-1158: exported so plan folds battery priority through the SAME consumption-aware seam.
|
|
206
|
+
export function pendingDaemonApprovalActions(events) {
|
|
203
207
|
const actions = pendingApprovalActions(events);
|
|
204
208
|
const rechecks = pendingRechecks(events);
|
|
205
209
|
for (const [id, action] of actions) {
|
|
@@ -310,6 +314,17 @@ function classifySignalOnlyTest(g) {
|
|
|
310
314
|
return;
|
|
311
315
|
g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
|
|
312
316
|
}
|
|
317
|
+
/** OBS-1106: ONE infrastructure predicate for classification, persistence and repair admission. A red
|
|
318
|
+
* that carries an infra fingerprint or classification without `meta.infra` (a legacy journal row, an
|
|
319
|
+
* oracle that named only its classification) is still a non-verdict about the work: it is normalized
|
|
320
|
+
* here BEFORE its journal row and before any park/repair seam reads it, so those seams can key on the
|
|
321
|
+
* same `isInfraResult` the gate cache keys on and never on one metadata field alone. */
|
|
322
|
+
function classifyInfraResult(g) {
|
|
323
|
+
classifySignalOnlyTest(g);
|
|
324
|
+
if (g.pass || g.meta?.infra === true || !isInfraResult(g))
|
|
325
|
+
return;
|
|
326
|
+
g.meta = { ...g.meta, classification: "infra", infra: true };
|
|
327
|
+
}
|
|
313
328
|
// v1.85 T3: the gates whose failure IS a deterministic measurement — a machine re-ran a command over a
|
|
314
329
|
// tree and printed the same bytes. Those are the failures the fingerprint cap governs (the ruling names
|
|
315
330
|
// it a "deterministic-gate" cap): a third identical answer to a question already answered twice is the
|
|
@@ -455,6 +470,15 @@ export const setApprovalWindowForTests = (ms) => { approvalWindowMs = ms; };
|
|
|
455
470
|
export const resetApprovalWindowForTests = () => { approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS; };
|
|
456
471
|
const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
|
|
457
472
|
const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
|
|
473
|
+
// OBS-1161: transient capacity ("Selected model is at capacity") — bounded same-seat requeues with a
|
|
474
|
+
// real wait between them, THEN a same-floor failover; never a demotion. The budget is per task and
|
|
475
|
+
// seat, counted from the journal's own capacity-requeue rows so a resume continues it, never restarts it.
|
|
476
|
+
const CAPACITY_REQUEUE_CAP = 2;
|
|
477
|
+
const CAPACITY_BACKOFF_MS = 60_000;
|
|
478
|
+
let capacityBackoffMs = CAPACITY_BACKOFF_MS;
|
|
479
|
+
/** Test seam — shrink the capacity backoff without minute-long sleeps. */
|
|
480
|
+
export function setCapacityBackoffMsForTests(ms) { capacityBackoffMs = ms; }
|
|
481
|
+
export function resetCapacityBackoffMsForTests() { capacityBackoffMs = CAPACITY_BACKOFF_MS; }
|
|
458
482
|
const NO_TRAILER_DEMOTION_STREAK = 2; // OBS-57: consecutive no-trailer windows demote a channel for the rest of the run
|
|
459
483
|
// OBS-117 (v1.71 T6): a worker pane that never prints a byte by T+60s after dispatch is a dead
|
|
460
484
|
// channel — don't burn the full stall window waiting for a silent launch failure. Checked on the
|
|
@@ -734,7 +758,8 @@ function lastVerifyCycle(events) {
|
|
|
734
758
|
* OBS-1077 close rider: what the engagement's LATEST verification cycle proved — its
|
|
735
759
|
* `tip-verify-start` row and what followed, never a commit comparison. Exactly one kind per close:
|
|
736
760
|
* fresh (every tip command ran AND passed), reused (an eligible cached cycle carried forward),
|
|
737
|
-
* failed, or incomplete (cancelled, cut short
|
|
761
|
+
* failed, or incomplete (cancelled, cut short or undelimited). Completed mixed cycles retain each
|
|
762
|
+
* gate's kind beside the whole-cycle reused kind. Nothing before the start row
|
|
738
763
|
* is read, so an unfinished cycle inherits nothing from an earlier green one.
|
|
739
764
|
*/
|
|
740
765
|
export function runEndTipProof(events) {
|
|
@@ -759,14 +784,25 @@ export function runEndTipProof(events) {
|
|
|
759
784
|
if ((events[start].data.cached === true) !== cachedRow || (cachedRow && carried !== rows.length))
|
|
760
785
|
return proof("incomplete");
|
|
761
786
|
// D-131: fresh only when EVERY command executed; one per-gate persisted verdict makes the cycle reused.
|
|
762
|
-
return
|
|
787
|
+
return {
|
|
788
|
+
...proof(carried > 0 ? "reused" : "fresh"),
|
|
789
|
+
gates: gates.map((gate) => ({
|
|
790
|
+
gate: String(gate),
|
|
791
|
+
kind: rows.some((r) => r.data.gate === gate && r.data.cached === true) ? "reused" : "fresh",
|
|
792
|
+
})),
|
|
793
|
+
};
|
|
763
794
|
}
|
|
764
795
|
/** The close notification's statement of the proof — one clause per kind. */
|
|
765
796
|
export function formatTipProof(p) {
|
|
766
797
|
const at = p.tip ? ` ${p.tip.slice(0, 12)}` : "";
|
|
798
|
+
const gateReading = p.gates?.map(({ gate, kind }) => `${gate}: ${kind === "fresh" ? "verified fresh" : "cached (reused) — carried, not re-run"}`).join("; ");
|
|
799
|
+
const suffix = gateReading ? `; ${gateReading}` : "";
|
|
800
|
+
if (p.kind === "reused" && p.gates?.some(({ kind }) => kind === "fresh")) {
|
|
801
|
+
return `tip proof: ${p.kind} — commit${at}; ${gateReading}`;
|
|
802
|
+
}
|
|
767
803
|
switch (p.kind) {
|
|
768
|
-
case "fresh": return `tip proof: fresh — every tip command ran and passed on${at || " the integration tip"}`;
|
|
769
|
-
case "reused": return `tip proof: reused — carried from verified commit${at}, commands not re-run`;
|
|
804
|
+
case "fresh": return `tip proof: fresh — every tip command ran and passed on${at || " the integration tip"}${suffix}`;
|
|
805
|
+
case "reused": return `tip proof: reused — carried from verified commit${at}, commands not re-run${suffix}`;
|
|
770
806
|
case "failed": return `tip proof: failed —${at ? ` commit${at}` : ""} latest verification cycle is red`;
|
|
771
807
|
case "incomplete": return `tip proof: incomplete —${at ? ` commit${at}` : ""} latest verification cycle did not finish`;
|
|
772
808
|
}
|
|
@@ -1476,6 +1512,124 @@ async function cherryPickCommits(wt, commits) {
|
|
|
1476
1512
|
}
|
|
1477
1513
|
return carried;
|
|
1478
1514
|
}
|
|
1515
|
+
const PRESERVED_REF_PREFIX = "refs/tickmarkr/preserved/";
|
|
1516
|
+
const preservedRefOf = (data) => [data.ref, data.preservedRef].find((v) => typeof v === "string" && v.startsWith(PRESERVED_REF_PREFIX));
|
|
1517
|
+
/** The attempt whose worker last launched into the task's current checkout; "unknown" once a
|
|
1518
|
+
* recreation replaced that tree without a new launch (recheck, gate-only restore) or when no
|
|
1519
|
+
* dispatch assignment is on record. Never the newest dispatch by itself. */
|
|
1520
|
+
export function knownProducer(events, taskId) {
|
|
1521
|
+
let dispatched = "unknown";
|
|
1522
|
+
let producer = "unknown";
|
|
1523
|
+
for (const row of events) {
|
|
1524
|
+
if (row.taskId !== taskId)
|
|
1525
|
+
continue;
|
|
1526
|
+
if (row.event === "task-dispatch") {
|
|
1527
|
+
const a = row.data.assignment;
|
|
1528
|
+
dispatched = typeof a?.adapter === "string" && typeof a?.model === "string"
|
|
1529
|
+
? { channel: `${a.adapter}:${a.model}`, attempt: typeof row.data.attempt === "number" ? row.data.attempt : 0 } : "unknown";
|
|
1530
|
+
}
|
|
1531
|
+
else if (row.event === "worker-launch")
|
|
1532
|
+
producer = dispatched;
|
|
1533
|
+
else if (row.event === "worktree-recreation")
|
|
1534
|
+
producer = "unknown";
|
|
1535
|
+
}
|
|
1536
|
+
return producer;
|
|
1537
|
+
}
|
|
1538
|
+
/** Distinct author channels of the subject for task-done/status/board rows: "unknown" replaces every
|
|
1539
|
+
* unresolvable owner (legacy unattributed preservation, dispatch without assignment). */
|
|
1540
|
+
const mergedAuthors = (authors) => [...new Set(authors.map((a) => a.startsWith("unknown author") ? "unknown" : a))].sort();
|
|
1541
|
+
/** Fold lifetime dispatch/carry evidence, independent of attempt budgets and routing exclusions.
|
|
1542
|
+
* Recreation rows name SOURCE hashes, so ownership is joined by stable patch identity. Each
|
|
1543
|
+
* attempt owns only what the next carry (or current subject) adds beyond its own incoming set.
|
|
1544
|
+
* Gate-only recreations do not start an attempt or transfer authorship to the restored seat.
|
|
1545
|
+
*/
|
|
1546
|
+
async function subjectAuthors(events, taskId, wt, base) {
|
|
1547
|
+
const cache = new Map();
|
|
1548
|
+
const patches = async (commits) => {
|
|
1549
|
+
const ids = new Set();
|
|
1550
|
+
for (const commit of commits) {
|
|
1551
|
+
if (!cache.has(commit)) {
|
|
1552
|
+
const diff = await shGit(`git show --format= --binary ${shq(commit)}`, wt);
|
|
1553
|
+
if (diff.code !== 0)
|
|
1554
|
+
throw new Error(`cannot read author patch ${commit}`);
|
|
1555
|
+
const id = execFileSync("git", ["patch-id", "--stable"], {
|
|
1556
|
+
cwd: wt, input: diff.stdout, encoding: "utf8", maxBuffer: 32 * 1024 * 1024,
|
|
1557
|
+
}).trim().split(/\s+/)[0];
|
|
1558
|
+
cache.set(commit, id || undefined); // an empty commit authored no patch
|
|
1559
|
+
}
|
|
1560
|
+
const id = cache.get(commit);
|
|
1561
|
+
if (id)
|
|
1562
|
+
ids.add(id);
|
|
1563
|
+
}
|
|
1564
|
+
return ids;
|
|
1565
|
+
};
|
|
1566
|
+
let current;
|
|
1567
|
+
let previous;
|
|
1568
|
+
let awaitingCarry = false;
|
|
1569
|
+
const owners = new Map();
|
|
1570
|
+
// Preserved engine commits carry their producing attempt on the row; a row without one is legacy
|
|
1571
|
+
// and stays explicitly unknown rather than inheriting the seat that later carried the patch.
|
|
1572
|
+
const preservedOwner = new Map();
|
|
1573
|
+
const attribute = (ids, attempt) => {
|
|
1574
|
+
for (const id of ids) {
|
|
1575
|
+
if (attempt?.incoming.has(id))
|
|
1576
|
+
continue;
|
|
1577
|
+
const authors = owners.get(id) ?? new Set();
|
|
1578
|
+
authors.add(preservedOwner.get(id) ?? attempt?.author ?? "unknown author (missing task-dispatch assignment)");
|
|
1579
|
+
owners.set(id, authors);
|
|
1580
|
+
}
|
|
1581
|
+
};
|
|
1582
|
+
try {
|
|
1583
|
+
for (const row of events) {
|
|
1584
|
+
if (row.taskId !== taskId)
|
|
1585
|
+
continue;
|
|
1586
|
+
const preserved = preservedRefOf(row.data);
|
|
1587
|
+
if (preserved) {
|
|
1588
|
+
// Only an engine preserve commit is owned by its row; a ref naming a worker's own commit keeps
|
|
1589
|
+
// that commit's dispatch attribution.
|
|
1590
|
+
const shown = await shGit(`git show -s ${shq(`--format=%H%n%s%n%(trailers:key=${PRESERVE_PRODUCER_TRAILER},valueonly)`)} ${shq(`${preserved}^{commit}`)}`, wt);
|
|
1591
|
+
const [commit, subject, trailer] = shown.stdout.trim().split("\n");
|
|
1592
|
+
if (shown.code === 0 && commit && subject === PRESERVE_COMMIT_SUBJECT) {
|
|
1593
|
+
// A row that merely mentions the ref (a park naming it) defers to the commit's own trailer;
|
|
1594
|
+
// only a commit with neither is legacy. A known owner is never downgraded by a later mention.
|
|
1595
|
+
const owner = typeof row.data.producer === "string" ? row.data.producer
|
|
1596
|
+
: trailer?.trim().replace(/ attempt \d+$/, "") || "unknown author (legacy unattributed preservation)";
|
|
1597
|
+
for (const id of await patches([commit])) {
|
|
1598
|
+
const prior = preservedOwner.get(id);
|
|
1599
|
+
if (prior === undefined || prior === "unknown" || prior.startsWith("unknown author"))
|
|
1600
|
+
preservedOwner.set(id, owner);
|
|
1601
|
+
}
|
|
1602
|
+
}
|
|
1603
|
+
}
|
|
1604
|
+
if (row.event === "task-dispatch") {
|
|
1605
|
+
previous = current;
|
|
1606
|
+
const a = row.data.assignment;
|
|
1607
|
+
current = { author: typeof a?.adapter === "string" && typeof a?.model === "string"
|
|
1608
|
+
? `${a.adapter}:${a.model}` : "unknown author (missing task-dispatch assignment)", incoming: new Set() };
|
|
1609
|
+
awaitingCarry = true;
|
|
1610
|
+
}
|
|
1611
|
+
else if (row.event === "worktree-recreation") {
|
|
1612
|
+
const carried = await patches(Array.isArray(row.data.carried) ? row.data.carried : []);
|
|
1613
|
+
attribute(carried, awaitingCarry ? previous : current);
|
|
1614
|
+
if (awaitingCarry && current)
|
|
1615
|
+
current.incoming = carried;
|
|
1616
|
+
awaitingCarry = false;
|
|
1617
|
+
}
|
|
1618
|
+
else if (["worker-launch", "worker-result", "gate-result", "task-human"].includes(row.event)) {
|
|
1619
|
+
// The initial checkout has no recreation row. Once work/gates start, a later
|
|
1620
|
+
// recreation is a restore of this attempt, not the input of a new dispatch.
|
|
1621
|
+
awaitingCarry = false;
|
|
1622
|
+
}
|
|
1623
|
+
}
|
|
1624
|
+
const subject = await patches(await commitsAheadOf(base, wt));
|
|
1625
|
+
attribute(subject, current);
|
|
1626
|
+
return [...new Set([...subject].flatMap((id) => [...(owners.get(id) ?? [])]))];
|
|
1627
|
+
}
|
|
1628
|
+
catch (error) {
|
|
1629
|
+
// An unreadable history cannot silently remove an author from the exclusion set.
|
|
1630
|
+
return [`unknown author (${String(error)})`];
|
|
1631
|
+
}
|
|
1632
|
+
}
|
|
1479
1633
|
// T7 (v1.86): a first run-end append that fails AFTER partial bytes landed leaves a torn tail at
|
|
1480
1634
|
// EOF with no newline; a blind retry would glue the run-end line onto those bytes and readJsonl's
|
|
1481
1635
|
// torn-line tolerance would drop the retry too — no terminal record despite a successful write.
|
|
@@ -1623,6 +1777,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1623
1777
|
let fatalPhase = "setup";
|
|
1624
1778
|
let deliberateTermination = false;
|
|
1625
1779
|
const fatalStop = new AbortController();
|
|
1780
|
+
const hostStop = new AbortController();
|
|
1781
|
+
const hostChecks = new Set();
|
|
1626
1782
|
const inflight = new Map();
|
|
1627
1783
|
let retireFatalSlots;
|
|
1628
1784
|
let branch = "";
|
|
@@ -1719,21 +1875,50 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1719
1875
|
if (owner) {
|
|
1720
1876
|
const processGroup = readOwnedProcessGroup(owner.groupFile);
|
|
1721
1877
|
let survivors;
|
|
1878
|
+
const strays = [];
|
|
1879
|
+
let session;
|
|
1722
1880
|
try {
|
|
1723
1881
|
const shared = processGroup !== undefined && [...workerOwners].some(([other, otherOwner]) => other !== slot && liveSlots.has(other) && readOwnedProcessGroup(otherOwner.groupFile) === processGroup);
|
|
1724
1882
|
if (shared)
|
|
1725
1883
|
throw new Error(`worker group ${processGroup} is shared with another live attempt`);
|
|
1726
|
-
|
|
1884
|
+
try {
|
|
1885
|
+
session = readFileSync(`${owner.groupFile}.session`, "utf8").trim() || undefined;
|
|
1886
|
+
}
|
|
1887
|
+
catch { /* pre-launch or older driver */ }
|
|
1888
|
+
let parent;
|
|
1889
|
+
try {
|
|
1890
|
+
const row = /^\s*(\d+)\s+(.+)$/.exec(readFileSync(`${owner.groupFile}.parent`, "utf8").trim());
|
|
1891
|
+
if (row)
|
|
1892
|
+
parent = { pid: Number(row[1]), startedAt: row[2].trim().replace(/\s+/g, " ") };
|
|
1893
|
+
}
|
|
1894
|
+
catch { /* a live dispatch root can still prove its parent */ }
|
|
1895
|
+
const others = [...workerOwners].filter(([other]) => other !== slot && liveSlots.has(other));
|
|
1896
|
+
survivors = await reapOwnedProcessGroup(processGroup, slot.cwd, {
|
|
1897
|
+
marker: owner.marker, session, parent, identities: owner.identities, descendants: owner.descendants, strays,
|
|
1898
|
+
excludedGroups: others.flatMap(([, other]) => {
|
|
1899
|
+
const group = readOwnedProcessGroup(other.groupFile);
|
|
1900
|
+
return group === undefined ? [] : [group];
|
|
1901
|
+
}),
|
|
1902
|
+
excludedWorktrees: others.map(([other]) => other.cwd).filter((cwd) => cwd !== slot.cwd),
|
|
1903
|
+
});
|
|
1727
1904
|
}
|
|
1728
1905
|
catch (error) {
|
|
1729
1906
|
journal.append("worker-process-reaped", owner.taskId, { slot: slot.name, attempt: owner.attempt,
|
|
1730
|
-
processGroup: processGroup ?? null, survivors: null, error: String(error) });
|
|
1907
|
+
processGroup: processGroup ?? null, strays, survivors: null, error: String(error) });
|
|
1908
|
+
workerOwners.delete(slot); // one reap row per attempt: a later close never re-sweeps
|
|
1731
1909
|
throw error;
|
|
1732
1910
|
}
|
|
1733
|
-
reapReports.set(slot, { processGroup: processGroup ?? null, survivors });
|
|
1911
|
+
reapReports.set(slot, { processGroup: processGroup ?? null, strays, survivors });
|
|
1734
1912
|
journal.append("worker-process-reaped", owner.taskId, {
|
|
1735
|
-
slot: slot.name, attempt: owner.attempt, processGroup: processGroup ?? null, survivors,
|
|
1913
|
+
slot: slot.name, attempt: owner.attempt, processGroup: processGroup ?? null, strays, survivors,
|
|
1736
1914
|
});
|
|
1915
|
+
// Every outcome retires the claim: one reap row per attempt, whatever the verdict.
|
|
1916
|
+
workerOwners.delete(slot);
|
|
1917
|
+
// Pre-launch and in-process drivers have no OS dispatch claim; their close owns retirement.
|
|
1918
|
+
// Once any dispatch ownership exists, an unreadable sweep must block the next gate.
|
|
1919
|
+
if (survivors === null && (processGroup !== undefined || session !== undefined || owner.descendants.size > 0)) {
|
|
1920
|
+
throw new Error(`worker group ${processGroup} cleanup unknown`);
|
|
1921
|
+
}
|
|
1737
1922
|
if (survivors && survivors.length > 0)
|
|
1738
1923
|
throw new Error(`worker group ${processGroup} survivors: ${survivors.join(", ")}`);
|
|
1739
1924
|
}
|
|
@@ -1832,6 +2017,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1832
2017
|
// new attempt while this reaper is still closing the old ones.
|
|
1833
2018
|
const termination = new Error(`terminated by ${sig}`);
|
|
1834
2019
|
abortRun(termination);
|
|
2020
|
+
hostStop.abort(termination);
|
|
2021
|
+
await Promise.allSettled(hostChecks);
|
|
1835
2022
|
if (activeTipVerify) {
|
|
1836
2023
|
activeTipVerify.controller.abort(termination);
|
|
1837
2024
|
await activeTipVerify.settled;
|
|
@@ -2024,32 +2211,143 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2024
2211
|
journal.append("watch-board-reopen-failed", undefined, { pane: loss.pane, attempt, error: reopened.error, ...(boardless ? { boardless: true } : {}) });
|
|
2025
2212
|
console.error(`tickmarkr: board not reopened (attempt ${attempt}): ${reopened.error}`);
|
|
2026
2213
|
};
|
|
2027
|
-
//
|
|
2028
|
-
const
|
|
2029
|
-
|
|
2030
|
-
|
|
2214
|
+
// Reference rows are the durable source of truth; old journals establish one on first resume.
|
|
2215
|
+
const priorReference = opts.resume ? [...journal.read()].reverse().find(e => ["host-reference", "host-reference-reset"].includes(e.event)
|
|
2216
|
+
&& typeof e.data.medianMs === "number" && Number.isFinite(e.data.medianMs) && e.data.medianMs > 0) : undefined;
|
|
2217
|
+
let hostReferenceMs = priorReference?.data.medianMs;
|
|
2218
|
+
const hostSignal = (signal) => AbortSignal.any([hostStop.signal, fatalStop.signal, signal, executionSignal()].filter((s) => !!s));
|
|
2219
|
+
const trackHost = (promise) => {
|
|
2220
|
+
hostChecks.add(promise);
|
|
2221
|
+
void promise.finally(() => hostChecks.delete(promise)).catch(() => { });
|
|
2222
|
+
return promise;
|
|
2223
|
+
};
|
|
2224
|
+
const recordReference = (observation) => {
|
|
2225
|
+
if (observation.medianMs === null)
|
|
2226
|
+
return;
|
|
2227
|
+
journal.append("host-reference", undefined, { ...observation });
|
|
2228
|
+
hostReferenceMs = observation.medianMs;
|
|
2229
|
+
};
|
|
2230
|
+
// Health and occupancy share a deadline, but only healthy occupancy gets the bounded fallback.
|
|
2231
|
+
const admitHost = async (taskId, signal, initial, resuming = false, gate, onWait) => {
|
|
2031
2232
|
const startedAt = Date.now();
|
|
2233
|
+
let lastCount = -1;
|
|
2234
|
+
let observation = initial;
|
|
2235
|
+
let first = true;
|
|
2236
|
+
let resetEligible = resuming && hostReferenceMs !== undefined;
|
|
2032
2237
|
for (;;) {
|
|
2033
|
-
signal
|
|
2034
|
-
fatalStop.signal.throwIfAborted();
|
|
2035
|
-
executionSignal()?.throwIfAborted();
|
|
2238
|
+
signal.throwIfAborted();
|
|
2036
2239
|
const count = await liveSuiteCount(repoRoot);
|
|
2037
|
-
|
|
2038
|
-
|
|
2039
|
-
|
|
2040
|
-
|
|
2240
|
+
signal.throwIfAborted();
|
|
2241
|
+
const remaining = suiteWaitCeilingMs - (Date.now() - startedAt);
|
|
2242
|
+
// Reserve a full bounded batch. A partial last batch would manufacture an unreadable
|
|
2243
|
+
// host at an otherwise healthy occupancy deadline. Retain the latest complete observation.
|
|
2244
|
+
const probeBudget = HOST_PROBE_SAMPLE_MS * HOST_PROBE_SAMPLES;
|
|
2245
|
+
if (!observation || (!first && remaining >= probeBudget)) {
|
|
2246
|
+
observation = await observeHost(signal, remaining > 0 ? Math.min(probeBudget, remaining) : undefined);
|
|
2247
|
+
journal.append("host-observation", taskId, { ...observation, referenceMs: hostReferenceMs ?? null, resuming });
|
|
2248
|
+
}
|
|
2249
|
+
first = false;
|
|
2250
|
+
if (hostReferenceMs === undefined && observation.medianMs !== null)
|
|
2251
|
+
recordReference(observation);
|
|
2252
|
+
const degraded = hostDegraded(observation, hostReferenceMs);
|
|
2253
|
+
resetEligible &&= count === 0 && observation.medianMs !== null && degraded;
|
|
2254
|
+
if (!degraded && (count === 0 || resuming))
|
|
2255
|
+
return count;
|
|
2256
|
+
onWait?.();
|
|
2257
|
+
if (degraded)
|
|
2258
|
+
journal.append("host-degraded", taskId, {
|
|
2259
|
+
...observation, referenceMs: hostReferenceMs ?? null, ...(gate ? { gate } : {}), count, waitedMs: Date.now() - startedAt,
|
|
2260
|
+
});
|
|
2261
|
+
else if (count !== lastCount)
|
|
2262
|
+
journal.append("suite-wait", taskId, { count, ...(gate ? { gate } : {}) });
|
|
2041
2263
|
lastCount = count;
|
|
2042
2264
|
if (Date.now() - startedAt >= suiteWaitCeilingMs) {
|
|
2265
|
+
if (degraded) {
|
|
2266
|
+
if (resetEligible) {
|
|
2267
|
+
// Append BOTH medians before adoption. An unreadable sample or live suite vetoes reset.
|
|
2268
|
+
journal.append("host-reference-reset", undefined, {
|
|
2269
|
+
referenceMs: hostReferenceMs, medianMs: observation.medianMs, waitedMs: Date.now() - startedAt,
|
|
2270
|
+
});
|
|
2271
|
+
hostReferenceMs = observation.medianMs;
|
|
2272
|
+
return count;
|
|
2273
|
+
}
|
|
2274
|
+
throw new HostDegradedError("host latency remained degraded or unreadable through suite-wait deadline");
|
|
2275
|
+
}
|
|
2043
2276
|
journal.append("suite-wait-ceiling", taskId, { count, waitedMs: Date.now() - startedAt });
|
|
2044
|
-
|
|
2045
|
-
count, occupancyCap: occupancyCapacity.forkCap, conservativeCap: conservativeCapacity.forkCap,
|
|
2046
|
-
});
|
|
2047
|
-
return await runWithVerificationBudget(conservativeCapacity, execute);
|
|
2277
|
+
return count;
|
|
2048
2278
|
}
|
|
2049
|
-
await new Promise((
|
|
2279
|
+
await new Promise((resolve, reject) => {
|
|
2280
|
+
const abort = () => { clearTimeout(timer); reject(signal.reason); };
|
|
2281
|
+
const timer = setTimeout(() => { signal.removeEventListener("abort", abort); resolve(); }, Math.min(SUITE_POLL_MS, Math.max(0, suiteWaitCeilingMs - (Date.now() - startedAt))));
|
|
2282
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
2283
|
+
if (signal.aborted)
|
|
2284
|
+
abort();
|
|
2285
|
+
});
|
|
2050
2286
|
}
|
|
2051
|
-
|
|
2052
|
-
|
|
2287
|
+
};
|
|
2288
|
+
const initializeHost = () => trackHost((async () => {
|
|
2289
|
+
const signal = hostSignal();
|
|
2290
|
+
const observation = await observeHost(signal);
|
|
2291
|
+
journal.append("host-observation", undefined, { ...observation, referenceMs: hostReferenceMs ?? null, resuming: !!opts.resume });
|
|
2292
|
+
if (hostReferenceMs === undefined)
|
|
2293
|
+
recordReference(observation);
|
|
2294
|
+
if (opts.resume || observation.medianMs === null)
|
|
2295
|
+
await admitHost(undefined, signal, observation, !!opts.resume);
|
|
2296
|
+
})());
|
|
2297
|
+
// Context carries attribution through gates and remote inference; only shell commands acquire.
|
|
2298
|
+
const commandLeases = new CommandLeases();
|
|
2299
|
+
// Only an unambiguous, currently open phase can attribute a command wait. Parallel siblings
|
|
2300
|
+
// deliberately leave gate absent; neither the last phase nor the last red is a safe substitute.
|
|
2301
|
+
const activeGatePhases = new Map();
|
|
2302
|
+
const withCommandContext = (taskId, run, signal = executionSignal()) => {
|
|
2303
|
+
let hostFailure;
|
|
2304
|
+
return runWithCommandLease((_command, execute) => {
|
|
2305
|
+
const active = taskId ? activeGatePhases.get(taskId) : undefined;
|
|
2306
|
+
const gate = active?.size === 1 ? [...active][0] : undefined;
|
|
2307
|
+
let waited = false;
|
|
2308
|
+
return commandLeases.run(async () => {
|
|
2309
|
+
if (hostFailure)
|
|
2310
|
+
throw hostFailure;
|
|
2311
|
+
let count;
|
|
2312
|
+
try {
|
|
2313
|
+
count = await trackHost(admitHost(taskId, hostSignal(signal), undefined, false, gate, () => { waited = true; }));
|
|
2314
|
+
}
|
|
2315
|
+
catch (error) {
|
|
2316
|
+
if (error instanceof HostDegradedError)
|
|
2317
|
+
hostFailure = error;
|
|
2318
|
+
throw error;
|
|
2319
|
+
}
|
|
2320
|
+
if (waited && taskId) {
|
|
2321
|
+
if (gate)
|
|
2322
|
+
journal.phaseStart(taskId, phaseForGate(gate), { gate, admitted: true });
|
|
2323
|
+
else
|
|
2324
|
+
journal.append("suite-admitted", taskId, {});
|
|
2325
|
+
}
|
|
2326
|
+
if (count > 0) {
|
|
2327
|
+
journal.append("suite-budget", taskId, {
|
|
2328
|
+
count, occupancyCap: occupancyCapacity.forkCap, conservativeCap: conservativeCapacity.forkCap,
|
|
2329
|
+
});
|
|
2330
|
+
return await runWithVerificationBudget(conservativeCapacity, execute);
|
|
2331
|
+
}
|
|
2332
|
+
return await execute();
|
|
2333
|
+
}, (count) => {
|
|
2334
|
+
waited = true;
|
|
2335
|
+
journal.append("suite-wait", taskId, { count, ...(gate ? { gate } : {}) });
|
|
2336
|
+
}, SUITE_POLL_MS, signal);
|
|
2337
|
+
}, async () => {
|
|
2338
|
+
try {
|
|
2339
|
+
const result = await run();
|
|
2340
|
+
// Some command oracles turn launch errors into results. Admission failure still parks infra.
|
|
2341
|
+
if (hostFailure)
|
|
2342
|
+
throw hostFailure;
|
|
2343
|
+
return result;
|
|
2344
|
+
}
|
|
2345
|
+
finally {
|
|
2346
|
+
if (taskId)
|
|
2347
|
+
activeGatePhases.delete(taskId);
|
|
2348
|
+
}
|
|
2349
|
+
});
|
|
2350
|
+
};
|
|
2053
2351
|
let baseRef;
|
|
2054
2352
|
let baseline;
|
|
2055
2353
|
let baselinePending = false;
|
|
@@ -2181,6 +2479,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2181
2479
|
});
|
|
2182
2480
|
runStarted = true;
|
|
2183
2481
|
await placeBoard();
|
|
2482
|
+
// A terminal resume with no commands has no execution to admit.
|
|
2483
|
+
if (Object.keys(commands).length > 0 || graph.tasks.some(t => ["pending", "running", "gated"].includes(t.status))) {
|
|
2484
|
+
baselineCapture = initializeHost();
|
|
2485
|
+
void baselineCapture.catch(() => { baselineFailed = true; });
|
|
2486
|
+
}
|
|
2184
2487
|
}
|
|
2185
2488
|
else {
|
|
2186
2489
|
baseRef = await gitHead(repoRoot);
|
|
@@ -2190,7 +2493,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2190
2493
|
// Workers can run beside capture, but no gate may observe an absent or partial baseline.
|
|
2191
2494
|
// Keep publication and warnings inside the same barrier as the suite's final verdict.
|
|
2192
2495
|
baselineCapture = withCommandContext(undefined, async () => {
|
|
2193
|
-
|
|
2496
|
+
// With no commands, capture is already complete: persist it before the probe can wait.
|
|
2497
|
+
// Otherwise a kill during startup can strand a resumable run without baseline.json.
|
|
2498
|
+
const emptyCapture = Object.keys(commands).length === 0 ? await captureBaseline(repoRoot, commands) : undefined;
|
|
2499
|
+
if (emptyCapture)
|
|
2500
|
+
writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(emptyCapture, null, 2));
|
|
2501
|
+
await initializeHost();
|
|
2502
|
+
const captured = emptyCapture ?? await captureBaseline(repoRoot, commands);
|
|
2194
2503
|
writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(captured, null, 2));
|
|
2195
2504
|
baseline = captured;
|
|
2196
2505
|
for (const warning of captured.warnings ?? [])
|
|
@@ -2436,8 +2745,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2436
2745
|
}
|
|
2437
2746
|
return false;
|
|
2438
2747
|
};
|
|
2748
|
+
// OBS-1158: admission reads the same journal snapshot the sweep folded, through the shared seam.
|
|
2749
|
+
let admissionPriority = batteryPriority(startupActions.values());
|
|
2750
|
+
const admissible = () => readyTasks(graph, admissionPriority);
|
|
2439
2751
|
const sweepLiveApprovals = () => {
|
|
2440
2752
|
const events = journal.read();
|
|
2753
|
+
admissionPriority = batteryPriority(pendingDaemonApprovalActions(events).values());
|
|
2441
2754
|
const approvals = events.slice(approvalSweepCursor)
|
|
2442
2755
|
.filter((e) => e.event === "task-approved" && e.taskId);
|
|
2443
2756
|
approvalSweepCursor = events.length;
|
|
@@ -2506,10 +2819,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2506
2819
|
// old path. A preservation failure throws and therefore leaves the old checkout in place. The
|
|
2507
2820
|
// row is deliberately written before the later worktree-recreation row so the journal cannot
|
|
2508
2821
|
// describe only the commits it carried while omitting uncommitted work the removal destroyed.
|
|
2822
|
+
const producerNow = () => knownProducer(journal.read(), t.id);
|
|
2509
2823
|
const recreateTaskWorktree = async (taskBranch, taskBase, priorWt) => {
|
|
2510
|
-
const
|
|
2824
|
+
const producer = producerNow();
|
|
2825
|
+
const ref = await preserveWorktree(priorWt, producer);
|
|
2511
2826
|
if (ref)
|
|
2512
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
2827
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
2513
2828
|
return driver.worktree(repoRoot, taskBranch, taskBase);
|
|
2514
2829
|
};
|
|
2515
2830
|
// A dead worker with a clean checkout still needs a durable recovery handle: there may be no
|
|
@@ -2542,7 +2857,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2542
2857
|
// IS the fix; the tried seed and the attempt-loop start close RES-01/RES-02 alongside it.
|
|
2543
2858
|
//
|
|
2544
2859
|
// v1.24 OBS-18: a task-approved{release:attempt-cap} zeros rs.attempts (fresh budget) and clears
|
|
2545
|
-
// lastAssignment while keeping tried
|
|
2860
|
+
// lastAssignment while keeping tried, so the restore below is skipped — after a
|
|
2546
2861
|
// fresh-budget release, prefer nextChannel over the surviving tried-list so burned channels are
|
|
2547
2862
|
// not re-tried first (consult bans / prior failovers survive the release).
|
|
2548
2863
|
const rs = resume.get(t.id);
|
|
@@ -2563,7 +2878,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2563
2878
|
to: channelKey(assignment),
|
|
2564
2879
|
reason: JSON.stringify(previousHints?.pin) !== JSON.stringify(t.routingHints?.pin) ? "pin changed" : "floor changed",
|
|
2565
2880
|
});
|
|
2566
|
-
|
|
2881
|
+
// OBS-1161: no `attempts > 0` guard — every release already clears lastAssignment in the replay,
|
|
2882
|
+
// and a lastAssignment at zero attempts is a first dispatch whose capacity requeue was taken back:
|
|
2883
|
+
// the seat is still in force, so restore it instead of failing over its own tried[] entry early.
|
|
2884
|
+
if (!hintsChanged && rs?.lastAssignment
|
|
2567
2885
|
&& channels.some((c) => channelKey(c) === channelKey(rs.lastAssignment))
|
|
2568
2886
|
&& !demotedChannels.has(channelKey(rs.lastAssignment))) {
|
|
2569
2887
|
assignment = rs.lastAssignment; // restore the consult-chosen assignment (bypasses route()'s static re-pick)
|
|
@@ -2635,7 +2953,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2635
2953
|
const round = await runGates(task, ctx);
|
|
2636
2954
|
let review = round.results.find((g) => g.gate === "review");
|
|
2637
2955
|
while (review?.meta?.noVerdict === true) {
|
|
2638
|
-
|
|
2956
|
+
// Recovery must honor the gate's floor, including the seats that just failed to
|
|
2957
|
+
// return a verdict; a below-floor alternative cannot replace the infra result.
|
|
2958
|
+
const { floor } = gateReviewerFloor(task, ctx.cfg, ctx.author, ctx.channels, [...(ctx.priorReviewers ?? []), ...badReviewers]);
|
|
2959
|
+
const next = pickReviewer(ctx.author, ctx.channels, badReviewers, cfg.review.prefer ?? [], floor, reviewHistory, undefined, demotedReviewers);
|
|
2639
2960
|
if (!next)
|
|
2640
2961
|
break;
|
|
2641
2962
|
journal.append("review-infra-retry", t.id, { reviewer: channelKey(next), cause: review.meta.cause });
|
|
@@ -2721,6 +3042,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2721
3042
|
dirtyWorktree: true, dirtyPaths: g.meta.paths,
|
|
2722
3043
|
...(typeof g.meta.culprit === "string" ? { culprit: g.meta.culprit } : {}),
|
|
2723
3044
|
...(typeof g.meta.preservedRef === "string" ? { preservedRef: g.meta.preservedRef } : {}),
|
|
3045
|
+
...(typeof g.meta.producer === "string" ? { producer: g.meta.producer } : {}),
|
|
3046
|
+
...(typeof g.meta.producerAttempt === "number" ? { producerAttempt: g.meta.producerAttempt } : {}),
|
|
2724
3047
|
} : {}),
|
|
2725
3048
|
...(cfg.executionPolicy && !g.pass ? { disposition: failureDisposition(g) } : {}),
|
|
2726
3049
|
...Object.fromEntries(["runnerInfraRerun", "hostStarvedRerun", "recoveryBlocked", "failingFiles", "selectionDecision", "failureEvidence"]
|
|
@@ -2744,6 +3067,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2744
3067
|
...(Array.isArray(g.meta?.selectedTests) ? { selectedTests: g.meta.selectedTests } : {}),
|
|
2745
3068
|
...(g.meta?.fullSuite === true ? { fullSuite: true } : {}),
|
|
2746
3069
|
...(g.meta?.reapedGroup === true ? { reapedGroup: true } : {}),
|
|
3070
|
+
...(g.gate === "review" && g.meta?.noEligibleReviewer === true ? {
|
|
3071
|
+
noEligibleReviewer: true, authorVendors: g.meta.authorVendors, unresolvedAuthors: g.meta.unresolvedAuthors,
|
|
3072
|
+
} : {}),
|
|
2747
3073
|
...(g.gate === "review" && typeof g.meta?.reviewer === "string" ? {
|
|
2748
3074
|
reviewer: g.meta.reviewer,
|
|
2749
3075
|
...Object.fromEntries(["cause", "seatAuthoredBytes", "bytes", "rawPath", "briefPath", "timeoutMs", "unparseable", "noVerdict", "resolved", "reraised", "reviewerFloor", "reviewerFloorCause", "reviewerTier"]
|
|
@@ -2771,13 +3097,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2771
3097
|
// every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
|
|
2772
3098
|
// seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
|
|
2773
3099
|
...gateMeasurement(g.meta),
|
|
2774
|
-
//
|
|
2775
|
-
...(g.
|
|
2776
|
-
|
|
2777
|
-
|
|
2778
|
-
|
|
2779
|
-
|
|
2780
|
-
}
|
|
3100
|
+
// Carried receipts retain their original invocation and artifact root.
|
|
3101
|
+
...(g.evidenceReceipt ? { evidenceReceipt: g.evidenceReceipt } : {}),
|
|
3102
|
+
...(g.evidenceReceipts ? { evidenceReceipts: g.evidenceReceipts } : {}),
|
|
3103
|
+
...(g.meta?.reused ? {
|
|
3104
|
+
reused: true,
|
|
3105
|
+
...(g.originRunRoot ? { originRunRoot: g.originRunRoot } : {}),
|
|
3106
|
+
} : Object.fromEntries(["nonce", "stdoutPath", "stderrPath", "classification"]
|
|
3107
|
+
.filter(key => g.meta?.[key] !== undefined).map(key => [key, g.meta[key]]))),
|
|
2781
3108
|
// T7: the capacity the gate's own command child ran under, lifted verbatim from the result
|
|
2782
3109
|
// the battery produced — read where the shell built that child's environment, never
|
|
2783
3110
|
// re-derived from the run's own budget, which would answer a different number than the
|
|
@@ -2820,6 +3147,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2820
3147
|
const parallelPending = new Set();
|
|
2821
3148
|
let heldParallel;
|
|
2822
3149
|
const notePhaseStart = (e) => {
|
|
3150
|
+
const active = activeGatePhases.get(t.id) ?? new Set();
|
|
3151
|
+
active.add(e.gate);
|
|
3152
|
+
activeGatePhases.set(t.id, active);
|
|
2823
3153
|
if (e.parentAt !== undefined)
|
|
2824
3154
|
parallelPending.add(e.gate);
|
|
2825
3155
|
};
|
|
@@ -2920,6 +3250,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2920
3250
|
}
|
|
2921
3251
|
return next;
|
|
2922
3252
|
};
|
|
3253
|
+
// OBS-1161: capacity requeues spent on a seat for this task since its last operator release —
|
|
3254
|
+
// journal-derived so the budget survives a resume instead of restarting with the process.
|
|
3255
|
+
const capacityRequeuesOn = (channel) => {
|
|
3256
|
+
const rows = journal.read().filter((e) => e.taskId === t.id);
|
|
3257
|
+
const since = rows.map((e) => e.event).lastIndexOf("task-approved");
|
|
3258
|
+
return rows.slice(since + 1).filter((e) => e.event === "capacity-requeue" && e.data.channel === channel).length;
|
|
3259
|
+
};
|
|
2923
3260
|
// OBS-202 (operator law: "you can spawn as many as you want"): channels are session FACTORIES,
|
|
2924
3261
|
// not consumed seats — a tried channel can always host a fresh worker session, and a fresh
|
|
2925
3262
|
// session carries none of the failed attempt's baggage. When the untried pool is empty, recycle
|
|
@@ -3005,7 +3342,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3005
3342
|
// OBS-130/T15: both gate-resume paths consume the persisted task branch with no worker dispatch.
|
|
3006
3343
|
// An operator approval skips its exact failed gate by authority; observed results skip only the
|
|
3007
3344
|
// contiguous green prefix whose recorded commit is still the task branch tip.
|
|
3008
|
-
|
|
3345
|
+
let satisfiedGate = satisfiedGates.get(t.id);
|
|
3009
3346
|
const replayedGates = fundedRerun ? undefined : replayedGateResults.get(t.id);
|
|
3010
3347
|
const recheck = approvalAction?.authority === "battery";
|
|
3011
3348
|
resumeGateReplay: if (satisfiedGate || replayedGates || recheck) {
|
|
@@ -3042,6 +3379,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3042
3379
|
// than against the stale worktree whose task-only history cannot see newly merged dependencies.
|
|
3043
3380
|
const currentTaskTip = await gitHead(wt);
|
|
3044
3381
|
const currentTaskSubject = await gateCommitSubject(taskBase, currentTaskTip, wt);
|
|
3382
|
+
// Recheck carries only review authority for the exact subject being gated, never tool greens.
|
|
3383
|
+
if (recheck) {
|
|
3384
|
+
satisfiedGate = journal.replaySatisfiedGates(new Map([[t.id, currentTaskSubject]])).get(t.id);
|
|
3385
|
+
}
|
|
3045
3386
|
journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
|
|
3046
3387
|
// OBS-212: same fail-closed rule as the dispatch path — but this path is worse, because it runs
|
|
3047
3388
|
// ONLY the gates after the approved one and then MERGES. T3 took it on run-20260728-110135:
|
|
@@ -3115,7 +3456,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3115
3456
|
const reused = [];
|
|
3116
3457
|
let remainingGates;
|
|
3117
3458
|
if (recheck) {
|
|
3118
|
-
remainingGates =
|
|
3459
|
+
remainingGates = declaredGates.filter((gate) => gate !== satisfiedGate);
|
|
3119
3460
|
}
|
|
3120
3461
|
else if (satisfiedGate) {
|
|
3121
3462
|
remainingGates = t.gates.filter((gate) => {
|
|
@@ -3212,10 +3553,19 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3212
3553
|
// deterministic-fingerprint occurrence/review round for budget accounting.
|
|
3213
3554
|
...(!satisfiedGate && !recheck ? { replayMeasurement: true } : {}),
|
|
3214
3555
|
};
|
|
3556
|
+
if (recheck && satisfiedGate === "review" && gateSubject.commit !== currentTaskSubject) {
|
|
3557
|
+
satisfiedGate = undefined;
|
|
3558
|
+
resumedTask.gates = declaredGates;
|
|
3559
|
+
}
|
|
3560
|
+
if (recheck && satisfiedGate === "review" && !resumedTask.gates.includes("review")) {
|
|
3561
|
+
journal.append("gate-waiver-carried", t.id, { gate: "review", commit: gateSubject.commit, carried: true, release: RECHECK_RELEASE });
|
|
3562
|
+
}
|
|
3215
3563
|
await trackedDriver.project?.(t.id, "in-review");
|
|
3216
3564
|
await waitForBaseline(t.id);
|
|
3217
3565
|
journal.phaseStart(t.id, "gates");
|
|
3218
|
-
const { results } = await withCommandContext(t.id, () => runReviewRecovery(resumedTask, {
|
|
3566
|
+
const { results } = await withCommandContext(t.id, async () => runReviewRecovery(resumedTask, {
|
|
3567
|
+
carriedAuthors: await subjectAuthors(journal.read(), t.id, wt, taskBase),
|
|
3568
|
+
producer: producerNow(),
|
|
3219
3569
|
carriedFindings: outstandingReviewFindings(journal.read(), t.id),
|
|
3220
3570
|
operatorContext,
|
|
3221
3571
|
worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
|
|
@@ -3234,7 +3584,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3234
3584
|
excludeReviewers: badReviewers,
|
|
3235
3585
|
reviewHistory, demotedReviewers, priorReviewers: taskReviewers(t.id),
|
|
3236
3586
|
// Leg-2 (OBS-1052): the run-scoped two-strike tally — without it retirement is inert and a
|
|
3237
|
-
// flaking seat is re-asked on every task.
|
|
3587
|
+
// flaking seat is re-asked on every task.
|
|
3238
3588
|
reviewNoVerdicts,
|
|
3239
3589
|
recheck, // OBS-1055: a recheck discards cached reds — the battery re-measures what the operator questioned
|
|
3240
3590
|
onGate: async (e) => {
|
|
@@ -3248,7 +3598,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3248
3598
|
return;
|
|
3249
3599
|
}
|
|
3250
3600
|
const g = e.result;
|
|
3251
|
-
|
|
3601
|
+
activeGatePhases.get(t.id)?.delete(e.gate);
|
|
3602
|
+
classifyInfraResult(g);
|
|
3252
3603
|
inParallelOrder(g.gate, () => {
|
|
3253
3604
|
journalGateResult(g);
|
|
3254
3605
|
noteReviewRetry(g);
|
|
@@ -3259,7 +3610,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3259
3610
|
});
|
|
3260
3611
|
},
|
|
3261
3612
|
}, false));
|
|
3262
|
-
results.forEach(
|
|
3613
|
+
results.forEach(classifyInfraResult);
|
|
3263
3614
|
if (pendingDaemonApprovalActions(journal.read()).get(t.id)?.authority === "battery") {
|
|
3264
3615
|
journal.append("recheck-battery", t.id, {
|
|
3265
3616
|
commit: gateSubject.commit,
|
|
@@ -3271,7 +3622,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3271
3622
|
graph = addEvidence(graph, t.id, { commits: approvedCommits, gateResults: results });
|
|
3272
3623
|
saveGraph(repoRoot, graph);
|
|
3273
3624
|
if (!results.every(gateSatisfied)) {
|
|
3274
|
-
const
|
|
3625
|
+
const unavailableReview = results.find((g) => gateFailed(g) && g.meta?.noEligibleReviewer === true);
|
|
3626
|
+
if (unavailableReview) {
|
|
3627
|
+
await park(t, gateFailApprovalReason(t.id, unavailableReview.details, true), "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
3628
|
+
return;
|
|
3629
|
+
}
|
|
3630
|
+
// OBS-1106: same predicate as classification — an infra replay is parked, never repaired.
|
|
3631
|
+
const infra = results.find((g) => gateFailed(g) && isInfraResult(g));
|
|
3275
3632
|
if (infra) {
|
|
3276
3633
|
await park(t, `${infra.gate}: ${infra.details}`, "infra", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
3277
3634
|
return;
|
|
@@ -3299,7 +3656,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3299
3656
|
// Observed green gates are only measurements. If the resumed suffix is red, preserve that
|
|
3300
3657
|
// result in the journal and return to the ordinary attempt/consult ladder, which rebuilds
|
|
3301
3658
|
// feedback from those rows. Only an operator-authorized gate release parks on a new red.
|
|
3302
|
-
if (!satisfiedGate) {
|
|
3659
|
+
if (!satisfiedGate || recheck) {
|
|
3303
3660
|
// OBS-1055: a recheck red funds a repair of the pin's own work. A pinned task's repair sits
|
|
3304
3661
|
// on the pin (OBS-1034's exemption keeps the tried list from excluding it); a pin the fleet
|
|
3305
3662
|
// cannot seat parks naming the pin — never the ladder.
|
|
@@ -3354,6 +3711,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3354
3711
|
saveGraph(repoRoot, graph);
|
|
3355
3712
|
journal.append("task-done", t.id, {
|
|
3356
3713
|
attempts: rs?.attempts ?? 0, assignment: gateAuthor, taskContentDigest: contentDigest,
|
|
3714
|
+
authors: mergedAuthors(await subjectAuthors(journal.read(), t.id, wt, taskBase)),
|
|
3357
3715
|
});
|
|
3358
3716
|
journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
|
|
3359
3717
|
await trackedDriver.project?.(t.id, "completed");
|
|
@@ -3706,6 +4064,20 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3706
4064
|
Object.defineProperty(workerSlotOpts, "agent", { value: assignment.adapter });
|
|
3707
4065
|
const slot = await trackedDriver.slot(wt, `${t.id}-worker-${assignment.adapter}-a${attempt}-${runTag}`, workerSlotOpts);
|
|
3708
4066
|
const sessionId = retryMode === "resume" ? priorSession.id : slot.name;
|
|
4067
|
+
const readResumeTranscript = () => {
|
|
4068
|
+
try {
|
|
4069
|
+
const sample = adapter.readSessionTranscript?.({ cwd: wt, id: sessionId });
|
|
4070
|
+
return sample && Number.isSafeInteger(sample.bytes) && sample.bytes >= 0 ? sample : null;
|
|
4071
|
+
}
|
|
4072
|
+
catch {
|
|
4073
|
+
return null;
|
|
4074
|
+
}
|
|
4075
|
+
};
|
|
4076
|
+
const resumeBaseline = retryMode === "resume" ? readResumeTranscript() : null;
|
|
4077
|
+
if (retryMode === "resume")
|
|
4078
|
+
journal.append("worker-resume-requested", t.id, {
|
|
4079
|
+
sessionId, attempt, workerDispatchOrdinal, baselineBytes: resumeBaseline?.bytes ?? null,
|
|
4080
|
+
});
|
|
3709
4081
|
const icmd = retryMode === "resume"
|
|
3710
4082
|
? adapter.resumeCommand(sessionId, promptFile, assignment.model)
|
|
3711
4083
|
: cfg.visibility.worker === "interactive" && driver.interactive
|
|
@@ -3731,11 +4103,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3731
4103
|
// Approval can reset the attempt counter while the old pane remains retained.
|
|
3732
4104
|
// Its ownership claim must survive a new engagement reusing the script path.
|
|
3733
4105
|
const groupFile = `${dispatchScript}.${nonce}.pgid`;
|
|
3734
|
-
workerOwners.set(slot, { taskId: t.id, attempt, groupFile });
|
|
4106
|
+
workerOwners.set(slot, { taskId: t.id, attempt, groupFile, marker: dispatchScript, identities: new Map(), descendants: new Map() });
|
|
3735
4107
|
writeFileSync(dispatchScript, [
|
|
4108
|
+
// Shell startup can swallow the driver's leading cd; the payload owns its checkout too.
|
|
4109
|
+
`cd ${shq(wt)} || exit 1`,
|
|
4110
|
+
`export ${VITEST_CACHE_ENV}=${shq(worktreeVitestCache(wt))}`,
|
|
3736
4111
|
// A driver may launch inside the daemon's group. That group is never worker-owned.
|
|
3737
4112
|
`worker_pgid=$(ps -o pgid= -p $$ 2>/dev/null); daemon_pgid=$(ps -o pgid= -p ${process.pid} 2>/dev/null)`,
|
|
3738
4113
|
`if [ -n "$worker_pgid" ] && [ -n "$daemon_pgid" ] && [ "$worker_pgid" != "$daemon_pgid" ]; then printf '%s\\n' "$worker_pgid" > ${shq(groupFile)}; fi`,
|
|
4114
|
+
`ps -o sess= -p $$ > ${shq(`${groupFile}.session`)} 2>/dev/null`,
|
|
4115
|
+
`ps -o pid=,lstart= -p $PPID > ${shq(`${groupFile}.parent`)} 2>/dev/null`,
|
|
3739
4116
|
"export BASH_SILENCE_DEPRECATION_WARNING=1",
|
|
3740
4117
|
bannerShell(),
|
|
3741
4118
|
`printf '%s\\n' 'TICKMARKR_DISPATCH_${nonce}'`,
|
|
@@ -3765,6 +4142,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3765
4142
|
journal.append("worker-launch", t.id, {
|
|
3766
4143
|
attempt,
|
|
3767
4144
|
retryMode,
|
|
4145
|
+
...(retryMode === "resume" ? { sessionId, workerDispatchOrdinal } : {}),
|
|
3768
4146
|
driver: trackedDriver.id,
|
|
3769
4147
|
slot: { ...slot },
|
|
3770
4148
|
workspace: driver.id === "herdr" ? process.env.HERDR_WORKSPACE_ID : undefined,
|
|
@@ -3804,7 +4182,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3804
4182
|
}
|
|
3805
4183
|
if (hasSeed || cpuAccountant !== undefined)
|
|
3806
4184
|
return;
|
|
3807
|
-
cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt, () => readOwnedProcessGroup(groupFile));
|
|
4185
|
+
cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt, () => readOwnedProcessGroup(groupFile), workerOwners.get(slot).descendants);
|
|
3808
4186
|
await cpuAccountant.start();
|
|
3809
4187
|
};
|
|
3810
4188
|
const readCpuLeg = () => {
|
|
@@ -3900,6 +4278,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3900
4278
|
let deadChannelKilled = false;
|
|
3901
4279
|
let hardTimedOut = false;
|
|
3902
4280
|
let quotaBannerKilled = false;
|
|
4281
|
+
let capacityBannerKilled = false; // OBS-1161: same banner filter and gates, transient outcome
|
|
3903
4282
|
let driverProbeFailed = false;
|
|
3904
4283
|
let heldLegs = [];
|
|
3905
4284
|
const noteDriverUnreadable = (error) => {
|
|
@@ -4228,19 +4607,32 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4228
4607
|
// filtered by identity, never by novelty — a banner already on screen at launch
|
|
4229
4608
|
// classifies exactly like one printed mid-attempt (T1 review: a novelty baseline
|
|
4230
4609
|
// exculpated the launch-throttle case forever).
|
|
4231
|
-
|
|
4610
|
+
// OBS-1161: a transient-capacity banner rides the SAME filter, streak and silence gates,
|
|
4611
|
+
// and concludes the attempt the same way — only its post-loop outcome differs (bounded
|
|
4612
|
+
// requeue on this seat, then same-floor failover, never demotion). Quota wins a tie.
|
|
4613
|
+
const bannerRows = stallSnapshotBannerRows(paneText);
|
|
4614
|
+
const quotaBanner = QUOTA_RE.exec(bannerRows);
|
|
4615
|
+
const bannerMatch = quotaBanner ?? CAPACITY_RE.exec(bannerRows);
|
|
4232
4616
|
if (bannerMatch)
|
|
4233
4617
|
quotaStreak++;
|
|
4234
4618
|
else
|
|
4235
4619
|
quotaStreak = 0;
|
|
4236
|
-
if (bannerMatch && !stallProgress.rowSignalSaturated && !nudgeFailed
|
|
4620
|
+
if (bannerMatch && !quotaBanner && !stallProgress.rowSignalSaturated && !nudgeFailed
|
|
4621
|
+
&& !(driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id) && (!nudged || nudgeDeadline !== undefined))
|
|
4622
|
+
&& readCpuLeg().state === "flat"
|
|
4623
|
+
&& quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
|
|
4624
|
+
capacityBannerKilled = true;
|
|
4625
|
+
journal.append("capacity-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched: bannerMatch[0], excerpt: bannerMatch.input, regex: CAPACITY_RE.source });
|
|
4626
|
+
break;
|
|
4627
|
+
}
|
|
4628
|
+
if (quotaBanner && !stallProgress.rowSignalSaturated && !nudgeFailed
|
|
4237
4629
|
&& !(driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id) && (!nudged || nudgeDeadline !== undefined))
|
|
4238
4630
|
&& readCpuLeg().state === "flat"
|
|
4239
4631
|
&& quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
|
|
4240
4632
|
// no `output =` here: the post-loop no-trailer tail re-reads the pane anyway, so an
|
|
4241
4633
|
// assignment would only split the classification read from the verdict read.
|
|
4242
4634
|
quotaBannerKilled = true;
|
|
4243
|
-
journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched:
|
|
4635
|
+
journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched: quotaBanner[0], excerpt: quotaBanner.input, regex: QUOTA_RE.source });
|
|
4244
4636
|
break;
|
|
4245
4637
|
}
|
|
4246
4638
|
// T1 (OBS-262): the `paged` latch is deleted — status is sampled EVERY slice (and
|
|
@@ -4396,7 +4788,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4396
4788
|
const ref = preservation.ref;
|
|
4397
4789
|
const reason = `worker is unambiguously dead: pane absent, process tree empty, and worktree unchanged; preserved at ${ref}`;
|
|
4398
4790
|
deadWorkerPark = { ref, reason };
|
|
4399
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
4791
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producerNow()) });
|
|
4400
4792
|
noteWorkerLiveness("worker-dead-held", {
|
|
4401
4793
|
slot: slot.name, attempt, reason: "unambiguous-worker-death", ref,
|
|
4402
4794
|
});
|
|
@@ -4768,9 +5160,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4768
5160
|
try {
|
|
4769
5161
|
await closeSlot(slot);
|
|
4770
5162
|
if (!workerFinished) {
|
|
4771
|
-
const
|
|
5163
|
+
const producer = producerNow();
|
|
5164
|
+
const ref = await preserveWorktree(wt, producer);
|
|
4772
5165
|
if (ref) {
|
|
4773
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
5166
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
4774
5167
|
reapedWorktreeRef = ref;
|
|
4775
5168
|
}
|
|
4776
5169
|
journal.append("worker-reaped-before-harvest", t.id, {
|
|
@@ -4786,12 +5179,24 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4786
5179
|
}
|
|
4787
5180
|
}
|
|
4788
5181
|
else if (keepOpen && (workerFinished || processExited || driver.id !== "subprocess")) {
|
|
5182
|
+
await reapWorker(slot);
|
|
4789
5183
|
keptSlots.push(slot);
|
|
4790
5184
|
supersededWorkerSlot = slot;
|
|
4791
5185
|
}
|
|
4792
5186
|
else {
|
|
4793
5187
|
await closeSlot(slot);
|
|
4794
5188
|
}
|
|
5189
|
+
if (retryMode === "resume") {
|
|
5190
|
+
const transcript = readResumeTranscript();
|
|
5191
|
+
const identity = resumeBaseline === null || transcript === null
|
|
5192
|
+
? "unknown" : transcript.bytes > resumeBaseline.bytes ? "confirmed" : "unconfirmed";
|
|
5193
|
+
journal.append("worker-resume-identity", t.id, {
|
|
5194
|
+
sessionId, attempt, workerDispatchOrdinal, identity,
|
|
5195
|
+
baselineBytes: resumeBaseline?.bytes ?? null, observedBytes: transcript?.bytes ?? null,
|
|
5196
|
+
// This is evidence of file growth, not an independent runtime identity handshake.
|
|
5197
|
+
assumption: "external runtime appends to the requested session's own transcript",
|
|
5198
|
+
});
|
|
5199
|
+
}
|
|
4795
5200
|
// SPEND-01: usage from the harness's own cwd-keyed structured store, read POST-HOC from disk —
|
|
4796
5201
|
// `wt` is this task's private worktree, so the path is unique; the read is sliced to records
|
|
4797
5202
|
// stamped at/after this attempt's dispatch instant. Never the harvested pane text, never the
|
|
@@ -4829,9 +5234,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4829
5234
|
stallReaps = stallSeat === seat ? stallReaps + 1 : 1;
|
|
4830
5235
|
stallSeat = seat;
|
|
4831
5236
|
if (stallReaps >= 2) {
|
|
4832
|
-
const
|
|
5237
|
+
const producer = producerNow();
|
|
5238
|
+
const ref = await preserveWorktree(wt, producer);
|
|
4833
5239
|
if (ref)
|
|
4834
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
5240
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
4835
5241
|
await park(t, `two consecutive stall reaps without a gate on seat ${seat}`, "stall", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { seat, stallReaps });
|
|
4836
5242
|
return;
|
|
4837
5243
|
}
|
|
@@ -4876,6 +5282,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4876
5282
|
const quotaMatch = (interactive ? !workerFinished : exitCode !== 0)
|
|
4877
5283
|
? QUOTA_RE.exec(stallSnapshotBannerRows(output))
|
|
4878
5284
|
: null;
|
|
5285
|
+
// OBS-1161: transient capacity reads the SAME tail under the same guards — a live idle banner
|
|
5286
|
+
// and a no-trailer capacity exit classify identically — through the parse-boundary rule that a
|
|
5287
|
+
// parsed verdict (either way) is work, so a trailer QUOTING the phrase never lands here.
|
|
5288
|
+
const capacityMatch = !quotaMatch && (interactive ? !workerFinished : exitCode !== 0)
|
|
5289
|
+
? classifyTransientCapacity({ ...preHarvestResult, raw: stallSnapshotBannerRows(output) })
|
|
5290
|
+
: null;
|
|
4879
5291
|
// Q-1: a graph pin is an operator instruction — a quota match ALONE never overrides it; only a
|
|
4880
5292
|
// channel-attributed error (auth/setup/outage/timeout, the typed dead-channel classes) may.
|
|
4881
5293
|
const pin = t.routingHints?.pin;
|
|
@@ -4886,7 +5298,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4886
5298
|
// demote the pin two tails in and the demotion re-dispatch would move the task off it.
|
|
4887
5299
|
if (preHarvestResult.ok && workerFinished)
|
|
4888
5300
|
noTrailerStreak.set(channelKey(assignment), 0);
|
|
4889
|
-
else if (!workerFinished && cause !== "provider-death" && !pinRefusesQuota) {
|
|
5301
|
+
else if (!workerFinished && cause !== "provider-death" && !pinRefusesQuota && !capacityMatch) {
|
|
4890
5302
|
const ck = channelKey(assignment);
|
|
4891
5303
|
const streak = (noTrailerStreak.get(ck) ?? 0) + 1;
|
|
4892
5304
|
noTrailerStreak.set(ck, streak);
|
|
@@ -4904,6 +5316,48 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4904
5316
|
attempt--;
|
|
4905
5317
|
continue;
|
|
4906
5318
|
}
|
|
5319
|
+
// OBS-1161: transient capacity → bounded same-seat requeue with backoff (no attempt burn, no
|
|
5320
|
+
// consult), then a same-floor failover exactly like quota — but the seat is NEVER demoted or
|
|
5321
|
+
// excluded: it is busy, not dead, and a later task may find it free. The budget is read from
|
|
5322
|
+
// the journal, never a loop-local counter, so a resume continues the count it left off at.
|
|
5323
|
+
if (capacityMatch) {
|
|
5324
|
+
const from = channelKey(assignment);
|
|
5325
|
+
const requeues = capacityRequeuesOn(from);
|
|
5326
|
+
const source = capacityBannerKilled ? "banner" : "exit";
|
|
5327
|
+
if (requeues < CAPACITY_REQUEUE_CAP) {
|
|
5328
|
+
journal.append("capacity-requeue", t.id, {
|
|
5329
|
+
attempt, requeue: requeues + 1, of: CAPACITY_REQUEUE_CAP, channel: from, assignment,
|
|
5330
|
+
matched: capacityMatch[0], source, backoffMs: capacityBackoffMs,
|
|
5331
|
+
});
|
|
5332
|
+
await new Promise((r) => setTimeout(r, capacityBackoffMs));
|
|
5333
|
+
attempt--;
|
|
5334
|
+
continue;
|
|
5335
|
+
}
|
|
5336
|
+
const next = failover("capacity-failover");
|
|
5337
|
+
journal.append("capacity-failover", t.id, {
|
|
5338
|
+
from, to: next ? channelKey(next) : null, matched: capacityMatch[0], source, requeues, cause: "capacity",
|
|
5339
|
+
});
|
|
5340
|
+
if (next) {
|
|
5341
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} capacity failover`, { tier: "attention" });
|
|
5342
|
+
if (!keepForever) {
|
|
5343
|
+
const idx = keptSlots.indexOf(slot);
|
|
5344
|
+
if (idx >= 0) {
|
|
5345
|
+
keptSlots.splice(idx, 1);
|
|
5346
|
+
try {
|
|
5347
|
+
await closeSlot(slot);
|
|
5348
|
+
}
|
|
5349
|
+
catch { /* cosmetic — reconcile is the backstop */ }
|
|
5350
|
+
}
|
|
5351
|
+
if (supersededWorkerSlot === slot)
|
|
5352
|
+
supersededWorkerSlot = undefined;
|
|
5353
|
+
}
|
|
5354
|
+
assignment = next;
|
|
5355
|
+
tried.push(channelKey(next));
|
|
5356
|
+
continue;
|
|
5357
|
+
}
|
|
5358
|
+
await park(t, `capacity exhausted on ${from} after ${requeues} requeues and no eligible channel at floor`, "quota", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { cause: "capacity", channel: from, requeues });
|
|
5359
|
+
return;
|
|
5360
|
+
}
|
|
4907
5361
|
// quota exhaustion → failover within floor; does NOT consume the ladder (spec §4)
|
|
4908
5362
|
// print: guarded on exit code — exit-0 output that merely MENTIONS "rate limit" must not failover
|
|
4909
5363
|
// interactive: a worker-CLAIMED trailer beats quota mentions; without one, quota text fails over
|
|
@@ -5067,7 +5521,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5067
5521
|
return;
|
|
5068
5522
|
}
|
|
5069
5523
|
const g = e.result;
|
|
5070
|
-
|
|
5524
|
+
activeGatePhases.get(t.id)?.delete(e.gate);
|
|
5525
|
+
classifyInfraResult(g);
|
|
5071
5526
|
inParallelOrder(g.gate, () => {
|
|
5072
5527
|
// GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
|
|
5073
5528
|
// which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
|
|
@@ -5123,6 +5578,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5123
5578
|
meta: Object.fromEntries(Object.entries(data).filter(([key]) => !GATE_TELEMETRY_KEYS.includes(key) && key !== "capacity")),
|
|
5124
5579
|
}));
|
|
5125
5580
|
commits = await commitsAheadOf(taskBase, wt);
|
|
5581
|
+
// OBS-1106: a replayed legacy infra row lacking `infra` is re-classified before it is
|
|
5582
|
+
// re-journaled and before the infra park below reads it.
|
|
5583
|
+
results.forEach(classifyInfraResult);
|
|
5126
5584
|
for (const g of results) {
|
|
5127
5585
|
journal.append("gate-replayed", t.id, {
|
|
5128
5586
|
attempt, priorAttempt: attempt - 1, gate: g.gate, commit: gateSubject.commit,
|
|
@@ -5133,7 +5591,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5133
5591
|
}
|
|
5134
5592
|
}
|
|
5135
5593
|
else {
|
|
5136
|
-
({ results, commits } = await withCommandContext(t.id, () => runReviewRecovery(t, {
|
|
5594
|
+
({ results, commits } = await withCommandContext(t.id, async () => runReviewRecovery(t, {
|
|
5595
|
+
carriedAuthors: await subjectAuthors(journal.read(), t.id, wt, taskBase),
|
|
5596
|
+
producer: producerNow(),
|
|
5137
5597
|
carriedFindings: outstandingFindings,
|
|
5138
5598
|
operatorContext,
|
|
5139
5599
|
worktree: wt, baseRef: taskBase, result, author: assignment,
|
|
@@ -5163,7 +5623,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5163
5623
|
onGate,
|
|
5164
5624
|
})));
|
|
5165
5625
|
}
|
|
5166
|
-
results.forEach(
|
|
5626
|
+
results.forEach(classifyInfraResult);
|
|
5167
5627
|
graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
|
|
5168
5628
|
saveGraph(repoRoot, graph);
|
|
5169
5629
|
if (results.some((g) => g.gate === "test" && !g.pass
|
|
@@ -5200,6 +5660,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5200
5660
|
saveGraph(repoRoot, graph);
|
|
5201
5661
|
journal.append("task-done", t.id, {
|
|
5202
5662
|
attempts: attempt + 1, assignment, taskContentDigest: contentDigest,
|
|
5663
|
+
authors: mergedAuthors(await subjectAuthors(journal.read(), t.id, wt, taskBase)),
|
|
5203
5664
|
});
|
|
5204
5665
|
journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
|
|
5205
5666
|
await trackedDriver.project?.(t.id, "completed");
|
|
@@ -5228,7 +5689,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5228
5689
|
// already journaled by onGate, before gateFails and before any ladder selection can fund an
|
|
5229
5690
|
// identical retry in the same environment. Parsed judge refusals never carry infra and keep
|
|
5230
5691
|
// flowing through the chargeable quality path below.
|
|
5231
|
-
const
|
|
5692
|
+
const unavailableReview = results.find((g) => gateFailed(g) && g.meta?.noEligibleReviewer === true);
|
|
5693
|
+
if (unavailableReview) {
|
|
5694
|
+
await park(t, gateFailApprovalReason(t.id, unavailableReview.details, true), "gate-fail", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
5695
|
+
return;
|
|
5696
|
+
}
|
|
5697
|
+
const infraFailure = results.find((g) => gateFailed(g) && isInfraResult(g));
|
|
5232
5698
|
if (infraFailure) {
|
|
5233
5699
|
await park(t, `${infraFailure.gate}: ${infraFailure.details}${infraFailure.meta?.recoveryBlocked ? ` — ${infraFailure.meta.recoveryBlocked}` : ""}`, "infra", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
5234
5700
|
return;
|
|
@@ -5445,7 +5911,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5445
5911
|
const holdEndCondition = (closing = true) => {
|
|
5446
5912
|
sweepLiveApprovals();
|
|
5447
5913
|
const freeSlots = Math.max(0, concurrency - inflight.size);
|
|
5448
|
-
const dispatchable =
|
|
5914
|
+
const dispatchable = admissible().filter((t) => !inflight.has(t.id));
|
|
5449
5915
|
if (freeSlots === 0 || dispatchable.length === 0)
|
|
5450
5916
|
return false;
|
|
5451
5917
|
if (closing) {
|
|
@@ -5465,7 +5931,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5465
5931
|
throw new Error(`terminated by ${termSignal}`);
|
|
5466
5932
|
await watchBoard();
|
|
5467
5933
|
sweepLiveApprovals();
|
|
5468
|
-
const ready =
|
|
5934
|
+
const ready = admissible()
|
|
5469
5935
|
.filter((t) => !inflight.has(t.id))
|
|
5470
5936
|
.slice(0, Math.max(0, concurrency - inflight.size));
|
|
5471
5937
|
for (const t of ready) {
|
|
@@ -5486,9 +5952,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5486
5952
|
if (fatalStop.signal.aborted)
|
|
5487
5953
|
return;
|
|
5488
5954
|
const cleanupEvidence = cleanupErrors.length ? { cleanupErrors } : {};
|
|
5955
|
+
if (err instanceof HostDegradedError) {
|
|
5956
|
+
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "host-degraded", ...cleanupEvidence });
|
|
5957
|
+
return;
|
|
5958
|
+
}
|
|
5489
5959
|
if (err instanceof HeldProbeExhausted) {
|
|
5490
5960
|
const wt = worktreePath(repoRoot, `${branch}--${t.id}`);
|
|
5491
|
-
|
|
5961
|
+
const producer = knownProducer(journal.read(), t.id);
|
|
5962
|
+
let ref = await withoutExecutionBudget(() => preserveWorktree(wt, producer));
|
|
5492
5963
|
if (!ref) {
|
|
5493
5964
|
const head = await gitHead(wt);
|
|
5494
5965
|
ref = `refs/tickmarkr/preserved/${head}`;
|
|
@@ -5496,15 +5967,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5496
5967
|
if (saved.code !== 0)
|
|
5497
5968
|
throw new Error(`could not preserve ${head}: ${saved.stderr}`);
|
|
5498
5969
|
}
|
|
5499
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
5970
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
5500
5971
|
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "transport-uncertain", ref, ...cleanupEvidence });
|
|
5501
5972
|
return;
|
|
5502
5973
|
}
|
|
5503
5974
|
if (err instanceof ExecutionBudgetExceeded) {
|
|
5504
5975
|
const wt = worktreePath(repoRoot, `${branch}--${t.id}`);
|
|
5505
|
-
const
|
|
5976
|
+
const producer = knownProducer(journal.read(), t.id);
|
|
5977
|
+
const ref = await preserveWorktree(wt, producer);
|
|
5506
5978
|
if (ref)
|
|
5507
|
-
journal.append("worktree-preserved", t.id, { ref });
|
|
5979
|
+
journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
|
|
5508
5980
|
const dispatch = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "task-dispatch");
|
|
5509
5981
|
await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "execution-budget-exhausted", limitMs: cfg.executionPolicy.taskExecutionLimitMs,
|
|
5510
5982
|
...cleanupEvidence,
|
|
@@ -5538,7 +6010,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5538
6010
|
journal.append("approval-window-start", undefined, { windowMs, parked: [...parked] });
|
|
5539
6011
|
// The narrator may itself append a decision at this boundary.
|
|
5540
6012
|
sweepLiveApprovals();
|
|
5541
|
-
if (
|
|
6013
|
+
if (admissible().length) {
|
|
5542
6014
|
approvalDeadline = undefined;
|
|
5543
6015
|
continue;
|
|
5544
6016
|
}
|
|
@@ -5588,7 +6060,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5588
6060
|
pending: pendingTasks(graph).map((t) => t.id),
|
|
5589
6061
|
};
|
|
5590
6062
|
fatalPhase = "baseline";
|
|
5591
|
-
await baselineCapture
|
|
6063
|
+
await baselineCapture.catch(error => {
|
|
6064
|
+
// A host park is resumable only after baseline publication. Before that, retain the
|
|
6065
|
+
// fatal baseline failure: resume requires baseline.json and cannot recover this capture.
|
|
6066
|
+
if (!(error instanceof HostDegradedError) || !existsSync(join(journal.dir, "baseline.json")))
|
|
6067
|
+
throw error;
|
|
6068
|
+
});
|
|
5592
6069
|
fatalPhase = "tip-verify";
|
|
5593
6070
|
// OBS-34: post-merge integration-tip verify — strict exit codes, no baseline forgiveness.
|
|
5594
6071
|
const lastMergedTask = [...journal.read()].reverse().find((e) => e.event === "merge" && e.taskId)?.taskId;
|
|
@@ -5598,7 +6075,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5598
6075
|
const checkApprovals = () => {
|
|
5599
6076
|
try {
|
|
5600
6077
|
sweepLiveApprovals();
|
|
5601
|
-
if (
|
|
6078
|
+
if (admissible().length)
|
|
5602
6079
|
controller.abort(cancellation);
|
|
5603
6080
|
if (termSignal)
|
|
5604
6081
|
controller.abort(new Error(`terminated by ${termSignal}`));
|