tickmarkr 2.1.4 → 2.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/stats.d.ts +21 -0
- package/dist/cli/commands/stats.js +210 -0
- package/dist/cli/commands/status.js +28 -3
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +3 -1
- package/dist/compile/collateral.d.ts +23 -2
- package/dist/compile/collateral.js +126 -11
- package/dist/compile/native.js +35 -3
- package/dist/gates/baseline.d.ts +23 -3
- package/dist/gates/baseline.js +102 -33
- package/dist/gates/llm.d.ts +1 -0
- package/dist/gates/llm.js +1 -0
- package/dist/gates/review.d.ts +9 -2
- package/dist/gates/review.js +51 -10
- package/dist/gates/run-gates.js +12 -1
- package/dist/run/daemon.js +325 -19
- package/dist/run/git.d.ts +50 -0
- package/dist/run/git.js +56 -2
- package/dist/run/journal.d.ts +28 -0
- package/dist/run/journal.js +137 -20
- package/dist/run/merge.js +13 -3
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +120 -3
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +63 -5
package/dist/run/daemon.js
CHANGED
|
@@ -19,9 +19,9 @@ import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHa
|
|
|
19
19
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
20
20
|
import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
21
21
|
import { runEnvironment } from "./environment.js";
|
|
22
|
-
import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, runWithForkBudget, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
22
|
+
import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, sameCapacity, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
23
23
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
24
|
-
import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
24
|
+
import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, renderStructuredReviewFinding, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
25
25
|
import { isDiffCapPark } from "../gates/review.js";
|
|
26
26
|
import { acquireApprovalSerialization, acquireRunLock, releaseRunLock } from "./lock.js";
|
|
27
27
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
@@ -242,6 +242,10 @@ function repairBrief(findings, diff, baseRef) {
|
|
|
242
242
|
// v1.70 T5: default request-changes rounds a task may draw before it parks. OBS-419 keeps this as the
|
|
243
243
|
// no-ceiling behavior; an operator may narrow only the next engagement on the approval that releases it.
|
|
244
244
|
const REVIEW_ROUND_CAP = 2;
|
|
245
|
+
// T2: one heading per fact. The first is owed a passing review; the second already drew one and was
|
|
246
|
+
// accepted with the reviewer's own rationale, which travels beside (but never changes) its identity.
|
|
247
|
+
const OUTSTANDING_FINDINGS_HEADING = "## Outstanding review findings — a review has NOT passed on these yet";
|
|
248
|
+
const DEFERRED_FINDINGS_HEADING = "## Deferred review findings — a reviewer ACCEPTED these with a rationale and did NOT block on them; do not re-litigate, fix only if your change touches them";
|
|
245
249
|
// OBS-419: the newest approval starts the current engagement, so it is also the sole authority for
|
|
246
250
|
// that engagement's optional ceiling. Stop at the newest approval even when the field is absent: a
|
|
247
251
|
// later ordinary release restores the module default instead of inheriting an older operator limit.
|
|
@@ -392,7 +396,7 @@ function lastVerifyCycle(events) {
|
|
|
392
396
|
if (e.event === "tip-verify-start") {
|
|
393
397
|
const { tip, cmdHash } = e.data;
|
|
394
398
|
cur = typeof tip === "string" && typeof cmdHash === "string"
|
|
395
|
-
? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false }
|
|
399
|
+
? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [e.data.capacity] }
|
|
396
400
|
: undefined;
|
|
397
401
|
afterRunEnd = false;
|
|
398
402
|
continue;
|
|
@@ -410,9 +414,14 @@ function lastVerifyCycle(events) {
|
|
|
410
414
|
continue;
|
|
411
415
|
}
|
|
412
416
|
if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
|
|
413
|
-
cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false };
|
|
417
|
+
cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [] };
|
|
414
418
|
}
|
|
415
419
|
afterRunEnd = false;
|
|
420
|
+
// T7: EVERY verdict row's own capacity, not just the start row's. The start row is a statement of
|
|
421
|
+
// intent written before a single command ran; the green this cache would carry forward lives on
|
|
422
|
+
// these rows, so a row whose capacity differs from the session's — or which is malformed — has to
|
|
423
|
+
// be able to sink the cycle by itself.
|
|
424
|
+
cur.capacities.push(e.data.capacity);
|
|
416
425
|
if (e.event === "tip-verify-failed")
|
|
417
426
|
cur.failed = true;
|
|
418
427
|
else {
|
|
@@ -440,23 +449,31 @@ function lastVerifyCycle(events) {
|
|
|
440
449
|
export async function verifyIntegrationTipCached(intWt, commands, journal, opts = {}) {
|
|
441
450
|
const cmdHash = commandsHash(commands);
|
|
442
451
|
const tip = await gitHead(intWt);
|
|
452
|
+
// T7: the capacity this session's verify children WOULD run under — the third thing a carried
|
|
453
|
+
// green must match, beside the tip and the command set. A cached verdict is the one place a green
|
|
454
|
+
// crosses a session boundary with nothing re-run, and a session resumed at a different concurrency
|
|
455
|
+
// divides the machine by a different number: that green was established in another world, so it is
|
|
456
|
+
// not carried forward and the commands run again. A pre-T7 cycle records no capacity and keeps
|
|
457
|
+
// exactly the behaviour it has today.
|
|
458
|
+
const capacity = resolvedCapacity();
|
|
443
459
|
const porcelain = await shGit("git status --porcelain", intWt);
|
|
444
460
|
const clean = porcelain.code === 0 && porcelain.stdout.trim() === "";
|
|
445
461
|
const last = lastVerifyCycle(journal.read());
|
|
446
462
|
const cached = last !== undefined && !last.failed && !last.forgiven && last.tip === tip && last.cmdHash === cmdHash
|
|
447
|
-
&& Object.keys(commands).every((g) => last.gates.has(g))
|
|
463
|
+
&& Object.keys(commands).every((g) => last.gates.has(g))
|
|
464
|
+
&& last.capacities.every((recorded) => sameCapacity(recorded, capacity));
|
|
448
465
|
// A pair can be verified red and then green without either SHA or command hash changing (for
|
|
449
466
|
// example, an external service or ignored fixture recovers). Delimit attempts explicitly so that
|
|
450
467
|
// the earlier red cannot remain latched into the later complete green cycle.
|
|
451
|
-
journal.append("tip-verify-start", undefined, { tip, cmdHash, gates: Object.keys(commands), cached: clean && cached });
|
|
468
|
+
journal.append("tip-verify-start", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands), cached: clean && cached });
|
|
452
469
|
if (clean && cached) {
|
|
453
|
-
journal.append("tip-verify-cached", undefined, { tip, cmdHash, gates: Object.keys(commands) });
|
|
470
|
+
journal.append("tip-verify-cached", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands) });
|
|
454
471
|
// The skip must not read as a red. Every surface derives the tip's verdict from this cycle's
|
|
455
472
|
// `tip-verify` events (cockpit derive.ts tipVerificationPassed: a run-end claiming "passed" with
|
|
456
473
|
// ZERO events is fail-closed to FALSE), so a carried-forward green still journals its per-gate
|
|
457
474
|
// pass — `cached: true` keeps it honest about not having re-run the command.
|
|
458
475
|
for (const gate of Object.keys(commands)) {
|
|
459
|
-
journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash });
|
|
476
|
+
journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash, capacity });
|
|
460
477
|
}
|
|
461
478
|
return false;
|
|
462
479
|
}
|
|
@@ -464,7 +481,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
464
481
|
for (const r of await verifyIntegrationTip(intWt, commands, journal.dir, opts.baseline)) {
|
|
465
482
|
if (r.pass) {
|
|
466
483
|
// Q121s: a forgiven pass journals its fingerprints — honest about what was carried, never a silent green.
|
|
467
|
-
journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash });
|
|
484
|
+
journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity });
|
|
468
485
|
}
|
|
469
486
|
else {
|
|
470
487
|
journal.append("tip-verify-failed", undefined, {
|
|
@@ -476,6 +493,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
476
493
|
lastMergedTask: opts.lastMergedTask,
|
|
477
494
|
tip,
|
|
478
495
|
cmdHash,
|
|
496
|
+
capacity,
|
|
479
497
|
});
|
|
480
498
|
tipFailed = true;
|
|
481
499
|
}
|
|
@@ -503,6 +521,35 @@ async function gateCommitSubject(base, head, wt) {
|
|
|
503
521
|
return head; // fail closed to the exact object id if canonicalization fails
|
|
504
522
|
return createHash("sha256").update(history.stdout).digest("hex");
|
|
505
523
|
}
|
|
524
|
+
// A CPU total cannot prove a tree empty: a newly launched live process can still have a measured
|
|
525
|
+
// total of zero at the host's clock resolution. This probe asks the narrower cardinality question
|
|
526
|
+
// OBS-737 needs. A readable process table with no marker root is measured empty; a failed or empty
|
|
527
|
+
// table is unmeasurable. Descendants are closed over PPID because not every child retains the
|
|
528
|
+
// dispatch-script marker in its own argv.
|
|
529
|
+
async function observeWorkerProcessTree(marker, cwd) {
|
|
530
|
+
const snapshot = await shGit("ps -Awwo pid=,ppid=,command=", cwd, 15_000);
|
|
531
|
+
if (snapshot.code !== 0)
|
|
532
|
+
return "unmeasurable";
|
|
533
|
+
const rows = [];
|
|
534
|
+
for (const line of snapshot.stdout.split("\n")) {
|
|
535
|
+
const match = /^\s*(\d+)\s+(\d+)\s+(.*)$/.exec(line);
|
|
536
|
+
if (match)
|
|
537
|
+
rows.push({ pid: match[1], ppid: match[2], command: match[3] });
|
|
538
|
+
}
|
|
539
|
+
if (rows.length === 0)
|
|
540
|
+
return "unmeasurable";
|
|
541
|
+
const tree = new Set(rows.filter((row) => row.command.includes(marker)).map((row) => row.pid));
|
|
542
|
+
for (let grew = true; grew;) {
|
|
543
|
+
grew = false;
|
|
544
|
+
for (const row of rows) {
|
|
545
|
+
if (!tree.has(row.pid) && tree.has(row.ppid)) {
|
|
546
|
+
tree.add(row.pid);
|
|
547
|
+
grew = true;
|
|
548
|
+
}
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
return tree.size === 0 ? "empty" : "running";
|
|
552
|
+
}
|
|
506
553
|
const OBSERVE_CHUNK_BYTES = 64 * 1024;
|
|
507
554
|
const OBSERVE_BUDGET_BYTES = 256 * 1024 * 1024;
|
|
508
555
|
let observeBudgetBytes = OBSERVE_BUDGET_BYTES;
|
|
@@ -587,6 +634,20 @@ async function boundedGitObservation(command, worktree, budget) {
|
|
|
587
634
|
return undefined;
|
|
588
635
|
return result.stdout;
|
|
589
636
|
}
|
|
637
|
+
// "No worktree delta" means no staged/unstaged/untracked bytes AND no commits beyond the task's
|
|
638
|
+
// dispatch base. Both reads are bounded and preserve the observer's third state: a failed probe is
|
|
639
|
+
// unreadable, never clean.
|
|
640
|
+
async function observeWorktreeDelta(base, worktree) {
|
|
641
|
+
let budget = observeBudgetBytes;
|
|
642
|
+
const status = await boundedGitObservation("GIT_OPTIONAL_LOCKS=0 git status --porcelain=v1 -z --untracked-files=all", worktree, budget);
|
|
643
|
+
if (status === undefined)
|
|
644
|
+
return "unreadable";
|
|
645
|
+
budget -= Buffer.byteLength(status);
|
|
646
|
+
const ahead = await boundedGitObservation(`GIT_OPTIONAL_LOCKS=0 git rev-list --count ${shq(base)}..HEAD`, worktree, budget);
|
|
647
|
+
if (ahead === undefined || !/^\d+$/.test(ahead.trim()))
|
|
648
|
+
return "unreadable";
|
|
649
|
+
return status.length === 0 && ahead.trim() === "0" ? "unchanged" : "changed";
|
|
650
|
+
}
|
|
590
651
|
// The signature has four git-owned inputs: HEAD, staged blob/mode/path identity, porcelain path and
|
|
591
652
|
// state, and the set of worktree paths whose bytes Git cannot supply. Those paths contribute their
|
|
592
653
|
// filesystem identity, mode, symlink text, and content. A failed or over-budget leg is a third
|
|
@@ -898,6 +959,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
898
959
|
const replayedGateResults = resumeLifecycleOpen
|
|
899
960
|
? journal.replayCurrentAttemptGateResults()
|
|
900
961
|
: new Map();
|
|
962
|
+
// T7: the capacity THIS session resolved — read inside the run's fork budget, so it is the number
|
|
963
|
+
// every shell this run spawns will divide the machine by. Recorded evidence from another session
|
|
964
|
+
// is only reusable against this.
|
|
965
|
+
const sessionCapacity = resolvedCapacity();
|
|
901
966
|
const replayedExclusions = opts.resume ? journal.replayExcludedChannels() : new Set();
|
|
902
967
|
if (opts.resume) {
|
|
903
968
|
// v1.53 T5: a superseded run is dead — resuming it beside its successor is the exact
|
|
@@ -1069,10 +1134,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1069
1134
|
};
|
|
1070
1135
|
// gateFails/consults are execTask-scoped counters passed in so a park row is a rich verified-failure
|
|
1071
1136
|
// observation (e.g. ladder-exhausted + gateFails:4); every task-human row has a closed kind, never prose alone.
|
|
1072
|
-
const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh") => {
|
|
1137
|
+
const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh", details = {}) => {
|
|
1073
1138
|
graph = setStatus(graph, t.id, "human");
|
|
1074
1139
|
saveGraph(repoRoot, graph);
|
|
1075
|
-
journal.append("task-human", t.id, { reason, kind });
|
|
1140
|
+
journal.append("task-human", t.id, { ...details, reason, kind });
|
|
1076
1141
|
if (assignment) {
|
|
1077
1142
|
// OBS-547: `metered` counts CHARGEABLE metered attempts, so an unchargeable dispatch passes 0 and
|
|
1078
1143
|
// the count is omitted rather than written as 0 or as `1` beside `attempts: 0` — a row claiming
|
|
@@ -1150,6 +1215,22 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1150
1215
|
journal.append("worktree-preserved", t.id, { ref });
|
|
1151
1216
|
return driver.worktree(repoRoot, taskBranch, taskBase);
|
|
1152
1217
|
};
|
|
1218
|
+
// A dead worker with a clean checkout still needs a durable recovery handle: there may be no
|
|
1219
|
+
// pane left to identify even the commit it was dispatched from. preserveWorktree deliberately
|
|
1220
|
+
// creates no ref for ordinary clean recreations, so this exceptional terminal path pins HEAD
|
|
1221
|
+
// explicitly under the same recovery namespace. Reconfirm the delta immediately before the
|
|
1222
|
+
// ref write; a change or unreadable recheck withdraws the park.
|
|
1223
|
+
const preserveDeadWorker = async (worktree, taskBase) => {
|
|
1224
|
+
const state = await observeWorktreeDelta(taskBase, worktree);
|
|
1225
|
+
if (state !== "unchanged")
|
|
1226
|
+
return { state };
|
|
1227
|
+
const head = await gitHead(worktree);
|
|
1228
|
+
const ref = `refs/tickmarkr/preserved/${head}`;
|
|
1229
|
+
const updated = await shGit(`git update-ref ${shq(ref)} ${shq(head)}`, worktree);
|
|
1230
|
+
if (updated.code !== 0)
|
|
1231
|
+
throw new Error(`could not preserve dead worker HEAD at ${ref}: ${updated.stderr || updated.stdout}`);
|
|
1232
|
+
return { state, ref };
|
|
1233
|
+
};
|
|
1153
1234
|
const r = route(t, cfg, channels, profile, undefined, demotedChannels);
|
|
1154
1235
|
for (const lint of r.lints)
|
|
1155
1236
|
journal.append("routing-lint", t.id, { lint });
|
|
@@ -1221,6 +1302,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1221
1302
|
let gateSubject;
|
|
1222
1303
|
const journalGateResult = (g) => {
|
|
1223
1304
|
const blocking = gateFailed(g) && (g.gate === "review" || g.gate === "acceptance");
|
|
1305
|
+
// T2: a review that PASSED while DEFERRING a concern still recorded a defect — the prompt
|
|
1306
|
+
// promises the deferral is recorded and never dropped, and a details string is not a record a
|
|
1307
|
+
// later round can match. The blocking projection above is the only writer of structured
|
|
1308
|
+
// findings today, so on this row it writes nothing and every structured reader goes blind.
|
|
1309
|
+
// Same shape, same identity, on the passing row: the verdict is untouched (`pass` stays true),
|
|
1310
|
+
// only the projection widens to the rows the reviewer itself classified as deferred.
|
|
1311
|
+
const deferred = !blocking && g.gate === "review" && g.pass === true
|
|
1312
|
+
? deferredReviewFindings(g.details)
|
|
1313
|
+
: [];
|
|
1224
1314
|
// R3 (OBS-186): a gate that DECLINED has no verdict to state, and this row is the ONE seam every
|
|
1225
1315
|
// fold outside this file shares. Writing `pass: false` for a decline is what turned a skip into
|
|
1226
1316
|
// a failure at all of them at once — the engagement round budget (reviewRoundsSinceApproval,
|
|
@@ -1265,6 +1355,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1265
1355
|
...(blocking ? {
|
|
1266
1356
|
taskContentDigest: contentDigest,
|
|
1267
1357
|
findings: structuredFindings(g.gate, g.details),
|
|
1358
|
+
} : deferred.length > 0 ? {
|
|
1359
|
+
taskContentDigest: contentDigest,
|
|
1360
|
+
findings: deferred,
|
|
1268
1361
|
} : {}),
|
|
1269
1362
|
// v2.0 T2 (OBS-554): the gate's OWN measurement, lifted verbatim from the meta run-gates
|
|
1270
1363
|
// stamped WHERE THE GATE RAN. Nothing here re-derives a duration by subtracting journal
|
|
@@ -1275,6 +1368,27 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1275
1368
|
// every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
|
|
1276
1369
|
// seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
|
|
1277
1370
|
...gateMeasurement(g.meta),
|
|
1371
|
+
// T7: the capacity the gate's own command child ran under, lifted verbatim from the result
|
|
1372
|
+
// the battery produced — read where the shell built that child's environment, never
|
|
1373
|
+
// re-derived from the run's own budget, which would answer a different number than the
|
|
1374
|
+
// operator's export did. It is this row's only COMPARABLE identity: the two load samples
|
|
1375
|
+
// beside it are endpoint reads of a gate whose interior neither of them saw, so no reader
|
|
1376
|
+
// compares them, and matching capacity never claims the machine was calm. A gate that ran
|
|
1377
|
+
// no command carries nothing — it divided nothing.
|
|
1378
|
+
//
|
|
1379
|
+
// `dirtiedBy` is the one hole in that lift: run-gates REPLACES a green battery verdict with
|
|
1380
|
+
// a refusal when the command left the worktree dirty (run-gates.ts, three sites: the legacy
|
|
1381
|
+
// batch, the per-command loop and the merge-candidate full suite), and the refusal is a
|
|
1382
|
+
// fresh verdict object carrying nothing off the result it replaced. That command's child DID
|
|
1383
|
+
// run, so its row still owes the world it ran in. `sessionCapacity` is that world and not a
|
|
1384
|
+
// re-derivation of it: it applies the same precedence `shell` does — an operator export
|
|
1385
|
+
// first — and was read inside the same fork budget every gate child of this run is spawned
|
|
1386
|
+
// under, so it is by construction the number that child received. The flag is set only where
|
|
1387
|
+
// a command of THIS gate ran and dirtied the tree, so the round-entry refusal (no command
|
|
1388
|
+
// ran) and the round-end withdrawal (lands on a gate that runs no command) still carry
|
|
1389
|
+
// nothing.
|
|
1390
|
+
...(g.capacity ? { capacity: g.capacity }
|
|
1391
|
+
: g.meta?.dirtiedBy === g.gate ? { capacity: sessionCapacity } : {}),
|
|
1278
1392
|
});
|
|
1279
1393
|
};
|
|
1280
1394
|
// R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
|
|
@@ -1555,7 +1669,21 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1555
1669
|
const canonicalCurrentCommit = currentTaskSubject === replayedGates.commit;
|
|
1556
1670
|
const recreatedLegacyCommit = priorTaskTip === replayedGates.commit
|
|
1557
1671
|
&& priorTaskSubject === currentTaskSubject;
|
|
1558
|
-
|
|
1672
|
+
// T7: the commit says the gates would inspect the same TREE; it says nothing about the
|
|
1673
|
+
// machine they measured it on. A resume is a new session and may have resolved a different
|
|
1674
|
+
// concurrency, so the rows behind a replayed green must also have been measured under the
|
|
1675
|
+
// capacity this session resolved — otherwise a contiguous green prefix spans two worlds.
|
|
1676
|
+
// Rows from before this stamp carry no capacity and replay exactly as they do today.
|
|
1677
|
+
const replayedCapacities = priorEvents
|
|
1678
|
+
.filter((e) => e.event === "gate-result" && e.taskId === t.id && e.data.commit === replayedGates.commit)
|
|
1679
|
+
.map((e) => e.data.capacity);
|
|
1680
|
+
const sameWorld = replayedCapacities.every((recorded) => sameCapacity(recorded, sessionCapacity));
|
|
1681
|
+
if (!sameWorld) {
|
|
1682
|
+
journal.append("gate-replay-capacity-changed", t.id, {
|
|
1683
|
+
commit: replayedGates.commit, recorded: replayedCapacities, resolved: sessionCapacity,
|
|
1684
|
+
});
|
|
1685
|
+
}
|
|
1686
|
+
let reusable = sameWorld && (exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit);
|
|
1559
1687
|
const reused = [];
|
|
1560
1688
|
const declaredGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
|
|
1561
1689
|
for (const gate of declaredGates) {
|
|
@@ -1768,10 +1896,37 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1768
1896
|
// task (journal.ts `outstandingReviewFindings`). Appended row-wise, because this round's own
|
|
1769
1897
|
// feedback or a repair brief may already quote a finding and repeating it helps no worker.
|
|
1770
1898
|
const outstandingFindings = outstandingReviewFindings(journaledSoFar, t.id);
|
|
1771
|
-
|
|
1772
|
-
|
|
1773
|
-
|
|
1774
|
-
|
|
1899
|
+
// T2: the two are different facts about the work and no one heading is true of both. A finding
|
|
1900
|
+
// the reviewer DEFERRED was accepted with a rationale by a review that did not block on it; a
|
|
1901
|
+
// blocking one is still waiting for a review to pass. Filing the deferral under the blocking
|
|
1902
|
+
// heading tells the next worker a passing review is owed on a concern that already drew one —
|
|
1903
|
+
// the exact falsehood this carry exists to remove, restated in the brief that carries it.
|
|
1904
|
+
//
|
|
1905
|
+
// The de-dup below is BLOCKING-ONLY on purpose. A review round's raw bytes quote every finding
|
|
1906
|
+
// it recorded, deferrals included, and those bytes ride into the very next dispatch under the
|
|
1907
|
+
// repair brief's "fix ONLY what these findings name" — so on the ordinary immediate retry the
|
|
1908
|
+
// deferral is already stated, and stated AS BLOCKING. Suppressing its heading there because it
|
|
1909
|
+
// is "already quoted" leaves exactly the falsehood. A quoted BLOCKING finding is quoted
|
|
1910
|
+
// truthfully, so that one still de-dups; a deferral is instead CUT from the raw bytes and
|
|
1911
|
+
// restated once, under the only heading true of it.
|
|
1912
|
+
const deferredRows = outstandingFindings.filter(isDeferredFinding);
|
|
1913
|
+
const withoutDeferrals = (text) => deferredRows.reduce((brief, finding) => brief.replaceAll(renderStructuredReviewFinding(finding), ""), text).replace(/\n{3,}/g, "\n\n").trim();
|
|
1914
|
+
feedback = withoutDeferrals(feedback);
|
|
1915
|
+
if (repairFindings !== undefined)
|
|
1916
|
+
repairFindings = withoutDeferrals(repairFindings);
|
|
1917
|
+
const briefs = [
|
|
1918
|
+
[OUTSTANDING_FINDINGS_HEADING, outstandingFindings.filter((f) => !isDeferredFinding(f) && !feedback.includes(f.note))],
|
|
1919
|
+
[DEFERRED_FINDINGS_HEADING, deferredRows],
|
|
1920
|
+
];
|
|
1921
|
+
for (const [heading, rows] of briefs) {
|
|
1922
|
+
if (rows.length === 0)
|
|
1923
|
+
continue;
|
|
1924
|
+
const brief = [heading, ...rows.map((f) => {
|
|
1925
|
+
const rationale = f.rationale === undefined
|
|
1926
|
+
? ""
|
|
1927
|
+
: `\n Rationale: ${f.rationale}`;
|
|
1928
|
+
return `- ${f.path}: ${f.note}${rationale}`;
|
|
1929
|
+
})].join("\n");
|
|
1775
1930
|
feedback = feedback ? `${feedback}\n\n${brief}` : brief;
|
|
1776
1931
|
}
|
|
1777
1932
|
retryMode = repairFindings
|
|
@@ -2088,6 +2243,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2088
2243
|
// subprocess tree that REACHED its exit marker, not about what the worker claimed.
|
|
2089
2244
|
let processExited = false;
|
|
2090
2245
|
let earlyLaunchDead = false;
|
|
2246
|
+
let deadWorkerPark;
|
|
2091
2247
|
let settleParsed;
|
|
2092
2248
|
let seedResult;
|
|
2093
2249
|
// v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
|
|
@@ -2227,6 +2383,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2227
2383
|
let quotaStreak = 0;
|
|
2228
2384
|
let rowSaturationHeld = false; // journaled once per attempt when the kill stands down
|
|
2229
2385
|
let cpuHeld = false; // likewise for the CPU leg's stand-down (OBS-548)
|
|
2386
|
+
let paneReadHeld = false;
|
|
2387
|
+
let paneStatusHeld = false;
|
|
2230
2388
|
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
2231
2389
|
const sliceStart = Date.now();
|
|
2232
2390
|
const remaining = stallWindowMs - (sliceStart - lastProgressAt);
|
|
@@ -2254,7 +2412,39 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2254
2412
|
break;
|
|
2255
2413
|
}
|
|
2256
2414
|
}
|
|
2257
|
-
|
|
2415
|
+
// A failed pane read is absence of evidence, never evidence of an absent pane. Keep the
|
|
2416
|
+
// rolling taskTimeoutMinutes window as the backstop and name the held probe once.
|
|
2417
|
+
let paneText;
|
|
2418
|
+
try {
|
|
2419
|
+
paneText = await driver.read(slot, PANE_READ_ROWS);
|
|
2420
|
+
}
|
|
2421
|
+
catch (error) {
|
|
2422
|
+
// Preserve the pre-existing exception path when the independent status probe still
|
|
2423
|
+
// sees a pane. The outer attempt finally owns accountant cleanup on that path. Only
|
|
2424
|
+
// an undetectable status makes the read failure relevant to the death detector, and
|
|
2425
|
+
// that genuinely unmeasurable pair fails open to the rolling timeout.
|
|
2426
|
+
let paneUndetectable = true;
|
|
2427
|
+
try {
|
|
2428
|
+
paneUndetectable = await driver.status(slot) === "unknown";
|
|
2429
|
+
}
|
|
2430
|
+
catch {
|
|
2431
|
+
// Two unreadable pane probes are still unmeasurable, never proof of death.
|
|
2432
|
+
}
|
|
2433
|
+
if (!paneUndetectable)
|
|
2434
|
+
throw error;
|
|
2435
|
+
if (!paneReadHeld) {
|
|
2436
|
+
paneReadHeld = true;
|
|
2437
|
+
journal.append("worker-dead-held", t.id, {
|
|
2438
|
+
slot: slot.name, attempt, reason: "pane-read-unreadable",
|
|
2439
|
+
error: error instanceof Error ? error.message : String(error),
|
|
2440
|
+
});
|
|
2441
|
+
}
|
|
2442
|
+
await armCpuLeg(false);
|
|
2443
|
+
const spent = Date.now() - sliceStart;
|
|
2444
|
+
if (spent < slice)
|
|
2445
|
+
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
2446
|
+
continue;
|
|
2447
|
+
}
|
|
2258
2448
|
if (paneText.length > 0)
|
|
2259
2449
|
everHadOutput = true;
|
|
2260
2450
|
// OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
|
|
@@ -2323,7 +2513,24 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2323
2513
|
// gate held. page on "idle" too: herdr's blocked-scrape is strict and proved flaky
|
|
2324
2514
|
// for TUI dialogs (live check: cursor's trust dialog scraped as idle).
|
|
2325
2515
|
// "unknown"/"working" never page.
|
|
2326
|
-
|
|
2516
|
+
let st;
|
|
2517
|
+
try {
|
|
2518
|
+
st = await driver.status(slot);
|
|
2519
|
+
}
|
|
2520
|
+
catch (error) {
|
|
2521
|
+
if (!paneStatusHeld) {
|
|
2522
|
+
paneStatusHeld = true;
|
|
2523
|
+
journal.append("worker-dead-held", t.id, {
|
|
2524
|
+
slot: slot.name, attempt, reason: "pane-status-unreadable",
|
|
2525
|
+
error: error instanceof Error ? error.message : String(error),
|
|
2526
|
+
});
|
|
2527
|
+
}
|
|
2528
|
+
await armCpuLeg(false);
|
|
2529
|
+
const spent = Date.now() - sliceStart;
|
|
2530
|
+
if (spent < slice)
|
|
2531
|
+
await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
|
|
2532
|
+
continue;
|
|
2533
|
+
}
|
|
2327
2534
|
if (st !== lastStatus) {
|
|
2328
2535
|
lastStatus = st;
|
|
2329
2536
|
journal.append("worker-status", t.id, { slot: slot.name, status: st, attempt });
|
|
@@ -2363,6 +2570,95 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2363
2570
|
// daemon has nothing left to do — the pane falls back under the fast-kill and page
|
|
2364
2571
|
// watchdogs like any other, instead of riding the whole rolling window untended.
|
|
2365
2572
|
const nudgePending = nudgeable && (!nudged || nudgeDeadline !== undefined);
|
|
2573
|
+
// OBS-737: disposition for the one fully measured death state. `unknown` alone is not
|
|
2574
|
+
// absence (status parsing can fail), and an empty read alone is not absence (a live pane
|
|
2575
|
+
// can be quiet); together they are the existing driver-level pane absence witness. The
|
|
2576
|
+
// process probe and both worktree observations retain their own third states. Only the
|
|
2577
|
+
// explicit conjunction parks; every other state falls through to the unchanged rolling
|
|
2578
|
+
// timeout below. Seeded launches are excluded because their process marker is knowingly
|
|
2579
|
+
// unmeasurable (armCpuLeg documents that contract above).
|
|
2580
|
+
const paneAbsentCandidate = st === "unknown" && paneText.trim().length === 0;
|
|
2581
|
+
// A subprocess can exit between waitOutput and read while its stdout is still draining.
|
|
2582
|
+
// Confirm emptiness in this same poll before paying for `ps`; then confirm once more
|
|
2583
|
+
// after the process probe yielded the event loop. Any bytes or read error withdraw the
|
|
2584
|
+
// absence witness, so a fast completed worker cannot be parked in that drain race.
|
|
2585
|
+
let paneAbsent = paneAbsentCandidate;
|
|
2586
|
+
if (paneAbsent) {
|
|
2587
|
+
try {
|
|
2588
|
+
const confirmation = await driver.read(slot, PANE_READ_ROWS);
|
|
2589
|
+
paneAbsent = confirmation.trim().length === 0;
|
|
2590
|
+
if (!paneAbsent) {
|
|
2591
|
+
everHadOutput = true;
|
|
2592
|
+
if (stallProgress.observe({ paneText: confirmation, contextTokens }))
|
|
2593
|
+
lastProgressAt = Date.now();
|
|
2594
|
+
}
|
|
2595
|
+
}
|
|
2596
|
+
catch (error) {
|
|
2597
|
+
paneAbsent = false;
|
|
2598
|
+
if (!paneReadHeld) {
|
|
2599
|
+
paneReadHeld = true;
|
|
2600
|
+
journal.append("worker-dead-held", t.id, {
|
|
2601
|
+
slot: slot.name, attempt, reason: "pane-read-unreadable",
|
|
2602
|
+
error: error instanceof Error ? error.message : String(error),
|
|
2603
|
+
});
|
|
2604
|
+
}
|
|
2605
|
+
}
|
|
2606
|
+
}
|
|
2607
|
+
const processTree = paneAbsent && worktreeSinceLaunch === "unchanged" && !hasSeed
|
|
2608
|
+
? await observeWorkerProcessTree(dispatchScript, wt)
|
|
2609
|
+
: "unmeasurable";
|
|
2610
|
+
if (processTree === "empty") {
|
|
2611
|
+
try {
|
|
2612
|
+
const confirmation = await driver.read(slot, PANE_READ_ROWS);
|
|
2613
|
+
paneAbsent = confirmation.trim().length === 0;
|
|
2614
|
+
if (!paneAbsent) {
|
|
2615
|
+
everHadOutput = true;
|
|
2616
|
+
if (stallProgress.observe({ paneText: confirmation, contextTokens }))
|
|
2617
|
+
lastProgressAt = Date.now();
|
|
2618
|
+
}
|
|
2619
|
+
}
|
|
2620
|
+
catch (error) {
|
|
2621
|
+
paneAbsent = false;
|
|
2622
|
+
if (!paneReadHeld) {
|
|
2623
|
+
paneReadHeld = true;
|
|
2624
|
+
journal.append("worker-dead-held", t.id, {
|
|
2625
|
+
slot: slot.name, attempt, reason: "pane-read-unreadable",
|
|
2626
|
+
error: error instanceof Error ? error.message : String(error),
|
|
2627
|
+
});
|
|
2628
|
+
}
|
|
2629
|
+
}
|
|
2630
|
+
}
|
|
2631
|
+
const worktreeDelta = processTree === "empty"
|
|
2632
|
+
&& paneAbsent ? await observeWorktreeDelta(taskBase, wt)
|
|
2633
|
+
: "unreadable";
|
|
2634
|
+
// The first process snapshot can race a just-starting child after the dispatch pane
|
|
2635
|
+
// disappeared. Re-read it after the pane and worktree legs have both held: preservation
|
|
2636
|
+
// is terminal, so a process appearing in that interval must withdraw the park rather
|
|
2637
|
+
// than be orphaned by it. The final worktree recheck remains inside preserveDeadWorker.
|
|
2638
|
+
const confirmedProcessTree = worktreeDelta === "unchanged"
|
|
2639
|
+
? await observeWorkerProcessTree(dispatchScript, wt)
|
|
2640
|
+
: "unmeasurable";
|
|
2641
|
+
const deathCertain = paneAbsent
|
|
2642
|
+
&& processTree === "empty"
|
|
2643
|
+
&& confirmedProcessTree === "empty"
|
|
2644
|
+
&& worktreeDelta === "unchanged";
|
|
2645
|
+
if (deathCertain) {
|
|
2646
|
+
const preservation = await preserveDeadWorker(wt, taskBase);
|
|
2647
|
+
if (!preservation.ref) {
|
|
2648
|
+
journal.append("worker-dead-held", t.id, {
|
|
2649
|
+
slot: slot.name, attempt, reason: `worktree-${preservation.state}`,
|
|
2650
|
+
});
|
|
2651
|
+
continue;
|
|
2652
|
+
}
|
|
2653
|
+
const ref = preservation.ref;
|
|
2654
|
+
const reason = `worker is unambiguously dead: pane absent, process tree empty, and worktree unchanged; preserved at ${ref}`;
|
|
2655
|
+
deadWorkerPark = { ref, reason };
|
|
2656
|
+
journal.append("worktree-preserved", t.id, { ref });
|
|
2657
|
+
journal.append("worker-dead-held", t.id, {
|
|
2658
|
+
slot: slot.name, attempt, reason: "unambiguous-worker-death", ref,
|
|
2659
|
+
});
|
|
2660
|
+
break;
|
|
2661
|
+
}
|
|
2366
2662
|
// T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
|
|
2367
2663
|
// never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
|
|
2368
2664
|
// the re-arm report on row growth once tokens stick, and contextTokens is sticky across
|
|
@@ -2536,7 +2832,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2536
2832
|
if (!finished && exitCode === null) {
|
|
2537
2833
|
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
2538
2834
|
timedOut = Date.now() - lastProgressAt >= stallWindowMs;
|
|
2539
|
-
|
|
2835
|
+
try {
|
|
2836
|
+
output = await driver.read(slot, PANE_READ_ROWS);
|
|
2837
|
+
}
|
|
2838
|
+
catch {
|
|
2839
|
+
// The poll loop already recorded the unreadable pane. Retain the last readable bytes
|
|
2840
|
+
// so this ambiguous path still reaches the ordinary timeout/consult backstop.
|
|
2841
|
+
}
|
|
2540
2842
|
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
2541
2843
|
const exit = exitRe.exec(output);
|
|
2542
2844
|
exitCode = exit ? Number(exit[1]) : null;
|
|
@@ -2670,6 +2972,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2670
2972
|
tokens = addUsage(tokens, attemptUsage);
|
|
2671
2973
|
metered++;
|
|
2672
2974
|
}
|
|
2975
|
+
if (deadWorkerPark) {
|
|
2976
|
+
await park(t, deadWorkerPark.reason, "stall", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { ref: deadWorkerPark.ref });
|
|
2977
|
+
return;
|
|
2978
|
+
}
|
|
2673
2979
|
let result = settleParsed ?? adapter.parse(output, nonce);
|
|
2674
2980
|
const workerFinished = finished;
|
|
2675
2981
|
const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut, deadChannel: deadChannelKilled });
|
package/dist/run/git.d.ts
CHANGED
|
@@ -34,6 +34,54 @@ export declare const deriveForkCap: (concurrency: number, cores?: number) => num
|
|
|
34
34
|
export declare const runWithForkBudget: <T>(concurrency: number, fn: () => Promise<T>) => Promise<T>;
|
|
35
35
|
/** The cap owned by the run on this async context; the standalone default outside one. */
|
|
36
36
|
export declare const resolvedForkCap: () => string;
|
|
37
|
+
/**
|
|
38
|
+
* T7: the CAPACITY a suite verdict was measured under — the fork cap the command's child actually
|
|
39
|
+
* received, and the core count that cap was divided from. Two verdicts are comparable only when both
|
|
40
|
+
* numbers match: a run resumed at a different concurrency divides the same machine by a different
|
|
41
|
+
* number, so a green measured in that other world is not evidence about this one.
|
|
42
|
+
*
|
|
43
|
+
* This pair is the WHOLE comparable identity, and the load averages a gate row already carries beside
|
|
44
|
+
* it are deliberately NOT part of it — no reader below ever feeds a load sample into the comparison.
|
|
45
|
+
* The capacity is deterministic and resolved here, where the child's environment is built. The load
|
|
46
|
+
* endpoints are neither: they are two samples taken at a gate's boundaries, and a gate's INTERIOR is
|
|
47
|
+
* invisible to them — on this milestone's own run a gate's interior reached well over twice what
|
|
48
|
+
* either of its own endpoints saw. So matching capacity establishes only that two measurements
|
|
49
|
+
* divided the same machine by the same number. It says nothing about whether the machine was calm.
|
|
50
|
+
*/
|
|
51
|
+
export interface RunCapacity {
|
|
52
|
+
forkCap: number;
|
|
53
|
+
cores: number;
|
|
54
|
+
}
|
|
55
|
+
export type CapacityRead = {
|
|
56
|
+
state: "present";
|
|
57
|
+
capacity: RunCapacity;
|
|
58
|
+
} | {
|
|
59
|
+
state: "absent";
|
|
60
|
+
} | {
|
|
61
|
+
state: "malformed";
|
|
62
|
+
};
|
|
63
|
+
/**
|
|
64
|
+
* Three states, never two. A record carrying NO capacity is an older record from before this stamp
|
|
65
|
+
* existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
|
|
66
|
+
* half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
|
|
67
|
+
* that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
|
|
68
|
+
*/
|
|
69
|
+
export declare function readCapacity(value: unknown): CapacityRead;
|
|
70
|
+
/**
|
|
71
|
+
* May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
|
|
72
|
+
* running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
|
|
73
|
+
* different capacity, or a current capacity the caller could not state → no.
|
|
74
|
+
*/
|
|
75
|
+
export declare function sameCapacity(recorded: unknown, current: RunCapacity | undefined): boolean;
|
|
76
|
+
export declare const describeCapacity: (value: unknown) => string;
|
|
77
|
+
/**
|
|
78
|
+
* The capacity a child spawned on THIS async context would receive: the same precedence `shell`
|
|
79
|
+
* applies below — an operator export of the cap wins over the run's own derived value — beside the
|
|
80
|
+
* cores it was divided from. A caller holding a command's own result reads the capacity off THAT
|
|
81
|
+
* result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
|
|
82
|
+
* decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
|
|
83
|
+
*/
|
|
84
|
+
export declare const resolvedCapacity: () => RunCapacity;
|
|
37
85
|
/** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
|
|
38
86
|
export declare const DEFAULT_SHELL_TIMEOUT_MS = 600000;
|
|
39
87
|
export interface ShResult {
|
|
@@ -42,6 +90,8 @@ export interface ShResult {
|
|
|
42
90
|
stderr: string;
|
|
43
91
|
timedOut?: boolean;
|
|
44
92
|
durationMs?: number;
|
|
93
|
+
/** T7: the capacity THIS child ran under, stamped where its environment was built (see `shell`). */
|
|
94
|
+
capacity?: RunCapacity;
|
|
45
95
|
}
|
|
46
96
|
export declare const setSpawnForTests: (fn: typeof spawn) => void;
|
|
47
97
|
export declare const resetSpawnForTests: () => void;
|