tickmarkr 1.96.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/brand.d.ts +31 -0
- package/dist/brand.js +44 -1
- package/dist/cli/commands/compile.js +32 -1
- package/dist/cli/commands/resume.js +9 -3
- package/dist/cli/commands/run.d.ts +61 -1
- package/dist/cli/commands/run.js +368 -17
- package/dist/cli/commands/status.js +303 -145
- package/dist/compile/collateral.d.ts +25 -0
- package/dist/compile/collateral.js +46 -11
- package/dist/compile/native.js +10 -0
- package/dist/drivers/herdr.d.ts +19 -13
- package/dist/drivers/herdr.js +88 -26
- package/dist/drivers/types.d.ts +2 -0
- package/dist/gates/acceptance.js +17 -7
- package/dist/gates/llm.d.ts +19 -0
- package/dist/gates/llm.js +104 -6
- package/dist/gates/run-gates.d.ts +18 -0
- package/dist/gates/run-gates.js +195 -29
- package/dist/gates/scope.d.ts +9 -1
- package/dist/gates/scope.js +22 -2
- package/dist/graph/graph.d.ts +1 -0
- package/dist/graph/graph.js +19 -2
- package/dist/report/compare.js +17 -2
- package/dist/run/daemon.d.ts +1 -8
- package/dist/run/daemon.js +231 -246
- package/dist/run/environment.d.ts +18 -1
- package/dist/run/environment.js +19 -2
- package/dist/run/journal.d.ts +50 -3
- package/dist/run/journal.js +181 -5
- package/dist/run/protocol.d.ts +4 -4
- package/dist/run/stall.d.ts +30 -0
- package/dist/run/stall.js +173 -0
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +20 -15
- package/skills/tickmarkr-overseer/scripts/seat-send.sh +98 -5
package/dist/run/daemon.js
CHANGED
|
@@ -8,26 +8,30 @@ import { classifyDeadChannel, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TR
|
|
|
8
8
|
import { allAdapters, getAdapter, probeAll, readDoctor, rolePools } from "../adapters/registry.js";
|
|
9
9
|
import { addUsage, channelKey, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
10
10
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
11
|
+
import { collateralHits } from "../compile/collateral.js";
|
|
11
12
|
import { DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
|
|
12
13
|
import { DeliveryReadinessError } from "../drivers/herdr.js";
|
|
13
14
|
import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
|
|
14
15
|
import { formatOwnedName } from "../drivers/types.js";
|
|
15
16
|
import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
|
|
16
17
|
import { runGates } from "../gates/run-gates.js";
|
|
17
|
-
import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus } from "../graph/graph.js";
|
|
18
|
+
import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest } from "../graph/graph.js";
|
|
18
19
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
19
20
|
import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
20
21
|
import { runEnvironment } from "./environment.js";
|
|
21
22
|
import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, runWithForkBudget, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
22
23
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
23
|
-
import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, pendingRepairFindings, phaseForGate, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
24
|
+
import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
24
25
|
import { isDiffCapPark } from "../gates/review.js";
|
|
25
26
|
import { acquireApprovalSerialization, acquireRunLock, releaseRunLock } from "./lock.js";
|
|
26
27
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
27
28
|
import { nextChannel, route } from "../route/router.js";
|
|
28
29
|
import { desiredPanes } from "./reconcile.js";
|
|
29
|
-
import { NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows } from "./stall.js";
|
|
30
|
+
import { harvestCpuFlatWindowMs, NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows, WorkerTreeCpuAccountant, } from "./stall.js";
|
|
30
31
|
import { armSupervision } from "./supervision.js";
|
|
32
|
+
// Compatibility exports for the daemon liveness tests and existing consumers. The implementation
|
|
33
|
+
// lives in stall.ts so gate dispatch can depend on it without importing the daemon.
|
|
34
|
+
export { harvestCpuFlatWindowMs, resetHarvestCpuFlatMsForTests, setHarvestCpuFlatMsForTests, workerTreeCpuMs, } from "./stall.js";
|
|
31
35
|
const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
|
|
32
36
|
// An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
|
|
33
37
|
// carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
|
|
@@ -182,6 +186,10 @@ function narrowRepairBattery(failing) {
|
|
|
182
186
|
return isOracleFailure(g) && g.meta?.unparseable !== true;
|
|
183
187
|
return false;
|
|
184
188
|
}
|
|
189
|
+
// v2.0 T2 (OBS-554): the measurement keys run-gates stamps on a GateResult, lifted verbatim onto the
|
|
190
|
+
// gate row. One list, one lift — both onGate sites record through the same helper.
|
|
191
|
+
const GATE_TELEMETRY_KEYS = ["durationMs", "load1Start", "load1End", "selectedDurationMs", "fullDurationMs", "invocations"];
|
|
192
|
+
const gateMeasurement = (meta = {}) => Object.fromEntries(GATE_TELEMETRY_KEYS.filter((k) => meta[k] !== undefined).map((k) => [k, meta[k]]));
|
|
185
193
|
/**
|
|
186
194
|
* T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
|
|
187
195
|
* review are now launched together, so a round can journal a failed review that the serial walk would
|
|
@@ -350,40 +358,6 @@ export function setHarvestSilentMsForTests(ms) {
|
|
|
350
358
|
export function resetHarvestSilentMsForTests() {
|
|
351
359
|
harvestSilentMs = HARVEST_SILENT_MS;
|
|
352
360
|
}
|
|
353
|
-
// A CPU delta needs two samples separated in WALL CLOCK, and the CPU clock is QUANTIZED: darwin's
|
|
354
|
-
// `ps` prints hundredths ("0:00.03"), linux's prints whole seconds ("00:00:01"). Equality across a
|
|
355
|
-
// window shorter than the quantum is not evidence of anything — a worker throttled to a low duty
|
|
356
|
-
// cycle accrues less than one tick per sample and reads flat while genuinely working. So the flat
|
|
357
|
-
// observation must span the LARGER of a floor and this many ticks of the clock actually in use:
|
|
358
|
-
// crossing 30 ticks means the tree burned <1 tick in 30, i.e. under ~3% of one core. On a
|
|
359
|
-
// hundredths host that is a 3s window; on a whole-second host it is 30s — still nothing against the
|
|
360
|
-
// ~15m redispatch it replaces. Resolution is read off the sampled rows, never assumed.
|
|
361
|
-
const HARVEST_CPU_FLAT_MS = 3_000;
|
|
362
|
-
const HARVEST_CPU_FLAT_TICKS = 30;
|
|
363
|
-
// Once the flat window opens, retain descendants often enough to observe brief tool processes that
|
|
364
|
-
// can start and exit between the daemon's ordinary wait slices. This sampler exists only during an
|
|
365
|
-
// eligible silence window; it is stopped on progress or as soon as the worker wait concludes.
|
|
366
|
-
const HARVEST_CPU_ACCOUNTING_POLL_MS = 100;
|
|
367
|
-
// T2 review (material): that 100ms cadence forks a shell plus `ps` ten times a second, and on a host
|
|
368
|
-
// where `ps` is unsupported or denied (the managed-sandbox class) EVERY sample fails — tens of
|
|
369
|
-
// thousands of processes per silent attempt, multiplied by daemon concurrency, for a probe that can
|
|
370
|
-
// never conclude anything. Persistent failure is structural, not transient, so the sampler STOPS
|
|
371
|
-
// after this many consecutive unreadable snapshots. It stays stopped for the silence window it was
|
|
372
|
-
// started for: read() then reports no CPU, the triad refuses to conclude and journals the gap, and a
|
|
373
|
-
// later window (after real progress clears the accountant) starts a fresh one that pays the same
|
|
374
|
-
// bounded probe again.
|
|
375
|
-
const HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP = 20;
|
|
376
|
-
let harvestCpuFlatMs;
|
|
377
|
-
export function harvestCpuFlatWindowMs(resolutionMs) {
|
|
378
|
-
return harvestCpuFlatMs ?? Math.max(HARVEST_CPU_FLAT_MS, resolutionMs * HARVEST_CPU_FLAT_TICKS);
|
|
379
|
-
}
|
|
380
|
-
/** Test seam — pin the flat window so a probe case need not sit through a real one. */
|
|
381
|
-
export function setHarvestCpuFlatMsForTests(ms) {
|
|
382
|
-
harvestCpuFlatMs = ms;
|
|
383
|
-
}
|
|
384
|
-
export function resetHarvestCpuFlatMsForTests() {
|
|
385
|
-
harvestCpuFlatMs = undefined;
|
|
386
|
-
}
|
|
387
361
|
// Once the silence gate is met the CPU probe owns the poll cadence: the trailer-wait slice is 30s,
|
|
388
362
|
// so two samples would otherwise cost a minute of wall clock apiece. Below the gate the only rule
|
|
389
363
|
// is not to sleep PAST it — at the shipped 5m gate that changes no slice a worker sees today.
|
|
@@ -509,167 +483,6 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
509
483
|
}
|
|
510
484
|
return tipFailed;
|
|
511
485
|
}
|
|
512
|
-
// `ps` CPU time: "[[dd-]hh:]mm:ss[.frac]" (darwin prints "0:00.03", linux "00:00:01", both print
|
|
513
|
-
// "1-02:03:04" past a day). Anything else is a header or a row this parser must not guess at.
|
|
514
|
-
// `frac` reports whether THIS host prints sub-second digits — the quantum the flat window is sized
|
|
515
|
-
// against, measured rather than assumed (a darwin sample is 10ms, a linux one 1000ms).
|
|
516
|
-
function parsePsCpu(raw) {
|
|
517
|
-
const m = /^(?:(\d+)-)?(?:(\d+):)?(\d+):(\d+(?:\.\d+)?)$/.exec(raw);
|
|
518
|
-
if (!m)
|
|
519
|
-
return undefined;
|
|
520
|
-
const ms = ((Number(m[1] ?? 0) * 24 + Number(m[2] ?? 0)) * 60 + Number(m[3])) * 60_000 + Math.round(Number(m[4]) * 1000);
|
|
521
|
-
return { ms, frac: m[4].includes(".") };
|
|
522
|
-
}
|
|
523
|
-
let linuxClockTickMs;
|
|
524
|
-
function linuxProcessCpuMs(pid, cwd) {
|
|
525
|
-
if (!existsSync("/proc/self/stat"))
|
|
526
|
-
return Promise.resolve(undefined);
|
|
527
|
-
// shGit, not sh: the accountant samples this path at a 100ms cadence, and a LOGIN shell would
|
|
528
|
-
// re-run the operator's profile (nvm/pyenv/direnv side effects included) on every sample.
|
|
529
|
-
linuxClockTickMs ??= shGit("getconf CLK_TCK", cwd, 15_000).then((r) => {
|
|
530
|
-
const ticks = r.code === 0 ? Number(r.stdout.trim()) : Number.NaN;
|
|
531
|
-
return Number.isFinite(ticks) && ticks > 0 ? 1_000 / ticks : undefined;
|
|
532
|
-
});
|
|
533
|
-
return linuxClockTickMs.then((resolutionMs) => {
|
|
534
|
-
if (resolutionMs === undefined)
|
|
535
|
-
return undefined;
|
|
536
|
-
try {
|
|
537
|
-
// `/proc/<pid>/stat` fields 14-17 are user/system jiffies for the process and its waited-for
|
|
538
|
-
// children. The child totals retain tools that start and exit wholly between live-tree polls.
|
|
539
|
-
// Split after the LAST ')' because comm may contain spaces or parentheses; field 3 is rest[0].
|
|
540
|
-
const stat = readFileSync(`/proc/${pid}/stat`, "utf8");
|
|
541
|
-
const fields = stat.slice(stat.lastIndexOf(")") + 2).trim().split(/\s+/);
|
|
542
|
-
const ticks = Number(fields[11]) + Number(fields[12]) + Number(fields[13]) + Number(fields[14]);
|
|
543
|
-
return Number.isFinite(ticks) ? { ms: ticks * resolutionMs, resolutionMs } : undefined;
|
|
544
|
-
}
|
|
545
|
-
catch {
|
|
546
|
-
return undefined; // process exited between ps ancestry capture and the precise CPU read
|
|
547
|
-
}
|
|
548
|
-
});
|
|
549
|
-
}
|
|
550
|
-
// T2 (OBS-264): the triad's CPU leg. Every non-seeded process of an attempt descends from that
|
|
551
|
-
// attempt's own dispatch script, whose path is unique — print, argv-interactive and resume launches
|
|
552
|
-
// all start there. (interactive-seed is intentionally fail-open below because its adapter-owned
|
|
553
|
-
// launch bypasses this script.) ONE `ps` snapshot finds the root and all current descendants:
|
|
554
|
-
// the agent CLI is a CHILD of the script's shell, so the root's own TIME never moves while the CLI
|
|
555
|
-
// thinks. `resolutionMs` is the sampled clock's quantum, which sizes the caller's flat window.
|
|
556
|
-
// Returns 0 when nothing matches: a worker whose process tree is gone is the strongest possible
|
|
557
|
-
// "not working". Returns undefined when the snapshot itself failed or parsed to nothing —
|
|
558
|
-
// unmeasurable CPU is never evidence a worker stopped, and the caller refuses to conclude on it.
|
|
559
|
-
async function workerTreeCpuSnapshot(marker, cwd) {
|
|
560
|
-
// shGit, not sh: same login-shell cost as the CLK_TCK probe above — `ps` needs no profile.
|
|
561
|
-
const snapshot = await shGit("ps -Awwo pid=,ppid=,time=,command=", cwd, 15_000);
|
|
562
|
-
if (snapshot.code !== 0)
|
|
563
|
-
return undefined;
|
|
564
|
-
const rows = [];
|
|
565
|
-
for (const line of snapshot.stdout.split("\n")) {
|
|
566
|
-
const m = /^\s*(\d+)\s+(\d+)\s+(\S+)\s+(.*)$/.exec(line);
|
|
567
|
-
if (!m)
|
|
568
|
-
continue;
|
|
569
|
-
const cpu = parsePsCpu(m[3]);
|
|
570
|
-
if (cpu !== undefined)
|
|
571
|
-
rows.push({ pid: m[1], ppid: m[2], cpuMs: cpu.ms, frac: cpu.frac, cmd: m[4] });
|
|
572
|
-
}
|
|
573
|
-
if (rows.length === 0)
|
|
574
|
-
return undefined;
|
|
575
|
-
const tree = new Set(rows.filter((p) => p.cmd.includes(marker)).map((p) => p.pid));
|
|
576
|
-
// `ps` output is not topologically ordered — relax the parent→child closure until it stops growing.
|
|
577
|
-
for (let grew = true; grew;) {
|
|
578
|
-
grew = false;
|
|
579
|
-
for (const p of rows) {
|
|
580
|
-
if (!tree.has(p.pid) && tree.has(p.ppid)) {
|
|
581
|
-
tree.add(p.pid);
|
|
582
|
-
grew = true;
|
|
583
|
-
}
|
|
584
|
-
}
|
|
585
|
-
}
|
|
586
|
-
const precise = new Map();
|
|
587
|
-
let preciseResolutionMs;
|
|
588
|
-
for (const p of rows) {
|
|
589
|
-
if (!tree.has(p.pid))
|
|
590
|
-
continue;
|
|
591
|
-
const cpu = await linuxProcessCpuMs(p.pid, cwd);
|
|
592
|
-
precise.set(p.pid, cpu?.ms ?? p.cpuMs);
|
|
593
|
-
if (cpu !== undefined)
|
|
594
|
-
preciseResolutionMs = cpu.resolutionMs;
|
|
595
|
-
}
|
|
596
|
-
// Even an empty worker tree needs the host's actual measurement quantum: on Linux the /proc
|
|
597
|
-
// jiffy clock remains available after the worker exits, while `ps time` only prints whole seconds.
|
|
598
|
-
if (preciseResolutionMs === undefined && existsSync("/proc/self/stat")) {
|
|
599
|
-
preciseResolutionMs = (await linuxProcessCpuMs(String(process.pid), cwd))?.resolutionMs;
|
|
600
|
-
}
|
|
601
|
-
return {
|
|
602
|
-
processes: precise,
|
|
603
|
-
resolutionMs: preciseResolutionMs ?? (rows.some((p) => p.frac) ? 10 : 1_000),
|
|
604
|
-
};
|
|
605
|
-
}
|
|
606
|
-
export async function workerTreeCpuMs(marker, cwd) {
|
|
607
|
-
const snapshot = await workerTreeCpuSnapshot(marker, cwd);
|
|
608
|
-
if (snapshot === undefined)
|
|
609
|
-
return undefined;
|
|
610
|
-
return {
|
|
611
|
-
ms: [...snapshot.processes.values()].reduce((sum, cpuMs) => sum + cpuMs, 0),
|
|
612
|
-
resolutionMs: snapshot.resolutionMs,
|
|
613
|
-
};
|
|
614
|
-
}
|
|
615
|
-
// Sparse live-tree totals forget a tool's CPU as soon as that tool exits. This attempt-local
|
|
616
|
-
// accountant instead adds each observed process's CPU DELTA to a monotonic total and replaces only
|
|
617
|
-
// the live-PID cursor on each sample. When a PID disappears, its contribution stays in `totalMs`;
|
|
618
|
-
// if that PID is later reused, its fresh total is added from zero because it left `live` in between.
|
|
619
|
-
class WorkerTreeCpuAccountant {
|
|
620
|
-
marker;
|
|
621
|
-
cwd;
|
|
622
|
-
active = false;
|
|
623
|
-
loop;
|
|
624
|
-
live = new Map();
|
|
625
|
-
totalMs = 0;
|
|
626
|
-
gaps = 0;
|
|
627
|
-
consecutiveGaps = 0;
|
|
628
|
-
latest;
|
|
629
|
-
constructor(marker, cwd) {
|
|
630
|
-
this.marker = marker;
|
|
631
|
-
this.cwd = cwd;
|
|
632
|
-
}
|
|
633
|
-
async sample() {
|
|
634
|
-
const snapshot = await workerTreeCpuSnapshot(this.marker, this.cwd);
|
|
635
|
-
if (snapshot === undefined) {
|
|
636
|
-
this.gaps++;
|
|
637
|
-
this.live.clear();
|
|
638
|
-
this.latest = undefined;
|
|
639
|
-
// Stop forking `ps` at 10Hz once the host has proved it cannot answer — see the cap's comment.
|
|
640
|
-
if (++this.consecutiveGaps >= HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP)
|
|
641
|
-
this.active = false;
|
|
642
|
-
return;
|
|
643
|
-
}
|
|
644
|
-
this.consecutiveGaps = 0;
|
|
645
|
-
for (const [pid, cpuMs] of snapshot.processes) {
|
|
646
|
-
const prior = this.live.get(pid);
|
|
647
|
-
this.totalMs += prior === undefined || cpuMs < prior ? cpuMs : cpuMs - prior;
|
|
648
|
-
}
|
|
649
|
-
this.live = snapshot.processes;
|
|
650
|
-
this.latest = { ms: this.totalMs, resolutionMs: snapshot.resolutionMs };
|
|
651
|
-
}
|
|
652
|
-
async start() {
|
|
653
|
-
if (this.active)
|
|
654
|
-
return;
|
|
655
|
-
this.active = true;
|
|
656
|
-
await this.sample();
|
|
657
|
-
this.loop = (async () => {
|
|
658
|
-
while (this.active) {
|
|
659
|
-
await new Promise((resolve) => setTimeout(resolve, HARVEST_CPU_ACCOUNTING_POLL_MS));
|
|
660
|
-
if (this.active)
|
|
661
|
-
await this.sample();
|
|
662
|
-
}
|
|
663
|
-
})();
|
|
664
|
-
}
|
|
665
|
-
read() {
|
|
666
|
-
return { cpu: this.latest, gaps: this.gaps };
|
|
667
|
-
}
|
|
668
|
-
async stop() {
|
|
669
|
-
this.active = false;
|
|
670
|
-
await this.loop;
|
|
671
|
-
}
|
|
672
|
-
}
|
|
673
486
|
async function commitsAheadOf(base, wt) {
|
|
674
487
|
const head = await gitHead(wt);
|
|
675
488
|
if (head === base)
|
|
@@ -925,6 +738,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
925
738
|
let releaseApprovalSerialization;
|
|
926
739
|
try {
|
|
927
740
|
let graph = loadGraph(repoRoot);
|
|
741
|
+
// One bounded snapshot supplies every dispatch. On resume the current journal still participates
|
|
742
|
+
// in the fold so its task-done/ordinary-approval events can retire older evidence, but findings it
|
|
743
|
+
// produced are suppressed: same-run feedback already replays those bytes and must not deliver them
|
|
744
|
+
// twice. Evidence from genuinely earlier runs remains available to a resumed fresh dispatch.
|
|
745
|
+
const priorRunEvidence = readPriorRunEvidence(repoRoot, graph.tasks, { suppressRunId: runId });
|
|
928
746
|
// GATE-FIX-4 defect 4 (no-op run refusal): a fresh run on a graph with nothing dispatchable used
|
|
929
747
|
// to journal {run-start, run-end} with zero dispatches — and downstream readers (greenness exit,
|
|
930
748
|
// status, notify) treat that run-end as completion, so an all-terminal graph "went green" having
|
|
@@ -1159,6 +977,19 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1159
977
|
for (const w of await detectVacuousOracles(repoRoot, graph.tasks))
|
|
1160
978
|
journal.append("baseline-warning", w.taskId, { ...w });
|
|
1161
979
|
}
|
|
980
|
+
// OBS-547: ONE full per-task collateral prediction for the whole run, computed here — uncapped, and
|
|
981
|
+
// handed whole to every scope gate below (never recomputed at a red, never the plan's 20-item view).
|
|
982
|
+
// Persisted beside baseline.json and RELOADED on resume (same reason baseline is): the prediction is
|
|
983
|
+
// pre-dispatch state, and rescanning a repository the run has since edited would let an offender flip
|
|
984
|
+
// between predicted and missed across stop/amend/resume. Missing file (pre-OBS-547 journal) ⇒ scan and
|
|
985
|
+
// pin it now, so every later resume of this run agrees with this one.
|
|
986
|
+
const collateralPath = join(journal.dir, "collateral.json");
|
|
987
|
+
const pinnedCollateral = opts.resume && existsSync(collateralPath)
|
|
988
|
+
? new Map(Object.entries(JSON.parse(readFileSync(collateralPath, "utf8"))))
|
|
989
|
+
: null;
|
|
990
|
+
const collateral = pinnedCollateral ?? collateralHits(graph.tasks, repoRoot);
|
|
991
|
+
if (!pinnedCollateral)
|
|
992
|
+
writeFileSync(collateralPath, JSON.stringify(Object.fromEntries(collateral), null, 2));
|
|
1162
993
|
// T6: open the narrator AFTER run-start/run-resume is journaled so the watch surface has a run to
|
|
1163
994
|
// show. driver.narrator is undefined on subprocess → no-op (subprocess spawns nothing). Swallowed:
|
|
1164
995
|
// a failed-to-open or later-dead watch pane never affects the run.
|
|
@@ -1249,11 +1080,60 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1249
1080
|
saveGraph(repoRoot, graph);
|
|
1250
1081
|
journal.append("task-human", t.id, { reason, kind });
|
|
1251
1082
|
if (assignment) {
|
|
1252
|
-
|
|
1083
|
+
// OBS-547: `metered` counts CHARGEABLE metered attempts, so an unchargeable dispatch passes 0 and
|
|
1084
|
+
// the count is omitted rather than written as 0 or as `1` beside `attempts: 0` — a row claiming
|
|
1085
|
+
// more metered attempts than it charges reports as "floor: 1/0 attempts metered". tokens stay
|
|
1086
|
+
// (the spend was real); tokens-without-a-count is the already-modelled degraded ⇒ floor row.
|
|
1087
|
+
journal.telemetry({ taskId: t.id, shape: t.shape, adapter: assignment.adapter, model: assignment.model, channel: assignment.channel, attempts, outcome: "human", durationMs: Date.now() - startMs, parkKind: kind, gateFails, consults, tokens, meteredAttempts: tokens && metered ? metered : undefined, retryMode });
|
|
1253
1088
|
}
|
|
1254
1089
|
await reconcile({ spareLiveLlm: true }); // task-human is a terminal event — sweep, sparing sibling tasks' live LLM panes
|
|
1255
1090
|
await driver.notify(`tickmarkr ${runId}: ${t.id} needs a human — ${reason}`, { tier: "attention" });
|
|
1256
1091
|
};
|
|
1092
|
+
// OBS-547: cross-reference a scope red against the prediction this run already computed. Every hard
|
|
1093
|
+
// offender predicted ⇒ an AUTHORING defect whose repair is pre-written: journal the classification
|
|
1094
|
+
// with the verbatim files[] lines and park unchargeable. It owes nothing to the quality machinery —
|
|
1095
|
+
// no chargeable attempt, no gateFails, no escalation, no ladder rung — and `scope-authoring` is what
|
|
1096
|
+
// makes a resume replay agree (journal.ts replayResumeState). One unpredicted offender keeps today's
|
|
1097
|
+
// chargeable behaviour and records the miss, so the lint's blind spots accumulate as evidence rather
|
|
1098
|
+
// than folklore. ONE disposition for BOTH paths that observe a red — the ordinary attempt and the
|
|
1099
|
+
// resume gate replay: which code path noticed the red must never decide who pays for it. Returns
|
|
1100
|
+
// true when it parked; the caller then returns without charging anything.
|
|
1101
|
+
// `attempts` is the count of CHARGEABLE attempts before this dispatch — never this dispatch's own
|
|
1102
|
+
// count. The two callers arrive at it differently (the attempt loop's index already excludes the
|
|
1103
|
+
// dispatch in flight; a gate replay's rs.attempts already includes it), and the ordinal journaled
|
|
1104
|
+
// below is derived from it here so both paths number the same dispatch identically.
|
|
1105
|
+
const dispositionScopeRed = async (t, results, assignment, attempts, startMs, gateFails, consults, tokens, metered, retryMode) => {
|
|
1106
|
+
const scopeRed = results.find((g) => g.gate === "scope" && gateFailed(g));
|
|
1107
|
+
const verdict = scopeRed?.meta?.collateral;
|
|
1108
|
+
if (!verdict)
|
|
1109
|
+
return false;
|
|
1110
|
+
if (verdict.authoring) {
|
|
1111
|
+
journal.append("scope-authoring", t.id, {
|
|
1112
|
+
gate: "scope",
|
|
1113
|
+
predicted: verdict.predicted,
|
|
1114
|
+
repair: verdict.repair,
|
|
1115
|
+
attempt: attempts + 1,
|
|
1116
|
+
chargeable: false,
|
|
1117
|
+
// The dispatch was PHYSICALLY metered even though nobody is charged for it. Physical metering
|
|
1118
|
+
// is recorded here, separately, so the telemetry row can stay chargeable-consistent: passing
|
|
1119
|
+
// this count on would print `meteredAttempts: 1` beside `attempts: 0`.
|
|
1120
|
+
...(tokens ? { tokens, meteredDispatches: metered } : {}),
|
|
1121
|
+
...(assignment ? { channel: channelKey(assignment) } : {}),
|
|
1122
|
+
});
|
|
1123
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} scope red was PREDICTED — authoring defect, no attempt charged`, { tier: "attention" });
|
|
1124
|
+
await park(t, `scope: every out-of-scope path was predicted by the collateral lint before dispatch — authoring defect, not a worker failure. Repair:\n${verdict.repair}`, "authoring", assignment, attempts, startMs, gateFails, consults, tokens, 0, retryMode);
|
|
1125
|
+
return true;
|
|
1126
|
+
}
|
|
1127
|
+
if (verdict.missed.length) {
|
|
1128
|
+
journal.append("collateral-miss", t.id, {
|
|
1129
|
+
gate: "scope",
|
|
1130
|
+
unpredicted: verdict.missed,
|
|
1131
|
+
predicted: verdict.predicted,
|
|
1132
|
+
attempt: attempts + 1,
|
|
1133
|
+
});
|
|
1134
|
+
}
|
|
1135
|
+
return false;
|
|
1136
|
+
};
|
|
1257
1137
|
const execTask = async (t) => {
|
|
1258
1138
|
const startMs = Date.now();
|
|
1259
1139
|
const taskTimeoutMinutes = t.timeoutMinutes ?? cfg.taskTimeoutMinutes;
|
|
@@ -1283,6 +1163,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1283
1163
|
// fresh-budget release, prefer nextChannel over the surviving tried-list so burned channels are
|
|
1284
1164
|
// not re-tried first (consult bans / prior failovers survive the release).
|
|
1285
1165
|
const rs = resume.get(t.id);
|
|
1166
|
+
const contentDigest = taskContentDigest(t);
|
|
1286
1167
|
if (rs?.lastAssignment && rs.attempts > 0
|
|
1287
1168
|
&& channels.some((c) => channelKey(c) === channelKey(rs.lastAssignment))
|
|
1288
1169
|
&& !demotedChannels.has(channelKey(rs.lastAssignment))) {
|
|
@@ -1376,7 +1257,19 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1376
1257
|
...(g.meta?.fullSuite === true ? { fullSuite: true } : {}),
|
|
1377
1258
|
// A finding's path is its own evidence path. Do not pass task scope here: a declaration says
|
|
1378
1259
|
// where work is allowed, not where this verdict found the defect.
|
|
1379
|
-
...(blocking ? {
|
|
1260
|
+
...(blocking ? {
|
|
1261
|
+
taskContentDigest: contentDigest,
|
|
1262
|
+
findings: structuredFindings(g.gate, g.details),
|
|
1263
|
+
} : {}),
|
|
1264
|
+
// v2.0 T2 (OBS-554): the gate's OWN measurement, lifted verbatim from the meta run-gates
|
|
1265
|
+
// stamped WHERE THE GATE RAN. Nothing here re-derives a duration by subtracting journal
|
|
1266
|
+
// timestamps — that would measure this row's queue as well as its work. It rides the
|
|
1267
|
+
// gate-result row itself because that IS the row a recalibration reads: a measurement kept
|
|
1268
|
+
// in a side stream is one join away from the verdict it explains, and the two can drift.
|
|
1269
|
+
// A gate run-gates did not measure contributes NO field rather than a fabricated zero — for
|
|
1270
|
+
// every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
|
|
1271
|
+
// seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
|
|
1272
|
+
...gateMeasurement(g.meta),
|
|
1380
1273
|
});
|
|
1381
1274
|
};
|
|
1382
1275
|
// R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
|
|
@@ -1418,9 +1311,18 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1418
1311
|
// in resume state. The journal already holds the upheld review's bytes; no reset of attempt or
|
|
1419
1312
|
// channel state can take them away, on any path, including `resume --retry-failed`.
|
|
1420
1313
|
const upheldFeedback = upheldFeedbackByTask(journal.read()).get(t.id) ?? rs?.upheldFeedback;
|
|
1421
|
-
|
|
1314
|
+
const carriedEvidence = priorRunEvidence.findings
|
|
1315
|
+
.filter((finding) => finding.taskId === t.id)
|
|
1316
|
+
.map(formatPriorFindingEvidence)
|
|
1317
|
+
.join("\n\n");
|
|
1318
|
+
const withCarriedEvidence = (brief) => {
|
|
1319
|
+
if (!carriedEvidence || brief.includes(carriedEvidence))
|
|
1320
|
+
return brief;
|
|
1321
|
+
return brief ? `${brief}\n\n${carriedEvidence}` : carriedEvidence;
|
|
1322
|
+
};
|
|
1323
|
+
let feedback = withCarriedEvidence(upheldFeedback
|
|
1422
1324
|
? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${upheldFeedback}`
|
|
1423
|
-
: "";
|
|
1325
|
+
: "");
|
|
1424
1326
|
let ladderIdx = 0;
|
|
1425
1327
|
let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
|
|
1426
1328
|
let gateFails = 0; // TEL-02: incremented ONLY where feedback is built from failing gates — never derived from attempts (quota failovers bump attempts too, Pitfall 6)
|
|
@@ -1677,6 +1579,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1677
1579
|
const { results } = await runGates(resumedTask, {
|
|
1678
1580
|
worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
|
|
1679
1581
|
commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
|
|
1582
|
+
collateral: collateral.get(t.id) ?? [],
|
|
1680
1583
|
// a recheck re-verifies a human's release: it never selects tests down, it runs the suite.
|
|
1681
1584
|
pipeline: "v185",
|
|
1682
1585
|
via: cfg.visibility.llm === "pane"
|
|
@@ -1710,6 +1613,20 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1710
1613
|
graph = addEvidence(graph, t.id, { commits: approvedCommits, gateResults: results });
|
|
1711
1614
|
saveGraph(repoRoot, graph);
|
|
1712
1615
|
if (!results.every(gateSatisfied)) {
|
|
1616
|
+
// OBS-547: disposition FIRST, and through the same helper the ordinary attempt path uses. A
|
|
1617
|
+
// crash between a journaled scope red and its classification must not turn a predicted red
|
|
1618
|
+
// into a charged one just because a resume is what observed it.
|
|
1619
|
+
// rs.attempts COUNTS the interrupted dispatch this replay is judging; the chargeable
|
|
1620
|
+
// attempts behind it are one fewer. Passing the count itself would number the same dispatch
|
|
1621
|
+
// one higher here than on the ordinary path and bill an attempt the fresh path forgives.
|
|
1622
|
+
// UNLESS an earlier resume already journaled the classification and died before its park:
|
|
1623
|
+
// replayResumeState() has then ALREADY taken that dispatch back, so subtracting again would
|
|
1624
|
+
// erase an EARLIER chargeable attempt and attribute the park to the assignment the rewind
|
|
1625
|
+
// restored. Ask the journal which state this is, and take the classified dispatch's own
|
|
1626
|
+
// assignment back with it.
|
|
1627
|
+
const classified = journal.classifiedDispatch(t.id);
|
|
1628
|
+
if (await dispositionScopeRed(t, results, classified?.assignment ?? gateAuthor, classified ? (rs?.attempts ?? 0) : Math.max(0, (rs?.attempts ?? 0) - 1), startMs, gateFails, consults, tokens, metered, retryMode))
|
|
1629
|
+
return;
|
|
1713
1630
|
gateFails++;
|
|
1714
1631
|
// Observed green gates are only measurements. If the resumed suffix is red, preserve that
|
|
1715
1632
|
// result in the journal and return to the ordinary attempt/consult ladder, which rebuilds
|
|
@@ -1734,7 +1651,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1734
1651
|
}
|
|
1735
1652
|
graph = setStatus(graph, t.id, "done");
|
|
1736
1653
|
saveGraph(repoRoot, graph);
|
|
1737
|
-
journal.append("task-done", t.id, {
|
|
1654
|
+
journal.append("task-done", t.id, {
|
|
1655
|
+
attempts: rs?.attempts ?? 0, assignment: gateAuthor, taskContentDigest: contentDigest,
|
|
1656
|
+
});
|
|
1738
1657
|
journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
|
|
1739
1658
|
journal.telemetry({
|
|
1740
1659
|
taskId: t.id, shape: t.shape, adapter: gateAuthor.adapter, model: gateAuthor.model,
|
|
@@ -1956,7 +1875,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1956
1875
|
const promptFile = writePrompt(journal.dir, t, attempt, feedback, nonce);
|
|
1957
1876
|
// OBS-56: state the non-interactive, one-pass finish contract and the OBS-54 stall budget in every
|
|
1958
1877
|
// worker prompt, not only consult retry guidance. Prepended so prompt.ts's completion trailer stays last.
|
|
1959
|
-
const workerContract = `## Harness contract\n- This harness is non-interactive: make one continuous pass; do not stop for questions or follow-up input.\n- You have a ${taskTimeoutMinutes} minute stall window.
|
|
1878
|
+
const workerContract = `## Harness contract\n- This harness is non-interactive: make one continuous pass; do not stop for questions or follow-up input.\n- You have a ${taskTimeoutMinutes} minute stall window. The gates run the full suite for you — never spend this window on one; commit your work and emit the completion trailer inside it.\n- Each test: acceptance criterion must exist as a vitest test whose OWN title (the leaf, not counting enclosing describe titles) is the criterion string verbatim — never shortened, never decorated. Nesting under describe() is allowed.`; // OBS-64; OBS-511: leaf-title rule stated where the worker reads it; OBS-548: the suite is the GATES' job — this repo's own suite outlasts the output-silence windows the daemon polices, so a worker obeying "budget the full suite" was killed by construction
|
|
1960
1879
|
// OBS-47: state the worktree layout contract in the worker prompt (cheap-tier workers were
|
|
1961
1880
|
// committing/deleting node_modules and tripping the scope gate). The harness re-asserts the link
|
|
1962
1881
|
// itself before gates regardless of what the worker does with it.
|
|
@@ -2023,30 +1942,34 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2023
1942
|
unmeasurableNoted = true;
|
|
2024
1943
|
journal.append("worker-harvest-unmeasurable", t.id, { slot: slot.name, attempt, reason });
|
|
2025
1944
|
};
|
|
2026
|
-
|
|
2027
|
-
|
|
1945
|
+
// The CPU leg needs a marker in the worker's own argv, and every launch path puts this
|
|
1946
|
+
// attempt's dispatch script there EXCEPT interactiveSeed: runInteractiveSeed launches the
|
|
1947
|
+
// TUI directly, by a command the ADAPTER owns (seed.launch(model)) which tickmarkr cannot
|
|
1948
|
+
// make attempt-unique and must deliver verbatim. A marker that matches nothing reads as
|
|
1949
|
+
// zero CPU — precisely the false "flat" that would harvest a worker mid-turn — so a seeded
|
|
1950
|
+
// attempt has no measurable CPU leg and neither reader ever concludes it. The other half of
|
|
1951
|
+
// OBS-264 is untouched there: when its window does expire with commits on the worktree, the
|
|
1952
|
+
// no-trailer tail gates them instead of buying a fresh worker to re-produce them.
|
|
1953
|
+
const armCpuLeg = async (armed) => {
|
|
1954
|
+
if (!armed) {
|
|
2028
1955
|
await cpuAccountant?.stop();
|
|
2029
1956
|
cpuAccountant = undefined;
|
|
2030
1957
|
cpuFlat = undefined;
|
|
2031
1958
|
cpuGapCount = 0;
|
|
2032
|
-
return
|
|
1959
|
+
return;
|
|
2033
1960
|
}
|
|
2034
|
-
|
|
2035
|
-
|
|
2036
|
-
|
|
2037
|
-
|
|
2038
|
-
|
|
2039
|
-
|
|
2040
|
-
// OBS-264 is untouched there: when its window does expire with commits on the worktree, the
|
|
2041
|
-
// no-trailer tail gates them instead of buying a fresh worker to re-produce them.
|
|
1961
|
+
if (hasSeed || cpuAccountant !== undefined)
|
|
1962
|
+
return;
|
|
1963
|
+
cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt);
|
|
1964
|
+
await cpuAccountant.start();
|
|
1965
|
+
};
|
|
1966
|
+
const readCpuLeg = () => {
|
|
2042
1967
|
if (hasSeed) {
|
|
2043
1968
|
noteUnmeasurable("interactive-seed launch is not in the probed process tree");
|
|
2044
|
-
return
|
|
2045
|
-
}
|
|
2046
|
-
if (cpuAccountant === undefined) {
|
|
2047
|
-
cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt);
|
|
2048
|
-
await cpuAccountant.start();
|
|
1969
|
+
return { state: "unmeasurable" };
|
|
2049
1970
|
}
|
|
1971
|
+
if (cpuAccountant === undefined)
|
|
1972
|
+
return { state: "unmeasurable" }; // no window open: nothing measured yet
|
|
2050
1973
|
const observation = cpuAccountant.read();
|
|
2051
1974
|
if (observation.gaps !== cpuGapCount) {
|
|
2052
1975
|
cpuGapCount = observation.gaps;
|
|
@@ -2060,21 +1983,29 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2060
1983
|
// structural hole as the seeded launch, and must not be the one that stays silent.
|
|
2061
1984
|
cpuFlat = undefined;
|
|
2062
1985
|
noteUnmeasurable("the worker process snapshot could not be read");
|
|
2063
|
-
return
|
|
1986
|
+
return { state: "unmeasurable" };
|
|
2064
1987
|
}
|
|
2065
1988
|
const now = Date.now();
|
|
2066
1989
|
if (cpu.ms !== cpuFlat?.ms) {
|
|
2067
1990
|
cpuFlat = { ms: cpu.ms, since: now };
|
|
2068
|
-
return
|
|
1991
|
+
return { state: "accruing" };
|
|
2069
1992
|
}
|
|
2070
|
-
|
|
1993
|
+
return now - cpuFlat.since < harvestCpuFlatWindowMs(cpu.resolutionMs)
|
|
1994
|
+
? { state: "accruing" }
|
|
1995
|
+
: { state: "flat", cpu };
|
|
1996
|
+
};
|
|
1997
|
+
const harvestConcludes = async (silentMs) => {
|
|
1998
|
+
if (silentMs < harvestSilentMs)
|
|
1999
|
+
return false;
|
|
2000
|
+
const leg = readCpuLeg();
|
|
2001
|
+
if (leg.state !== "flat")
|
|
2071
2002
|
return false;
|
|
2072
2003
|
const carried = await commitsAheadOf(taskBase, wt);
|
|
2073
2004
|
if (carried.length === 0)
|
|
2074
2005
|
return false; // nothing landed: not this branch's population
|
|
2075
2006
|
journal.append("worker-harvest", t.id, {
|
|
2076
2007
|
slot: slot.name, attempt, commits: carried.length,
|
|
2077
|
-
silentMs, cpuMs: cpu.ms, cpuResolutionMs: cpu.resolutionMs,
|
|
2008
|
+
silentMs, cpuMs: leg.cpu.ms, cpuResolutionMs: leg.cpu.resolutionMs,
|
|
2078
2009
|
});
|
|
2079
2010
|
return true;
|
|
2080
2011
|
};
|
|
@@ -2117,6 +2048,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2117
2048
|
let output;
|
|
2118
2049
|
let exitCode;
|
|
2119
2050
|
let timedOut = false;
|
|
2051
|
+
// OBS-548 addendum: the mechanism that fired is what the repair brief and the consult read.
|
|
2052
|
+
// A fast-kill lands with the rolling window nowhere near expiry (measured: 693 s of silence
|
|
2053
|
+
// inside a 1,800,000 ms window), so the trailer-less tail's stall-timeout fallback named a
|
|
2054
|
+
// window that could not have killed anything and briefed the next worker to fix a stall that
|
|
2055
|
+
// never happened. Fourth conflation in the taxonomy OBS-53 opened.
|
|
2056
|
+
let deadChannelKilled = false;
|
|
2120
2057
|
// T2 review: print mode's "the exit marker appeared". Kept apart from `finished` (the
|
|
2121
2058
|
// trailer) but still needed by the keepPanes decision below, whose contract is about a
|
|
2122
2059
|
// subprocess tree that REACHED its exit marker, not about what the worker claimed.
|
|
@@ -2152,7 +2089,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2152
2089
|
keptSlots.push(slot);
|
|
2153
2090
|
else
|
|
2154
2091
|
await closeSlot(slot);
|
|
2155
|
-
feedback = `delivery readiness failed after ${error.waitedMs}ms; pane transcript:\n${error.transcript}
|
|
2092
|
+
feedback = withCarriedEvidence(`delivery readiness failed after ${error.waitedMs}ms; pane transcript:\n${error.transcript}`);
|
|
2156
2093
|
const step = r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
|
|
2157
2094
|
journal.append("escalation", t.id, { step, attempt: attempt + 1 });
|
|
2158
2095
|
await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
|
|
@@ -2260,6 +2197,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2260
2197
|
// lastProgressAt — see the kill below.
|
|
2261
2198
|
let quotaStreak = 0;
|
|
2262
2199
|
let rowSaturationHeld = false; // journaled once per attempt when the kill stands down
|
|
2200
|
+
let cpuHeld = false; // likewise for the CPU leg's stand-down (OBS-548)
|
|
2263
2201
|
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
2264
2202
|
const sliceStart = Date.now();
|
|
2265
2203
|
const remaining = stallWindowMs - (sliceStart - lastProgressAt);
|
|
@@ -2396,6 +2334,30 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2396
2334
|
// daemon has nothing left to do — the pane falls back under the fast-kill and page
|
|
2397
2335
|
// watchdogs like any other, instead of riding the whole rolling window untended.
|
|
2398
2336
|
const nudgePending = nudgeable && (!nudged || nudgeDeadline !== undefined);
|
|
2337
|
+
// T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
|
|
2338
|
+
// never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
|
|
2339
|
+
// the re-arm report on row growth once tokens stick, and contextTokens is sticky across
|
|
2340
|
+
// read misses, so a metered non-nudgeable adapter (pi) streaming rows under a stale
|
|
2341
|
+
// counter presented a frozen lastProgressAt and was killed mid-work. lastRowGrowthAt is
|
|
2342
|
+
// recorded on every high-water advance, suppressed or not; token growth already rides
|
|
2343
|
+
// lastProgressAt. Either one advancing is output growth.
|
|
2344
|
+
const lastOutputGrowthAt = Math.max(stallProgress.lastRowGrowthAt ?? 0, lastProgressAt);
|
|
2345
|
+
// Everything the fast-kill can decide from the CHANNEL, evaluated before the CPU leg is
|
|
2346
|
+
// asked for anything: the kill's own probe costs a `ps` every 100ms, so an ineligible
|
|
2347
|
+
// slice must not pay for it. The CPU reading is consulted at the kill itself, below.
|
|
2348
|
+
const fastKillEligible = !stallProgress.rowSignalSaturated
|
|
2349
|
+
&& !nudgePending && !nudgeFailed
|
|
2350
|
+
&& sliceNow - lastOutputGrowthAt >= deadChannelFastKillMs
|
|
2351
|
+
&& worktreeSinceLaunch === "unchanged";
|
|
2352
|
+
const harvestCpuEligible = !nudgePending && !nudgeFailed
|
|
2353
|
+
&& sliceNow - lastProgressAt >= harvestSilentMs;
|
|
2354
|
+
// OBS-548: ONE accountant serves both readers, so its lifecycle is decided once per
|
|
2355
|
+
// slice from BOTH windows. Arming per-reader let the harvest's own "not silent enough
|
|
2356
|
+
// yet" branch stop the sampler the fast-kill had just started, and a restarted
|
|
2357
|
+
// accountant re-bases its monotonic total — the leg could then never read flat. The
|
|
2358
|
+
// accountant also carries the readers' nudge holds: when neither reader may conclude,
|
|
2359
|
+
// there is no reason to fork its 10 Hz process snapshots through the remaining window.
|
|
2360
|
+
await armCpuLeg(harvestCpuEligible || fastKillEligible);
|
|
2399
2361
|
// T2 (OBS-264): the liveness triad CONCLUDES the wait on finished work — commits ahead
|
|
2400
2362
|
// of the task base (this attempt's own AND any carried forward — see the eligibility
|
|
2401
2363
|
// comment at the dispatch site), a flat worker-tree CPU delta, and >= harvestSilentMs
|
|
@@ -2408,10 +2370,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2408
2370
|
// claude-code worker before the rescue could fire, leaving T1's nudge dead code for
|
|
2409
2371
|
// exactly the committed-and-stalled population OBS-264 is about. Holding concludes at
|
|
2410
2372
|
// ~14m (nudge + grace) rather than ~36m — nearly all of the OBS-264 win, and a worker
|
|
2411
|
-
// that only needed a submit answers with a full trailer instead of partial work.
|
|
2412
|
-
// ANSWERED
|
|
2413
|
-
//
|
|
2414
|
-
|
|
2373
|
+
// that only needed a submit answers with a full trailer instead of partial work.
|
|
2374
|
+
// An ANSWERED nudge leaves nothing pending, so the triad governs again; the hold is on a
|
|
2375
|
+
// pending daemon ACTION, never on the adapter being nudgeable.
|
|
2376
|
+
// OBS-548: an UNDELIVERABLE nudge holds too. A worker inside one long foreground command
|
|
2377
|
+
// has no input box, so it is the population that CANNOT be nudged and is also the one
|
|
2378
|
+
// most likely to be legitimately silent — letting its delivery failure lift the hold made
|
|
2379
|
+
// the failure itself the trigger. `:357-359` already reads a false driver.nudge return as
|
|
2380
|
+
// a delivery outcome, "not proof of an unreachable channel"; unreachable is not at rest.
|
|
2381
|
+
if (!nudgePending && !nudgeFailed && await harvestConcludes(sliceNow - lastProgressAt))
|
|
2415
2382
|
break;
|
|
2416
2383
|
// T1 (R1 dead-channel fast-kill): no trailer, an unchanged launch tree, and no output growth
|
|
2417
2384
|
// for the fast-kill window — the channel is dead, so conclude NOW
|
|
@@ -2443,20 +2410,30 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2443
2410
|
rowSaturationHeld = true;
|
|
2444
2411
|
journal.append("worker-dead-held", t.id, { slot: slot.name, attempt, reason: "row-signal-saturated" });
|
|
2445
2412
|
}
|
|
2446
|
-
//
|
|
2447
|
-
//
|
|
2448
|
-
//
|
|
2449
|
-
//
|
|
2450
|
-
//
|
|
2451
|
-
//
|
|
2452
|
-
//
|
|
2453
|
-
|
|
2454
|
-
if (
|
|
2455
|
-
|
|
2456
|
-
|
|
2457
|
-
|
|
2458
|
-
|
|
2459
|
-
|
|
2413
|
+
// OBS-548: the FOURTH leg, and the one the daemon already measured. The three legs above
|
|
2414
|
+
// are all channel-side: they say a pane stopped talking. The tree's CPU says whether
|
|
2415
|
+
// anything is still WORKING, and the harvest triad forty lines up refuses exactly this
|
|
2416
|
+
// conclusion without it — "a worker that is merely thinking still burns CPU and is never
|
|
2417
|
+
// concluded here". A live CPU delta is a live worker, by construction; an UNMEASURABLE
|
|
2418
|
+
// reading keeps the fail-open contract the probe already states (undefined is never
|
|
2419
|
+
// evidence a worker stopped), so on a host whose `ps` cannot be read the fast-kill stands
|
|
2420
|
+
// down and the rolling window owns the pane, exactly as pre-T1.
|
|
2421
|
+
if (fastKillEligible) {
|
|
2422
|
+
const leg = readCpuLeg();
|
|
2423
|
+
if (leg.state === "flat") {
|
|
2424
|
+
deadChannelKilled = true;
|
|
2425
|
+
journal.append("worker-dead", t.id, {
|
|
2426
|
+
slot: slot.name, attempt, silentMs: sliceNow - lastOutputGrowthAt,
|
|
2427
|
+
cpuMs: leg.cpu.ms, cpuResolutionMs: leg.cpu.resolutionMs,
|
|
2428
|
+
});
|
|
2429
|
+
break;
|
|
2430
|
+
}
|
|
2431
|
+
if (!cpuHeld) {
|
|
2432
|
+
cpuHeld = true;
|
|
2433
|
+
journal.append("worker-dead-held", t.id, {
|
|
2434
|
+
slot: slot.name, attempt, reason: leg.state === "accruing" ? "cpu-accruing" : "cpu-unmeasurable",
|
|
2435
|
+
});
|
|
2436
|
+
}
|
|
2460
2437
|
}
|
|
2461
2438
|
if (nudgeable && !nudged && sliceNow - lastProgressAt >= nudgeAfterSilentMs) {
|
|
2462
2439
|
nudged = true;
|
|
@@ -2623,6 +2600,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2623
2600
|
// steer, driver.nudge is never consulted on this path), so there is no pending daemon
|
|
2624
2601
|
// action for the triad to preempt — the asymmetry with the interactive call site above is
|
|
2625
2602
|
// the absence of the thing being held for, not an oversight.
|
|
2603
|
+
await armCpuLeg(Date.now() - lastProgressAt >= harvestSilentMs);
|
|
2626
2604
|
if (await harvestConcludes(Date.now() - lastProgressAt))
|
|
2627
2605
|
break;
|
|
2628
2606
|
}
|
|
@@ -2665,7 +2643,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2665
2643
|
}
|
|
2666
2644
|
let result = settleParsed ?? adapter.parse(output, nonce);
|
|
2667
2645
|
const workerFinished = finished;
|
|
2668
|
-
const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut });
|
|
2646
|
+
const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut, deadChannel: deadChannelKilled });
|
|
2669
2647
|
journal.append("worker-result", t.id, {
|
|
2670
2648
|
ok: result.ok, summary: result.summary, deviations: result.deviations, finished: workerFinished, exitCode,
|
|
2671
2649
|
mode: interactive ? "interactive" : "print", ...(workerCause ? { cause: workerCause } : {}),
|
|
@@ -2888,6 +2866,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2888
2866
|
({ results, commits } = await runGates(t, {
|
|
2889
2867
|
worktree: wt, baseRef: taskBase, result, author: assignment,
|
|
2890
2868
|
commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
|
|
2869
|
+
collateral: collateral.get(t.id) ?? [],
|
|
2891
2870
|
pipeline: "v185", selectTests: !testGateFailed,
|
|
2892
2871
|
via: cfg.visibility.llm === "pane"
|
|
2893
2872
|
? {
|
|
@@ -2930,7 +2909,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2930
2909
|
}
|
|
2931
2910
|
graph = setStatus(graph, t.id, "done");
|
|
2932
2911
|
saveGraph(repoRoot, graph);
|
|
2933
|
-
journal.append("task-done", t.id, {
|
|
2912
|
+
journal.append("task-done", t.id, {
|
|
2913
|
+
attempts: attempt + 1, assignment, taskContentDigest: contentDigest,
|
|
2914
|
+
});
|
|
2934
2915
|
journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
|
|
2935
2916
|
// firstAttemptOk/gateFails/consults are recorded FACTS, not policy — a parkKind:"stall" row is
|
|
2936
2917
|
// recorded but NOT quality-negative in v1.6; Phase 12 owns reward policy, so flipping it later needs zero data migration.
|
|
@@ -2962,11 +2943,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2962
2943
|
await park(t, `${infraFailure.gate}: ${infraFailure.details}`, "infra", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
2963
2944
|
return;
|
|
2964
2945
|
}
|
|
2946
|
+
// OBS-547: who pays for this red is decided by the run's collateral prediction (see
|
|
2947
|
+
// dispositionScopeRed) — an authoring defect parks unchargeable before any accounting below.
|
|
2948
|
+
if (await dispositionScopeRed(t, results, assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode))
|
|
2949
|
+
return;
|
|
2965
2950
|
gateFails++; // this attempt's gates failed — the one place quality degradation is verified (never inferred from attempts)
|
|
2966
2951
|
// v1.53 T3: prefer the CLI's own session id captured from this attempt's output (kimi's resume
|
|
2967
2952
|
// trailer) over the harness slot name; absent hook or no capture keeps today's slot-name id.
|
|
2968
2953
|
retrySession = { channel: channelKey(assignment), id: adapter.sessionIdFrom?.(output) ?? sessionId, contextTokens };
|
|
2969
|
-
feedback = results.filter(gateFailed).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
|
|
2954
|
+
feedback = withCarriedEvidence(results.filter(gateFailed).map((g) => `${g.gate}: ${g.details}`).join("\n\n"));
|
|
2970
2955
|
// OBS-189/G3 (park-economics patch): a request-changes review is a findings brief, not a worker
|
|
2971
2956
|
// defect — the fix attempt stays on the same channel with the findings as feedback and consumes
|
|
2972
2957
|
// no escalation-ladder rung. Bounded by the engagement round cap at the top of this loop.
|