tickmarkr 1.81.0 → 1.84.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/brand.d.ts +4 -1
- package/dist/brand.js +46 -7
- package/dist/cli/commands/approve.js +53 -8
- package/dist/cli/commands/plan.js +17 -2
- package/dist/cli/commands/report.js +15 -6
- package/dist/cli/commands/ui.js +13 -1
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +1 -1
- package/dist/compile/collateral.d.ts +19 -0
- package/dist/compile/collateral.js +90 -0
- package/dist/compile/index.js +18 -4
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +19 -0
- package/dist/drivers/types.d.ts +1 -0
- package/dist/gates/acceptance.js +50 -8
- package/dist/gates/llm.js +34 -19
- package/dist/gates/review.d.ts +9 -1
- package/dist/gates/review.js +173 -7
- package/dist/gates/run-gates.d.ts +1 -0
- package/dist/gates/run-gates.js +18 -1
- package/dist/report/bundle.d.ts +6 -1
- package/dist/report/bundle.js +13 -3
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +188 -27
- package/dist/run/journal.d.ts +5 -0
- package/dist/run/journal.js +71 -1
- package/dist/tui/cockpit/capture.d.ts +107 -1
- package/dist/tui/cockpit/capture.js +287 -10
- package/dist/tui/cockpit/components.d.ts +40 -47
- package/dist/tui/cockpit/components.js +66 -50
- package/dist/tui/cockpit/derive.d.ts +73 -1
- package/dist/tui/cockpit/derive.js +211 -15
- package/dist/tui/cockpit/keys.d.ts +114 -19
- package/dist/tui/cockpit/keys.js +413 -44
- package/dist/tui/cockpit/layout.d.ts +231 -0
- package/dist/tui/cockpit/layout.js +307 -0
- package/dist/tui/cockpit/live.d.ts +65 -3
- package/dist/tui/cockpit/live.js +418 -48
- package/dist/tui/cockpit/pointer.d.ts +261 -0
- package/dist/tui/cockpit/pointer.js +610 -0
- package/dist/tui/cockpit/run-cockpit.d.ts +93 -6
- package/dist/tui/cockpit/run-cockpit.js +805 -164
- package/dist/tui/cockpit/setup-cockpit.d.ts +146 -1
- package/dist/tui/cockpit/setup-cockpit.js +337 -5
- package/dist/tui/cockpit/views.d.ts +72 -0
- package/dist/tui/cockpit/views.js +56 -0
- package/dist/tui/cockpit/width.d.ts +105 -0
- package/dist/tui/cockpit/width.js +276 -0
- package/fixtures/speckit-sample/tasks.md +2 -2
- package/package.json +3 -1
package/dist/run/daemon.js
CHANGED
|
@@ -20,7 +20,8 @@ import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
|
20
20
|
import { runEnvironment } from "./environment.js";
|
|
21
21
|
import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
22
22
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
23
|
-
import { classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId, phaseForGate, recordedTaskFailureKind } from "./journal.js";
|
|
23
|
+
import { classifyTaskFailure, classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId, phaseForGate, recordedTaskFailureKind, reviewRoundsSinceApproval } from "./journal.js";
|
|
24
|
+
import { isDiffCapPark } from "../gates/review.js";
|
|
24
25
|
import { acquireRunLock, releaseRunLock } from "./lock.js";
|
|
25
26
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
26
27
|
import { nextChannel, route } from "../route/router.js";
|
|
@@ -71,7 +72,10 @@ const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can neve
|
|
|
71
72
|
// v1.70 T5: request-changes review rounds a single task may draw before it parks for a human decision
|
|
72
73
|
// instead of cycling. Well below MAX_ATTEMPTS so review non-convergence is caught long before the
|
|
73
74
|
// global cap. ponytail: literal constant; lift to cfg.review.roundCap only if a second knob-turner appears.
|
|
74
|
-
|
|
75
|
+
// OBS-189 (park-economics patch): 3 → 2, counted per ENGAGEMENT (since the newest operator approval,
|
|
76
|
+
// reviewRoundsSinceApproval) — a park is now cheap (approve/--uphold both cost one decision, never a
|
|
77
|
+
// run), so non-convergence parks earlier and an upheld task re-enters with a fresh round budget.
|
|
78
|
+
const REVIEW_ROUND_CAP = 2;
|
|
75
79
|
const BLOCKED_POLL_MS = 30_000; // between trailer-wait slices, check whether the pane is blocked on a prompt
|
|
76
80
|
const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
|
|
77
81
|
const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
|
|
@@ -88,6 +92,30 @@ export function setEarlyLaunchLivenessMsForTests(ms) {
|
|
|
88
92
|
export function resetEarlyLaunchLivenessMsForTests() {
|
|
89
93
|
earlyLaunchLivenessMs = EARLY_LAUNCH_LIVENESS_MS;
|
|
90
94
|
}
|
|
95
|
+
// OBS-201: the liveness nudge — the daemon's ACTIVE response to an idle worker holding no trailer,
|
|
96
|
+
// replacing page-a-human-then-burn-the-window (289 of 692 worker-minutes in one measured day).
|
|
97
|
+
// Gate: herdr classifies the pane idle AND the monotonic tracker has seen nothing for
|
|
98
|
+
// NUDGE_AFTER_SILENT_MS (a worker grinding inside a tool run is `working` and never reaches the
|
|
99
|
+
// gate). One nudge per attempt; if the grace passes still idle with no progress, the wait concludes
|
|
100
|
+
// as a stall NOW and the consult sees the un-answered nudge instead of an hour of silence.
|
|
101
|
+
// Allowlist: claude-code first (steering path proven, OBS-122); widen per adapter only with a
|
|
102
|
+
// captured occupied-frame fixture (OBS-181 scar). The message builder takes no nonce — the
|
|
103
|
+
// self-reference guard holds by construction, an echoed bare token can never match the wait regex.
|
|
104
|
+
export const NUDGEABLE_ADAPTERS = new Set(["claude-code"]);
|
|
105
|
+
export const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the task is complete, print your TICKMARKR_RESULT completion trailer exactly as specified in your prompt now. If not, state your next concrete action and continue working.";
|
|
106
|
+
const NUDGE_AFTER_SILENT_MS = 3 * 60_000;
|
|
107
|
+
const WORKER_NUDGE_GRACE_MS = 4 * 60_000;
|
|
108
|
+
let nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
|
|
109
|
+
let workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
|
|
110
|
+
/** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
|
|
111
|
+
export function setNudgeTimingForTests(silentMs, graceMs) {
|
|
112
|
+
nudgeAfterSilentMs = silentMs;
|
|
113
|
+
workerNudgeGraceMs = graceMs;
|
|
114
|
+
}
|
|
115
|
+
export function resetNudgeTimingForTests() {
|
|
116
|
+
nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
|
|
117
|
+
workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
|
|
118
|
+
}
|
|
91
119
|
async function commitsAheadOf(base, wt) {
|
|
92
120
|
const head = await gitHead(wt);
|
|
93
121
|
if (head === base)
|
|
@@ -168,6 +196,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
168
196
|
status: (s) => driver.status(s),
|
|
169
197
|
read: (s, n) => driver.read(s, n),
|
|
170
198
|
...(driver.sendKey ? { sendKey: driver.sendKey.bind(driver) } : {}),
|
|
199
|
+
...(driver.nudge ? { nudge: driver.nudge.bind(driver) } : {}),
|
|
171
200
|
notify: (m, o) => driver.notify(m, o),
|
|
172
201
|
close: closeSlot,
|
|
173
202
|
worktree: (r, b, base) => driver.worktree(r, b, base),
|
|
@@ -453,13 +482,26 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
453
482
|
const badReviewers = []; // v1.1: reviewer channels that produced unparseable output for this task
|
|
454
483
|
// v1.70 T5 (review-convergence): failed review rounds this task has drawn, counted from the per-task
|
|
455
484
|
// review history already in the journal — the SAME review gate-result stream onGate reads to grow the
|
|
456
|
-
// reviewer-exclusion list (badReviewers), never a second parallel counter.
|
|
457
|
-
//
|
|
458
|
-
//
|
|
459
|
-
const reviewRoundsDrawn = () => journal.read()
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
485
|
+
// reviewer-exclusion list (badReviewers), never a second parallel counter. OBS-189: scoped to the
|
|
486
|
+
// current engagement — an operator approval (uphold or accept) resets the round budget, so an upheld
|
|
487
|
+
// task can dispatch its funded attempt instead of re-parking against the whole journal's history.
|
|
488
|
+
const reviewRoundsDrawn = () => reviewRoundsSinceApproval(journal.read(), t.id);
|
|
489
|
+
// OBS-193: journal the in-gate review retry (mirrors judge-retry) and exclude the flaked seat from
|
|
490
|
+
// later attempts' reviewer picks. One helper, called from both onGate sites (satisfied-gate + main).
|
|
491
|
+
const noteReviewRetry = (g) => {
|
|
492
|
+
const rr = g.meta?.reviewRetry;
|
|
493
|
+
if (g.gate === "review" && rr && typeof rr.flaked === "string" && typeof rr.retried === "string") {
|
|
494
|
+
journal.append("review-retry", t.id, {
|
|
495
|
+
gate: "review", flaked: rr.flaked, retried: rr.retried,
|
|
496
|
+
...(g.meta?.unparseable === true ? { secondUnparseable: true } : {}),
|
|
497
|
+
});
|
|
498
|
+
badReviewers.push(rr.flaked);
|
|
499
|
+
}
|
|
500
|
+
};
|
|
501
|
+
// OBS-189: the operator upheld the reviewer — the findings ARE the brief for this funded attempt.
|
|
502
|
+
let feedback = rs?.upheldFeedback
|
|
503
|
+
? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${rs.upheldFeedback}`
|
|
504
|
+
: "";
|
|
463
505
|
let ladderIdx = 0;
|
|
464
506
|
let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
|
|
465
507
|
let gateFails = 0; // TEL-02: incremented ONLY where feedback is built from failing gates — never derived from attempts (quota failovers bump attempts too, Pitfall 6)
|
|
@@ -484,6 +526,23 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
484
526
|
}
|
|
485
527
|
return next;
|
|
486
528
|
};
|
|
529
|
+
// OBS-202 (operator law: "you can spawn as many as you want"): channels are session FACTORIES,
|
|
530
|
+
// not consumed seats — a tried channel can always host a fresh worker session, and a fresh
|
|
531
|
+
// session carries none of the failed attempt's baggage. When the untried pool is empty, recycle
|
|
532
|
+
// the best LIVE channel (preferring a different seat than the current one) instead of parking
|
|
533
|
+
// on artificial scarcity; MAX_ATTEMPTS and the review round cap are the real bounds. Only a
|
|
534
|
+
// fleet whose every channel is DEMOTED (verified dead: auth/setup/provider outage) has nothing
|
|
535
|
+
// left to spawn — that case alone still parks.
|
|
536
|
+
const failoverOrRecycle = (site) => {
|
|
537
|
+
const next = failover(site);
|
|
538
|
+
if (next)
|
|
539
|
+
return next;
|
|
540
|
+
const recycled = nextChannel(assignment, t, cfg, channels, [channelKey(assignment)], profile, demotedChannels)
|
|
541
|
+
?? (demotedChannels.has(channelKey(assignment)) ? null : assignment);
|
|
542
|
+
if (recycled)
|
|
543
|
+
journal.append("channel-recycle", t.id, { site, channel: channelKey(recycled) });
|
|
544
|
+
return recycled;
|
|
545
|
+
};
|
|
487
546
|
const runConsult = (trigger, transcript, diffOrFeedback, gates) => {
|
|
488
547
|
consults++;
|
|
489
548
|
return consult({
|
|
@@ -527,13 +586,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
527
586
|
}
|
|
528
587
|
}
|
|
529
588
|
}
|
|
530
|
-
const next =
|
|
589
|
+
const next = failoverOrRecycle("consult-reroute");
|
|
531
590
|
if (next) {
|
|
532
591
|
assignment = next;
|
|
533
|
-
|
|
592
|
+
const k = channelKey(next);
|
|
593
|
+
if (!tried.includes(k))
|
|
594
|
+
tried.push(k);
|
|
534
595
|
return true;
|
|
535
596
|
}
|
|
536
|
-
await park(t, "consult said reroute but every channel is
|
|
597
|
+
await park(t, "consult said reroute but every channel is demoted (verified dead) — nothing left to spawn", "reroute-exhausted", assignment, attempts, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
537
598
|
return false;
|
|
538
599
|
}
|
|
539
600
|
await park(t, `consult verdict: ${v.action} — ${v.notes}`, trigger, assignment, attempts, startMs, gateFails, consults, tokens, metered, retryMode); // decompose|human
|
|
@@ -554,6 +615,23 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
554
615
|
const wt = await driver.worktree(repoRoot, taskBranch, taskBase);
|
|
555
616
|
const carriedCommits = await cherryPickCommits(wt, commitsToCarry);
|
|
556
617
|
journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
|
|
618
|
+
// OBS-212: same fail-closed rule as the dispatch path — but this path is worse, because it runs
|
|
619
|
+
// ONLY the gates after the approved one and then MERGES. T3 took it on run-20260728-110135:
|
|
620
|
+
// approved past review at 11:22, recreated at 12:50, and phase-start{gates} / phase-start{merge}
|
|
621
|
+
// landed in the same second with zero gate-result events. Work missing here is merged unverified.
|
|
622
|
+
{
|
|
623
|
+
const present = new Set(carriedCommits);
|
|
624
|
+
for (const h of commitsToCarry) {
|
|
625
|
+
if (!present.has(h) && (await shGit(`git merge-base --is-ancestor ${shq(h)} HEAD`, wt)).code === 0) {
|
|
626
|
+
present.add(h);
|
|
627
|
+
}
|
|
628
|
+
}
|
|
629
|
+
const lost = commitsToCarry.filter((h) => !present.has(h));
|
|
630
|
+
if (lost.length > 0) {
|
|
631
|
+
await park(t, `carry lost ${lost.length} of ${commitsToCarry.length} verified commit(s) recreating the worktree for approved gate ${satisfiedGate} (first missing: ${lost[0].slice(0, 10)}) — refusing to merge a tree that is missing landed work`, "infra", assignment, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
632
|
+
return;
|
|
633
|
+
}
|
|
634
|
+
}
|
|
557
635
|
if (!linkNodeModules(repoRoot, wt, { force: true })) {
|
|
558
636
|
await park(t, "environmental: node_modules link could not be re-asserted before gates (OBS-47)", "setup", assignment, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
559
637
|
return;
|
|
@@ -586,7 +664,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
586
664
|
journal.phaseStart(t.id, "gates");
|
|
587
665
|
const { results } = await runGates(resumedTask, {
|
|
588
666
|
worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
|
|
589
|
-
commands, baseline, channels, adapters, cfg,
|
|
667
|
+
commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
|
|
590
668
|
via: cfg.visibility.llm === "pane"
|
|
591
669
|
? {
|
|
592
670
|
driver: trackedDriver,
|
|
@@ -607,6 +685,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
607
685
|
gate: g.gate, pass: g.pass, details: g.details,
|
|
608
686
|
...(g.meta?.skipped === true ? { skipped: true } : {}),
|
|
609
687
|
});
|
|
688
|
+
noteReviewRetry(g);
|
|
610
689
|
if (g.gate === "review" && !g.pass && /unparseable/.test(g.details)
|
|
611
690
|
&& typeof g.meta?.reviewer === "string") {
|
|
612
691
|
badReviewers.push(g.meta.reviewer);
|
|
@@ -669,11 +748,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
669
748
|
}
|
|
670
749
|
// v1.70 T5: a task that has already drawn REVIEW_ROUND_CAP request-changes review rounds parks for
|
|
671
750
|
// a human decision instead of dispatching another round. Condition on the review history (data),
|
|
672
|
-
// never the code path
|
|
673
|
-
//
|
|
674
|
-
//
|
|
675
|
-
if (attempt > 0 &&
|
|
676
|
-
await park(t, `review round cap (${REVIEW_ROUND_CAP}) reached —
|
|
751
|
+
// never the code path. OBS-189: the count is engagement-scoped — any operator approval resets it
|
|
752
|
+
// by construction (reviewRoundsSinceApproval), so the old blanket approved-task exemption is gone
|
|
753
|
+
// and an upheld task that STILL cannot converge re-parks after another full round budget.
|
|
754
|
+
if (attempt > 0 && reviewRoundsDrawn() >= REVIEW_ROUND_CAP) {
|
|
755
|
+
await park(t, `review round cap (${REVIEW_ROUND_CAP}) reached this engagement — \`tickmarkr approve\` accepts the diff past review; \`tickmarkr approve --uphold\` funds one fixed attempt carrying the findings`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
677
756
|
return;
|
|
678
757
|
}
|
|
679
758
|
// OBS-57: a demoted channel must not be re-dispatched on consult retry or provider requeue.
|
|
@@ -736,6 +815,27 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
736
815
|
presentCommits.add(h);
|
|
737
816
|
}
|
|
738
817
|
}
|
|
818
|
+
// OBS-212: losing carried work is WORK DESTRUCTION, never an ordinary continue. Every commit in
|
|
819
|
+
// commitsToCarry is one a prior attempt landed and the gates verified. If a commit is neither
|
|
820
|
+
// cherry-picked nor already an ancestor of the new base, dispatching a worker onto this tree
|
|
821
|
+
// silently makes it re-buy verified work — the failure this cherry-pick exists to prevent
|
|
822
|
+
// (OBS-58, comment above) happening silently inside the mechanism itself. Measured on
|
|
823
|
+
// run-20260728-110135: T2 carried 0 of 17 at 17:07 and T1 carried 0 of 15 at 21:37, twenty
|
|
824
|
+
// minutes after T2's merge advanced the integration tip. Nothing failed and nothing parked;
|
|
825
|
+
// the run simply re-paid for the work. Fail closed instead — a visible park beats a silent loss.
|
|
826
|
+
// On the DISPATCH path a drop is not always a defect: a human can re-pend a task with new
|
|
827
|
+
// intent, and the superseded commit then SHOULD be left behind (pinned by the worktree-cleanup
|
|
828
|
+
// resume test, where T1 and T2 both write shared.txt and the loser is re-scripted). We cannot
|
|
829
|
+
// tell supersession from destruction here, so this path stays loud rather than fail-closed —
|
|
830
|
+
// the harm that cost this run ~9 hours was the SILENCE, not the drop. The merge path, where a
|
|
831
|
+
// drop is never legitimate, does fail closed (see the satisfied-gate site above).
|
|
832
|
+
const lostCommits = commitsToCarry.filter((h) => !presentCommits.has(h));
|
|
833
|
+
if (lostCommits.length > 0) {
|
|
834
|
+
journal.append("work-loss", t.id, {
|
|
835
|
+
site: "dispatch", base: taskBase, attempted: commitsToCarry, carried: carriedCommits, lost: lostCommits,
|
|
836
|
+
});
|
|
837
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} lost ${lostCommits.length} of ${commitsToCarry.length} landed commit(s) recreating its worktree — it will re-do that work`, { tier: "attention" });
|
|
838
|
+
}
|
|
739
839
|
if (feedback || priorNamed.length > 0) {
|
|
740
840
|
feedback = augmentRetryBrief(feedback, { attempted: commitsToCarry, carried: carriedCommits, present: presentCommits });
|
|
741
841
|
}
|
|
@@ -908,6 +1008,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
908
1008
|
// v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
|
|
909
1009
|
// Any other blocked/idle dialog still pages the operator (paged latch below).
|
|
910
1010
|
let trustAnswered = false;
|
|
1011
|
+
// OBS-201: one liveness nudge per attempt; the grace deadline is its OWN timer, never the
|
|
1012
|
+
// stall window (the nudge's pane echo is absorbed before it starts, or the echo itself
|
|
1013
|
+
// would reset the window and make the early conclusion unreachable).
|
|
1014
|
+
let nudged = false;
|
|
1015
|
+
let nudgeDeadline;
|
|
911
1016
|
finished = false;
|
|
912
1017
|
exitCode = null;
|
|
913
1018
|
// OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
|
|
@@ -980,9 +1085,53 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
980
1085
|
/* read/send failed — fall through to page the operator */
|
|
981
1086
|
}
|
|
982
1087
|
}
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
1088
|
+
// OBS-201: the daemon ACTS on an idle pane before paging anyone. Gate: idle AND the
|
|
1089
|
+
// monotonic tracker silent ≥ the nudge threshold (a worker inside a tool run reads as
|
|
1090
|
+
// `working` and never gets here; a briefly-settling TUI hasn't been silent long enough).
|
|
1091
|
+
const nudgeable = st === "idle" && !!driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id);
|
|
1092
|
+
if (nudgeable && !nudged) {
|
|
1093
|
+
if (Date.now() - lastProgressAt >= nudgeAfterSilentMs) {
|
|
1094
|
+
nudged = true;
|
|
1095
|
+
if (await driver.nudge(slot, WORKER_NUDGE_MESSAGE)) {
|
|
1096
|
+
// absorb the nudge's own echo BEFORE arming the grace timer — post-nudge progress
|
|
1097
|
+
// is measured against this baseline, not against the echo.
|
|
1098
|
+
const echo = await driver.read(slot, 1000);
|
|
1099
|
+
stallProgress.observe({ paneText: echo, contextTokens });
|
|
1100
|
+
nudgeDeadline = Date.now() + workerNudgeGraceMs;
|
|
1101
|
+
journal.append("worker-nudge", t.id, { slot: slot.name, attempt });
|
|
1102
|
+
}
|
|
1103
|
+
else {
|
|
1104
|
+
journal.append("worker-nudge-failed", t.id, { slot: slot.name, attempt });
|
|
1105
|
+
// nudge undeliverable — fall back to today's behavior: page once, window backstop
|
|
1106
|
+
paged = true;
|
|
1107
|
+
await driver.notify(`tickmarkr ${runId}: ${slot.name} looks idle without finishing — check its pane`, { tier: "attention" });
|
|
1108
|
+
}
|
|
1109
|
+
}
|
|
1110
|
+
// idle but not yet silent past the threshold: neither nudge nor page this slice
|
|
1111
|
+
}
|
|
1112
|
+
else if (nudgeable && nudged && nudgeDeadline !== undefined) {
|
|
1113
|
+
if (Date.now() >= nudgeDeadline && Date.now() - lastProgressAt >= workerNudgeGraceMs) {
|
|
1114
|
+
// grace spent, still idle, no post-nudge progress: re-harvest once (the trailer may
|
|
1115
|
+
// have landed between polls), then conclude the wait as a stall NOW — the consult
|
|
1116
|
+
// sees the un-answered nudge instead of the remainder of the window.
|
|
1117
|
+
nudgeDeadline = undefined;
|
|
1118
|
+
output = await driver.read(slot, 1000);
|
|
1119
|
+
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
1120
|
+
const exit = exitRe.exec(output);
|
|
1121
|
+
if (finished || exit) {
|
|
1122
|
+
exitCode = exit ? Number(exit[1]) : null;
|
|
1123
|
+
await sampleContext();
|
|
1124
|
+
break;
|
|
1125
|
+
}
|
|
1126
|
+
journal.append("worker-nudge-expired", t.id, { slot: slot.name, attempt, graceMs: workerNudgeGraceMs });
|
|
1127
|
+
lastProgressAt = Date.now() - stallWindowMs; // the existing harvest/classify tail runs unmodified
|
|
1128
|
+
}
|
|
1129
|
+
}
|
|
1130
|
+
else {
|
|
1131
|
+
paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
|
|
1132
|
+
const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
|
|
1133
|
+
await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
|
|
1134
|
+
}
|
|
986
1135
|
}
|
|
987
1136
|
// a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
|
|
988
1137
|
const spent = Date.now() - sliceStart;
|
|
@@ -1172,7 +1321,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1172
1321
|
const from = channelKey(assignment);
|
|
1173
1322
|
demotedChannels.add(from); // excluded for later attempts AND later tasks in this run
|
|
1174
1323
|
journal.append("channel-exclusion", t.id, { channel: from, reason: dead, kind: "dead-channel" });
|
|
1175
|
-
const next =
|
|
1324
|
+
const next = failoverOrRecycle("dead-channel");
|
|
1176
1325
|
journal.append("dead-channel-failover", t.id, { reason: dead, from, to: next ? channelKey(next) : null });
|
|
1177
1326
|
if (next) {
|
|
1178
1327
|
await driver.notify(`tickmarkr ${runId}: ${t.id} dead channel (${dead}) failover`, { tier: "attention" });
|
|
@@ -1255,6 +1404,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1255
1404
|
}
|
|
1256
1405
|
}
|
|
1257
1406
|
journal.append("gate-result", t.id, { gate: g.gate, pass: g.pass, details: g.details, ...(g.meta?.skipped === true ? { skipped: true } : {}) });
|
|
1407
|
+
noteReviewRetry(g);
|
|
1258
1408
|
// v1.1 failover: never re-ask a reviewer channel that produced garbage for this task
|
|
1259
1409
|
if (g.gate === "review" && !g.pass && /unparseable/.test(g.details) && typeof g.meta?.reviewer === "string") {
|
|
1260
1410
|
badReviewers.push(g.meta.reviewer);
|
|
@@ -1267,7 +1417,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1267
1417
|
journal.phaseStart(t.id, "gates");
|
|
1268
1418
|
({ results, commits } = await runGates(t, {
|
|
1269
1419
|
worktree: wt, baseRef: taskBase, result, author: assignment,
|
|
1270
|
-
commands, baseline, channels, adapters, cfg,
|
|
1420
|
+
commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
|
|
1271
1421
|
via: cfg.visibility.llm === "pane"
|
|
1272
1422
|
? {
|
|
1273
1423
|
driver: trackedDriver,
|
|
@@ -1335,8 +1485,19 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1335
1485
|
// trailer) over the harness slot name; absent hook or no capture keeps today's slot-name id.
|
|
1336
1486
|
retrySession = { channel: channelKey(assignment), id: adapter.sessionIdFrom?.(output) ?? sessionId, contextTokens };
|
|
1337
1487
|
feedback = results.filter((g) => !g.pass).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
|
|
1338
|
-
|
|
1339
|
-
|
|
1488
|
+
// OBS-189/G3 (park-economics patch): a request-changes review is a findings brief, not a worker
|
|
1489
|
+
// defect — the fix attempt stays on the same channel with the findings as feedback and consumes
|
|
1490
|
+
// no escalation-ladder rung. Bounded by the engagement round cap at the top of this loop.
|
|
1491
|
+
// Unparseable verdicts (already retried in-gate, OBS-193) and diff-cap trips (the diff cannot
|
|
1492
|
+
// shrink by retrying, OBS-48) fall through to the ladder unchanged. Review runs last, so a
|
|
1493
|
+
// failed review with every other gate green is exactly "the work landed, the reviewer objects".
|
|
1494
|
+
const reviewFail = results.find((g) => g.gate === "review" && !g.pass);
|
|
1495
|
+
const reviewFixRetry = reviewFail !== undefined
|
|
1496
|
+
&& reviewFail.meta?.unparseable !== true
|
|
1497
|
+
&& !isDiffCapPark(reviewFail)
|
|
1498
|
+
&& results.every((g) => g.pass || g.gate === "review");
|
|
1499
|
+
const step = reviewFixRetry ? "retry" : r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
|
|
1500
|
+
journal.append("escalation", t.id, { step, attempt: attempt + 1, ...(reviewFixRetry ? { reviewFix: true } : {}) });
|
|
1340
1501
|
await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
|
|
1341
1502
|
if (step === "retry")
|
|
1342
1503
|
continue;
|
|
@@ -1374,8 +1535,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1374
1535
|
.catch(async (err) => {
|
|
1375
1536
|
const taskEvents = journal.read().filter((e) => e.taskId === t.id);
|
|
1376
1537
|
const dispatch = [...taskEvents].reverse().find((e) => e.event === "task-dispatch");
|
|
1377
|
-
|
|
1378
|
-
const kind =
|
|
1538
|
+
// OBS-206: shared rule with `resume --retry-failed` — see classifyTaskFailure.
|
|
1539
|
+
const kind = classifyTaskFailure(taskEvents);
|
|
1379
1540
|
const attempts = dispatch && Number.isInteger(dispatch.data.attempt) ? dispatch.data.attempt : 0;
|
|
1380
1541
|
graph = setStatus(graph, t.id, "failed");
|
|
1381
1542
|
saveGraph(repoRoot, graph);
|
package/dist/run/journal.d.ts
CHANGED
|
@@ -16,11 +16,15 @@ export interface ResumeState {
|
|
|
16
16
|
attempts: number;
|
|
17
17
|
tried: string[];
|
|
18
18
|
lastAssignment?: Assignment;
|
|
19
|
+
upheldFeedback?: string;
|
|
19
20
|
}
|
|
20
21
|
export declare const CHANNEL_EXCLUSION_KINDS: readonly ["dead-channel"];
|
|
21
22
|
export type ChannelExclusionKind = (typeof CHANNEL_EXCLUSION_KINDS)[number];
|
|
22
23
|
export declare const ATTEMPT_CAP_RELEASE: "attempt-cap";
|
|
23
24
|
export declare const GATE_SATISFIED_RELEASE: "gate-satisfied";
|
|
25
|
+
export declare const REVIEW_UPHELD_RELEASE: "review-upheld";
|
|
26
|
+
export declare const RECHECK_RELEASE: "recheck";
|
|
27
|
+
export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string): number;
|
|
24
28
|
export declare const PARK_KINDS: readonly ["human-gate", "ladder-exhausted", "attempt-cap", "gate-fail", "quota", "reroute-exhausted", "setup", "stall", "merge-conflict", "tip-moved", "infra", "dispatch"];
|
|
25
29
|
export type ParkKind = (typeof PARK_KINDS)[number];
|
|
26
30
|
export declare const RETRY_MODES: readonly ["resume", "fresh"];
|
|
@@ -28,6 +32,7 @@ export type RetryMode = (typeof RETRY_MODES)[number];
|
|
|
28
32
|
export declare const WORKER_RESULT_CAUSES: readonly ["provider-death", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer"];
|
|
29
33
|
export type WorkerResultCause = (typeof WORKER_RESULT_CAUSES)[number];
|
|
30
34
|
export declare function isQualityFailureParkKind(kind: ParkKind): boolean;
|
|
35
|
+
export declare function classifyTaskFailure(taskEvents: JournalEvent[]): ParkKind;
|
|
31
36
|
export declare function recordedTaskFailureKind(events: JournalEvent[], taskId: string): ParkKind | undefined;
|
|
32
37
|
export declare function runHasEnded(events: JournalEvent[]): boolean;
|
|
33
38
|
/** OBS-53: classify worker-result failures so retries and routing see the true signal, not one lumped bucket. */
|
package/dist/run/journal.js
CHANGED
|
@@ -44,6 +44,32 @@ export const ATTEMPT_CAP_RELEASE = "attempt-cap";
|
|
|
44
44
|
// OBS-130: task-approved carries the exact failed gate the operator satisfied. The release tag keeps
|
|
45
45
|
// ordinary humanGate and attempt-cap approvals byte-compatible while making this authority explicit.
|
|
46
46
|
export const GATE_SATISFIED_RELEASE = "gate-satisfied";
|
|
47
|
+
// OBS-189: the second human decision a review park needs. `approve` (gate-satisfied) accepts the diff
|
|
48
|
+
// the reviewer rejected; `approve --uphold` sides WITH the reviewer and funds one fixed worker attempt
|
|
49
|
+
// carrying the findings — a park costs an attempt, never a fresh run that re-executes green tasks.
|
|
50
|
+
export const REVIEW_UPHELD_RELEASE = "review-upheld";
|
|
51
|
+
// OBS-203: the third decision a gate-fail park needs. Plain approve WAIVES the failed gate (and every
|
|
52
|
+
// gate before it — daemon.ts remainingGates), which is wrong when the gate failed against a stale task
|
|
53
|
+
// DECLARATION rather than a bad diff: amend the spec's files[], recompile, and the gate now passes
|
|
54
|
+
// honestly. This release re-dispatches with the whole gate suite intact and no gate marked satisfied,
|
|
55
|
+
// so the corrected declaration is the thing that earns the green. Budget semantics match attempt-cap
|
|
56
|
+
// (fresh attempts, tried survives) because the park cost the task its remaining budget.
|
|
57
|
+
export const RECHECK_RELEASE = "recheck";
|
|
58
|
+
// OBS-189: review rounds are scoped to the current ENGAGEMENT — the stretch since the newest operator
|
|
59
|
+
// approval for the task. A whole-journal count re-parks an upheld task before its funded attempt can
|
|
60
|
+
// dispatch (measured live on run-20260726-213539), making a fresh journal the only escape.
|
|
61
|
+
export function reviewRoundsSinceApproval(events, taskId) {
|
|
62
|
+
let rounds = 0;
|
|
63
|
+
for (const e of events) {
|
|
64
|
+
if (e.taskId !== taskId)
|
|
65
|
+
continue;
|
|
66
|
+
if (e.event === "task-approved")
|
|
67
|
+
rounds = 0;
|
|
68
|
+
else if (e.event === "gate-result" && e.data.gate === "review" && e.data.pass === false)
|
|
69
|
+
rounds++;
|
|
70
|
+
}
|
|
71
|
+
return rounds;
|
|
72
|
+
}
|
|
47
73
|
// Fail-closed shape for a dispatched assignment (journal.ts:75-90 posture): a malformed assignment in
|
|
48
74
|
// one dispatch degrades that single task toward today's behavior — counts toward attempts, contributes
|
|
49
75
|
// nothing to tried, poisons only lastAssignment — never crashes resume, never poisons other tasks.
|
|
@@ -66,6 +92,25 @@ export function isQualityFailureParkKind(kind) {
|
|
|
66
92
|
outcome: "human", durationMs: 0, parkKind: kind,
|
|
67
93
|
}) === 0;
|
|
68
94
|
}
|
|
95
|
+
// OBS-206: ONE classification rule, called at write time by the daemon's failure handler and at read
|
|
96
|
+
// time by `resume --retry-failed`, so the two can never drift and a label written by an older binary
|
|
97
|
+
// can never outvote the evidence sitting in the same journal. `taskEvents` is this task's events up to
|
|
98
|
+
// (not including) the failure. The gate evidence is scoped to the CURRENT attempt: a whole-history
|
|
99
|
+
// scan asks "did this task EVER fail a gate", which mislabels an infra death as a verified failure for
|
|
100
|
+
// any task that ever failed one gate — and, since only "dispatch" is retryable, locks it out forever.
|
|
101
|
+
export function classifyTaskFailure(taskEvents) {
|
|
102
|
+
let dispatchIdx = -1;
|
|
103
|
+
for (let i = taskEvents.length - 1; i >= 0; i--) {
|
|
104
|
+
if (taskEvents[i].event === "task-dispatch") {
|
|
105
|
+
dispatchIdx = i;
|
|
106
|
+
break;
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
if (dispatchIdx < 0)
|
|
110
|
+
return "infra"; // never dispatched — nothing to retry
|
|
111
|
+
return taskEvents.slice(dispatchIdx + 1)
|
|
112
|
+
.some((e) => e.event === "gate-result" && e.data.pass === false) ? "gate-fail" : "dispatch";
|
|
113
|
+
}
|
|
69
114
|
// The newest terminal event owns the cause. Unknown/malformed kinds fail toward undefined so a
|
|
70
115
|
// task-failed row keeps the legacy red treatment instead of being mistaken for recoverable noise.
|
|
71
116
|
export function recordedTaskFailureKind(events, taskId) {
|
|
@@ -73,6 +118,12 @@ export function recordedTaskFailureKind(events, taskId) {
|
|
|
73
118
|
const e = events[i];
|
|
74
119
|
if (e.taskId !== taskId || (e.event !== "task-human" && e.event !== "task-failed"))
|
|
75
120
|
continue;
|
|
121
|
+
// A park kind is a daemon DECISION and is read exactly as recorded. A failure kind is a
|
|
122
|
+
// CLASSIFICATION over evidence that is still in this journal, so it is re-derived (OBS-206) —
|
|
123
|
+
// recomputation agrees with every correctly-written label and repairs the wrong ones in place.
|
|
124
|
+
if (e.event === "task-failed") {
|
|
125
|
+
return classifyTaskFailure(events.slice(0, i).filter((x) => x.taskId === taskId));
|
|
126
|
+
}
|
|
76
127
|
return typeof e.data.kind === "string" && PARK_KINDS.includes(e.data.kind)
|
|
77
128
|
? e.data.kind
|
|
78
129
|
: undefined;
|
|
@@ -461,9 +512,14 @@ export class Journal {
|
|
|
461
512
|
replayResumeState() {
|
|
462
513
|
const m = new Map();
|
|
463
514
|
const pendingReroute = new Set(); // reroute verdicts not yet cleared by a later dispatch
|
|
515
|
+
const lastReviewFail = new Map(); // OBS-189: newest failed review details per task
|
|
464
516
|
for (const e of this.read()) {
|
|
465
517
|
if (!e.taskId)
|
|
466
518
|
continue;
|
|
519
|
+
if (e.event === "gate-result" && e.data.gate === "review" && e.data.pass === false
|
|
520
|
+
&& typeof e.data.details === "string") {
|
|
521
|
+
lastReviewFail.set(e.taskId, e.data.details);
|
|
522
|
+
}
|
|
467
523
|
if (e.event === "task-dispatch") {
|
|
468
524
|
// A subsequent dispatch clears the pending reroute — the reroute was acted on pre-kill.
|
|
469
525
|
pendingReroute.delete(e.taskId);
|
|
@@ -490,7 +546,8 @@ export class Journal {
|
|
|
490
546
|
// A reroute bans the in-force channel; retry/decompose/human verdicts ban nothing (D-03).
|
|
491
547
|
pendingReroute.add(e.taskId);
|
|
492
548
|
}
|
|
493
|
-
else if (e.event === "task-approved"
|
|
549
|
+
else if (e.event === "task-approved"
|
|
550
|
+
&& (e.data.release === ATTEMPT_CAP_RELEASE || e.data.release === RECHECK_RELEASE)) {
|
|
494
551
|
// v1.24 OBS-18: operator released an attempt-cap park. Pre-v1.24 task-approved events have no
|
|
495
552
|
// `release` key ⇒ this branch never fires (corpus criterion: identical statuses + resume state).
|
|
496
553
|
// attempts reset to 0 so the daemon's attempt-cap check does not re-park in the same tick;
|
|
@@ -503,6 +560,19 @@ export class Journal {
|
|
|
503
560
|
st.lastAssignment = undefined;
|
|
504
561
|
}
|
|
505
562
|
}
|
|
563
|
+
else if (e.event === "task-approved" && e.data.release === REVIEW_UPHELD_RELEASE) {
|
|
564
|
+
// OBS-189: the operator upheld the reviewer. Same budget semantics as the attempt-cap release
|
|
565
|
+
// (fresh attempts, tried survives, no burned-channel restore) PLUS the findings carry: the
|
|
566
|
+
// newest failed review's details ride into the funded attempt as its retry brief.
|
|
567
|
+
const st = m.get(e.taskId);
|
|
568
|
+
if (st) {
|
|
569
|
+
st.attempts = 0;
|
|
570
|
+
st.lastAssignment = undefined;
|
|
571
|
+
const details = lastReviewFail.get(e.taskId);
|
|
572
|
+
if (details)
|
|
573
|
+
st.upheldFeedback = details;
|
|
574
|
+
}
|
|
575
|
+
}
|
|
506
576
|
}
|
|
507
577
|
// Trailing-reroute edge (D-01 kill between verdict and dispatch): a reroute verdict with NO
|
|
508
578
|
// subsequent dispatch means the last-dispatched channel is itself banned — add it to tried (if
|