tickmarkr 1.75.0 → 1.76.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/kimi.d.ts +5 -0
- package/dist/adapters/kimi.js +30 -10
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +29 -2
- package/dist/drivers/herdr.d.ts +2 -0
- package/dist/drivers/herdr.js +48 -6
- package/dist/run/daemon.js +25 -30
- package/dist/run/stall.d.ts +21 -4
- package/dist/run/stall.js +43 -13
- package/package.json +1 -1
package/dist/adapters/kimi.d.ts
CHANGED
|
@@ -3,6 +3,11 @@ import { type Assignment, type WorkerAdapter, type WorkerResult } from "./types.
|
|
|
3
3
|
export declare function kimiAuthed(credentialsText: string, nowMs: number): boolean;
|
|
4
4
|
export declare function parseKimiModels(raw: string): string[];
|
|
5
5
|
export declare function parseKimiResult(raw: string, nonce: string): WorkerResult;
|
|
6
|
+
export interface KimiDoctorTurnResult {
|
|
7
|
+
ok: boolean;
|
|
8
|
+
evidence: string;
|
|
9
|
+
}
|
|
10
|
+
export declare function probeKimiDoctorTurn(cwd: string): Promise<KimiDoctorTurnResult>;
|
|
6
11
|
export declare function kimiSessionId(output: string): string | undefined;
|
|
7
12
|
export declare function kimiBannerModel(banner: string): string | undefined;
|
|
8
13
|
export declare function kimiBannerSessionId(banner: string): string | undefined;
|
package/dist/adapters/kimi.js
CHANGED
|
@@ -64,6 +64,28 @@ export function parseKimiResult(raw, nonce) {
|
|
|
64
64
|
const stripped = raw.split("\n").map((l) => l.replace(/^[\s]*[•*-]\s+/, "")).join("\n");
|
|
65
65
|
return parseWorkerResult(stripped, nonce);
|
|
66
66
|
}
|
|
67
|
+
const KIMI_DOCTOR_TURN_MODEL = "kimi-code/k3";
|
|
68
|
+
const KIMI_DOCTOR_TURN_PROMPT = "Reply with exactly OK and nothing else.";
|
|
69
|
+
const KIMI_DOCTOR_TURN_TIMEOUT_MS = 60000;
|
|
70
|
+
// OBS-141: intentionally separate from probe() so plan/run remain free file checks. Only doctor
|
|
71
|
+
// calls this one-turn contract probe; its test seam stubs sh and never launches a real agent CLI.
|
|
72
|
+
export async function probeKimiDoctorTurn(cwd) {
|
|
73
|
+
const command = `kimi -p ${shq(KIMI_DOCTOR_TURN_PROMPT)} --model ${shq(KIMI_DOCTOR_TURN_MODEL)} --output-format text`;
|
|
74
|
+
const r = await sh(command, cwd, KIMI_DOCTOR_TURN_TIMEOUT_MS);
|
|
75
|
+
if (r.timedOut) {
|
|
76
|
+
return { ok: false, evidence: `turn timed out after ${KIMI_DOCTOR_TURN_TIMEOUT_MS}ms` };
|
|
77
|
+
}
|
|
78
|
+
const output = `${r.stderr}\n${r.stdout}`.trim().replace(/\s+/g, " ");
|
|
79
|
+
if (r.code !== 0) {
|
|
80
|
+
return { ok: false, evidence: output || `turn exited ${r.code}` };
|
|
81
|
+
}
|
|
82
|
+
const returnedOk = r.stdout.split("\n")
|
|
83
|
+
.map((line) => line.replace(/^[\s]*[•*-]\s+/, "").trim())
|
|
84
|
+
.includes("OK");
|
|
85
|
+
return returnedOk
|
|
86
|
+
? { ok: true, evidence: `model turn returned OK with ${KIMI_DOCTOR_TURN_MODEL}` }
|
|
87
|
+
: { ok: false, evidence: `turn returned no exact OK answer${output ? `: ${output}` : ""}` };
|
|
88
|
+
}
|
|
67
89
|
// v1.53 T3: session-id capture from the run-output trailer — every `kimi -p` run (fresh or resumed)
|
|
68
90
|
// ends with `To resume this session: kimi -r session_<uuid>` (live probe 2026-07-18). Anchored full
|
|
69
91
|
// line only: prompt/model prose can contain lookalike text, and the anchored charset keeps a
|
|
@@ -89,23 +111,20 @@ export function kimiSessionId(output) {
|
|
|
89
111
|
}
|
|
90
112
|
return id;
|
|
91
113
|
}
|
|
92
|
-
// v1.69 T6: the native TUI takes -m <alias>, where config.toml aliases are the bare model suffix of
|
|
93
|
-
// the tickmarkr channel id (live probe 2026-07-22). Keep the mapping explicit and localized.
|
|
94
|
-
function kimiAlias(model) {
|
|
95
|
-
return model.replace(/^kimi-code\//, "");
|
|
96
|
-
}
|
|
97
114
|
// v1.69 T7: the cold-start banner prints the model alias and session id. Parse them from the
|
|
98
|
-
// banner text already captured for the readiness match — no new probe, no extra dispatch.
|
|
115
|
+
// banner text already captured for the readiness match — no new probe, no extra dispatch. Kimi
|
|
116
|
+
// 0.29.0 may print either the full config key or its display suffix; normalize both idempotently
|
|
117
|
+
// to the full channel identifier routing uses.
|
|
99
118
|
const BANNER_MODEL_RE = /^Model:\s*(.+)$/m;
|
|
100
119
|
const BANNER_SESSION_RE = /^Session:\s*(session_[0-9a-f-]+)$/m;
|
|
101
120
|
export function kimiBannerModel(banner) {
|
|
102
121
|
const m = BANNER_MODEL_RE.exec(banner);
|
|
103
122
|
if (!m)
|
|
104
123
|
return undefined;
|
|
105
|
-
const
|
|
106
|
-
if (!
|
|
124
|
+
const printedModel = m[1].trim();
|
|
125
|
+
if (!printedModel)
|
|
107
126
|
return undefined;
|
|
108
|
-
return `kimi-code/${
|
|
127
|
+
return printedModel.startsWith("kimi-code/") ? printedModel : `kimi-code/${printedModel}`;
|
|
109
128
|
}
|
|
110
129
|
export function kimiBannerSessionId(banner) {
|
|
111
130
|
return BANNER_SESSION_RE.exec(banner)?.[1];
|
|
@@ -126,7 +145,8 @@ export const KIMI_INPUT_BOX = declareInputBox("kimi", {
|
|
|
126
145
|
// Shared launch-then-seed surface (T6) + banner confirm (T7/T2). One definition so the adapter
|
|
127
146
|
// property and the daemon's generic runInteractiveSeed path cannot drift.
|
|
128
147
|
const KIMI_SEED = {
|
|
129
|
-
|
|
148
|
+
// v1.76 T2 / OBS-141: 0.29.0 resolves only the full config.toml model key on the first turn.
|
|
149
|
+
launch: (model) => `kimi -y -m ${shq(model)}`,
|
|
130
150
|
readinessMatch: "Send /help for help information.",
|
|
131
151
|
seedLine: (promptFile) => `Read ${promptFile} and do exactly what it says.`,
|
|
132
152
|
confirmBanner: confirmKimiSeedBanner,
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import type { WorkerAdapter } from "../../adapters/types.js";
|
|
2
|
+
import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
|
|
2
3
|
export type DoctorOpts = {
|
|
3
4
|
banner?: boolean;
|
|
5
|
+
kimiTurnProbe?: (cwd: string) => Promise<KimiDoctorTurnResult>;
|
|
4
6
|
};
|
|
5
7
|
export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
|
|
@@ -6,6 +6,7 @@ import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
|
|
|
6
6
|
import { declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
|
|
7
7
|
import { DEFAULT_CONFIG, loadConfig, overlayPreferShapes } from "../../config/config.js";
|
|
8
8
|
import { HerdrDriver } from "../../drivers/herdr.js";
|
|
9
|
+
import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
|
|
9
10
|
import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
|
|
10
11
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
11
12
|
const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
|
|
@@ -20,6 +21,26 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
20
21
|
console.error("probing installed agent CLIs — one short LLM call per configured model, may take a minute...");
|
|
21
22
|
const probeProgressTTY = process.stderr.isTTY === true;
|
|
22
23
|
const health = await probeAll(adapters, { cwd });
|
|
24
|
+
const kimiAdapter = adapters.find((a) => a.id === kimi.id);
|
|
25
|
+
const kimiTurnEnabled = kimiAdapter !== undefined
|
|
26
|
+
&& (kimiAdapter === kimi || opts.kimiTurnProbe !== undefined);
|
|
27
|
+
if (kimiTurnEnabled) {
|
|
28
|
+
const h = health.kimi;
|
|
29
|
+
if (h.installed && h.authed) {
|
|
30
|
+
let turn;
|
|
31
|
+
try {
|
|
32
|
+
turn = await (opts.kimiTurnProbe ?? probeKimiDoctorTurn)(cwd);
|
|
33
|
+
}
|
|
34
|
+
catch (e) {
|
|
35
|
+
turn = { ok: false, evidence: e instanceof Error ? e.message : String(e) };
|
|
36
|
+
}
|
|
37
|
+
health.kimi = {
|
|
38
|
+
...h,
|
|
39
|
+
authed: turn.ok,
|
|
40
|
+
note: `${h.note ? `${h.note}; ` : ""}${turn.ok ? turn.evidence : `model turn failed: ${turn.evidence}`}`,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
}
|
|
23
44
|
// MODEL-02: detect models where the adapter exposes a list surface, BEFORE writing doctor.json (write once, below).
|
|
24
45
|
// Fail OPEN — the inverse of gates' fail-closed: detection is advisory, so a broken list surface NEVER fails doctor.
|
|
25
46
|
for (const a of adapters) {
|
|
@@ -35,14 +56,20 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
35
56
|
}
|
|
36
57
|
catch { /* fail open: leave models as-is, doctor stays healthy */ }
|
|
37
58
|
}
|
|
38
|
-
|
|
59
|
+
// A free Kimi auth failure or failed earned-green turn must not spend more probes. Every other
|
|
60
|
+
// adapter keeps the exact existing model-sweep path.
|
|
61
|
+
const modelProbeAdapters = kimiTurnEnabled && health.kimi.authed === false
|
|
62
|
+
? adapters.filter((a) => a !== kimiAdapter)
|
|
63
|
+
: adapters;
|
|
64
|
+
await probeModels(cfg, cwd, modelProbeAdapters, health, probeProgressTTY
|
|
39
65
|
? (adapter, model, status, durationMs) => console.error(` ${adapter}:${model} ${status} (${(durationMs / 1000).toFixed(1)}s)`)
|
|
40
66
|
: undefined);
|
|
41
67
|
writeDoctor(cwd, health);
|
|
42
68
|
const rows = adapters.map((a) => {
|
|
43
69
|
const h = health[a.id];
|
|
44
70
|
const state = !h.installed ? "not installed" : `${h.version ?? "installed"}${h.note ? ` (${h.note})` : ""}`;
|
|
45
|
-
|
|
71
|
+
const healthy = h.installed && (a.id !== kimi.id || h.authed);
|
|
72
|
+
return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
|
|
46
73
|
});
|
|
47
74
|
// v1.48 T1: advisory sweep for known agent CLIs with no adapter — never written to doctor.json health.
|
|
48
75
|
rows.push(...detectCandidateClis().map(({ binary, version }) => alignedStatusRow("warn", binary, `detected: ${version ?? "version unknown"} (no tickmarkr adapter — not routable)`)));
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -24,6 +24,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
24
24
|
private reserveDispatch;
|
|
25
25
|
private verifyPaneIdentityBinding;
|
|
26
26
|
private deliveryMatches;
|
|
27
|
+
private submissionRegistered;
|
|
27
28
|
static available(): boolean;
|
|
28
29
|
private herdr;
|
|
29
30
|
private namedPaneId;
|
|
@@ -40,6 +41,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
40
41
|
private joinGroup;
|
|
41
42
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
42
43
|
private deliver;
|
|
44
|
+
private submitVerifiedDelivery;
|
|
43
45
|
private settleDeliveryLine;
|
|
44
46
|
private deliveryReadMatches;
|
|
45
47
|
private waitOk;
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -8,6 +8,7 @@ export const TRAILER_SAFE_FLOOR_COLS = 108;
|
|
|
8
8
|
export const TRAILER_WIDTH_MARGIN = 2; // cols below (floor + margin) refuse a rightward first split
|
|
9
9
|
// OBS-85 verified delivery: bounded type→read-back→enter attempts before failing closed.
|
|
10
10
|
export const DELIVERY_ATTEMPTS = 3;
|
|
11
|
+
const DELIVERY_SUBMIT_ATTEMPTS = 2; // initial Enter + one evidence-backed re-press (OBS-140)
|
|
11
12
|
const DELIVERY_VERIFY_TIMEOUT_MS = 2000; // per attempt — a paste that hasn't rendered in 2s is retyped
|
|
12
13
|
const DELIVERY_READ_LINES = 80;
|
|
13
14
|
const DELIVERY_SETTLE_READ_ATTEMPTS = 6;
|
|
@@ -95,6 +96,22 @@ export class HerdrDriver {
|
|
|
95
96
|
const needle = norm(cmd);
|
|
96
97
|
return needle.length > 0 && hay.includes(needle);
|
|
97
98
|
}
|
|
99
|
+
// Submission succeeds when the typed prompt disappears, or when it has moved above a fresh
|
|
100
|
+
// adapter-declared input box (the prompt is now transcript, not input). Shell-line delivery uses
|
|
101
|
+
// the same normalized seam: a prompt still at the bottom ends the pane text; output/a fresh prompt
|
|
102
|
+
// after it proves Enter registered. No adapter-specific fingerprint lives in the driver.
|
|
103
|
+
submissionRegistered(transcript, cmd, inputBox) {
|
|
104
|
+
const norm = (s) => s.replace(/\s+/g, "");
|
|
105
|
+
const hay = norm(transcript);
|
|
106
|
+
const needle = norm(cmd);
|
|
107
|
+
const promptAt = hay.lastIndexOf(needle);
|
|
108
|
+
if (needle.length === 0 || promptAt < 0)
|
|
109
|
+
return true;
|
|
110
|
+
if (inputBox && matchesInputBox(transcript, inputBox)) {
|
|
111
|
+
return hay.lastIndexOf(norm(inputBox.fingerprint)) > promptAt;
|
|
112
|
+
}
|
|
113
|
+
return promptAt + needle.length < hay.length;
|
|
114
|
+
}
|
|
98
115
|
static available() {
|
|
99
116
|
return process.env.HERDR_ENV === "1";
|
|
100
117
|
}
|
|
@@ -419,9 +436,7 @@ export class HerdrDriver {
|
|
|
419
436
|
throw new Error(`herdr pane send-text failed: ${typed.stderr || typed.stdout}`);
|
|
420
437
|
const back = await this.herdr(`pane wait-output ${shq(pane)} --match ${shq(cmd)} --timeout ${DELIVERY_VERIFY_TIMEOUT_MS}`, slot.cwd, DELIVERY_VERIFY_TIMEOUT_MS + 15_000);
|
|
421
438
|
if (this.waitOk(back.code, back.stdout) || await this.deliveryReadMatches(pane, cmd, slot.cwd)) {
|
|
422
|
-
|
|
423
|
-
if (enter.code !== 0)
|
|
424
|
-
throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
|
|
439
|
+
await this.submitVerifiedDelivery(slot, cmd, pane);
|
|
425
440
|
return;
|
|
426
441
|
}
|
|
427
442
|
// capture the corrupted delivery BEFORE clearing it — the OBS-85 byte-level evidence
|
|
@@ -429,13 +444,40 @@ export class HerdrDriver {
|
|
|
429
444
|
}
|
|
430
445
|
throw new Error(`herdr delivery corrupted after ${DELIVERY_ATTEMPTS} attempts — enter never pressed (OBS-85); pane transcript:\n${transcript}`);
|
|
431
446
|
}
|
|
432
|
-
async
|
|
447
|
+
async submitVerifiedDelivery(slot, cmd, pane) {
|
|
448
|
+
let transcript = "";
|
|
449
|
+
const inputBox = this.inputBoxes.get(slot);
|
|
450
|
+
for (let attempt = 0; attempt < DELIVERY_SUBMIT_ATTEMPTS; attempt++) {
|
|
451
|
+
const enter = await this.herdr(`pane send-keys ${shq(pane)} Enter`, slot.cwd);
|
|
452
|
+
if (enter.code !== 0)
|
|
453
|
+
throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
|
|
454
|
+
// Reuse the existing settle-read window. A first-read success returns before any timer; only
|
|
455
|
+
// a prompt that still occupies the delivery target spends the bounded settle window. This
|
|
456
|
+
// verification always completes before a possible re-press, so a slow submit cannot duplicate.
|
|
457
|
+
const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript, inputBox, (candidate) => this.submissionRegistered(candidate, cmd, inputBox));
|
|
458
|
+
transcript = settled.transcript;
|
|
459
|
+
if (settled.ok)
|
|
460
|
+
return;
|
|
461
|
+
if (settled.readFailed) {
|
|
462
|
+
throw new Error(`herdr delivery corrupted — submission verification failed, refusing to re-press Enter (OBS-140); pane transcript:\n${transcript}`);
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
throw new Error(`herdr delivery corrupted after ${DELIVERY_SUBMIT_ATTEMPTS} submit attempts — submission never registered (OBS-140); pane transcript:\n${transcript}`);
|
|
466
|
+
}
|
|
467
|
+
async settleDeliveryLine(pane, cwd, initialTranscript, inputBox, accept) {
|
|
433
468
|
let transcript = initialTranscript;
|
|
434
469
|
for (let readAttempt = 0; readAttempt < DELIVERY_SETTLE_READ_ATTEMPTS; readAttempt++) {
|
|
435
470
|
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
|
|
436
471
|
if (read.code !== 0)
|
|
437
|
-
return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false };
|
|
438
|
-
if (read.stdout
|
|
472
|
+
return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false, readFailed: true };
|
|
473
|
+
if (accept?.(read.stdout)) {
|
|
474
|
+
return {
|
|
475
|
+
ok: true,
|
|
476
|
+
transcript: read.stdout,
|
|
477
|
+
recognizedInputBox: inputBox !== undefined && matchesInputBox(read.stdout, inputBox),
|
|
478
|
+
};
|
|
479
|
+
}
|
|
480
|
+
if (accept === undefined && read.stdout === transcript) {
|
|
439
481
|
return {
|
|
440
482
|
ok: true,
|
|
441
483
|
transcript,
|
package/dist/run/daemon.js
CHANGED
|
@@ -24,7 +24,7 @@ import { acquireRunLock, releaseRunLock } from "./lock.js";
|
|
|
24
24
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
25
25
|
import { nextChannel, route } from "../route/router.js";
|
|
26
26
|
import { desiredPanes } from "./reconcile.js";
|
|
27
|
-
import {
|
|
27
|
+
import { StallProgressTracker } from "./stall.js";
|
|
28
28
|
const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
|
|
29
29
|
// An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
|
|
30
30
|
// carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
|
|
@@ -795,13 +795,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
795
795
|
// single site, so a test can reason about it; keep Date.now() out of profile.ts (still pure) and
|
|
796
796
|
// out of adapter module scope (the cursor is a parameter, threaded from the daemon).
|
|
797
797
|
const attemptStart = Date.now();
|
|
798
|
-
// v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing
|
|
799
|
-
//
|
|
798
|
+
// v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing worker
|
|
799
|
+
// wait slices — never a new timer loop. null/unknown usage fails OPEN
|
|
800
800
|
// (never treated as over-threshold). Journal + notify fire at most once while the value stays high.
|
|
801
801
|
let contextWarned = false;
|
|
802
802
|
let contextTokens;
|
|
803
803
|
const sampleContext = async () => {
|
|
804
|
-
if (
|
|
804
|
+
if (!adapter.contextUsage)
|
|
805
805
|
return;
|
|
806
806
|
let usage = null;
|
|
807
807
|
try {
|
|
@@ -814,7 +814,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
814
814
|
if (!usage || typeof usage.tokens !== "number" || !Number.isFinite(usage.tokens))
|
|
815
815
|
return;
|
|
816
816
|
contextTokens = usage.tokens; // last known valid sample, including under-threshold resume candidates
|
|
817
|
-
if (usage.tokens < cfg.contextWarnTokens)
|
|
817
|
+
if (contextWarned || usage.tokens < cfg.contextWarnTokens)
|
|
818
818
|
return;
|
|
819
819
|
contextWarned = true;
|
|
820
820
|
lastContextTokens = usage.tokens;
|
|
@@ -862,15 +862,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
862
862
|
// OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
|
|
863
863
|
// stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
|
|
864
864
|
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
865
|
-
//
|
|
866
|
-
//
|
|
867
|
-
// trailer detection, harvest, paging, and quota checks all read the raw pane.
|
|
865
|
+
// v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
|
|
866
|
+
// the stall clock. Raw pane differences are terminal chrome until proven otherwise.
|
|
868
867
|
let everHadOutput = output.length > 0;
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
868
|
+
const stallProgress = new StallProgressTracker();
|
|
869
|
+
stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
|
|
870
|
+
let lastProgressAt = Date.now();
|
|
871
|
+
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
872
872
|
const sliceStart = Date.now();
|
|
873
|
-
const remaining = stallWindowMs - (sliceStart -
|
|
873
|
+
const remaining = stallWindowMs - (sliceStart - lastProgressAt);
|
|
874
874
|
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
875
875
|
if (!everHadOutput) {
|
|
876
876
|
const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
|
|
@@ -893,11 +893,6 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
893
893
|
const paneText = await driver.read(slot, 1000);
|
|
894
894
|
if (paneText.length > 0)
|
|
895
895
|
everHadOutput = true;
|
|
896
|
-
const currentStallSnapshot = normalizeStallSnapshot(paneText);
|
|
897
|
-
if (currentStallSnapshot !== lastStallSnapshot) {
|
|
898
|
-
lastStallSnapshot = currentStallSnapshot;
|
|
899
|
-
lastOutputAt = Date.now();
|
|
900
|
-
}
|
|
901
896
|
// OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
|
|
902
897
|
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
903
898
|
earlyLaunchDead = true;
|
|
@@ -906,6 +901,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
906
901
|
}
|
|
907
902
|
// v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
|
|
908
903
|
await sampleContext();
|
|
904
|
+
if (stallProgress.observe({ paneText, contextTokens }))
|
|
905
|
+
lastProgressAt = Date.now();
|
|
909
906
|
// page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
|
|
910
907
|
// (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
|
|
911
908
|
// a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
|
|
@@ -943,7 +940,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
943
940
|
}
|
|
944
941
|
if (!finished && exitCode === null) {
|
|
945
942
|
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
946
|
-
timedOut = Date.now() -
|
|
943
|
+
timedOut = Date.now() - lastProgressAt >= stallWindowMs;
|
|
947
944
|
output = await driver.read(slot, 1000);
|
|
948
945
|
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
949
946
|
const exit = exitRe.exec(output);
|
|
@@ -981,16 +978,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
981
978
|
else {
|
|
982
979
|
await driver.run(slot, paneDispatchCommand(dispatchScript));
|
|
983
980
|
// OBS-54: headless workers have the same output-inactivity budget as visible panes.
|
|
984
|
-
//
|
|
985
|
-
// exhaust the budget here too; harvest below still reads the raw pane.
|
|
981
|
+
// v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
|
|
986
982
|
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
987
983
|
const initialPane = await driver.read(slot, 500);
|
|
988
984
|
let everHadOutput = initialPane.length > 0;
|
|
989
|
-
|
|
990
|
-
|
|
985
|
+
const stallProgress = new StallProgressTracker();
|
|
986
|
+
stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
|
|
987
|
+
let lastProgressAt = Date.now();
|
|
991
988
|
finished = false;
|
|
992
|
-
while (Date.now() -
|
|
993
|
-
const remaining = stallWindowMs - (Date.now() -
|
|
989
|
+
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
990
|
+
const remaining = stallWindowMs - (Date.now() - lastProgressAt);
|
|
994
991
|
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
995
992
|
if (!everHadOutput) {
|
|
996
993
|
const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
|
|
@@ -1004,19 +1001,17 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1004
1001
|
const paneText = await driver.read(slot, 500);
|
|
1005
1002
|
if (paneText.length > 0)
|
|
1006
1003
|
everHadOutput = true;
|
|
1007
|
-
const currentStallSnapshot = normalizeStallSnapshot(paneText);
|
|
1008
|
-
if (currentStallSnapshot !== lastStallSnapshot) {
|
|
1009
|
-
lastStallSnapshot = currentStallSnapshot;
|
|
1010
|
-
lastOutputAt = Date.now();
|
|
1011
|
-
}
|
|
1012
1004
|
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
1013
1005
|
earlyLaunchDead = true;
|
|
1014
1006
|
break;
|
|
1015
1007
|
}
|
|
1008
|
+
await sampleContext();
|
|
1009
|
+
if (stallProgress.observe({ paneText, contextTokens }))
|
|
1010
|
+
lastProgressAt = Date.now();
|
|
1016
1011
|
}
|
|
1017
1012
|
output = await driver.read(slot, 500);
|
|
1018
1013
|
exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
|
|
1019
|
-
timedOut = !finished && Date.now() -
|
|
1014
|
+
timedOut = !finished && Date.now() - lastProgressAt >= stallWindowMs;
|
|
1020
1015
|
}
|
|
1021
1016
|
// SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
|
|
1022
1017
|
// shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
|
package/dist/run/stall.d.ts
CHANGED
|
@@ -1,8 +1,25 @@
|
|
|
1
|
-
/** Normalize
|
|
2
|
-
* waitOutput, and paging read the raw text
|
|
3
|
-
* CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
|
|
4
|
-
* equal are the same frame modulo spinner presentation; any other byte difference is activity. */
|
|
1
|
+
/** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
|
|
2
|
+
* parsing, harvest, waitOutput, and paging always read the raw text. */
|
|
5
3
|
export declare function normalizeStallSnapshot(text: string): string;
|
|
4
|
+
export interface StallProgressSample {
|
|
5
|
+
paneText: string;
|
|
6
|
+
seedSubmitted?: boolean;
|
|
7
|
+
contextTokens?: number;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Monotonic worker-progress measure for the stall watchdog.
|
|
11
|
+
*
|
|
12
|
+
* Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
|
|
13
|
+
* evidence of work. A rendered transcript is only known to have grown when it occupies more
|
|
14
|
+
* non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
|
|
15
|
+
* advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
|
|
16
|
+
*/
|
|
17
|
+
export declare class StallProgressTracker {
|
|
18
|
+
private transcriptRows;
|
|
19
|
+
private seedSubmitted;
|
|
20
|
+
private contextTokens;
|
|
21
|
+
observe(sample: StallProgressSample): boolean;
|
|
22
|
+
}
|
|
6
23
|
/** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
|
|
7
24
|
* seam exists for fault injection in tests only — production callers pass text alone. */
|
|
8
25
|
export declare function filterLlmTranscript(text: string, classify?: (t: string) => string): string;
|
package/dist/run/stall.js
CHANGED
|
@@ -1,12 +1,9 @@
|
|
|
1
|
-
// OBS-82:
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
// design: an allowlist MISS degrades to today's recoverable no-reap behavior, while an over-broad
|
|
8
|
-
// deletion would reap a healthy worker — a new failure class. Grow the allowlist only with
|
|
9
|
-
// captured evidence (tests/fixtures/codex-mcp-spinner/).
|
|
1
|
+
// OBS-82: normalize known presentation tokens before measuring transcript extent or filtering an
|
|
2
|
+
// LLM-bound transcript. This remains a closed allowlist — ANSI/VT escapes, braille-range spinner
|
|
3
|
+
// glyphs, and elapsed-time tokens bound to time-unit suffixes. Every other byte passes through
|
|
4
|
+
// identical. v1.76 deliberately stopped treating arbitrary normalized byte changes as progress:
|
|
5
|
+
// StallProgressTracker below requires monotonic evidence, so an unknown repaint fails closed toward
|
|
6
|
+
// a recoverable consult instead of holding the watchdog silent.
|
|
10
7
|
// CSI (with intermediates), OSC (BEL- or ST-terminated), DCS/SOS/PM/APC strings, single-char
|
|
11
8
|
// escapes, and charset selection — the raw-pty forms; herdr pane reads are already rendered.
|
|
12
9
|
// eslint-disable-next-line no-control-regex
|
|
@@ -16,13 +13,46 @@ const SPINNER_RE = /[⠀-⣿]/g;
|
|
|
16
13
|
// A digit run (optionally decimal) bound directly to a time-unit suffix, standing alone as a
|
|
17
14
|
// word: 9s, 41s, 3m, 1h, 800ms. Never bare digits — "(6/7)" and "5 of 7" stay change-sensitive.
|
|
18
15
|
const ELAPSED_RE = /(?<![\w.])\d+(?:\.\d+)?(?:ms|[hms])(?!\w)/g;
|
|
19
|
-
/** Normalize
|
|
20
|
-
* waitOutput, and paging read the raw text
|
|
21
|
-
* CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
|
|
22
|
-
* equal are the same frame modulo spinner presentation; any other byte difference is activity. */
|
|
16
|
+
/** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
|
|
17
|
+
* parsing, harvest, waitOutput, and paging always read the raw text. */
|
|
23
18
|
export function normalizeStallSnapshot(text) {
|
|
24
19
|
return text.replace(ANSI_RE, "").replace(SPINNER_RE, "").replace(ELAPSED_RE, "");
|
|
25
20
|
}
|
|
21
|
+
/**
|
|
22
|
+
* Monotonic worker-progress measure for the stall watchdog.
|
|
23
|
+
*
|
|
24
|
+
* Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
|
|
25
|
+
* evidence of work. A rendered transcript is only known to have grown when it occupies more
|
|
26
|
+
* non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
|
|
27
|
+
* advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
|
|
28
|
+
*/
|
|
29
|
+
export class StallProgressTracker {
|
|
30
|
+
transcriptRows = 0;
|
|
31
|
+
seedSubmitted = false;
|
|
32
|
+
contextTokens;
|
|
33
|
+
observe(sample) {
|
|
34
|
+
let advanced = false;
|
|
35
|
+
const rows = normalizeStallSnapshot(sample.paneText)
|
|
36
|
+
.split("\n")
|
|
37
|
+
.filter((line) => line.trim().length > 0)
|
|
38
|
+
.length;
|
|
39
|
+
if (rows > this.transcriptRows) {
|
|
40
|
+
this.transcriptRows = rows;
|
|
41
|
+
advanced = true;
|
|
42
|
+
}
|
|
43
|
+
if (sample.seedSubmitted && !this.seedSubmitted) {
|
|
44
|
+
this.seedSubmitted = true;
|
|
45
|
+
advanced = true;
|
|
46
|
+
}
|
|
47
|
+
const tokens = sample.contextTokens;
|
|
48
|
+
if (tokens !== undefined && Number.isFinite(tokens)) {
|
|
49
|
+
if (tokens > (this.contextTokens ?? 0))
|
|
50
|
+
advanced = true;
|
|
51
|
+
this.contextTokens = Math.max(this.contextTokens ?? 0, tokens);
|
|
52
|
+
}
|
|
53
|
+
return advanced;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
26
56
|
// ─── v1.65 T2: LLM-bound transcript filter ──────────────────────────────────────────────────────
|
|
27
57
|
// Consult dossiers and gate prompts pay tokens per transcript byte, so LLM-bound text runs through
|
|
28
58
|
// a per-line classifier: carriage-return overwrite churn keeps only the final paint, lines that are
|