tickmarkr 1.75.0 → 1.77.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/kimi.d.ts +5 -0
- package/dist/adapters/kimi.js +54 -13
- package/dist/adapters/types.d.ts +5 -0
- package/dist/adapters/types.js +4 -1
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +29 -2
- package/dist/drivers/herdr.d.ts +10 -0
- package/dist/drivers/herdr.js +140 -11
- package/dist/run/daemon.js +88 -33
- package/dist/run/stall.d.ts +21 -4
- package/dist/run/stall.js +43 -13
- package/package.json +1 -1
package/dist/adapters/kimi.d.ts
CHANGED
|
@@ -3,6 +3,11 @@ import { type Assignment, type WorkerAdapter, type WorkerResult } from "./types.
|
|
|
3
3
|
export declare function kimiAuthed(credentialsText: string, nowMs: number): boolean;
|
|
4
4
|
export declare function parseKimiModels(raw: string): string[];
|
|
5
5
|
export declare function parseKimiResult(raw: string, nonce: string): WorkerResult;
|
|
6
|
+
export interface KimiDoctorTurnResult {
|
|
7
|
+
ok: boolean;
|
|
8
|
+
evidence: string;
|
|
9
|
+
}
|
|
10
|
+
export declare function probeKimiDoctorTurn(cwd: string): Promise<KimiDoctorTurnResult>;
|
|
6
11
|
export declare function kimiSessionId(output: string): string | undefined;
|
|
7
12
|
export declare function kimiBannerModel(banner: string): string | undefined;
|
|
8
13
|
export declare function kimiBannerSessionId(banner: string): string | undefined;
|
package/dist/adapters/kimi.js
CHANGED
|
@@ -64,6 +64,28 @@ export function parseKimiResult(raw, nonce) {
|
|
|
64
64
|
const stripped = raw.split("\n").map((l) => l.replace(/^[\s]*[•*-]\s+/, "")).join("\n");
|
|
65
65
|
return parseWorkerResult(stripped, nonce);
|
|
66
66
|
}
|
|
67
|
+
const KIMI_DOCTOR_TURN_MODEL = "kimi-code/k3";
|
|
68
|
+
const KIMI_DOCTOR_TURN_PROMPT = "Reply with exactly OK and nothing else.";
|
|
69
|
+
const KIMI_DOCTOR_TURN_TIMEOUT_MS = 60000;
|
|
70
|
+
// OBS-141: intentionally separate from probe() so plan/run remain free file checks. Only doctor
|
|
71
|
+
// calls this one-turn contract probe; its test seam stubs sh and never launches a real agent CLI.
|
|
72
|
+
export async function probeKimiDoctorTurn(cwd) {
|
|
73
|
+
const command = `kimi -p ${shq(KIMI_DOCTOR_TURN_PROMPT)} --model ${shq(KIMI_DOCTOR_TURN_MODEL)} --output-format text`;
|
|
74
|
+
const r = await sh(command, cwd, KIMI_DOCTOR_TURN_TIMEOUT_MS);
|
|
75
|
+
if (r.timedOut) {
|
|
76
|
+
return { ok: false, evidence: `turn timed out after ${KIMI_DOCTOR_TURN_TIMEOUT_MS}ms` };
|
|
77
|
+
}
|
|
78
|
+
const output = `${r.stderr}\n${r.stdout}`.trim().replace(/\s+/g, " ");
|
|
79
|
+
if (r.code !== 0) {
|
|
80
|
+
return { ok: false, evidence: output || `turn exited ${r.code}` };
|
|
81
|
+
}
|
|
82
|
+
const returnedOk = r.stdout.split("\n")
|
|
83
|
+
.map((line) => line.replace(/^[\s]*[•*-]\s+/, "").trim())
|
|
84
|
+
.includes("OK");
|
|
85
|
+
return returnedOk
|
|
86
|
+
? { ok: true, evidence: `model turn returned OK with ${KIMI_DOCTOR_TURN_MODEL}` }
|
|
87
|
+
: { ok: false, evidence: `turn returned no exact OK answer${output ? `: ${output}` : ""}` };
|
|
88
|
+
}
|
|
67
89
|
// v1.53 T3: session-id capture from the run-output trailer — every `kimi -p` run (fresh or resumed)
|
|
68
90
|
// ends with `To resume this session: kimi -r session_<uuid>` (live probe 2026-07-18). Anchored full
|
|
69
91
|
// line only: prompt/model prose can contain lookalike text, and the anchored charset keeps a
|
|
@@ -89,23 +111,20 @@ export function kimiSessionId(output) {
|
|
|
89
111
|
}
|
|
90
112
|
return id;
|
|
91
113
|
}
|
|
92
|
-
// v1.69 T6: the native TUI takes -m <alias>, where config.toml aliases are the bare model suffix of
|
|
93
|
-
// the tickmarkr channel id (live probe 2026-07-22). Keep the mapping explicit and localized.
|
|
94
|
-
function kimiAlias(model) {
|
|
95
|
-
return model.replace(/^kimi-code\//, "");
|
|
96
|
-
}
|
|
97
114
|
// v1.69 T7: the cold-start banner prints the model alias and session id. Parse them from the
|
|
98
|
-
// banner text already captured for the readiness match — no new probe, no extra dispatch.
|
|
115
|
+
// banner text already captured for the readiness match — no new probe, no extra dispatch. Kimi
|
|
116
|
+
// 0.29.0 may print either the full config key or its display suffix; normalize both idempotently
|
|
117
|
+
// to the full channel identifier routing uses.
|
|
99
118
|
const BANNER_MODEL_RE = /^Model:\s*(.+)$/m;
|
|
100
119
|
const BANNER_SESSION_RE = /^Session:\s*(session_[0-9a-f-]+)$/m;
|
|
101
120
|
export function kimiBannerModel(banner) {
|
|
102
121
|
const m = BANNER_MODEL_RE.exec(banner);
|
|
103
122
|
if (!m)
|
|
104
123
|
return undefined;
|
|
105
|
-
const
|
|
106
|
-
if (!
|
|
124
|
+
const printedModel = m[1].trim();
|
|
125
|
+
if (!printedModel)
|
|
107
126
|
return undefined;
|
|
108
|
-
return `kimi-code/${
|
|
127
|
+
return printedModel.startsWith("kimi-code/") ? printedModel : `kimi-code/${printedModel}`;
|
|
109
128
|
}
|
|
110
129
|
export function kimiBannerSessionId(banner) {
|
|
111
130
|
return BANNER_SESSION_RE.exec(banner)?.[1];
|
|
@@ -118,15 +137,37 @@ export function confirmKimiSeedBanner(banner, assignedModel) {
|
|
|
118
137
|
}
|
|
119
138
|
return { ok: true, sessionId: kimiBannerSessionId(banner) };
|
|
120
139
|
}
|
|
121
|
-
|
|
122
|
-
|
|
140
|
+
const kimiTuiLaunch = (model) => `kimi -y -m ${shq(model)}`;
|
|
141
|
+
const KIMI_ANSI_SGR_RE = /\u001B\[[0-9;]*m/g;
|
|
142
|
+
const KIMI_EDITOR_TOP_RE = /^╭─+╮$/;
|
|
143
|
+
const KIMI_EDITOR_INPUT_RE = /^│ > .*│$/;
|
|
144
|
+
const KIMI_EDITOR_EMPTY_RE = /^│ > +│$/;
|
|
145
|
+
const KIMI_EDITOR_BOTTOM_RE = /^╰─+╯$/;
|
|
146
|
+
function matchesKimiEditorBox(paneText, empty) {
|
|
147
|
+
const lines = paneText.replace(KIMI_ANSI_SGR_RE, "").split("\n");
|
|
148
|
+
const input = empty ? KIMI_EDITOR_EMPTY_RE : KIMI_EDITOR_INPUT_RE;
|
|
149
|
+
return lines.some((line, index) => input.test(line)
|
|
150
|
+
&& index > 0
|
|
151
|
+
&& index + 1 < lines.length
|
|
152
|
+
&& KIMI_EDITOR_TOP_RE.test(lines[index - 1])
|
|
153
|
+
&& KIMI_EDITOR_BOTTOM_RE.test(lines[index + 1]));
|
|
154
|
+
}
|
|
155
|
+
// v1.75 T1 / OBS-136: Kimi's actual editor is a three-line bordered box whose content row starts
|
|
156
|
+
// `│ > `. The welcome panel separately contains "Send /help..." and is only launch-liveness
|
|
157
|
+
// evidence (OBS-142), so it must never satisfy this declaration. The adapter owns the full matcher,
|
|
158
|
+
// empty-state matcher, launch distinction, and bounded cold-start window; the driver stays generic.
|
|
123
159
|
export const KIMI_INPUT_BOX = declareInputBox("kimi", {
|
|
124
|
-
fingerprint: "
|
|
160
|
+
fingerprint: "│ > ",
|
|
161
|
+
match: (paneText) => matchesKimiEditorBox(paneText, false),
|
|
162
|
+
emptyMatch: (paneText) => matchesKimiEditorBox(paneText, true),
|
|
163
|
+
launchCommand: (command) => command.startsWith("kimi -y -m "),
|
|
164
|
+
readinessTimeoutMs: 15_000,
|
|
125
165
|
});
|
|
126
166
|
// Shared launch-then-seed surface (T6) + banner confirm (T7/T2). One definition so the adapter
|
|
127
167
|
// property and the daemon's generic runInteractiveSeed path cannot drift.
|
|
128
168
|
const KIMI_SEED = {
|
|
129
|
-
|
|
169
|
+
// v1.76 T2 / OBS-141: 0.29.0 resolves only the full config.toml model key on the first turn.
|
|
170
|
+
launch: kimiTuiLaunch,
|
|
130
171
|
readinessMatch: "Send /help for help information.",
|
|
131
172
|
seedLine: (promptFile) => `Read ${promptFile} and do exactly what it says.`,
|
|
132
173
|
confirmBanner: confirmKimiSeedBanner,
|
package/dist/adapters/types.d.ts
CHANGED
|
@@ -84,10 +84,15 @@ export interface TrustDialog {
|
|
|
84
84
|
export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): boolean;
|
|
85
85
|
export interface InputBox {
|
|
86
86
|
fingerprint: string;
|
|
87
|
+
match?(paneText: string): boolean;
|
|
88
|
+
emptyMatch?(paneText: string): boolean;
|
|
89
|
+
launchCommand?(command: string): boolean;
|
|
90
|
+
readinessTimeoutMs?: number;
|
|
87
91
|
}
|
|
88
92
|
export declare function declareInputBox(adapterId: string, inputBox: InputBox): InputBox;
|
|
89
93
|
export declare function declaredInputBoxForWorkerName(workerName: string): InputBox | undefined;
|
|
90
94
|
export declare function matchesInputBox(paneText: string, inputBox: InputBox): boolean;
|
|
95
|
+
export declare function matchesEmptyInputBox(paneText: string, inputBox: InputBox): boolean;
|
|
91
96
|
export interface WorkerAdapter {
|
|
92
97
|
id: string;
|
|
93
98
|
vendor: string;
|
package/dist/adapters/types.js
CHANGED
|
@@ -40,7 +40,10 @@ export function declaredInputBoxForWorkerName(workerName) {
|
|
|
40
40
|
return adapterId === undefined ? undefined : inputBoxes.get(adapterId);
|
|
41
41
|
}
|
|
42
42
|
export function matchesInputBox(paneText, inputBox) {
|
|
43
|
-
return paneText.includes(inputBox.fingerprint);
|
|
43
|
+
return inputBox.match?.(paneText) ?? paneText.includes(inputBox.fingerprint);
|
|
44
|
+
}
|
|
45
|
+
export function matchesEmptyInputBox(paneText, inputBox) {
|
|
46
|
+
return inputBox.emptyMatch?.(paneText) === true;
|
|
44
47
|
}
|
|
45
48
|
export function channelsFromConfig(adapterId, cfg) {
|
|
46
49
|
const e = cfg.tiers[adapterId];
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import type { WorkerAdapter } from "../../adapters/types.js";
|
|
2
|
+
import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
|
|
2
3
|
export type DoctorOpts = {
|
|
3
4
|
banner?: boolean;
|
|
5
|
+
kimiTurnProbe?: (cwd: string) => Promise<KimiDoctorTurnResult>;
|
|
4
6
|
};
|
|
5
7
|
export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
|
|
@@ -6,6 +6,7 @@ import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
|
|
|
6
6
|
import { declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
|
|
7
7
|
import { DEFAULT_CONFIG, loadConfig, overlayPreferShapes } from "../../config/config.js";
|
|
8
8
|
import { HerdrDriver } from "../../drivers/herdr.js";
|
|
9
|
+
import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
|
|
9
10
|
import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
|
|
10
11
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
11
12
|
const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
|
|
@@ -20,6 +21,26 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
20
21
|
console.error("probing installed agent CLIs — one short LLM call per configured model, may take a minute...");
|
|
21
22
|
const probeProgressTTY = process.stderr.isTTY === true;
|
|
22
23
|
const health = await probeAll(adapters, { cwd });
|
|
24
|
+
const kimiAdapter = adapters.find((a) => a.id === kimi.id);
|
|
25
|
+
const kimiTurnEnabled = kimiAdapter !== undefined
|
|
26
|
+
&& (kimiAdapter === kimi || opts.kimiTurnProbe !== undefined);
|
|
27
|
+
if (kimiTurnEnabled) {
|
|
28
|
+
const h = health.kimi;
|
|
29
|
+
if (h.installed && h.authed) {
|
|
30
|
+
let turn;
|
|
31
|
+
try {
|
|
32
|
+
turn = await (opts.kimiTurnProbe ?? probeKimiDoctorTurn)(cwd);
|
|
33
|
+
}
|
|
34
|
+
catch (e) {
|
|
35
|
+
turn = { ok: false, evidence: e instanceof Error ? e.message : String(e) };
|
|
36
|
+
}
|
|
37
|
+
health.kimi = {
|
|
38
|
+
...h,
|
|
39
|
+
authed: turn.ok,
|
|
40
|
+
note: `${h.note ? `${h.note}; ` : ""}${turn.ok ? turn.evidence : `model turn failed: ${turn.evidence}`}`,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
}
|
|
23
44
|
// MODEL-02: detect models where the adapter exposes a list surface, BEFORE writing doctor.json (write once, below).
|
|
24
45
|
// Fail OPEN — the inverse of gates' fail-closed: detection is advisory, so a broken list surface NEVER fails doctor.
|
|
25
46
|
for (const a of adapters) {
|
|
@@ -35,14 +56,20 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
35
56
|
}
|
|
36
57
|
catch { /* fail open: leave models as-is, doctor stays healthy */ }
|
|
37
58
|
}
|
|
38
|
-
|
|
59
|
+
// A free Kimi auth failure or failed earned-green turn must not spend more probes. Every other
|
|
60
|
+
// adapter keeps the exact existing model-sweep path.
|
|
61
|
+
const modelProbeAdapters = kimiTurnEnabled && health.kimi.authed === false
|
|
62
|
+
? adapters.filter((a) => a !== kimiAdapter)
|
|
63
|
+
: adapters;
|
|
64
|
+
await probeModels(cfg, cwd, modelProbeAdapters, health, probeProgressTTY
|
|
39
65
|
? (adapter, model, status, durationMs) => console.error(` ${adapter}:${model} ${status} (${(durationMs / 1000).toFixed(1)}s)`)
|
|
40
66
|
: undefined);
|
|
41
67
|
writeDoctor(cwd, health);
|
|
42
68
|
const rows = adapters.map((a) => {
|
|
43
69
|
const h = health[a.id];
|
|
44
70
|
const state = !h.installed ? "not installed" : `${h.version ?? "installed"}${h.note ? ` (${h.note})` : ""}`;
|
|
45
|
-
|
|
71
|
+
const healthy = h.installed && (a.id !== kimi.id || h.authed);
|
|
72
|
+
return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
|
|
46
73
|
});
|
|
47
74
|
// v1.48 T1: advisory sweep for known agent CLIs with no adapter — never written to doctor.json health.
|
|
48
75
|
rows.push(...detectCandidateClis().map(({ binary, version }) => alignedStatusRow("warn", binary, `detected: ${version ?? "version unknown"} (no tickmarkr adapter — not routable)`)));
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -2,6 +2,12 @@ import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "
|
|
|
2
2
|
export declare const TRAILER_SAFE_FLOOR_COLS = 108;
|
|
3
3
|
export declare const TRAILER_WIDTH_MARGIN = 2;
|
|
4
4
|
export declare const DELIVERY_ATTEMPTS = 3;
|
|
5
|
+
export declare class DeliveryReadinessError extends Error {
|
|
6
|
+
readonly waitedMs: number;
|
|
7
|
+
readonly transcript: string;
|
|
8
|
+
readonly phase = "READINESS";
|
|
9
|
+
constructor(waitedMs: number, transcript: string);
|
|
10
|
+
}
|
|
5
11
|
/** First-generation join direction from measured trailer-safe floor (43-MEASUREMENT.md). */
|
|
6
12
|
export declare function workerSplitDirection(paneCols: number | null, safeFloor?: number, margin?: number): "right" | "down";
|
|
7
13
|
export declare class HerdrDriver implements ExecutorDriver {
|
|
@@ -24,6 +30,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
24
30
|
private reserveDispatch;
|
|
25
31
|
private verifyPaneIdentityBinding;
|
|
26
32
|
private deliveryMatches;
|
|
33
|
+
private submissionRegistered;
|
|
27
34
|
static available(): boolean;
|
|
28
35
|
private herdr;
|
|
29
36
|
private namedPaneId;
|
|
@@ -40,7 +47,10 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
40
47
|
private joinGroup;
|
|
41
48
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
42
49
|
private deliver;
|
|
50
|
+
private deliverPersistentShellCommand;
|
|
51
|
+
private submitVerifiedDelivery;
|
|
43
52
|
private settleDeliveryLine;
|
|
53
|
+
private awaitDeliveryReadiness;
|
|
44
54
|
private deliveryReadMatches;
|
|
45
55
|
private waitOk;
|
|
46
56
|
waitOutput(slot: Slot, pattern: string, timeoutMs: number, opts?: {
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { declaredInputBoxForWorkerName, matchesInputBox, shq } from "../adapters/types.js";
|
|
1
|
+
import { declaredInputBoxForWorkerName, matchesEmptyInputBox, matchesInputBox, shq } from "../adapters/types.js";
|
|
2
2
|
import { PANE_IDENTITY_ENV, paneIdentityLine } from "../brand.js";
|
|
3
3
|
import { createWorktree, sh } from "../run/git.js";
|
|
4
4
|
import { herdrSealShellPrefix } from "./subprocess.js";
|
|
@@ -8,10 +8,26 @@ export const TRAILER_SAFE_FLOOR_COLS = 108;
|
|
|
8
8
|
export const TRAILER_WIDTH_MARGIN = 2; // cols below (floor + margin) refuse a rightward first split
|
|
9
9
|
// OBS-85 verified delivery: bounded type→read-back→enter attempts before failing closed.
|
|
10
10
|
export const DELIVERY_ATTEMPTS = 3;
|
|
11
|
+
const DELIVERY_SUBMIT_ATTEMPTS = 2; // initial Enter + one evidence-backed re-press (OBS-140)
|
|
11
12
|
const DELIVERY_VERIFY_TIMEOUT_MS = 2000; // per attempt — a paste that hasn't rendered in 2s is retyped
|
|
12
13
|
const DELIVERY_READ_LINES = 80;
|
|
13
14
|
const DELIVERY_SETTLE_READ_ATTEMPTS = 6;
|
|
14
15
|
const DELIVERY_SETTLE_POLL_MS = 100;
|
|
16
|
+
const DELIVERY_READINESS_TIMEOUT_MS = 1_000;
|
|
17
|
+
// OBS-142: a typed identity lets the daemon's next task distinguish cold-start variance from
|
|
18
|
+
// structural driver faults without parsing prose. The message also carries the bounded wait and
|
|
19
|
+
// final pane evidence so a terminal failure is diagnosable on its own.
|
|
20
|
+
export class DeliveryReadinessError extends Error {
|
|
21
|
+
waitedMs;
|
|
22
|
+
transcript;
|
|
23
|
+
phase = "READINESS";
|
|
24
|
+
constructor(waitedMs, transcript) {
|
|
25
|
+
super(`herdr delivery READINESS failed after ${waitedMs}ms — interface never became interactive (OBS-142); pane transcript:\n${transcript}`);
|
|
26
|
+
this.waitedMs = waitedMs;
|
|
27
|
+
this.transcript = transcript;
|
|
28
|
+
this.name = "DeliveryReadinessError";
|
|
29
|
+
}
|
|
30
|
+
}
|
|
15
31
|
/** First-generation join direction from measured trailer-safe floor (43-MEASUREMENT.md). */
|
|
16
32
|
export function workerSplitDirection(paneCols, safeFloor = TRAILER_SAFE_FLOOR_COLS, margin = TRAILER_WIDTH_MARGIN) {
|
|
17
33
|
if (paneCols == null || paneCols <= 0)
|
|
@@ -95,6 +111,23 @@ export class HerdrDriver {
|
|
|
95
111
|
const needle = norm(cmd);
|
|
96
112
|
return needle.length > 0 && hay.includes(needle);
|
|
97
113
|
}
|
|
114
|
+
// Submission requires positive post-Enter evidence. A prompt echoed above a fresh input target
|
|
115
|
+
// proves registration; a declared box that is visibly empty after the verified paste also proves
|
|
116
|
+
// it consumed the prompt. Prompt absence alone is never success (OBS-142).
|
|
117
|
+
submissionRegistered(transcript, cmd, inputBox) {
|
|
118
|
+
const norm = (s) => s.replace(/\s+/g, "");
|
|
119
|
+
const hay = norm(transcript);
|
|
120
|
+
const needle = norm(cmd);
|
|
121
|
+
const promptAt = hay.lastIndexOf(needle);
|
|
122
|
+
if (needle.length === 0)
|
|
123
|
+
return false;
|
|
124
|
+
if (promptAt < 0)
|
|
125
|
+
return inputBox !== undefined && matchesEmptyInputBox(transcript, inputBox);
|
|
126
|
+
if (inputBox && matchesInputBox(transcript, inputBox)) {
|
|
127
|
+
return hay.lastIndexOf(norm(inputBox.fingerprint)) > promptAt;
|
|
128
|
+
}
|
|
129
|
+
return promptAt + needle.length < hay.length;
|
|
130
|
+
}
|
|
98
131
|
static available() {
|
|
99
132
|
return process.env.HERDR_ENV === "1";
|
|
100
133
|
}
|
|
@@ -392,7 +425,8 @@ export class HerdrDriver {
|
|
|
392
425
|
this.deliveredPanes.set(slot, paneId);
|
|
393
426
|
});
|
|
394
427
|
}
|
|
395
|
-
async deliver(slot, cmd, pane) {
|
|
428
|
+
async deliver(slot, cmd, pane, verifySubmission = true) {
|
|
429
|
+
const readiness = await this.awaitDeliveryReadiness(slot, cmd, pane);
|
|
396
430
|
let transcript = "";
|
|
397
431
|
for (let attempt = 0; attempt < DELIVERY_ATTEMPTS; attempt++) {
|
|
398
432
|
if (attempt > 0) {
|
|
@@ -419,9 +453,14 @@ export class HerdrDriver {
|
|
|
419
453
|
throw new Error(`herdr pane send-text failed: ${typed.stderr || typed.stdout}`);
|
|
420
454
|
const back = await this.herdr(`pane wait-output ${shq(pane)} --match ${shq(cmd)} --timeout ${DELIVERY_VERIFY_TIMEOUT_MS}`, slot.cwd, DELIVERY_VERIFY_TIMEOUT_MS + 15_000);
|
|
421
455
|
if (this.waitOk(back.code, back.stdout) || await this.deliveryReadMatches(pane, cmd, slot.cwd)) {
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
456
|
+
if (verifySubmission) {
|
|
457
|
+
await this.submitVerifiedDelivery(slot, cmd, pane, readiness);
|
|
458
|
+
}
|
|
459
|
+
else {
|
|
460
|
+
const enter = await this.herdr(`pane send-keys ${shq(pane)} Enter`, slot.cwd);
|
|
461
|
+
if (enter.code !== 0)
|
|
462
|
+
throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
|
|
463
|
+
}
|
|
425
464
|
return;
|
|
426
465
|
}
|
|
427
466
|
// capture the corrupted delivery BEFORE clearing it — the OBS-85 byte-level evidence
|
|
@@ -429,13 +468,57 @@ export class HerdrDriver {
|
|
|
429
468
|
}
|
|
430
469
|
throw new Error(`herdr delivery corrupted after ${DELIVERY_ATTEMPTS} attempts — enter never pressed (OBS-85); pane transcript:\n${transcript}`);
|
|
431
470
|
}
|
|
432
|
-
|
|
471
|
+
// The narrator launches a perpetual shell watch, not an adapter-backed input interface. Keep
|
|
472
|
+
// OBS-85's readiness + paste read-back and the historical single Enter, while worker/gate runs
|
|
473
|
+
// continue through positive post-Enter evidence in submitVerifiedDelivery.
|
|
474
|
+
async deliverPersistentShellCommand(slot, cmd) {
|
|
475
|
+
return this.deliveryQueue(async () => {
|
|
476
|
+
const pane = await this.paneId(slot);
|
|
477
|
+
await this.deliver(slot, cmd, pane, false);
|
|
478
|
+
this.deliveredPanes.set(slot, pane);
|
|
479
|
+
});
|
|
480
|
+
}
|
|
481
|
+
async submitVerifiedDelivery(slot, cmd, pane, readiness) {
|
|
482
|
+
let transcript = "";
|
|
483
|
+
const inputBox = this.inputBoxes.get(slot);
|
|
484
|
+
// The base window preserves OBS-140's bounded behavior. A slow readiness observation grants
|
|
485
|
+
// the same measured time once more for submit paint, capped by the readiness bound itself.
|
|
486
|
+
const baseVerifyMs = (DELIVERY_SETTLE_READ_ATTEMPTS - 1) * DELIVERY_SETTLE_POLL_MS;
|
|
487
|
+
const verifyWindowMs = Math.min(baseVerifyMs + readiness.timeoutMs, baseVerifyMs + readiness.waitedMs);
|
|
488
|
+
for (let attempt = 0; attempt < DELIVERY_SUBMIT_ATTEMPTS; attempt++) {
|
|
489
|
+
const enter = await this.herdr(`pane send-keys ${shq(pane)} Enter`, slot.cwd);
|
|
490
|
+
if (enter.code !== 0)
|
|
491
|
+
throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
|
|
492
|
+
// Reuse the existing settle-read window. A first-read success returns before any timer; only
|
|
493
|
+
// a prompt that still occupies the delivery target spends the bounded settle window. This
|
|
494
|
+
// verification always completes before a possible re-press, so a slow submit cannot duplicate.
|
|
495
|
+
const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript, inputBox, (candidate) => this.submissionRegistered(candidate, cmd, inputBox), verifyWindowMs);
|
|
496
|
+
transcript = settled.transcript;
|
|
497
|
+
if (settled.ok)
|
|
498
|
+
return;
|
|
499
|
+
if (settled.readFailed) {
|
|
500
|
+
throw new Error(`herdr delivery corrupted — submission verification failed, refusing to re-press Enter (OBS-140); pane transcript:\n${transcript}`);
|
|
501
|
+
}
|
|
502
|
+
}
|
|
503
|
+
throw new Error(`herdr delivery corrupted after ${DELIVERY_SUBMIT_ATTEMPTS} submit attempts — submission never registered (OBS-140); pane transcript:\n${transcript}`);
|
|
504
|
+
}
|
|
505
|
+
async settleDeliveryLine(pane, cwd, initialTranscript, inputBox, accept, settleWindowMs) {
|
|
433
506
|
let transcript = initialTranscript;
|
|
434
|
-
|
|
507
|
+
const readAttempts = settleWindowMs === undefined
|
|
508
|
+
? DELIVERY_SETTLE_READ_ATTEMPTS
|
|
509
|
+
: Math.max(DELIVERY_SETTLE_READ_ATTEMPTS, Math.floor(settleWindowMs / DELIVERY_SETTLE_POLL_MS) + 1);
|
|
510
|
+
for (let readAttempt = 0; readAttempt < readAttempts; readAttempt++) {
|
|
435
511
|
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
|
|
436
512
|
if (read.code !== 0)
|
|
437
|
-
return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false };
|
|
438
|
-
if (read.stdout
|
|
513
|
+
return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false, readFailed: true };
|
|
514
|
+
if (accept?.(read.stdout)) {
|
|
515
|
+
return {
|
|
516
|
+
ok: true,
|
|
517
|
+
transcript: read.stdout,
|
|
518
|
+
recognizedInputBox: inputBox !== undefined && matchesInputBox(read.stdout, inputBox),
|
|
519
|
+
};
|
|
520
|
+
}
|
|
521
|
+
if (accept === undefined && read.stdout === transcript) {
|
|
439
522
|
return {
|
|
440
523
|
ok: true,
|
|
441
524
|
transcript,
|
|
@@ -443,12 +526,58 @@ export class HerdrDriver {
|
|
|
443
526
|
};
|
|
444
527
|
}
|
|
445
528
|
transcript = read.stdout;
|
|
446
|
-
if (readAttempt <
|
|
529
|
+
if (readAttempt < readAttempts - 1) {
|
|
447
530
|
await new Promise((resolve) => setTimeout(resolve, DELIVERY_SETTLE_POLL_MS));
|
|
448
531
|
}
|
|
449
532
|
}
|
|
450
533
|
return { ok: false, transcript, recognizedInputBox: false };
|
|
451
534
|
}
|
|
535
|
+
async awaitDeliveryReadiness(slot, cmd, pane) {
|
|
536
|
+
const inputBox = this.inputBoxes.get(slot);
|
|
537
|
+
const requireInputBox = inputBox !== undefined && inputBox.launchCommand?.(cmd) !== true;
|
|
538
|
+
const timeoutMs = inputBox?.readinessTimeoutMs ?? DELIVERY_READINESS_TIMEOUT_MS;
|
|
539
|
+
const started = Date.now();
|
|
540
|
+
let previous;
|
|
541
|
+
let transcript = "";
|
|
542
|
+
let reads = 0;
|
|
543
|
+
while (true) {
|
|
544
|
+
const elapsedBeforeRead = Date.now() - started;
|
|
545
|
+
const remainingBeforeRead = timeoutMs - elapsedBeforeRead;
|
|
546
|
+
if (remainingBeforeRead <= 0)
|
|
547
|
+
throw new DeliveryReadinessError(elapsedBeforeRead, transcript);
|
|
548
|
+
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, slot.cwd, Math.max(1, Math.ceil(remainingBeforeRead)));
|
|
549
|
+
reads++;
|
|
550
|
+
transcript = read.stdout || transcript;
|
|
551
|
+
const waitedMs = Date.now() - started;
|
|
552
|
+
if (read.code !== 0) {
|
|
553
|
+
// Once at least one valid frame has been observed, a read that consumes the remainder of
|
|
554
|
+
// the readiness budget is the bounded window expiring, not a new submission/protocol
|
|
555
|
+
// identity. A first-read timeout and every non-timeout read failure remain structural.
|
|
556
|
+
if (read.timedOut && previous !== undefined && waitedMs >= timeoutMs) {
|
|
557
|
+
throw new DeliveryReadinessError(waitedMs, transcript);
|
|
558
|
+
}
|
|
559
|
+
const detail = read.stderr || read.stdout || `exit ${read.code}`;
|
|
560
|
+
throw new Error(`herdr pane read failed during readiness${read.timedOut ? " (timed out)" : ""}: ${detail}`);
|
|
561
|
+
}
|
|
562
|
+
if (previous !== undefined) {
|
|
563
|
+
const stableFrame = read.stdout === previous;
|
|
564
|
+
const targetReady = stableFrame && (requireInputBox
|
|
565
|
+
? matchesInputBox(previous, inputBox) && matchesInputBox(read.stdout, inputBox)
|
|
566
|
+
: !this.deliveryMatches(read.stdout, cmd));
|
|
567
|
+
if (targetReady)
|
|
568
|
+
return { waitedMs, timeoutMs, transcript: read.stdout };
|
|
569
|
+
}
|
|
570
|
+
previous = read.stdout;
|
|
571
|
+
const remaining = timeoutMs - waitedMs;
|
|
572
|
+
if (remaining <= 0)
|
|
573
|
+
throw new DeliveryReadinessError(waitedMs, transcript);
|
|
574
|
+
// The second read is immediate: an already-painted stable target proves readiness with no
|
|
575
|
+
// added delay. Only observed change spends a poll interval, and every path remains bounded.
|
|
576
|
+
if (reads > 1) {
|
|
577
|
+
await new Promise((resolve) => setTimeout(resolve, Math.min(DELIVERY_SETTLE_POLL_MS, remaining)));
|
|
578
|
+
}
|
|
579
|
+
}
|
|
580
|
+
}
|
|
452
581
|
async deliveryReadMatches(pane, cmd, cwd) {
|
|
453
582
|
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
|
|
454
583
|
return read.code === 0 && this.deliveryMatches(read.stdout, cmd);
|
|
@@ -638,7 +767,7 @@ export class HerdrDriver {
|
|
|
638
767
|
const s = await this.watchSlot(cwd, name);
|
|
639
768
|
this.watches.set(name, s);
|
|
640
769
|
try {
|
|
641
|
-
await this.
|
|
770
|
+
await this.deliverPersistentShellCommand(s, command);
|
|
642
771
|
}
|
|
643
772
|
catch (err) {
|
|
644
773
|
this.watches.delete(name);
|
package/dist/run/daemon.js
CHANGED
|
@@ -9,6 +9,7 @@ import { allAdapters, discoverChannels, getAdapter, probeAll, readDoctor } from
|
|
|
9
9
|
import { addUsage, channelKey, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
10
10
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
11
11
|
import { globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
|
|
12
|
+
import { DeliveryReadinessError } from "../drivers/herdr.js";
|
|
12
13
|
import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
|
|
13
14
|
import { formatOwnedName } from "../drivers/types.js";
|
|
14
15
|
import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
|
|
@@ -24,7 +25,7 @@ import { acquireRunLock, releaseRunLock } from "./lock.js";
|
|
|
24
25
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
25
26
|
import { nextChannel, route } from "../route/router.js";
|
|
26
27
|
import { desiredPanes } from "./reconcile.js";
|
|
27
|
-
import {
|
|
28
|
+
import { StallProgressTracker } from "./stall.js";
|
|
28
29
|
const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
|
|
29
30
|
// An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
|
|
30
31
|
// carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
|
|
@@ -795,13 +796,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
795
796
|
// single site, so a test can reason about it; keep Date.now() out of profile.ts (still pure) and
|
|
796
797
|
// out of adapter module scope (the cursor is a parameter, threaded from the daemon).
|
|
797
798
|
const attemptStart = Date.now();
|
|
798
|
-
// v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing
|
|
799
|
-
//
|
|
799
|
+
// v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing worker
|
|
800
|
+
// wait slices — never a new timer loop. null/unknown usage fails OPEN
|
|
800
801
|
// (never treated as over-threshold). Journal + notify fire at most once while the value stays high.
|
|
801
802
|
let contextWarned = false;
|
|
802
803
|
let contextTokens;
|
|
803
804
|
const sampleContext = async () => {
|
|
804
|
-
if (
|
|
805
|
+
if (!adapter.contextUsage)
|
|
805
806
|
return;
|
|
806
807
|
let usage = null;
|
|
807
808
|
try {
|
|
@@ -814,7 +815,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
814
815
|
if (!usage || typeof usage.tokens !== "number" || !Number.isFinite(usage.tokens))
|
|
815
816
|
return;
|
|
816
817
|
contextTokens = usage.tokens; // last known valid sample, including under-threshold resume candidates
|
|
817
|
-
if (usage.tokens < cfg.contextWarnTokens)
|
|
818
|
+
if (contextWarned || usage.tokens < cfg.contextWarnTokens)
|
|
818
819
|
return;
|
|
819
820
|
contextWarned = true;
|
|
820
821
|
lastContextTokens = usage.tokens;
|
|
@@ -833,6 +834,38 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
833
834
|
let earlyLaunchDead = false;
|
|
834
835
|
let settleParsed;
|
|
835
836
|
let seedResult;
|
|
837
|
+
const handleDeliveryReadiness = async (error) => {
|
|
838
|
+
journal.append("delivery-readiness-failed", t.id, {
|
|
839
|
+
attempt,
|
|
840
|
+
waitedMs: error.waitedMs,
|
|
841
|
+
transcript: error.transcript,
|
|
842
|
+
});
|
|
843
|
+
if (keepOpen)
|
|
844
|
+
keptSlots.push(slot);
|
|
845
|
+
else
|
|
846
|
+
await closeSlot(slot);
|
|
847
|
+
feedback = `delivery readiness failed after ${error.waitedMs}ms; pane transcript:\n${error.transcript}`;
|
|
848
|
+
const step = r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
|
|
849
|
+
journal.append("escalation", t.id, { step, attempt: attempt + 1 });
|
|
850
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
|
|
851
|
+
if (step === "retry")
|
|
852
|
+
return true;
|
|
853
|
+
if (step === "escalate") {
|
|
854
|
+
const next = failover("escalate");
|
|
855
|
+
if (next) {
|
|
856
|
+
assignment = next;
|
|
857
|
+
tried.push(channelKey(next));
|
|
858
|
+
return true;
|
|
859
|
+
}
|
|
860
|
+
// no channel left — fall through to a consult
|
|
861
|
+
}
|
|
862
|
+
if (step === "escalate" || step === "consult") {
|
|
863
|
+
const v = await runConsult("delivery-readiness", error.transcript, feedback, []);
|
|
864
|
+
return applyVerdict(v, attempt + 1, "dispatch");
|
|
865
|
+
}
|
|
866
|
+
await park(t, "escalation ladder exhausted", "ladder-exhausted", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
867
|
+
return false;
|
|
868
|
+
};
|
|
836
869
|
if (interactive) {
|
|
837
870
|
// v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
|
|
838
871
|
// The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
|
|
@@ -842,11 +875,29 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
842
875
|
// v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
|
|
843
876
|
// then fall through to the normal trailer harvest. A failed seed is recorded as a finished
|
|
844
877
|
// failure rather than allowed to race the trailer wait.
|
|
845
|
-
|
|
878
|
+
try {
|
|
879
|
+
seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
|
|
880
|
+
}
|
|
881
|
+
catch (error) {
|
|
882
|
+
if (!(error instanceof DeliveryReadinessError))
|
|
883
|
+
throw error;
|
|
884
|
+
if (await handleDeliveryReadiness(error))
|
|
885
|
+
continue attempts;
|
|
886
|
+
return;
|
|
887
|
+
}
|
|
846
888
|
output = seedResult.output;
|
|
847
889
|
}
|
|
848
890
|
else {
|
|
849
|
-
|
|
891
|
+
try {
|
|
892
|
+
await driver.run(slot, paneDispatchCommand(dispatchScript));
|
|
893
|
+
}
|
|
894
|
+
catch (error) {
|
|
895
|
+
if (!(error instanceof DeliveryReadinessError))
|
|
896
|
+
throw error;
|
|
897
|
+
if (await handleDeliveryReadiness(error))
|
|
898
|
+
continue attempts;
|
|
899
|
+
return;
|
|
900
|
+
}
|
|
850
901
|
output = await driver.read(slot, 1000);
|
|
851
902
|
}
|
|
852
903
|
if (seedResult?.seedFailed) {
|
|
@@ -862,15 +913,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
862
913
|
// OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
|
|
863
914
|
// stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
|
|
864
915
|
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
865
|
-
//
|
|
866
|
-
//
|
|
867
|
-
// trailer detection, harvest, paging, and quota checks all read the raw pane.
|
|
916
|
+
// v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
|
|
917
|
+
// the stall clock. Raw pane differences are terminal chrome until proven otherwise.
|
|
868
918
|
let everHadOutput = output.length > 0;
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
919
|
+
const stallProgress = new StallProgressTracker();
|
|
920
|
+
stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
|
|
921
|
+
let lastProgressAt = Date.now();
|
|
922
|
+
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
872
923
|
const sliceStart = Date.now();
|
|
873
|
-
const remaining = stallWindowMs - (sliceStart -
|
|
924
|
+
const remaining = stallWindowMs - (sliceStart - lastProgressAt);
|
|
874
925
|
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
875
926
|
if (!everHadOutput) {
|
|
876
927
|
const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
|
|
@@ -893,11 +944,6 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
893
944
|
const paneText = await driver.read(slot, 1000);
|
|
894
945
|
if (paneText.length > 0)
|
|
895
946
|
everHadOutput = true;
|
|
896
|
-
const currentStallSnapshot = normalizeStallSnapshot(paneText);
|
|
897
|
-
if (currentStallSnapshot !== lastStallSnapshot) {
|
|
898
|
-
lastStallSnapshot = currentStallSnapshot;
|
|
899
|
-
lastOutputAt = Date.now();
|
|
900
|
-
}
|
|
901
947
|
// OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
|
|
902
948
|
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
903
949
|
earlyLaunchDead = true;
|
|
@@ -906,6 +952,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
906
952
|
}
|
|
907
953
|
// v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
|
|
908
954
|
await sampleContext();
|
|
955
|
+
if (stallProgress.observe({ paneText, contextTokens }))
|
|
956
|
+
lastProgressAt = Date.now();
|
|
909
957
|
// page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
|
|
910
958
|
// (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
|
|
911
959
|
// a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
|
|
@@ -943,7 +991,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
943
991
|
}
|
|
944
992
|
if (!finished && exitCode === null) {
|
|
945
993
|
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
946
|
-
timedOut = Date.now() -
|
|
994
|
+
timedOut = Date.now() - lastProgressAt >= stallWindowMs;
|
|
947
995
|
output = await driver.read(slot, 1000);
|
|
948
996
|
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
949
997
|
const exit = exitRe.exec(output);
|
|
@@ -979,18 +1027,27 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
979
1027
|
}
|
|
980
1028
|
}
|
|
981
1029
|
else {
|
|
982
|
-
|
|
1030
|
+
try {
|
|
1031
|
+
await driver.run(slot, paneDispatchCommand(dispatchScript));
|
|
1032
|
+
}
|
|
1033
|
+
catch (error) {
|
|
1034
|
+
if (!(error instanceof DeliveryReadinessError))
|
|
1035
|
+
throw error;
|
|
1036
|
+
if (await handleDeliveryReadiness(error))
|
|
1037
|
+
continue attempts;
|
|
1038
|
+
return;
|
|
1039
|
+
}
|
|
983
1040
|
// OBS-54: headless workers have the same output-inactivity budget as visible panes.
|
|
984
|
-
//
|
|
985
|
-
// exhaust the budget here too; harvest below still reads the raw pane.
|
|
1041
|
+
// v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
|
|
986
1042
|
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
987
1043
|
const initialPane = await driver.read(slot, 500);
|
|
988
1044
|
let everHadOutput = initialPane.length > 0;
|
|
989
|
-
|
|
990
|
-
|
|
1045
|
+
const stallProgress = new StallProgressTracker();
|
|
1046
|
+
stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
|
|
1047
|
+
let lastProgressAt = Date.now();
|
|
991
1048
|
finished = false;
|
|
992
|
-
while (Date.now() -
|
|
993
|
-
const remaining = stallWindowMs - (Date.now() -
|
|
1049
|
+
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
1050
|
+
const remaining = stallWindowMs - (Date.now() - lastProgressAt);
|
|
994
1051
|
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
995
1052
|
if (!everHadOutput) {
|
|
996
1053
|
const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
|
|
@@ -1004,19 +1061,17 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1004
1061
|
const paneText = await driver.read(slot, 500);
|
|
1005
1062
|
if (paneText.length > 0)
|
|
1006
1063
|
everHadOutput = true;
|
|
1007
|
-
const currentStallSnapshot = normalizeStallSnapshot(paneText);
|
|
1008
|
-
if (currentStallSnapshot !== lastStallSnapshot) {
|
|
1009
|
-
lastStallSnapshot = currentStallSnapshot;
|
|
1010
|
-
lastOutputAt = Date.now();
|
|
1011
|
-
}
|
|
1012
1064
|
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
1013
1065
|
earlyLaunchDead = true;
|
|
1014
1066
|
break;
|
|
1015
1067
|
}
|
|
1068
|
+
await sampleContext();
|
|
1069
|
+
if (stallProgress.observe({ paneText, contextTokens }))
|
|
1070
|
+
lastProgressAt = Date.now();
|
|
1016
1071
|
}
|
|
1017
1072
|
output = await driver.read(slot, 500);
|
|
1018
1073
|
exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
|
|
1019
|
-
timedOut = !finished && Date.now() -
|
|
1074
|
+
timedOut = !finished && Date.now() - lastProgressAt >= stallWindowMs;
|
|
1020
1075
|
}
|
|
1021
1076
|
// SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
|
|
1022
1077
|
// shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
|
package/dist/run/stall.d.ts
CHANGED
|
@@ -1,8 +1,25 @@
|
|
|
1
|
-
/** Normalize
|
|
2
|
-
* waitOutput, and paging read the raw text
|
|
3
|
-
* CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
|
|
4
|
-
* equal are the same frame modulo spinner presentation; any other byte difference is activity. */
|
|
1
|
+
/** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
|
|
2
|
+
* parsing, harvest, waitOutput, and paging always read the raw text. */
|
|
5
3
|
export declare function normalizeStallSnapshot(text: string): string;
|
|
4
|
+
export interface StallProgressSample {
|
|
5
|
+
paneText: string;
|
|
6
|
+
seedSubmitted?: boolean;
|
|
7
|
+
contextTokens?: number;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Monotonic worker-progress measure for the stall watchdog.
|
|
11
|
+
*
|
|
12
|
+
* Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
|
|
13
|
+
* evidence of work. A rendered transcript is only known to have grown when it occupies more
|
|
14
|
+
* non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
|
|
15
|
+
* advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
|
|
16
|
+
*/
|
|
17
|
+
export declare class StallProgressTracker {
|
|
18
|
+
private transcriptRows;
|
|
19
|
+
private seedSubmitted;
|
|
20
|
+
private contextTokens;
|
|
21
|
+
observe(sample: StallProgressSample): boolean;
|
|
22
|
+
}
|
|
6
23
|
/** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
|
|
7
24
|
* seam exists for fault injection in tests only — production callers pass text alone. */
|
|
8
25
|
export declare function filterLlmTranscript(text: string, classify?: (t: string) => string): string;
|
package/dist/run/stall.js
CHANGED
|
@@ -1,12 +1,9 @@
|
|
|
1
|
-
// OBS-82:
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
// design: an allowlist MISS degrades to today's recoverable no-reap behavior, while an over-broad
|
|
8
|
-
// deletion would reap a healthy worker — a new failure class. Grow the allowlist only with
|
|
9
|
-
// captured evidence (tests/fixtures/codex-mcp-spinner/).
|
|
1
|
+
// OBS-82: normalize known presentation tokens before measuring transcript extent or filtering an
|
|
2
|
+
// LLM-bound transcript. This remains a closed allowlist — ANSI/VT escapes, braille-range spinner
|
|
3
|
+
// glyphs, and elapsed-time tokens bound to time-unit suffixes. Every other byte passes through
|
|
4
|
+
// identical. v1.76 deliberately stopped treating arbitrary normalized byte changes as progress:
|
|
5
|
+
// StallProgressTracker below requires monotonic evidence, so an unknown repaint fails closed toward
|
|
6
|
+
// a recoverable consult instead of holding the watchdog silent.
|
|
10
7
|
// CSI (with intermediates), OSC (BEL- or ST-terminated), DCS/SOS/PM/APC strings, single-char
|
|
11
8
|
// escapes, and charset selection — the raw-pty forms; herdr pane reads are already rendered.
|
|
12
9
|
// eslint-disable-next-line no-control-regex
|
|
@@ -16,13 +13,46 @@ const SPINNER_RE = /[⠀-⣿]/g;
|
|
|
16
13
|
// A digit run (optionally decimal) bound directly to a time-unit suffix, standing alone as a
|
|
17
14
|
// word: 9s, 41s, 3m, 1h, 800ms. Never bare digits — "(6/7)" and "5 of 7" stay change-sensitive.
|
|
18
15
|
const ELAPSED_RE = /(?<![\w.])\d+(?:\.\d+)?(?:ms|[hms])(?!\w)/g;
|
|
19
|
-
/** Normalize
|
|
20
|
-
* waitOutput, and paging read the raw text
|
|
21
|
-
* CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
|
|
22
|
-
* equal are the same frame modulo spinner presentation; any other byte difference is activity. */
|
|
16
|
+
/** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
|
|
17
|
+
* parsing, harvest, waitOutput, and paging always read the raw text. */
|
|
23
18
|
export function normalizeStallSnapshot(text) {
|
|
24
19
|
return text.replace(ANSI_RE, "").replace(SPINNER_RE, "").replace(ELAPSED_RE, "");
|
|
25
20
|
}
|
|
21
|
+
/**
|
|
22
|
+
* Monotonic worker-progress measure for the stall watchdog.
|
|
23
|
+
*
|
|
24
|
+
* Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
|
|
25
|
+
* evidence of work. A rendered transcript is only known to have grown when it occupies more
|
|
26
|
+
* non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
|
|
27
|
+
* advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
|
|
28
|
+
*/
|
|
29
|
+
export class StallProgressTracker {
|
|
30
|
+
transcriptRows = 0;
|
|
31
|
+
seedSubmitted = false;
|
|
32
|
+
contextTokens;
|
|
33
|
+
observe(sample) {
|
|
34
|
+
let advanced = false;
|
|
35
|
+
const rows = normalizeStallSnapshot(sample.paneText)
|
|
36
|
+
.split("\n")
|
|
37
|
+
.filter((line) => line.trim().length > 0)
|
|
38
|
+
.length;
|
|
39
|
+
if (rows > this.transcriptRows) {
|
|
40
|
+
this.transcriptRows = rows;
|
|
41
|
+
advanced = true;
|
|
42
|
+
}
|
|
43
|
+
if (sample.seedSubmitted && !this.seedSubmitted) {
|
|
44
|
+
this.seedSubmitted = true;
|
|
45
|
+
advanced = true;
|
|
46
|
+
}
|
|
47
|
+
const tokens = sample.contextTokens;
|
|
48
|
+
if (tokens !== undefined && Number.isFinite(tokens)) {
|
|
49
|
+
if (tokens > (this.contextTokens ?? 0))
|
|
50
|
+
advanced = true;
|
|
51
|
+
this.contextTokens = Math.max(this.contextTokens ?? 0, tokens);
|
|
52
|
+
}
|
|
53
|
+
return advanced;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
26
56
|
// ─── v1.65 T2: LLM-bound transcript filter ──────────────────────────────────────────────────────
|
|
27
57
|
// Consult dossiers and gate prompts pay tokens per transcript byte, so LLM-bound text runs through
|
|
28
58
|
// a per-line classifier: carriage-return overwrite churn keeps only the final paint, lines that are
|