tickmarkr 1.75.0 → 1.76.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,6 +3,11 @@ import { type Assignment, type WorkerAdapter, type WorkerResult } from "./types.
3
3
  export declare function kimiAuthed(credentialsText: string, nowMs: number): boolean;
4
4
  export declare function parseKimiModels(raw: string): string[];
5
5
  export declare function parseKimiResult(raw: string, nonce: string): WorkerResult;
6
+ export interface KimiDoctorTurnResult {
7
+ ok: boolean;
8
+ evidence: string;
9
+ }
10
+ export declare function probeKimiDoctorTurn(cwd: string): Promise<KimiDoctorTurnResult>;
6
11
  export declare function kimiSessionId(output: string): string | undefined;
7
12
  export declare function kimiBannerModel(banner: string): string | undefined;
8
13
  export declare function kimiBannerSessionId(banner: string): string | undefined;
@@ -64,6 +64,28 @@ export function parseKimiResult(raw, nonce) {
64
64
  const stripped = raw.split("\n").map((l) => l.replace(/^[\s]*[•*-]\s+/, "")).join("\n");
65
65
  return parseWorkerResult(stripped, nonce);
66
66
  }
67
+ const KIMI_DOCTOR_TURN_MODEL = "kimi-code/k3";
68
+ const KIMI_DOCTOR_TURN_PROMPT = "Reply with exactly OK and nothing else.";
69
+ const KIMI_DOCTOR_TURN_TIMEOUT_MS = 60000;
70
+ // OBS-141: intentionally separate from probe() so plan/run remain free file checks. Only doctor
71
+ // calls this one-turn contract probe; its test seam stubs sh and never launches a real agent CLI.
72
+ export async function probeKimiDoctorTurn(cwd) {
73
+ const command = `kimi -p ${shq(KIMI_DOCTOR_TURN_PROMPT)} --model ${shq(KIMI_DOCTOR_TURN_MODEL)} --output-format text`;
74
+ const r = await sh(command, cwd, KIMI_DOCTOR_TURN_TIMEOUT_MS);
75
+ if (r.timedOut) {
76
+ return { ok: false, evidence: `turn timed out after ${KIMI_DOCTOR_TURN_TIMEOUT_MS}ms` };
77
+ }
78
+ const output = `${r.stderr}\n${r.stdout}`.trim().replace(/\s+/g, " ");
79
+ if (r.code !== 0) {
80
+ return { ok: false, evidence: output || `turn exited ${r.code}` };
81
+ }
82
+ const returnedOk = r.stdout.split("\n")
83
+ .map((line) => line.replace(/^[\s]*[•*-]\s+/, "").trim())
84
+ .includes("OK");
85
+ return returnedOk
86
+ ? { ok: true, evidence: `model turn returned OK with ${KIMI_DOCTOR_TURN_MODEL}` }
87
+ : { ok: false, evidence: `turn returned no exact OK answer${output ? `: ${output}` : ""}` };
88
+ }
67
89
  // v1.53 T3: session-id capture from the run-output trailer — every `kimi -p` run (fresh or resumed)
68
90
  // ends with `To resume this session: kimi -r session_<uuid>` (live probe 2026-07-18). Anchored full
69
91
  // line only: prompt/model prose can contain lookalike text, and the anchored charset keeps a
@@ -89,23 +111,20 @@ export function kimiSessionId(output) {
89
111
  }
90
112
  return id;
91
113
  }
92
- // v1.69 T6: the native TUI takes -m <alias>, where config.toml aliases are the bare model suffix of
93
- // the tickmarkr channel id (live probe 2026-07-22). Keep the mapping explicit and localized.
94
- function kimiAlias(model) {
95
- return model.replace(/^kimi-code\//, "");
96
- }
97
114
  // v1.69 T7: the cold-start banner prints the model alias and session id. Parse them from the
98
- // banner text already captured for the readiness match — no new probe, no extra dispatch.
115
+ // banner text already captured for the readiness match — no new probe, no extra dispatch. Kimi
116
+ // 0.29.0 may print either the full config key or its display suffix; normalize both idempotently
117
+ // to the full channel identifier routing uses.
99
118
  const BANNER_MODEL_RE = /^Model:\s*(.+)$/m;
100
119
  const BANNER_SESSION_RE = /^Session:\s*(session_[0-9a-f-]+)$/m;
101
120
  export function kimiBannerModel(banner) {
102
121
  const m = BANNER_MODEL_RE.exec(banner);
103
122
  if (!m)
104
123
  return undefined;
105
- const alias = m[1].trim();
106
- if (!alias)
124
+ const printedModel = m[1].trim();
125
+ if (!printedModel)
107
126
  return undefined;
108
- return `kimi-code/${alias}`;
127
+ return printedModel.startsWith("kimi-code/") ? printedModel : `kimi-code/${printedModel}`;
109
128
  }
110
129
  export function kimiBannerSessionId(banner) {
111
130
  return BANNER_SESSION_RE.exec(banner)?.[1];
@@ -126,7 +145,8 @@ export const KIMI_INPUT_BOX = declareInputBox("kimi", {
126
145
  // Shared launch-then-seed surface (T6) + banner confirm (T7/T2). One definition so the adapter
127
146
  // property and the daemon's generic runInteractiveSeed path cannot drift.
128
147
  const KIMI_SEED = {
129
- launch: (model) => `kimi -y -m ${shq(kimiAlias(model))}`,
148
+ // v1.76 T2 / OBS-141: 0.29.0 resolves only the full config.toml model key on the first turn.
149
+ launch: (model) => `kimi -y -m ${shq(model)}`,
130
150
  readinessMatch: "Send /help for help information.",
131
151
  seedLine: (promptFile) => `Read ${promptFile} and do exactly what it says.`,
132
152
  confirmBanner: confirmKimiSeedBanner,
@@ -1,5 +1,7 @@
1
1
  import type { WorkerAdapter } from "../../adapters/types.js";
2
+ import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
2
3
  export type DoctorOpts = {
3
4
  banner?: boolean;
5
+ kimiTurnProbe?: (cwd: string) => Promise<KimiDoctorTurnResult>;
4
6
  };
5
7
  export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
@@ -6,6 +6,7 @@ import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
6
6
  import { declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
7
7
  import { DEFAULT_CONFIG, loadConfig, overlayPreferShapes } from "../../config/config.js";
8
8
  import { HerdrDriver } from "../../drivers/herdr.js";
9
+ import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
9
10
  import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
10
11
  const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
11
12
  const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
@@ -20,6 +21,26 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
20
21
  console.error("probing installed agent CLIs — one short LLM call per configured model, may take a minute...");
21
22
  const probeProgressTTY = process.stderr.isTTY === true;
22
23
  const health = await probeAll(adapters, { cwd });
24
+ const kimiAdapter = adapters.find((a) => a.id === kimi.id);
25
+ const kimiTurnEnabled = kimiAdapter !== undefined
26
+ && (kimiAdapter === kimi || opts.kimiTurnProbe !== undefined);
27
+ if (kimiTurnEnabled) {
28
+ const h = health.kimi;
29
+ if (h.installed && h.authed) {
30
+ let turn;
31
+ try {
32
+ turn = await (opts.kimiTurnProbe ?? probeKimiDoctorTurn)(cwd);
33
+ }
34
+ catch (e) {
35
+ turn = { ok: false, evidence: e instanceof Error ? e.message : String(e) };
36
+ }
37
+ health.kimi = {
38
+ ...h,
39
+ authed: turn.ok,
40
+ note: `${h.note ? `${h.note}; ` : ""}${turn.ok ? turn.evidence : `model turn failed: ${turn.evidence}`}`,
41
+ };
42
+ }
43
+ }
23
44
  // MODEL-02: detect models where the adapter exposes a list surface, BEFORE writing doctor.json (write once, below).
24
45
  // Fail OPEN — the inverse of gates' fail-closed: detection is advisory, so a broken list surface NEVER fails doctor.
25
46
  for (const a of adapters) {
@@ -35,14 +56,20 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
35
56
  }
36
57
  catch { /* fail open: leave models as-is, doctor stays healthy */ }
37
58
  }
38
- await probeModels(cfg, cwd, adapters, health, probeProgressTTY
59
+ // A free Kimi auth failure or failed earned-green turn must not spend more probes. Every other
60
+ // adapter keeps the exact existing model-sweep path.
61
+ const modelProbeAdapters = kimiTurnEnabled && health.kimi.authed === false
62
+ ? adapters.filter((a) => a !== kimiAdapter)
63
+ : adapters;
64
+ await probeModels(cfg, cwd, modelProbeAdapters, health, probeProgressTTY
39
65
  ? (adapter, model, status, durationMs) => console.error(` ${adapter}:${model} ${status} (${(durationMs / 1000).toFixed(1)}s)`)
40
66
  : undefined);
41
67
  writeDoctor(cwd, health);
42
68
  const rows = adapters.map((a) => {
43
69
  const h = health[a.id];
44
70
  const state = !h.installed ? "not installed" : `${h.version ?? "installed"}${h.note ? ` (${h.note})` : ""}`;
45
- return alignedStatusRow(h.installed ? "pass" : "fail", a.id, state);
71
+ const healthy = h.installed && (a.id !== kimi.id || h.authed);
72
+ return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
46
73
  });
47
74
  // v1.48 T1: advisory sweep for known agent CLIs with no adapter — never written to doctor.json health.
48
75
  rows.push(...detectCandidateClis().map(({ binary, version }) => alignedStatusRow("warn", binary, `detected: ${version ?? "version unknown"} (no tickmarkr adapter — not routable)`)));
@@ -24,6 +24,7 @@ export declare class HerdrDriver implements ExecutorDriver {
24
24
  private reserveDispatch;
25
25
  private verifyPaneIdentityBinding;
26
26
  private deliveryMatches;
27
+ private submissionRegistered;
27
28
  static available(): boolean;
28
29
  private herdr;
29
30
  private namedPaneId;
@@ -40,6 +41,7 @@ export declare class HerdrDriver implements ExecutorDriver {
40
41
  private joinGroup;
41
42
  run(slot: Slot, cmd: string): Promise<void>;
42
43
  private deliver;
44
+ private submitVerifiedDelivery;
43
45
  private settleDeliveryLine;
44
46
  private deliveryReadMatches;
45
47
  private waitOk;
@@ -8,6 +8,7 @@ export const TRAILER_SAFE_FLOOR_COLS = 108;
8
8
  export const TRAILER_WIDTH_MARGIN = 2; // cols below (floor + margin) refuse a rightward first split
9
9
  // OBS-85 verified delivery: bounded type→read-back→enter attempts before failing closed.
10
10
  export const DELIVERY_ATTEMPTS = 3;
11
+ const DELIVERY_SUBMIT_ATTEMPTS = 2; // initial Enter + one evidence-backed re-press (OBS-140)
11
12
  const DELIVERY_VERIFY_TIMEOUT_MS = 2000; // per attempt — a paste that hasn't rendered in 2s is retyped
12
13
  const DELIVERY_READ_LINES = 80;
13
14
  const DELIVERY_SETTLE_READ_ATTEMPTS = 6;
@@ -95,6 +96,22 @@ export class HerdrDriver {
95
96
  const needle = norm(cmd);
96
97
  return needle.length > 0 && hay.includes(needle);
97
98
  }
99
+ // Submission succeeds when the typed prompt disappears, or when it has moved above a fresh
100
+ // adapter-declared input box (the prompt is now transcript, not input). Shell-line delivery uses
101
+ // the same normalized seam: a prompt still at the bottom ends the pane text; output/a fresh prompt
102
+ // after it proves Enter registered. No adapter-specific fingerprint lives in the driver.
103
+ submissionRegistered(transcript, cmd, inputBox) {
104
+ const norm = (s) => s.replace(/\s+/g, "");
105
+ const hay = norm(transcript);
106
+ const needle = norm(cmd);
107
+ const promptAt = hay.lastIndexOf(needle);
108
+ if (needle.length === 0 || promptAt < 0)
109
+ return true;
110
+ if (inputBox && matchesInputBox(transcript, inputBox)) {
111
+ return hay.lastIndexOf(norm(inputBox.fingerprint)) > promptAt;
112
+ }
113
+ return promptAt + needle.length < hay.length;
114
+ }
98
115
  static available() {
99
116
  return process.env.HERDR_ENV === "1";
100
117
  }
@@ -419,9 +436,7 @@ export class HerdrDriver {
419
436
  throw new Error(`herdr pane send-text failed: ${typed.stderr || typed.stdout}`);
420
437
  const back = await this.herdr(`pane wait-output ${shq(pane)} --match ${shq(cmd)} --timeout ${DELIVERY_VERIFY_TIMEOUT_MS}`, slot.cwd, DELIVERY_VERIFY_TIMEOUT_MS + 15_000);
421
438
  if (this.waitOk(back.code, back.stdout) || await this.deliveryReadMatches(pane, cmd, slot.cwd)) {
422
- const enter = await this.herdr(`pane send-keys ${shq(pane)} Enter`, slot.cwd);
423
- if (enter.code !== 0)
424
- throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
439
+ await this.submitVerifiedDelivery(slot, cmd, pane);
425
440
  return;
426
441
  }
427
442
  // capture the corrupted delivery BEFORE clearing it — the OBS-85 byte-level evidence
@@ -429,13 +444,40 @@ export class HerdrDriver {
429
444
  }
430
445
  throw new Error(`herdr delivery corrupted after ${DELIVERY_ATTEMPTS} attempts — enter never pressed (OBS-85); pane transcript:\n${transcript}`);
431
446
  }
432
- async settleDeliveryLine(pane, cwd, initialTranscript, inputBox) {
447
+ async submitVerifiedDelivery(slot, cmd, pane) {
448
+ let transcript = "";
449
+ const inputBox = this.inputBoxes.get(slot);
450
+ for (let attempt = 0; attempt < DELIVERY_SUBMIT_ATTEMPTS; attempt++) {
451
+ const enter = await this.herdr(`pane send-keys ${shq(pane)} Enter`, slot.cwd);
452
+ if (enter.code !== 0)
453
+ throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
454
+ // Reuse the existing settle-read window. A first-read success returns before any timer; only
455
+ // a prompt that still occupies the delivery target spends the bounded settle window. This
456
+ // verification always completes before a possible re-press, so a slow submit cannot duplicate.
457
+ const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript, inputBox, (candidate) => this.submissionRegistered(candidate, cmd, inputBox));
458
+ transcript = settled.transcript;
459
+ if (settled.ok)
460
+ return;
461
+ if (settled.readFailed) {
462
+ throw new Error(`herdr delivery corrupted — submission verification failed, refusing to re-press Enter (OBS-140); pane transcript:\n${transcript}`);
463
+ }
464
+ }
465
+ throw new Error(`herdr delivery corrupted after ${DELIVERY_SUBMIT_ATTEMPTS} submit attempts — submission never registered (OBS-140); pane transcript:\n${transcript}`);
466
+ }
467
+ async settleDeliveryLine(pane, cwd, initialTranscript, inputBox, accept) {
433
468
  let transcript = initialTranscript;
434
469
  for (let readAttempt = 0; readAttempt < DELIVERY_SETTLE_READ_ATTEMPTS; readAttempt++) {
435
470
  const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
436
471
  if (read.code !== 0)
437
- return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false };
438
- if (read.stdout === transcript) {
472
+ return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false, readFailed: true };
473
+ if (accept?.(read.stdout)) {
474
+ return {
475
+ ok: true,
476
+ transcript: read.stdout,
477
+ recognizedInputBox: inputBox !== undefined && matchesInputBox(read.stdout, inputBox),
478
+ };
479
+ }
480
+ if (accept === undefined && read.stdout === transcript) {
439
481
  return {
440
482
  ok: true,
441
483
  transcript,
@@ -24,7 +24,7 @@ import { acquireRunLock, releaseRunLock } from "./lock.js";
24
24
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
25
25
  import { nextChannel, route } from "../route/router.js";
26
26
  import { desiredPanes } from "./reconcile.js";
27
- import { normalizeStallSnapshot } from "./stall.js";
27
+ import { StallProgressTracker } from "./stall.js";
28
28
  const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
29
29
  // An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
30
30
  // carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
@@ -795,13 +795,13 @@ export async function runDaemon(repoRoot, opts = {}) {
795
795
  // single site, so a test can reason about it; keep Date.now() out of profile.ts (still pure) and
796
796
  // out of adapter module scope (the cursor is a parameter, threaded from the daemon).
797
797
  const attemptStart = Date.now();
798
- // v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing poll
799
- // seams (interactive wait slices) — never a new timer loop. null/unknown usage fails OPEN
798
+ // v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing worker
799
+ // wait slices — never a new timer loop. null/unknown usage fails OPEN
800
800
  // (never treated as over-threshold). Journal + notify fire at most once while the value stays high.
801
801
  let contextWarned = false;
802
802
  let contextTokens;
803
803
  const sampleContext = async () => {
804
- if (contextWarned || !adapter.contextUsage)
804
+ if (!adapter.contextUsage)
805
805
  return;
806
806
  let usage = null;
807
807
  try {
@@ -814,7 +814,7 @@ export async function runDaemon(repoRoot, opts = {}) {
814
814
  if (!usage || typeof usage.tokens !== "number" || !Number.isFinite(usage.tokens))
815
815
  return;
816
816
  contextTokens = usage.tokens; // last known valid sample, including under-threshold resume candidates
817
- if (usage.tokens < cfg.contextWarnTokens)
817
+ if (contextWarned || usage.tokens < cfg.contextWarnTokens)
818
818
  return;
819
819
  contextWarned = true;
820
820
  lastContextTokens = usage.tokens;
@@ -862,15 +862,15 @@ export async function runDaemon(repoRoot, opts = {}) {
862
862
  // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
863
863
  // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
864
864
  const stallWindowMs = taskTimeoutMinutes * 60_000;
865
- // OBS-82: the stall clock compares NORMALIZED snapshots so a spinner glyph/elapsed-time
866
- // repaint is silence, not activity. ONLY this inactivity compare sees normalized text —
867
- // trailer detection, harvest, paging, and quota checks all read the raw pane.
865
+ // v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
866
+ // the stall clock. Raw pane differences are terminal chrome until proven otherwise.
868
867
  let everHadOutput = output.length > 0;
869
- let lastStallSnapshot = normalizeStallSnapshot(output);
870
- let lastOutputAt = Date.now();
871
- while (Date.now() - lastOutputAt < stallWindowMs) {
868
+ const stallProgress = new StallProgressTracker();
869
+ stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
870
+ let lastProgressAt = Date.now();
871
+ while (Date.now() - lastProgressAt < stallWindowMs) {
872
872
  const sliceStart = Date.now();
873
- const remaining = stallWindowMs - (sliceStart - lastOutputAt);
873
+ const remaining = stallWindowMs - (sliceStart - lastProgressAt);
874
874
  let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
875
875
  if (!everHadOutput) {
876
876
  const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
@@ -893,11 +893,6 @@ export async function runDaemon(repoRoot, opts = {}) {
893
893
  const paneText = await driver.read(slot, 1000);
894
894
  if (paneText.length > 0)
895
895
  everHadOutput = true;
896
- const currentStallSnapshot = normalizeStallSnapshot(paneText);
897
- if (currentStallSnapshot !== lastStallSnapshot) {
898
- lastStallSnapshot = currentStallSnapshot;
899
- lastOutputAt = Date.now();
900
- }
901
896
  // OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
902
897
  if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
903
898
  earlyLaunchDead = true;
@@ -906,6 +901,8 @@ export async function runDaemon(repoRoot, opts = {}) {
906
901
  }
907
902
  // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
908
903
  await sampleContext();
904
+ if (stallProgress.observe({ paneText, contextTokens }))
905
+ lastProgressAt = Date.now();
909
906
  // page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
910
907
  // (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
911
908
  // a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
@@ -943,7 +940,7 @@ export async function runDaemon(repoRoot, opts = {}) {
943
940
  }
944
941
  if (!finished && exitCode === null) {
945
942
  // timed out (or only ever saw false positives): harvest whatever the pane holds now
946
- timedOut = Date.now() - lastOutputAt >= stallWindowMs;
943
+ timedOut = Date.now() - lastProgressAt >= stallWindowMs;
947
944
  output = await driver.read(slot, 1000);
948
945
  finished = new RegExp(trailerPattern(nonce)).test(output);
949
946
  const exit = exitRe.exec(output);
@@ -981,16 +978,16 @@ export async function runDaemon(repoRoot, opts = {}) {
981
978
  else {
982
979
  await driver.run(slot, paneDispatchCommand(dispatchScript));
983
980
  // OBS-54: headless workers have the same output-inactivity budget as visible panes.
984
- // OBS-82: same normalized-snapshot compare as the interactive site — spinner-only repaints
985
- // exhaust the budget here too; harvest below still reads the raw pane.
981
+ // v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
986
982
  const stallWindowMs = taskTimeoutMinutes * 60_000;
987
983
  const initialPane = await driver.read(slot, 500);
988
984
  let everHadOutput = initialPane.length > 0;
989
- let lastStallSnapshot = normalizeStallSnapshot(initialPane);
990
- let lastOutputAt = Date.now();
985
+ const stallProgress = new StallProgressTracker();
986
+ stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
987
+ let lastProgressAt = Date.now();
991
988
  finished = false;
992
- while (Date.now() - lastOutputAt < stallWindowMs) {
993
- const remaining = stallWindowMs - (Date.now() - lastOutputAt);
989
+ while (Date.now() - lastProgressAt < stallWindowMs) {
990
+ const remaining = stallWindowMs - (Date.now() - lastProgressAt);
994
991
  let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
995
992
  if (!everHadOutput) {
996
993
  const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
@@ -1004,19 +1001,17 @@ export async function runDaemon(repoRoot, opts = {}) {
1004
1001
  const paneText = await driver.read(slot, 500);
1005
1002
  if (paneText.length > 0)
1006
1003
  everHadOutput = true;
1007
- const currentStallSnapshot = normalizeStallSnapshot(paneText);
1008
- if (currentStallSnapshot !== lastStallSnapshot) {
1009
- lastStallSnapshot = currentStallSnapshot;
1010
- lastOutputAt = Date.now();
1011
- }
1012
1004
  if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
1013
1005
  earlyLaunchDead = true;
1014
1006
  break;
1015
1007
  }
1008
+ await sampleContext();
1009
+ if (stallProgress.observe({ paneText, contextTokens }))
1010
+ lastProgressAt = Date.now();
1016
1011
  }
1017
1012
  output = await driver.read(slot, 500);
1018
1013
  exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
1019
- timedOut = !finished && Date.now() - lastOutputAt >= stallWindowMs;
1014
+ timedOut = !finished && Date.now() - lastProgressAt >= stallWindowMs;
1020
1015
  }
1021
1016
  // SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
1022
1017
  // shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
@@ -1,8 +1,25 @@
1
- /** Normalize one pane snapshot for the stall-inactivity compare (trailer parsing, harvest,
2
- * waitOutput, and paging read the raw text; the LLM transcript filter below reuses this to
3
- * CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
4
- * equal are the same frame modulo spinner presentation; any other byte difference is activity. */
1
+ /** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
2
+ * parsing, harvest, waitOutput, and paging always read the raw text. */
5
3
  export declare function normalizeStallSnapshot(text: string): string;
4
+ export interface StallProgressSample {
5
+ paneText: string;
6
+ seedSubmitted?: boolean;
7
+ contextTokens?: number;
8
+ }
9
+ /**
10
+ * Monotonic worker-progress measure for the stall watchdog.
11
+ *
12
+ * Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
13
+ * evidence of work. A rendered transcript is only known to have grown when it occupies more
14
+ * non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
15
+ * advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
16
+ */
17
+ export declare class StallProgressTracker {
18
+ private transcriptRows;
19
+ private seedSubmitted;
20
+ private contextTokens;
21
+ observe(sample: StallProgressSample): boolean;
22
+ }
6
23
  /** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
7
24
  * seam exists for fault injection in tests only — production callers pass text alone. */
8
25
  export declare function filterLlmTranscript(text: string, classify?: (t: string) => string): string;
package/dist/run/stall.js CHANGED
@@ -1,12 +1,9 @@
1
- // OBS-82: codex's MCP-startup spinner repaints a braille glyph + elapsed-time cell forever, so the
2
- // daemon's raw snapshot compare reads a wedged pane as active and the stall clock never fires.
3
- // This normalizer deletes ONLY presentation tokens from a closed allowlist — ANSI/VT escape
4
- // sequences, braille-range spinner glyphs, and elapsed-time tokens bound to time-unit suffixes.
5
- // Every other byte passes through identical: words, paths, server names, and progress counts
6
- // (a five-of-seven counter change IS activity) all remain change-sensitive. The asymmetry is the
7
- // design: an allowlist MISS degrades to today's recoverable no-reap behavior, while an over-broad
8
- // deletion would reap a healthy worker — a new failure class. Grow the allowlist only with
9
- // captured evidence (tests/fixtures/codex-mcp-spinner/).
1
+ // OBS-82: normalize known presentation tokens before measuring transcript extent or filtering an
2
+ // LLM-bound transcript. This remains a closed allowlist — ANSI/VT escapes, braille-range spinner
3
+ // glyphs, and elapsed-time tokens bound to time-unit suffixes. Every other byte passes through
4
+ // identical. v1.76 deliberately stopped treating arbitrary normalized byte changes as progress:
5
+ // StallProgressTracker below requires monotonic evidence, so an unknown repaint fails closed toward
6
+ // a recoverable consult instead of holding the watchdog silent.
10
7
  // CSI (with intermediates), OSC (BEL- or ST-terminated), DCS/SOS/PM/APC strings, single-char
11
8
  // escapes, and charset selection — the raw-pty forms; herdr pane reads are already rendered.
12
9
  // eslint-disable-next-line no-control-regex
@@ -16,13 +13,46 @@ const SPINNER_RE = /[⠀-⣿]/g;
16
13
  // A digit run (optionally decimal) bound directly to a time-unit suffix, standing alone as a
17
14
  // word: 9s, 41s, 3m, 1h, 800ms. Never bare digits — "(6/7)" and "5 of 7" stay change-sensitive.
18
15
  const ELAPSED_RE = /(?<![\w.])\d+(?:\.\d+)?(?:ms|[hms])(?!\w)/g;
19
- /** Normalize one pane snapshot for the stall-inactivity compare (trailer parsing, harvest,
20
- * waitOutput, and paging read the raw text; the LLM transcript filter below reuses this to
21
- * CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
22
- * equal are the same frame modulo spinner presentation; any other byte difference is activity. */
16
+ /** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
17
+ * parsing, harvest, waitOutput, and paging always read the raw text. */
23
18
  export function normalizeStallSnapshot(text) {
24
19
  return text.replace(ANSI_RE, "").replace(SPINNER_RE, "").replace(ELAPSED_RE, "");
25
20
  }
21
+ /**
22
+ * Monotonic worker-progress measure for the stall watchdog.
23
+ *
24
+ * Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
25
+ * evidence of work. A rendered transcript is only known to have grown when it occupies more
26
+ * non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
27
+ * advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
28
+ */
29
+ export class StallProgressTracker {
30
+ transcriptRows = 0;
31
+ seedSubmitted = false;
32
+ contextTokens;
33
+ observe(sample) {
34
+ let advanced = false;
35
+ const rows = normalizeStallSnapshot(sample.paneText)
36
+ .split("\n")
37
+ .filter((line) => line.trim().length > 0)
38
+ .length;
39
+ if (rows > this.transcriptRows) {
40
+ this.transcriptRows = rows;
41
+ advanced = true;
42
+ }
43
+ if (sample.seedSubmitted && !this.seedSubmitted) {
44
+ this.seedSubmitted = true;
45
+ advanced = true;
46
+ }
47
+ const tokens = sample.contextTokens;
48
+ if (tokens !== undefined && Number.isFinite(tokens)) {
49
+ if (tokens > (this.contextTokens ?? 0))
50
+ advanced = true;
51
+ this.contextTokens = Math.max(this.contextTokens ?? 0, tokens);
52
+ }
53
+ return advanced;
54
+ }
55
+ }
26
56
  // ─── v1.65 T2: LLM-bound transcript filter ──────────────────────────────────────────────────────
27
57
  // Consult dossiers and gate prompts pay tokens per transcript byte, so LLM-bound text runs through
28
58
  // a per-line classifier: carriage-return overwrite churn keeps only the final paint, lines that are
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "1.75.0",
3
+ "version": "1.76.0",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",