tickmarkr 1.74.0 → 1.76.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,5 @@
1
- import { type AuthHealth, type WorkerAdapter } from "./types.js";
1
+ import { type AuthHealth, type TrustDialog, type WorkerAdapter } from "./types.js";
2
+ export declare const CLAUDE_TRUST_DIALOG: TrustDialog;
2
3
  export declare function claudeSlug(real: string): string;
3
4
  export declare function probeVersion(bin: string): AuthHealth;
4
5
  export declare const claudeCode: WorkerAdapter;
@@ -20,6 +20,12 @@ import { channelsFromConfig, shq, TokenUsageSchema } from "./types.js";
20
20
  // Phase 18's operator-price × tokens derivation, not a CLI claim.
21
21
  const MAX_SESSION_FILES = 20; // newest-first; a long-lived project dir can hold many sessions
22
22
  const MAX_SESSION_BYTES = 8_000_000; // per-file cap; a runaway JSONL cannot make the read unbounded
23
+ // v1.75 T2 / OBS-137: current Claude Code workspace-trust prompt (2.1.218). The full question
24
+ // distinguishes this startup gate from routine agent text; Enter accepts the selected trust option.
25
+ export const CLAUDE_TRUST_DIALOG = {
26
+ fingerprint: "Quick safety check: Is this a project you created or one you trust?",
27
+ key: "Enter",
28
+ };
23
29
  export function claudeSlug(real) {
24
30
  return real.replace(/[^A-Za-z0-9]/g, "-");
25
31
  }
@@ -54,13 +60,11 @@ export const claudeCode = {
54
60
  // and --mcp-config is VARIADIC — a positional after it is eaten as a config-file path, so another
55
61
  // flag must always follow the value, never the prompt.
56
62
  headlessCommand: (promptFile, model) => `claude -p "$(cat ${shq(promptFile)})" --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text`,
57
- // HYG-03: the residual first-entry dialog on an interactive TUI is the workspace TRUST dialog (not MCP
58
- // config loading) — CLI-imposed, no flag to pre-accept, only store is claude's global last-writer-wins
59
- // ~/.claude.json keyed on the exact path. Closed WON'T-FIX (decision B, 2026-07-10): tickmarkr writes nothing
60
- // to that file (a seed races claude's own writes, nondeterministically). Amortizes to one operator dismissal
61
- // per stable worktree path; blocked-pane paging surfaces it. Do NOT change this command to "fix" the dialog —
62
- // see .planning/REQUIREMENTS.md HYG-03 and 21-02-LIVE-CHECK.md. Revisit if upstream ships a --trust flag.
63
+ // HYG-03 / OBS-137: the residual first-entry dialog is workspace trust, not MCP config loading.
64
+ // Claude's only store is global last-writer-wins ~/.claude.json, so tickmarkr still does not seed it;
65
+ // the daemon safely answers only the exact adapter-declared dialog once per slot.
63
66
  interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
67
+ trustDialog: CLAUDE_TRUST_DIALOG,
64
68
  resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
65
69
  invoke(task, _cwd, a, ctx) {
66
70
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
@@ -1,8 +1,9 @@
1
- import { type TrustVerdict, type WorkerAdapter } from "./types.js";
1
+ import { type TrustDialog, type TrustVerdict, type WorkerAdapter } from "./types.js";
2
2
  export declare function readCodexModelsCache(path?: string): {
3
3
  models: string[];
4
4
  fetchedAt?: string;
5
5
  };
6
+ export declare const CODEX_TRUST_DIALOG: TrustDialog;
6
7
  export declare function seedCodexTrust(repoRoot: string, configPath?: string): TrustVerdict;
7
8
  export declare function hasCodexTrustedProject(text: string, root: string): boolean;
8
9
  export declare function codexConfigMcpServerNames(configPath?: string): string[];
@@ -89,6 +89,12 @@ const GITDIR_WRITABLE = `-c "sandbox_workspace_write.writable_roots=[\\"$(git re
89
89
  // -s/--sandbox workspace-write sandbox (deliberately NOT --dangerously-bypass-approvals-and-sandbox,
90
90
  // which would drop the sandbox). Listed by `codex --help` and `codex exec --help` (verified 2026-07-23).
91
91
  const CODEX_HOOK_TRUST = "--dangerously-bypass-hook-trust";
92
+ // v1.75 T2 / OBS-137: current Codex workspace-trust prompt (0.144.6). The exact heading
93
+ // is distinct from normal agent output; Enter accepts the selected "Yes, continue" option.
94
+ export const CODEX_TRUST_DIALOG = {
95
+ fingerprint: "Do you trust the contents of this directory?",
96
+ key: "Enter",
97
+ };
92
98
  // v1.22 T5 / OBS-16: codex keys trust on absolute path under [projects."<root>"] trust_level="trusted"
93
99
  // in ~/.codex/config.toml (CODEX_HOME relocates the dir). Worktrees inherit parent-project trust when
94
100
  // the REPO ROOT is trusted — seed the root once, cover every future worktree. Idempotent: a second
@@ -196,6 +202,7 @@ export const codex = {
196
202
  // v1.22 T5: seed [projects."<repoRoot>"] trust_level="trusted" so fresh worktrees never stall on
197
203
  // "Do you trust this directory?" (OBS-16). doctor-only side effect.
198
204
  trust: (repoRoot) => seedCodexTrust(repoRoot),
205
+ trustDialog: CODEX_TRUST_DIALOG,
199
206
  // v1.5 MODEL-01: file read only (no `codex models` subcommand exists, verified 2026-07-10).
200
207
  // Already fails OPEN to [] internally — advisory detection, unlike gates' fail-closed.
201
208
  listModels: async () => readCodexModelsCache().models,
@@ -3,6 +3,11 @@ import { type Assignment, type WorkerAdapter, type WorkerResult } from "./types.
3
3
  export declare function kimiAuthed(credentialsText: string, nowMs: number): boolean;
4
4
  export declare function parseKimiModels(raw: string): string[];
5
5
  export declare function parseKimiResult(raw: string, nonce: string): WorkerResult;
6
+ export interface KimiDoctorTurnResult {
7
+ ok: boolean;
8
+ evidence: string;
9
+ }
10
+ export declare function probeKimiDoctorTurn(cwd: string): Promise<KimiDoctorTurnResult>;
6
11
  export declare function kimiSessionId(output: string): string | undefined;
7
12
  export declare function kimiBannerModel(banner: string): string | undefined;
8
13
  export declare function kimiBannerSessionId(banner: string): string | undefined;
@@ -20,6 +25,7 @@ export interface KimiInteractiveSeedResult {
20
25
  seedError?: string;
21
26
  sessionId?: string;
22
27
  }
28
+ export declare const KIMI_INPUT_BOX: import("./types.js").InputBox;
23
29
  export declare function runKimiInteractiveSeed(opts: {
24
30
  driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
25
31
  slot: Slot;
@@ -5,7 +5,7 @@ import { probeVersion } from "./claude-code.js";
5
5
  import { parseWorkerResult } from "./prompt.js";
6
6
  import { runInteractiveSeed } from "../run/interactive-seed.js";
7
7
  import { sh } from "../run/git.js";
8
- import { channelsFromConfig, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
8
+ import { channelsFromConfig, declareInputBox, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
9
9
  // KIMI-03 → v1.58 T5: the "no harness-readable counter" block (research F-6, 2026-07-17) is
10
10
  // LIFTED for collectUsage — kimi 0.27.0 writes a wire journal per agent at
11
11
  // ~/.kimi-code/sessions/<wd>/session_<uuid>/agents/<agent>/wire.jsonl, and ~/.kimi-code/
@@ -64,6 +64,28 @@ export function parseKimiResult(raw, nonce) {
64
64
  const stripped = raw.split("\n").map((l) => l.replace(/^[\s]*[•*-]\s+/, "")).join("\n");
65
65
  return parseWorkerResult(stripped, nonce);
66
66
  }
67
+ const KIMI_DOCTOR_TURN_MODEL = "kimi-code/k3";
68
+ const KIMI_DOCTOR_TURN_PROMPT = "Reply with exactly OK and nothing else.";
69
+ const KIMI_DOCTOR_TURN_TIMEOUT_MS = 60000;
70
+ // OBS-141: intentionally separate from probe() so plan/run remain free file checks. Only doctor
71
+ // calls this one-turn contract probe; its test seam stubs sh and never launches a real agent CLI.
72
+ export async function probeKimiDoctorTurn(cwd) {
73
+ const command = `kimi -p ${shq(KIMI_DOCTOR_TURN_PROMPT)} --model ${shq(KIMI_DOCTOR_TURN_MODEL)} --output-format text`;
74
+ const r = await sh(command, cwd, KIMI_DOCTOR_TURN_TIMEOUT_MS);
75
+ if (r.timedOut) {
76
+ return { ok: false, evidence: `turn timed out after ${KIMI_DOCTOR_TURN_TIMEOUT_MS}ms` };
77
+ }
78
+ const output = `${r.stderr}\n${r.stdout}`.trim().replace(/\s+/g, " ");
79
+ if (r.code !== 0) {
80
+ return { ok: false, evidence: output || `turn exited ${r.code}` };
81
+ }
82
+ const returnedOk = r.stdout.split("\n")
83
+ .map((line) => line.replace(/^[\s]*[•*-]\s+/, "").trim())
84
+ .includes("OK");
85
+ return returnedOk
86
+ ? { ok: true, evidence: `model turn returned OK with ${KIMI_DOCTOR_TURN_MODEL}` }
87
+ : { ok: false, evidence: `turn returned no exact OK answer${output ? `: ${output}` : ""}` };
88
+ }
67
89
  // v1.53 T3: session-id capture from the run-output trailer — every `kimi -p` run (fresh or resumed)
68
90
  // ends with `To resume this session: kimi -r session_<uuid>` (live probe 2026-07-18). Anchored full
69
91
  // line only: prompt/model prose can contain lookalike text, and the anchored charset keeps a
@@ -89,23 +111,20 @@ export function kimiSessionId(output) {
89
111
  }
90
112
  return id;
91
113
  }
92
- // v1.69 T6: the native TUI takes -m <alias>, where config.toml aliases are the bare model suffix of
93
- // the tickmarkr channel id (live probe 2026-07-22). Keep the mapping explicit and localized.
94
- function kimiAlias(model) {
95
- return model.replace(/^kimi-code\//, "");
96
- }
97
114
  // v1.69 T7: the cold-start banner prints the model alias and session id. Parse them from the
98
- // banner text already captured for the readiness match — no new probe, no extra dispatch.
115
+ // banner text already captured for the readiness match — no new probe, no extra dispatch. Kimi
116
+ // 0.29.0 may print either the full config key or its display suffix; normalize both idempotently
117
+ // to the full channel identifier routing uses.
99
118
  const BANNER_MODEL_RE = /^Model:\s*(.+)$/m;
100
119
  const BANNER_SESSION_RE = /^Session:\s*(session_[0-9a-f-]+)$/m;
101
120
  export function kimiBannerModel(banner) {
102
121
  const m = BANNER_MODEL_RE.exec(banner);
103
122
  if (!m)
104
123
  return undefined;
105
- const alias = m[1].trim();
106
- if (!alias)
124
+ const printedModel = m[1].trim();
125
+ if (!printedModel)
107
126
  return undefined;
108
- return `kimi-code/${alias}`;
127
+ return printedModel.startsWith("kimi-code/") ? printedModel : `kimi-code/${printedModel}`;
109
128
  }
110
129
  export function kimiBannerSessionId(banner) {
111
130
  return BANNER_SESSION_RE.exec(banner)?.[1];
@@ -118,10 +137,16 @@ export function confirmKimiSeedBanner(banner, assignedModel) {
118
137
  }
119
138
  return { ok: true, sessionId: kimiBannerSessionId(banner) };
120
139
  }
140
+ // v1.75 T1 / OBS-136: this readiness line is rendered inside Kimi Code's bordered steady-state
141
+ // input box. The adapter owns the fingerprint; the driver only consults declarations generically.
142
+ export const KIMI_INPUT_BOX = declareInputBox("kimi", {
143
+ fingerprint: "Send /help for help information.",
144
+ });
121
145
  // Shared launch-then-seed surface (T6) + banner confirm (T7/T2). One definition so the adapter
122
146
  // property and the daemon's generic runInteractiveSeed path cannot drift.
123
147
  const KIMI_SEED = {
124
- launch: (model) => `kimi -y -m ${shq(kimiAlias(model))}`,
148
+ // v1.76 T2 / OBS-141: 0.29.0 resolves only the full config.toml model key on the first turn.
149
+ launch: (model) => `kimi -y -m ${shq(model)}`,
125
150
  readinessMatch: "Send /help for help information.",
126
151
  seedLine: (promptFile) => `Read ${promptFile} and do exactly what it says.`,
127
152
  confirmBanner: confirmKimiSeedBanner,
@@ -165,6 +190,7 @@ export const kimi = {
165
190
  // prompt as one user turn. Banner model/session confirmation runs on the daemon's generic
166
191
  // runInteractiveSeed path via confirmBanner on KIMI_SEED (not a separate dispatch helper).
167
192
  interactiveSeed: KIMI_SEED,
193
+ inputBox: KIMI_INPUT_BOX,
168
194
  // v1.53 T3 resume — live-probed 2026-07-18: `-p` + `-S <id>` compose cleanly (no OBS-67-class
169
195
  // flag rejection) and the resumed session carries prior conversation state. `-S <id>` is the
170
196
  // deterministic form; `-c` rejected as primary — cwd-keyed, nondeterministic under worktree
@@ -82,6 +82,12 @@ export interface TrustDialog {
82
82
  key: string;
83
83
  }
84
84
  export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): boolean;
85
+ export interface InputBox {
86
+ fingerprint: string;
87
+ }
88
+ export declare function declareInputBox(adapterId: string, inputBox: InputBox): InputBox;
89
+ export declare function declaredInputBoxForWorkerName(workerName: string): InputBox | undefined;
90
+ export declare function matchesInputBox(paneText: string, inputBox: InputBox): boolean;
85
91
  export interface WorkerAdapter {
86
92
  id: string;
87
93
  vendor: string;
@@ -105,6 +111,7 @@ export interface WorkerAdapter {
105
111
  contextUsage?(session: SessionRef): ContextUsage | null;
106
112
  trust?(repoRoot: string): TrustVerdict;
107
113
  trustDialog?: TrustDialog;
114
+ inputBox?: InputBox;
108
115
  hardcodedFlags?: {
109
116
  binary: string;
110
117
  flags: string[];
@@ -30,6 +30,18 @@ export function modelAuthed(health, model, allowUnverifiedModels = false) {
30
30
  export function matchesTrustDialog(paneText, dialog) {
31
31
  return paneText.includes(dialog.fingerprint);
32
32
  }
33
+ const inputBoxes = new Map();
34
+ export function declareInputBox(adapterId, inputBox) {
35
+ inputBoxes.set(adapterId, inputBox);
36
+ return inputBox;
37
+ }
38
+ export function declaredInputBoxForWorkerName(workerName) {
39
+ const adapterId = /^.+-worker-(.+)-a\d+-.+$/.exec(workerName)?.[1];
40
+ return adapterId === undefined ? undefined : inputBoxes.get(adapterId);
41
+ }
42
+ export function matchesInputBox(paneText, inputBox) {
43
+ return paneText.includes(inputBox.fingerprint);
44
+ }
33
45
  export function channelsFromConfig(adapterId, cfg) {
34
46
  const e = cfg.tiers[adapterId];
35
47
  if (!e)
@@ -1,5 +1,7 @@
1
1
  import type { WorkerAdapter } from "../../adapters/types.js";
2
+ import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
2
3
  export type DoctorOpts = {
3
4
  banner?: boolean;
5
+ kimiTurnProbe?: (cwd: string) => Promise<KimiDoctorTurnResult>;
4
6
  };
5
7
  export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
@@ -6,6 +6,7 @@ import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
6
6
  import { declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
7
7
  import { DEFAULT_CONFIG, loadConfig, overlayPreferShapes } from "../../config/config.js";
8
8
  import { HerdrDriver } from "../../drivers/herdr.js";
9
+ import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
9
10
  import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
10
11
  const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
11
12
  const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
@@ -20,6 +21,26 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
20
21
  console.error("probing installed agent CLIs — one short LLM call per configured model, may take a minute...");
21
22
  const probeProgressTTY = process.stderr.isTTY === true;
22
23
  const health = await probeAll(adapters, { cwd });
24
+ const kimiAdapter = adapters.find((a) => a.id === kimi.id);
25
+ const kimiTurnEnabled = kimiAdapter !== undefined
26
+ && (kimiAdapter === kimi || opts.kimiTurnProbe !== undefined);
27
+ if (kimiTurnEnabled) {
28
+ const h = health.kimi;
29
+ if (h.installed && h.authed) {
30
+ let turn;
31
+ try {
32
+ turn = await (opts.kimiTurnProbe ?? probeKimiDoctorTurn)(cwd);
33
+ }
34
+ catch (e) {
35
+ turn = { ok: false, evidence: e instanceof Error ? e.message : String(e) };
36
+ }
37
+ health.kimi = {
38
+ ...h,
39
+ authed: turn.ok,
40
+ note: `${h.note ? `${h.note}; ` : ""}${turn.ok ? turn.evidence : `model turn failed: ${turn.evidence}`}`,
41
+ };
42
+ }
43
+ }
23
44
  // MODEL-02: detect models where the adapter exposes a list surface, BEFORE writing doctor.json (write once, below).
24
45
  // Fail OPEN — the inverse of gates' fail-closed: detection is advisory, so a broken list surface NEVER fails doctor.
25
46
  for (const a of adapters) {
@@ -35,14 +56,20 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
35
56
  }
36
57
  catch { /* fail open: leave models as-is, doctor stays healthy */ }
37
58
  }
38
- await probeModels(cfg, cwd, adapters, health, probeProgressTTY
59
+ // A free Kimi auth failure or failed earned-green turn must not spend more probes. Every other
60
+ // adapter keeps the exact existing model-sweep path.
61
+ const modelProbeAdapters = kimiTurnEnabled && health.kimi.authed === false
62
+ ? adapters.filter((a) => a !== kimiAdapter)
63
+ : adapters;
64
+ await probeModels(cfg, cwd, modelProbeAdapters, health, probeProgressTTY
39
65
  ? (adapter, model, status, durationMs) => console.error(` ${adapter}:${model} ${status} (${(durationMs / 1000).toFixed(1)}s)`)
40
66
  : undefined);
41
67
  writeDoctor(cwd, health);
42
68
  const rows = adapters.map((a) => {
43
69
  const h = health[a.id];
44
70
  const state = !h.installed ? "not installed" : `${h.version ?? "installed"}${h.note ? ` (${h.note})` : ""}`;
45
- return alignedStatusRow(h.installed ? "pass" : "fail", a.id, state);
71
+ const healthy = h.installed && (a.id !== kimi.id || h.authed);
72
+ return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
46
73
  });
47
74
  // v1.48 T1: advisory sweep for known agent CLIs with no adapter — never written to doctor.json health.
48
75
  rows.push(...detectCandidateClis().map(({ binary, version }) => alignedStatusRow("warn", binary, `detected: ${version ?? "version unknown"} (no tickmarkr adapter — not routable)`)));
@@ -57,6 +57,14 @@ function withModeLine(yaml, mode) {
57
57
  return yaml.replace(/^routing:$/m, `routing:\n mode: ${mode}`);
58
58
  return `routing:\n mode: ${mode}\n${yaml}`;
59
59
  }
60
+ function formatFleetSteering(cfg) {
61
+ const blocks = [];
62
+ if (cfg.review.prefer?.length)
63
+ blocks.push(`review:\n prefer: ${JSON.stringify(cfg.review.prefer)}`);
64
+ if (cfg.consult.prefer?.length)
65
+ blocks.push(`consult:\n prefer: ${JSON.stringify(cfg.consult.prefer)}`);
66
+ return blocks.length ? `${blocks.join("\n")}\n` : "";
67
+ }
60
68
  // v1.51 T4: one gloss per routing mode on the fleet mode screen — mirrors the preset compiler.
61
69
  const MODE_GLOSS = {
62
70
  "partner-led": "every shape frontier · explore off",
@@ -83,7 +91,9 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
83
91
  const rm = resolveRunMode(cwd, { globalDir });
84
92
  const body = formatFleetPrint(cwd, { globalDir });
85
93
  const nl = body.indexOf("\n");
86
- return `${body.slice(0, nl)}\n# mode: ${rm.mode.mode} (${rm.source})${body.slice(nl)}`;
94
+ // Steering comes from the same resolved config snapshot the editor consumes below,
95
+ // not from another parse of either raw overlay.
96
+ return `${body.slice(0, nl)}\n# mode: ${rm.mode.mode} (${rm.source})${body.slice(nl)}${formatFleetSteering(rm.cfg)}`;
87
97
  }
88
98
  if (!interactive)
89
99
  return { out: NON_TTY_MSG, code: 1 };
@@ -14,6 +14,7 @@ export declare class HerdrDriver implements ExecutorDriver {
14
14
  private deliverySerial;
15
15
  private dispatchLeases;
16
16
  private deliveredPanes;
17
+ private inputBoxes;
17
18
  private ws;
18
19
  private callerPane;
19
20
  private watches;
@@ -23,6 +24,7 @@ export declare class HerdrDriver implements ExecutorDriver {
23
24
  private reserveDispatch;
24
25
  private verifyPaneIdentityBinding;
25
26
  private deliveryMatches;
27
+ private submissionRegistered;
26
28
  static available(): boolean;
27
29
  private herdr;
28
30
  private namedPaneId;
@@ -39,6 +41,7 @@ export declare class HerdrDriver implements ExecutorDriver {
39
41
  private joinGroup;
40
42
  run(slot: Slot, cmd: string): Promise<void>;
41
43
  private deliver;
44
+ private submitVerifiedDelivery;
42
45
  private settleDeliveryLine;
43
46
  private deliveryReadMatches;
44
47
  private waitOk;
@@ -1,4 +1,4 @@
1
- import { shq } from "../adapters/types.js";
1
+ import { declaredInputBoxForWorkerName, matchesInputBox, shq } from "../adapters/types.js";
2
2
  import { PANE_IDENTITY_ENV, paneIdentityLine } from "../brand.js";
3
3
  import { createWorktree, sh } from "../run/git.js";
4
4
  import { herdrSealShellPrefix } from "./subprocess.js";
@@ -8,6 +8,7 @@ export const TRAILER_SAFE_FLOOR_COLS = 108;
8
8
  export const TRAILER_WIDTH_MARGIN = 2; // cols below (floor + margin) refuse a rightward first split
9
9
  // OBS-85 verified delivery: bounded type→read-back→enter attempts before failing closed.
10
10
  export const DELIVERY_ATTEMPTS = 3;
11
+ const DELIVERY_SUBMIT_ATTEMPTS = 2; // initial Enter + one evidence-backed re-press (OBS-140)
11
12
  const DELIVERY_VERIFY_TIMEOUT_MS = 2000; // per attempt — a paste that hasn't rendered in 2s is retyped
12
13
  const DELIVERY_READ_LINES = 80;
13
14
  const DELIVERY_SETTLE_READ_ATTEMPTS = 6;
@@ -32,6 +33,7 @@ export class HerdrDriver {
32
33
  deliverySerial = Promise.resolve();
33
34
  dispatchLeases = new WeakMap();
34
35
  deliveredPanes = new WeakMap();
36
+ inputBoxes = new WeakMap();
35
37
  // VIS-10: the run's workspace id, captured once at construction (the daemon inherits it from the
36
38
  // operator's env before the driver is built). Required at slot() time, never in the constructor —
37
39
  // pickDriver and its unit test construct HerdrDriver without env, so slot() is the trust gate.
@@ -94,6 +96,22 @@ export class HerdrDriver {
94
96
  const needle = norm(cmd);
95
97
  return needle.length > 0 && hay.includes(needle);
96
98
  }
99
+ // Submission succeeds when the typed prompt disappears, or when it has moved above a fresh
100
+ // adapter-declared input box (the prompt is now transcript, not input). Shell-line delivery uses
101
+ // the same normalized seam: a prompt still at the bottom ends the pane text; output/a fresh prompt
102
+ // after it proves Enter registered. No adapter-specific fingerprint lives in the driver.
103
+ submissionRegistered(transcript, cmd, inputBox) {
104
+ const norm = (s) => s.replace(/\s+/g, "");
105
+ const hay = norm(transcript);
106
+ const needle = norm(cmd);
107
+ const promptAt = hay.lastIndexOf(needle);
108
+ if (needle.length === 0 || promptAt < 0)
109
+ return true;
110
+ if (inputBox && matchesInputBox(transcript, inputBox)) {
111
+ return hay.lastIndexOf(norm(inputBox.fingerprint)) > promptAt;
112
+ }
113
+ return promptAt + needle.length < hay.length;
114
+ }
97
115
  static available() {
98
116
  return process.env.HERDR_ENV === "1";
99
117
  }
@@ -142,6 +160,7 @@ export class HerdrDriver {
142
160
  }
143
161
  }
144
162
  async slot(cwd, name, opts) {
163
+ const inputBox = declaredInputBoxForWorkerName(name);
145
164
  // T1 ownership contract: `opts.owned` (T2 call sites) names the pane canonically —
146
165
  // tickmarkr:<role>:<taskId>:<attempt>:<runId>. Without it, `name` passes through byte-identical
147
166
  // (today's legacy daemon/gates/consult shapes) — canonicalizeLegacyName (types.ts) is what lets
@@ -154,12 +173,13 @@ export class HerdrDriver {
154
173
  // Production dispatch names are canonical even when the gate call site supplies the already-
155
174
  // formatted name rather than SlotOpts.owned. Hold one lease across slot() → run(); legacy/manual
156
175
  // slots retain their existing allocation-only semantics for compatibility.
157
- if (parseOwnedName(resolved))
158
- return this.reserveDispatch(allocate);
176
+ const slot = parseOwnedName(resolved) ? await this.reserveDispatch(allocate) : await allocate();
177
+ if (inputBox)
178
+ this.inputBoxes.set(slot, inputBox);
159
179
  // group wins if both are set (a group tab is already stage-labeled; passing both is a caller bug).
160
180
  // label (without group) → dedicated labeled tab via tabSlot's third param: no groups-map entry, no
161
181
  // refcount, no groupSerial, no degrade latch — dedicated tabs have no shared state to guard (SUP-01).
162
- return allocate(); // label undefined → defaults to name (today's behavior)
182
+ return slot; // label undefined → defaults to name (today's behavior)
163
183
  }
164
184
  // today's per-slot tab path, plus the VIS-04 orphan reap
165
185
  // label defaults to the slot name; group tabs pass the STAGE name instead — a first-member label
@@ -397,25 +417,26 @@ export class HerdrDriver {
397
417
  // line only after two consecutive pane reads agree; an already-stable frame returns on the
398
418
  // first fresh read without a timer. A changing pane is bounded and preserves OBS-85's
399
419
  // fail-closed error instead of guessing from an adapter fingerprint.
400
- const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript);
420
+ const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript, this.inputBoxes.get(slot));
401
421
  transcript = settled.transcript;
402
422
  if (!settled.ok) {
403
423
  throw new Error(`herdr delivery clear failed — refusing to retype onto a corrupted line (OBS-85); pane transcript:\n${transcript}`);
404
424
  }
405
- // Clear the corrupted input line before retyping; a failed clear must NOT be retyped onto —
406
- // corrupt-prefix + clean-retype would concatenate and false-verify by containment.
407
- const cleared = await this.herdr(`pane send-keys ${shq(pane)} C-u`, slot.cwd);
408
- if (cleared.code !== 0)
409
- throw new Error(`herdr delivery clear failed — refusing to retype onto a corrupted line (OBS-85); pane transcript:\n${transcript}`);
425
+ if (!settled.recognizedInputBox) {
426
+ // Clear the corrupted shell input line before retyping; a failed clear must NOT be retyped
427
+ // onto — corrupt-prefix + clean-retype would concatenate and false-verify by containment.
428
+ // A stable adapter-declared input box is already an empty legitimate delivery target.
429
+ const cleared = await this.herdr(`pane send-keys ${shq(pane)} C-u`, slot.cwd);
430
+ if (cleared.code !== 0)
431
+ throw new Error(`herdr delivery clear failed — refusing to retype onto a corrupted line (OBS-85); pane transcript:\n${transcript}`);
432
+ }
410
433
  }
411
434
  const typed = await this.herdr(`pane send-text ${shq(pane)} ${shq(cmd)}`, slot.cwd);
412
435
  if (typed.code !== 0)
413
436
  throw new Error(`herdr pane send-text failed: ${typed.stderr || typed.stdout}`);
414
437
  const back = await this.herdr(`pane wait-output ${shq(pane)} --match ${shq(cmd)} --timeout ${DELIVERY_VERIFY_TIMEOUT_MS}`, slot.cwd, DELIVERY_VERIFY_TIMEOUT_MS + 15_000);
415
438
  if (this.waitOk(back.code, back.stdout) || await this.deliveryReadMatches(pane, cmd, slot.cwd)) {
416
- const enter = await this.herdr(`pane send-keys ${shq(pane)} Enter`, slot.cwd);
417
- if (enter.code !== 0)
418
- throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
439
+ await this.submitVerifiedDelivery(slot, cmd, pane);
419
440
  return;
420
441
  }
421
442
  // capture the corrupted delivery BEFORE clearing it — the OBS-85 byte-level evidence
@@ -423,20 +444,52 @@ export class HerdrDriver {
423
444
  }
424
445
  throw new Error(`herdr delivery corrupted after ${DELIVERY_ATTEMPTS} attempts — enter never pressed (OBS-85); pane transcript:\n${transcript}`);
425
446
  }
426
- async settleDeliveryLine(pane, cwd, initialTranscript) {
447
+ async submitVerifiedDelivery(slot, cmd, pane) {
448
+ let transcript = "";
449
+ const inputBox = this.inputBoxes.get(slot);
450
+ for (let attempt = 0; attempt < DELIVERY_SUBMIT_ATTEMPTS; attempt++) {
451
+ const enter = await this.herdr(`pane send-keys ${shq(pane)} Enter`, slot.cwd);
452
+ if (enter.code !== 0)
453
+ throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
454
+ // Reuse the existing settle-read window. A first-read success returns before any timer; only
455
+ // a prompt that still occupies the delivery target spends the bounded settle window. This
456
+ // verification always completes before a possible re-press, so a slow submit cannot duplicate.
457
+ const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript, inputBox, (candidate) => this.submissionRegistered(candidate, cmd, inputBox));
458
+ transcript = settled.transcript;
459
+ if (settled.ok)
460
+ return;
461
+ if (settled.readFailed) {
462
+ throw new Error(`herdr delivery corrupted — submission verification failed, refusing to re-press Enter (OBS-140); pane transcript:\n${transcript}`);
463
+ }
464
+ }
465
+ throw new Error(`herdr delivery corrupted after ${DELIVERY_SUBMIT_ATTEMPTS} submit attempts — submission never registered (OBS-140); pane transcript:\n${transcript}`);
466
+ }
467
+ async settleDeliveryLine(pane, cwd, initialTranscript, inputBox, accept) {
427
468
  let transcript = initialTranscript;
428
469
  for (let readAttempt = 0; readAttempt < DELIVERY_SETTLE_READ_ATTEMPTS; readAttempt++) {
429
470
  const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
430
471
  if (read.code !== 0)
431
- return { ok: false, transcript: read.stdout || transcript };
432
- if (read.stdout === transcript)
433
- return { ok: true, transcript };
472
+ return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false, readFailed: true };
473
+ if (accept?.(read.stdout)) {
474
+ return {
475
+ ok: true,
476
+ transcript: read.stdout,
477
+ recognizedInputBox: inputBox !== undefined && matchesInputBox(read.stdout, inputBox),
478
+ };
479
+ }
480
+ if (accept === undefined && read.stdout === transcript) {
481
+ return {
482
+ ok: true,
483
+ transcript,
484
+ recognizedInputBox: inputBox !== undefined && matchesInputBox(transcript, inputBox),
485
+ };
486
+ }
434
487
  transcript = read.stdout;
435
488
  if (readAttempt < DELIVERY_SETTLE_READ_ATTEMPTS - 1) {
436
489
  await new Promise((resolve) => setTimeout(resolve, DELIVERY_SETTLE_POLL_MS));
437
490
  }
438
491
  }
439
- return { ok: false, transcript };
492
+ return { ok: false, transcript, recognizedInputBox: false };
440
493
  }
441
494
  async deliveryReadMatches(pane, cmd, cwd) {
442
495
  const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
@@ -24,7 +24,7 @@ import { acquireRunLock, releaseRunLock } from "./lock.js";
24
24
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
25
25
  import { nextChannel, route } from "../route/router.js";
26
26
  import { desiredPanes } from "./reconcile.js";
27
- import { normalizeStallSnapshot } from "./stall.js";
27
+ import { StallProgressTracker } from "./stall.js";
28
28
  const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
29
29
  // An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
30
30
  // carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
@@ -795,13 +795,13 @@ export async function runDaemon(repoRoot, opts = {}) {
795
795
  // single site, so a test can reason about it; keep Date.now() out of profile.ts (still pure) and
796
796
  // out of adapter module scope (the cursor is a parameter, threaded from the daemon).
797
797
  const attemptStart = Date.now();
798
- // v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing poll
799
- // seams (interactive wait slices) — never a new timer loop. null/unknown usage fails OPEN
798
+ // v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing worker
799
+ // wait slices — never a new timer loop. null/unknown usage fails OPEN
800
800
  // (never treated as over-threshold). Journal + notify fire at most once while the value stays high.
801
801
  let contextWarned = false;
802
802
  let contextTokens;
803
803
  const sampleContext = async () => {
804
- if (contextWarned || !adapter.contextUsage)
804
+ if (!adapter.contextUsage)
805
805
  return;
806
806
  let usage = null;
807
807
  try {
@@ -814,7 +814,7 @@ export async function runDaemon(repoRoot, opts = {}) {
814
814
  if (!usage || typeof usage.tokens !== "number" || !Number.isFinite(usage.tokens))
815
815
  return;
816
816
  contextTokens = usage.tokens; // last known valid sample, including under-threshold resume candidates
817
- if (usage.tokens < cfg.contextWarnTokens)
817
+ if (contextWarned || usage.tokens < cfg.contextWarnTokens)
818
818
  return;
819
819
  contextWarned = true;
820
820
  lastContextTokens = usage.tokens;
@@ -862,15 +862,15 @@ export async function runDaemon(repoRoot, opts = {}) {
862
862
  // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
863
863
  // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
864
864
  const stallWindowMs = taskTimeoutMinutes * 60_000;
865
- // OBS-82: the stall clock compares NORMALIZED snapshots so a spinner glyph/elapsed-time
866
- // repaint is silence, not activity. ONLY this inactivity compare sees normalized text —
867
- // trailer detection, harvest, paging, and quota checks all read the raw pane.
865
+ // v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
866
+ // the stall clock. Raw pane differences are terminal chrome until proven otherwise.
868
867
  let everHadOutput = output.length > 0;
869
- let lastStallSnapshot = normalizeStallSnapshot(output);
870
- let lastOutputAt = Date.now();
871
- while (Date.now() - lastOutputAt < stallWindowMs) {
868
+ const stallProgress = new StallProgressTracker();
869
+ stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
870
+ let lastProgressAt = Date.now();
871
+ while (Date.now() - lastProgressAt < stallWindowMs) {
872
872
  const sliceStart = Date.now();
873
- const remaining = stallWindowMs - (sliceStart - lastOutputAt);
873
+ const remaining = stallWindowMs - (sliceStart - lastProgressAt);
874
874
  let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
875
875
  if (!everHadOutput) {
876
876
  const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
@@ -893,11 +893,6 @@ export async function runDaemon(repoRoot, opts = {}) {
893
893
  const paneText = await driver.read(slot, 1000);
894
894
  if (paneText.length > 0)
895
895
  everHadOutput = true;
896
- const currentStallSnapshot = normalizeStallSnapshot(paneText);
897
- if (currentStallSnapshot !== lastStallSnapshot) {
898
- lastStallSnapshot = currentStallSnapshot;
899
- lastOutputAt = Date.now();
900
- }
901
896
  // OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
902
897
  if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
903
898
  earlyLaunchDead = true;
@@ -906,6 +901,8 @@ export async function runDaemon(repoRoot, opts = {}) {
906
901
  }
907
902
  // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
908
903
  await sampleContext();
904
+ if (stallProgress.observe({ paneText, contextTokens }))
905
+ lastProgressAt = Date.now();
909
906
  // page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
910
907
  // (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
911
908
  // a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
@@ -943,7 +940,7 @@ export async function runDaemon(repoRoot, opts = {}) {
943
940
  }
944
941
  if (!finished && exitCode === null) {
945
942
  // timed out (or only ever saw false positives): harvest whatever the pane holds now
946
- timedOut = Date.now() - lastOutputAt >= stallWindowMs;
943
+ timedOut = Date.now() - lastProgressAt >= stallWindowMs;
947
944
  output = await driver.read(slot, 1000);
948
945
  finished = new RegExp(trailerPattern(nonce)).test(output);
949
946
  const exit = exitRe.exec(output);
@@ -981,16 +978,16 @@ export async function runDaemon(repoRoot, opts = {}) {
981
978
  else {
982
979
  await driver.run(slot, paneDispatchCommand(dispatchScript));
983
980
  // OBS-54: headless workers have the same output-inactivity budget as visible panes.
984
- // OBS-82: same normalized-snapshot compare as the interactive site — spinner-only repaints
985
- // exhaust the budget here too; harvest below still reads the raw pane.
981
+ // v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
986
982
  const stallWindowMs = taskTimeoutMinutes * 60_000;
987
983
  const initialPane = await driver.read(slot, 500);
988
984
  let everHadOutput = initialPane.length > 0;
989
- let lastStallSnapshot = normalizeStallSnapshot(initialPane);
990
- let lastOutputAt = Date.now();
985
+ const stallProgress = new StallProgressTracker();
986
+ stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
987
+ let lastProgressAt = Date.now();
991
988
  finished = false;
992
- while (Date.now() - lastOutputAt < stallWindowMs) {
993
- const remaining = stallWindowMs - (Date.now() - lastOutputAt);
989
+ while (Date.now() - lastProgressAt < stallWindowMs) {
990
+ const remaining = stallWindowMs - (Date.now() - lastProgressAt);
994
991
  let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
995
992
  if (!everHadOutput) {
996
993
  const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
@@ -1004,19 +1001,17 @@ export async function runDaemon(repoRoot, opts = {}) {
1004
1001
  const paneText = await driver.read(slot, 500);
1005
1002
  if (paneText.length > 0)
1006
1003
  everHadOutput = true;
1007
- const currentStallSnapshot = normalizeStallSnapshot(paneText);
1008
- if (currentStallSnapshot !== lastStallSnapshot) {
1009
- lastStallSnapshot = currentStallSnapshot;
1010
- lastOutputAt = Date.now();
1011
- }
1012
1004
  if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
1013
1005
  earlyLaunchDead = true;
1014
1006
  break;
1015
1007
  }
1008
+ await sampleContext();
1009
+ if (stallProgress.observe({ paneText, contextTokens }))
1010
+ lastProgressAt = Date.now();
1016
1011
  }
1017
1012
  output = await driver.read(slot, 500);
1018
1013
  exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
1019
- timedOut = !finished && Date.now() - lastOutputAt >= stallWindowMs;
1014
+ timedOut = !finished && Date.now() - lastProgressAt >= stallWindowMs;
1020
1015
  }
1021
1016
  // SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
1022
1017
  // shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
@@ -1,8 +1,25 @@
1
- /** Normalize one pane snapshot for the stall-inactivity compare (trailer parsing, harvest,
2
- * waitOutput, and paging read the raw text; the LLM transcript filter below reuses this to
3
- * CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
4
- * equal are the same frame modulo spinner presentation; any other byte difference is activity. */
1
+ /** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
2
+ * parsing, harvest, waitOutput, and paging always read the raw text. */
5
3
  export declare function normalizeStallSnapshot(text: string): string;
4
+ export interface StallProgressSample {
5
+ paneText: string;
6
+ seedSubmitted?: boolean;
7
+ contextTokens?: number;
8
+ }
9
+ /**
10
+ * Monotonic worker-progress measure for the stall watchdog.
11
+ *
12
+ * Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
13
+ * evidence of work. A rendered transcript is only known to have grown when it occupies more
14
+ * non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
15
+ * advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
16
+ */
17
+ export declare class StallProgressTracker {
18
+ private transcriptRows;
19
+ private seedSubmitted;
20
+ private contextTokens;
21
+ observe(sample: StallProgressSample): boolean;
22
+ }
6
23
  /** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
7
24
  * seam exists for fault injection in tests only — production callers pass text alone. */
8
25
  export declare function filterLlmTranscript(text: string, classify?: (t: string) => string): string;
package/dist/run/stall.js CHANGED
@@ -1,12 +1,9 @@
1
- // OBS-82: codex's MCP-startup spinner repaints a braille glyph + elapsed-time cell forever, so the
2
- // daemon's raw snapshot compare reads a wedged pane as active and the stall clock never fires.
3
- // This normalizer deletes ONLY presentation tokens from a closed allowlist — ANSI/VT escape
4
- // sequences, braille-range spinner glyphs, and elapsed-time tokens bound to time-unit suffixes.
5
- // Every other byte passes through identical: words, paths, server names, and progress counts
6
- // (a five-of-seven counter change IS activity) all remain change-sensitive. The asymmetry is the
7
- // design: an allowlist MISS degrades to today's recoverable no-reap behavior, while an over-broad
8
- // deletion would reap a healthy worker — a new failure class. Grow the allowlist only with
9
- // captured evidence (tests/fixtures/codex-mcp-spinner/).
1
+ // OBS-82: normalize known presentation tokens before measuring transcript extent or filtering an
2
+ // LLM-bound transcript. This remains a closed allowlist — ANSI/VT escapes, braille-range spinner
3
+ // glyphs, and elapsed-time tokens bound to time-unit suffixes. Every other byte passes through
4
+ // identical. v1.76 deliberately stopped treating arbitrary normalized byte changes as progress:
5
+ // StallProgressTracker below requires monotonic evidence, so an unknown repaint fails closed toward
6
+ // a recoverable consult instead of holding the watchdog silent.
10
7
  // CSI (with intermediates), OSC (BEL- or ST-terminated), DCS/SOS/PM/APC strings, single-char
11
8
  // escapes, and charset selection — the raw-pty forms; herdr pane reads are already rendered.
12
9
  // eslint-disable-next-line no-control-regex
@@ -16,13 +13,46 @@ const SPINNER_RE = /[⠀-⣿]/g;
16
13
  // A digit run (optionally decimal) bound directly to a time-unit suffix, standing alone as a
17
14
  // word: 9s, 41s, 3m, 1h, 800ms. Never bare digits — "(6/7)" and "5 of 7" stay change-sensitive.
18
15
  const ELAPSED_RE = /(?<![\w.])\d+(?:\.\d+)?(?:ms|[hms])(?!\w)/g;
19
- /** Normalize one pane snapshot for the stall-inactivity compare (trailer parsing, harvest,
20
- * waitOutput, and paging read the raw text; the LLM transcript filter below reuses this to
21
- * CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
22
- * equal are the same frame modulo spinner presentation; any other byte difference is activity. */
16
+ /** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
17
+ * parsing, harvest, waitOutput, and paging always read the raw text. */
23
18
  export function normalizeStallSnapshot(text) {
24
19
  return text.replace(ANSI_RE, "").replace(SPINNER_RE, "").replace(ELAPSED_RE, "");
25
20
  }
21
+ /**
22
+ * Monotonic worker-progress measure for the stall watchdog.
23
+ *
24
+ * Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
25
+ * evidence of work. A rendered transcript is only known to have grown when it occupies more
26
+ * non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
27
+ * advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
28
+ */
29
+ export class StallProgressTracker {
30
+ transcriptRows = 0;
31
+ seedSubmitted = false;
32
+ contextTokens;
33
+ observe(sample) {
34
+ let advanced = false;
35
+ const rows = normalizeStallSnapshot(sample.paneText)
36
+ .split("\n")
37
+ .filter((line) => line.trim().length > 0)
38
+ .length;
39
+ if (rows > this.transcriptRows) {
40
+ this.transcriptRows = rows;
41
+ advanced = true;
42
+ }
43
+ if (sample.seedSubmitted && !this.seedSubmitted) {
44
+ this.seedSubmitted = true;
45
+ advanced = true;
46
+ }
47
+ const tokens = sample.contextTokens;
48
+ if (tokens !== undefined && Number.isFinite(tokens)) {
49
+ if (tokens > (this.contextTokens ?? 0))
50
+ advanced = true;
51
+ this.contextTokens = Math.max(this.contextTokens ?? 0, tokens);
52
+ }
53
+ return advanced;
54
+ }
55
+ }
26
56
  // ─── v1.65 T2: LLM-bound transcript filter ──────────────────────────────────────────────────────
27
57
  // Consult dossiers and gate prompts pay tokens per transcript byte, so LLM-bound text runs through
28
58
  // a per-line classifier: carriage-return overwrite churn keeps only the final paint, lines that are
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "1.74.0",
3
+ "version": "1.76.0",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -15,8 +15,9 @@ When working in a multi-agent terminal environment, decide your role before star
15
15
  - **Orchestrator:** your session was started to execute the mission. Rename your own tab/pane `ORCH · <version>` (short labels: ≤20 chars, `ROLE · token`) and run the loop below.
16
16
  - **Supervisor with a live orchestrator:** do not start a second run. Relay the mission to the existing orchestrator with a [verified handoff](#verified-handoffs-agent-to-agent-messaging), then supervise it as OVERSEER.
17
17
  - **Primary session without an orchestrator:** rename your own tab `OVERSEER · <version>` and your agent `overseer`, spawn one child orchestration session with your host's launch form, label its tab `ORCH · <version>` and name its agent, give it the mission and these rules verbatim, then supervise it. Do not drive a duplicate single-tier run yourself. Before spawning, confirm any PREVIOUS orchestrator has stood down (monitors stopped, input box empty — dim ghost-text suggestions are UI, not queued input; ANSI-verify before alarming) and close its tab — the journal, records, and ledger hold the story; scrollback is disposable.
18
- - **Claude Code:** `herdr agent start orchestrator --cwd <repo> --no-focus -- claude --permission-mode bypassPermissions`
19
- - **Codex:** `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
18
+ - **Spawning on current herdr is two-step** — the one-shot `agent start --cwd/--tab/--no-focus` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138). First create the pane: `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` (tab create does not steal focus unless `--focus` is passed; parse `result.root_pane.pane_id` from its JSON), then start the agent in it:
19
+ - **Claude Code:** `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions`
20
+ - **Codex:** `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
20
21
  - **Auxiliary agents you spawn (consultants, reviewers, scouts) follow the same forms.** Never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions` (tickmarkr's own adapter uses exactly this for workers, judges, and consults). A read-only codex consultant may use `--sandbox read-only`; any codex session that must touch git needs the unsandboxed flag above.
21
22
 
22
23
  Outside a multi-agent terminal environment, run the loop directly.
@@ -14,8 +14,9 @@ When working in a multi-agent terminal environment, decide your role before star
14
14
  - **Orchestrator:** your session was started to execute the mission. Rename your own tab/pane `ORCH · <version>` (short labels: ≤20 chars, `ROLE · token`) and run the loop below.
15
15
  - **Supervisor with a live orchestrator:** do not start a second run. Relay the mission to the existing orchestrator with a [verified handoff](#verified-handoffs-agent-to-agent-messaging), then supervise it as OVERSEER.
16
16
  - **Primary session without an orchestrator:** rename your own tab `OVERSEER · <version>` and your agent `overseer`, spawn one child orchestration session with your host's launch form, label its tab `ORCH · <version>` and name its agent, give it the mission and these rules verbatim, then supervise it. Do not drive a duplicate single-tier run yourself. Before spawning, confirm any PREVIOUS orchestrator has [stood down](#stand-down-mission-end-and-retirement) and close its tab.
17
- - **Claude Code:** `herdr agent start orchestrator --cwd <repo> --no-focus -- claude --permission-mode bypassPermissions`
18
- - **Codex:** `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
17
+ - **Spawning on current herdr is two-step** — the one-shot `agent start --cwd/--tab/--no-focus` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138). First create the pane: `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` (tab create does not steal focus unless `--focus` is passed; parse `result.root_pane.pane_id` from its JSON), then start the agent in it:
18
+ - **Claude Code:** `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions`
19
+ - **Codex:** `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
19
20
  - **Auxiliary agents you spawn (consultants, reviewers, scouts) follow the same forms.** Never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions` (tickmarkr's own adapter uses exactly this for workers, judges, and consults). A read-only codex consultant may use `--sandbox read-only`; any codex session that must touch git needs the unsandboxed flag above.
20
21
 
21
22
  Outside a multi-agent terminal environment, run the loop directly.
@@ -29,7 +29,7 @@ Requires `HERDR_ENV=1`; if unset, say so and stop.
29
29
  main name plus at most ONE hot-state token. Vocabulary: ORCH carries the milestone and progress
30
30
  fraction (`ORCH · v1.19 4/5`, updated on every task-done); WORKERS carries the task token (tickmarkr
31
31
  updates it). Never long context strings or ✓-chains.
32
- 2. **Orchestrator**: Launch the orchestrator with your agent host. For Claude Code, use `herdr agent start orchestrator --cwd <repo> --no-focus -- claude --permission-mode bypassPermissions` (pin a strong model with `--model <m>` if the operator has a policy). For Codex, use `herdr agent start orchestrator --cwd <repo> --no-focus -- codex --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions`, and a read-only codex consultant may use `--sandbox read-only`.
32
+ 2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions`, and a read-only codex consultant may use `--sandbox read-only`.
33
33
  3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
34
34
  truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
35
35
  (inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then send one line: