tickmarkr 1.74.0 → 1.76.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code.d.ts +2 -1
- package/dist/adapters/claude-code.js +10 -6
- package/dist/adapters/codex.d.ts +2 -1
- package/dist/adapters/codex.js +7 -0
- package/dist/adapters/kimi.d.ts +6 -0
- package/dist/adapters/kimi.js +37 -11
- package/dist/adapters/types.d.ts +7 -0
- package/dist/adapters/types.js +12 -0
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +29 -2
- package/dist/cli/commands/fleet.js +11 -1
- package/dist/drivers/herdr.d.ts +3 -0
- package/dist/drivers/herdr.js +71 -18
- package/dist/run/daemon.js +25 -30
- package/dist/run/stall.d.ts +21 -4
- package/dist/run/stall.js +43 -13
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +3 -2
- package/skills/tickmarkr-loop/SKILL.md +3 -2
- package/skills/tickmarkr-overseer/SKILL.md +1 -1
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { type AuthHealth, type WorkerAdapter } from "./types.js";
|
|
1
|
+
import { type AuthHealth, type TrustDialog, type WorkerAdapter } from "./types.js";
|
|
2
|
+
export declare const CLAUDE_TRUST_DIALOG: TrustDialog;
|
|
2
3
|
export declare function claudeSlug(real: string): string;
|
|
3
4
|
export declare function probeVersion(bin: string): AuthHealth;
|
|
4
5
|
export declare const claudeCode: WorkerAdapter;
|
|
@@ -20,6 +20,12 @@ import { channelsFromConfig, shq, TokenUsageSchema } from "./types.js";
|
|
|
20
20
|
// Phase 18's operator-price × tokens derivation, not a CLI claim.
|
|
21
21
|
const MAX_SESSION_FILES = 20; // newest-first; a long-lived project dir can hold many sessions
|
|
22
22
|
const MAX_SESSION_BYTES = 8_000_000; // per-file cap; a runaway JSONL cannot make the read unbounded
|
|
23
|
+
// v1.75 T2 / OBS-137: current Claude Code workspace-trust prompt (2.1.218). The full question
|
|
24
|
+
// distinguishes this startup gate from routine agent text; Enter accepts the selected trust option.
|
|
25
|
+
export const CLAUDE_TRUST_DIALOG = {
|
|
26
|
+
fingerprint: "Quick safety check: Is this a project you created or one you trust?",
|
|
27
|
+
key: "Enter",
|
|
28
|
+
};
|
|
23
29
|
export function claudeSlug(real) {
|
|
24
30
|
return real.replace(/[^A-Za-z0-9]/g, "-");
|
|
25
31
|
}
|
|
@@ -54,13 +60,11 @@ export const claudeCode = {
|
|
|
54
60
|
// and --mcp-config is VARIADIC — a positional after it is eaten as a config-file path, so another
|
|
55
61
|
// flag must always follow the value, never the prompt.
|
|
56
62
|
headlessCommand: (promptFile, model) => `claude -p "$(cat ${shq(promptFile)})" --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text`,
|
|
57
|
-
// HYG-03: the residual first-entry dialog
|
|
58
|
-
//
|
|
59
|
-
//
|
|
60
|
-
// to that file (a seed races claude's own writes, nondeterministically). Amortizes to one operator dismissal
|
|
61
|
-
// per stable worktree path; blocked-pane paging surfaces it. Do NOT change this command to "fix" the dialog —
|
|
62
|
-
// see .planning/REQUIREMENTS.md HYG-03 and 21-02-LIVE-CHECK.md. Revisit if upstream ships a --trust flag.
|
|
63
|
+
// HYG-03 / OBS-137: the residual first-entry dialog is workspace trust, not MCP config loading.
|
|
64
|
+
// Claude's only store is global last-writer-wins ~/.claude.json, so tickmarkr still does not seed it;
|
|
65
|
+
// the daemon safely answers only the exact adapter-declared dialog once per slot.
|
|
63
66
|
interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
67
|
+
trustDialog: CLAUDE_TRUST_DIALOG,
|
|
64
68
|
resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
65
69
|
invoke(task, _cwd, a, ctx) {
|
|
66
70
|
return { command: this.headlessCommand(ctx.promptFile, a.model) };
|
package/dist/adapters/codex.d.ts
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
|
-
import { type TrustVerdict, type WorkerAdapter } from "./types.js";
|
|
1
|
+
import { type TrustDialog, type TrustVerdict, type WorkerAdapter } from "./types.js";
|
|
2
2
|
export declare function readCodexModelsCache(path?: string): {
|
|
3
3
|
models: string[];
|
|
4
4
|
fetchedAt?: string;
|
|
5
5
|
};
|
|
6
|
+
export declare const CODEX_TRUST_DIALOG: TrustDialog;
|
|
6
7
|
export declare function seedCodexTrust(repoRoot: string, configPath?: string): TrustVerdict;
|
|
7
8
|
export declare function hasCodexTrustedProject(text: string, root: string): boolean;
|
|
8
9
|
export declare function codexConfigMcpServerNames(configPath?: string): string[];
|
package/dist/adapters/codex.js
CHANGED
|
@@ -89,6 +89,12 @@ const GITDIR_WRITABLE = `-c "sandbox_workspace_write.writable_roots=[\\"$(git re
|
|
|
89
89
|
// -s/--sandbox workspace-write sandbox (deliberately NOT --dangerously-bypass-approvals-and-sandbox,
|
|
90
90
|
// which would drop the sandbox). Listed by `codex --help` and `codex exec --help` (verified 2026-07-23).
|
|
91
91
|
const CODEX_HOOK_TRUST = "--dangerously-bypass-hook-trust";
|
|
92
|
+
// v1.75 T2 / OBS-137: current Codex workspace-trust prompt (0.144.6). The exact heading
|
|
93
|
+
// is distinct from normal agent output; Enter accepts the selected "Yes, continue" option.
|
|
94
|
+
export const CODEX_TRUST_DIALOG = {
|
|
95
|
+
fingerprint: "Do you trust the contents of this directory?",
|
|
96
|
+
key: "Enter",
|
|
97
|
+
};
|
|
92
98
|
// v1.22 T5 / OBS-16: codex keys trust on absolute path under [projects."<root>"] trust_level="trusted"
|
|
93
99
|
// in ~/.codex/config.toml (CODEX_HOME relocates the dir). Worktrees inherit parent-project trust when
|
|
94
100
|
// the REPO ROOT is trusted — seed the root once, cover every future worktree. Idempotent: a second
|
|
@@ -196,6 +202,7 @@ export const codex = {
|
|
|
196
202
|
// v1.22 T5: seed [projects."<repoRoot>"] trust_level="trusted" so fresh worktrees never stall on
|
|
197
203
|
// "Do you trust this directory?" (OBS-16). doctor-only side effect.
|
|
198
204
|
trust: (repoRoot) => seedCodexTrust(repoRoot),
|
|
205
|
+
trustDialog: CODEX_TRUST_DIALOG,
|
|
199
206
|
// v1.5 MODEL-01: file read only (no `codex models` subcommand exists, verified 2026-07-10).
|
|
200
207
|
// Already fails OPEN to [] internally — advisory detection, unlike gates' fail-closed.
|
|
201
208
|
listModels: async () => readCodexModelsCache().models,
|
package/dist/adapters/kimi.d.ts
CHANGED
|
@@ -3,6 +3,11 @@ import { type Assignment, type WorkerAdapter, type WorkerResult } from "./types.
|
|
|
3
3
|
export declare function kimiAuthed(credentialsText: string, nowMs: number): boolean;
|
|
4
4
|
export declare function parseKimiModels(raw: string): string[];
|
|
5
5
|
export declare function parseKimiResult(raw: string, nonce: string): WorkerResult;
|
|
6
|
+
export interface KimiDoctorTurnResult {
|
|
7
|
+
ok: boolean;
|
|
8
|
+
evidence: string;
|
|
9
|
+
}
|
|
10
|
+
export declare function probeKimiDoctorTurn(cwd: string): Promise<KimiDoctorTurnResult>;
|
|
6
11
|
export declare function kimiSessionId(output: string): string | undefined;
|
|
7
12
|
export declare function kimiBannerModel(banner: string): string | undefined;
|
|
8
13
|
export declare function kimiBannerSessionId(banner: string): string | undefined;
|
|
@@ -20,6 +25,7 @@ export interface KimiInteractiveSeedResult {
|
|
|
20
25
|
seedError?: string;
|
|
21
26
|
sessionId?: string;
|
|
22
27
|
}
|
|
28
|
+
export declare const KIMI_INPUT_BOX: import("./types.js").InputBox;
|
|
23
29
|
export declare function runKimiInteractiveSeed(opts: {
|
|
24
30
|
driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
|
|
25
31
|
slot: Slot;
|
package/dist/adapters/kimi.js
CHANGED
|
@@ -5,7 +5,7 @@ import { probeVersion } from "./claude-code.js";
|
|
|
5
5
|
import { parseWorkerResult } from "./prompt.js";
|
|
6
6
|
import { runInteractiveSeed } from "../run/interactive-seed.js";
|
|
7
7
|
import { sh } from "../run/git.js";
|
|
8
|
-
import { channelsFromConfig, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
|
|
8
|
+
import { channelsFromConfig, declareInputBox, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
|
|
9
9
|
// KIMI-03 → v1.58 T5: the "no harness-readable counter" block (research F-6, 2026-07-17) is
|
|
10
10
|
// LIFTED for collectUsage — kimi 0.27.0 writes a wire journal per agent at
|
|
11
11
|
// ~/.kimi-code/sessions/<wd>/session_<uuid>/agents/<agent>/wire.jsonl, and ~/.kimi-code/
|
|
@@ -64,6 +64,28 @@ export function parseKimiResult(raw, nonce) {
|
|
|
64
64
|
const stripped = raw.split("\n").map((l) => l.replace(/^[\s]*[•*-]\s+/, "")).join("\n");
|
|
65
65
|
return parseWorkerResult(stripped, nonce);
|
|
66
66
|
}
|
|
67
|
+
const KIMI_DOCTOR_TURN_MODEL = "kimi-code/k3";
|
|
68
|
+
const KIMI_DOCTOR_TURN_PROMPT = "Reply with exactly OK and nothing else.";
|
|
69
|
+
const KIMI_DOCTOR_TURN_TIMEOUT_MS = 60000;
|
|
70
|
+
// OBS-141: intentionally separate from probe() so plan/run remain free file checks. Only doctor
|
|
71
|
+
// calls this one-turn contract probe; its test seam stubs sh and never launches a real agent CLI.
|
|
72
|
+
export async function probeKimiDoctorTurn(cwd) {
|
|
73
|
+
const command = `kimi -p ${shq(KIMI_DOCTOR_TURN_PROMPT)} --model ${shq(KIMI_DOCTOR_TURN_MODEL)} --output-format text`;
|
|
74
|
+
const r = await sh(command, cwd, KIMI_DOCTOR_TURN_TIMEOUT_MS);
|
|
75
|
+
if (r.timedOut) {
|
|
76
|
+
return { ok: false, evidence: `turn timed out after ${KIMI_DOCTOR_TURN_TIMEOUT_MS}ms` };
|
|
77
|
+
}
|
|
78
|
+
const output = `${r.stderr}\n${r.stdout}`.trim().replace(/\s+/g, " ");
|
|
79
|
+
if (r.code !== 0) {
|
|
80
|
+
return { ok: false, evidence: output || `turn exited ${r.code}` };
|
|
81
|
+
}
|
|
82
|
+
const returnedOk = r.stdout.split("\n")
|
|
83
|
+
.map((line) => line.replace(/^[\s]*[•*-]\s+/, "").trim())
|
|
84
|
+
.includes("OK");
|
|
85
|
+
return returnedOk
|
|
86
|
+
? { ok: true, evidence: `model turn returned OK with ${KIMI_DOCTOR_TURN_MODEL}` }
|
|
87
|
+
: { ok: false, evidence: `turn returned no exact OK answer${output ? `: ${output}` : ""}` };
|
|
88
|
+
}
|
|
67
89
|
// v1.53 T3: session-id capture from the run-output trailer — every `kimi -p` run (fresh or resumed)
|
|
68
90
|
// ends with `To resume this session: kimi -r session_<uuid>` (live probe 2026-07-18). Anchored full
|
|
69
91
|
// line only: prompt/model prose can contain lookalike text, and the anchored charset keeps a
|
|
@@ -89,23 +111,20 @@ export function kimiSessionId(output) {
|
|
|
89
111
|
}
|
|
90
112
|
return id;
|
|
91
113
|
}
|
|
92
|
-
// v1.69 T6: the native TUI takes -m <alias>, where config.toml aliases are the bare model suffix of
|
|
93
|
-
// the tickmarkr channel id (live probe 2026-07-22). Keep the mapping explicit and localized.
|
|
94
|
-
function kimiAlias(model) {
|
|
95
|
-
return model.replace(/^kimi-code\//, "");
|
|
96
|
-
}
|
|
97
114
|
// v1.69 T7: the cold-start banner prints the model alias and session id. Parse them from the
|
|
98
|
-
// banner text already captured for the readiness match — no new probe, no extra dispatch.
|
|
115
|
+
// banner text already captured for the readiness match — no new probe, no extra dispatch. Kimi
|
|
116
|
+
// 0.29.0 may print either the full config key or its display suffix; normalize both idempotently
|
|
117
|
+
// to the full channel identifier routing uses.
|
|
99
118
|
const BANNER_MODEL_RE = /^Model:\s*(.+)$/m;
|
|
100
119
|
const BANNER_SESSION_RE = /^Session:\s*(session_[0-9a-f-]+)$/m;
|
|
101
120
|
export function kimiBannerModel(banner) {
|
|
102
121
|
const m = BANNER_MODEL_RE.exec(banner);
|
|
103
122
|
if (!m)
|
|
104
123
|
return undefined;
|
|
105
|
-
const
|
|
106
|
-
if (!
|
|
124
|
+
const printedModel = m[1].trim();
|
|
125
|
+
if (!printedModel)
|
|
107
126
|
return undefined;
|
|
108
|
-
return `kimi-code/${
|
|
127
|
+
return printedModel.startsWith("kimi-code/") ? printedModel : `kimi-code/${printedModel}`;
|
|
109
128
|
}
|
|
110
129
|
export function kimiBannerSessionId(banner) {
|
|
111
130
|
return BANNER_SESSION_RE.exec(banner)?.[1];
|
|
@@ -118,10 +137,16 @@ export function confirmKimiSeedBanner(banner, assignedModel) {
|
|
|
118
137
|
}
|
|
119
138
|
return { ok: true, sessionId: kimiBannerSessionId(banner) };
|
|
120
139
|
}
|
|
140
|
+
// v1.75 T1 / OBS-136: this readiness line is rendered inside Kimi Code's bordered steady-state
|
|
141
|
+
// input box. The adapter owns the fingerprint; the driver only consults declarations generically.
|
|
142
|
+
export const KIMI_INPUT_BOX = declareInputBox("kimi", {
|
|
143
|
+
fingerprint: "Send /help for help information.",
|
|
144
|
+
});
|
|
121
145
|
// Shared launch-then-seed surface (T6) + banner confirm (T7/T2). One definition so the adapter
|
|
122
146
|
// property and the daemon's generic runInteractiveSeed path cannot drift.
|
|
123
147
|
const KIMI_SEED = {
|
|
124
|
-
|
|
148
|
+
// v1.76 T2 / OBS-141: 0.29.0 resolves only the full config.toml model key on the first turn.
|
|
149
|
+
launch: (model) => `kimi -y -m ${shq(model)}`,
|
|
125
150
|
readinessMatch: "Send /help for help information.",
|
|
126
151
|
seedLine: (promptFile) => `Read ${promptFile} and do exactly what it says.`,
|
|
127
152
|
confirmBanner: confirmKimiSeedBanner,
|
|
@@ -165,6 +190,7 @@ export const kimi = {
|
|
|
165
190
|
// prompt as one user turn. Banner model/session confirmation runs on the daemon's generic
|
|
166
191
|
// runInteractiveSeed path via confirmBanner on KIMI_SEED (not a separate dispatch helper).
|
|
167
192
|
interactiveSeed: KIMI_SEED,
|
|
193
|
+
inputBox: KIMI_INPUT_BOX,
|
|
168
194
|
// v1.53 T3 resume — live-probed 2026-07-18: `-p` + `-S <id>` compose cleanly (no OBS-67-class
|
|
169
195
|
// flag rejection) and the resumed session carries prior conversation state. `-S <id>` is the
|
|
170
196
|
// deterministic form; `-c` rejected as primary — cwd-keyed, nondeterministic under worktree
|
package/dist/adapters/types.d.ts
CHANGED
|
@@ -82,6 +82,12 @@ export interface TrustDialog {
|
|
|
82
82
|
key: string;
|
|
83
83
|
}
|
|
84
84
|
export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): boolean;
|
|
85
|
+
export interface InputBox {
|
|
86
|
+
fingerprint: string;
|
|
87
|
+
}
|
|
88
|
+
export declare function declareInputBox(adapterId: string, inputBox: InputBox): InputBox;
|
|
89
|
+
export declare function declaredInputBoxForWorkerName(workerName: string): InputBox | undefined;
|
|
90
|
+
export declare function matchesInputBox(paneText: string, inputBox: InputBox): boolean;
|
|
85
91
|
export interface WorkerAdapter {
|
|
86
92
|
id: string;
|
|
87
93
|
vendor: string;
|
|
@@ -105,6 +111,7 @@ export interface WorkerAdapter {
|
|
|
105
111
|
contextUsage?(session: SessionRef): ContextUsage | null;
|
|
106
112
|
trust?(repoRoot: string): TrustVerdict;
|
|
107
113
|
trustDialog?: TrustDialog;
|
|
114
|
+
inputBox?: InputBox;
|
|
108
115
|
hardcodedFlags?: {
|
|
109
116
|
binary: string;
|
|
110
117
|
flags: string[];
|
package/dist/adapters/types.js
CHANGED
|
@@ -30,6 +30,18 @@ export function modelAuthed(health, model, allowUnverifiedModels = false) {
|
|
|
30
30
|
export function matchesTrustDialog(paneText, dialog) {
|
|
31
31
|
return paneText.includes(dialog.fingerprint);
|
|
32
32
|
}
|
|
33
|
+
const inputBoxes = new Map();
|
|
34
|
+
export function declareInputBox(adapterId, inputBox) {
|
|
35
|
+
inputBoxes.set(adapterId, inputBox);
|
|
36
|
+
return inputBox;
|
|
37
|
+
}
|
|
38
|
+
export function declaredInputBoxForWorkerName(workerName) {
|
|
39
|
+
const adapterId = /^.+-worker-(.+)-a\d+-.+$/.exec(workerName)?.[1];
|
|
40
|
+
return adapterId === undefined ? undefined : inputBoxes.get(adapterId);
|
|
41
|
+
}
|
|
42
|
+
export function matchesInputBox(paneText, inputBox) {
|
|
43
|
+
return paneText.includes(inputBox.fingerprint);
|
|
44
|
+
}
|
|
33
45
|
export function channelsFromConfig(adapterId, cfg) {
|
|
34
46
|
const e = cfg.tiers[adapterId];
|
|
35
47
|
if (!e)
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import type { WorkerAdapter } from "../../adapters/types.js";
|
|
2
|
+
import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
|
|
2
3
|
export type DoctorOpts = {
|
|
3
4
|
banner?: boolean;
|
|
5
|
+
kimiTurnProbe?: (cwd: string) => Promise<KimiDoctorTurnResult>;
|
|
4
6
|
};
|
|
5
7
|
export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
|
|
@@ -6,6 +6,7 @@ import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
|
|
|
6
6
|
import { declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
|
|
7
7
|
import { DEFAULT_CONFIG, loadConfig, overlayPreferShapes } from "../../config/config.js";
|
|
8
8
|
import { HerdrDriver } from "../../drivers/herdr.js";
|
|
9
|
+
import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
|
|
9
10
|
import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
|
|
10
11
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
11
12
|
const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
|
|
@@ -20,6 +21,26 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
20
21
|
console.error("probing installed agent CLIs — one short LLM call per configured model, may take a minute...");
|
|
21
22
|
const probeProgressTTY = process.stderr.isTTY === true;
|
|
22
23
|
const health = await probeAll(adapters, { cwd });
|
|
24
|
+
const kimiAdapter = adapters.find((a) => a.id === kimi.id);
|
|
25
|
+
const kimiTurnEnabled = kimiAdapter !== undefined
|
|
26
|
+
&& (kimiAdapter === kimi || opts.kimiTurnProbe !== undefined);
|
|
27
|
+
if (kimiTurnEnabled) {
|
|
28
|
+
const h = health.kimi;
|
|
29
|
+
if (h.installed && h.authed) {
|
|
30
|
+
let turn;
|
|
31
|
+
try {
|
|
32
|
+
turn = await (opts.kimiTurnProbe ?? probeKimiDoctorTurn)(cwd);
|
|
33
|
+
}
|
|
34
|
+
catch (e) {
|
|
35
|
+
turn = { ok: false, evidence: e instanceof Error ? e.message : String(e) };
|
|
36
|
+
}
|
|
37
|
+
health.kimi = {
|
|
38
|
+
...h,
|
|
39
|
+
authed: turn.ok,
|
|
40
|
+
note: `${h.note ? `${h.note}; ` : ""}${turn.ok ? turn.evidence : `model turn failed: ${turn.evidence}`}`,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
}
|
|
23
44
|
// MODEL-02: detect models where the adapter exposes a list surface, BEFORE writing doctor.json (write once, below).
|
|
24
45
|
// Fail OPEN — the inverse of gates' fail-closed: detection is advisory, so a broken list surface NEVER fails doctor.
|
|
25
46
|
for (const a of adapters) {
|
|
@@ -35,14 +56,20 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
35
56
|
}
|
|
36
57
|
catch { /* fail open: leave models as-is, doctor stays healthy */ }
|
|
37
58
|
}
|
|
38
|
-
|
|
59
|
+
// A free Kimi auth failure or failed earned-green turn must not spend more probes. Every other
|
|
60
|
+
// adapter keeps the exact existing model-sweep path.
|
|
61
|
+
const modelProbeAdapters = kimiTurnEnabled && health.kimi.authed === false
|
|
62
|
+
? adapters.filter((a) => a !== kimiAdapter)
|
|
63
|
+
: adapters;
|
|
64
|
+
await probeModels(cfg, cwd, modelProbeAdapters, health, probeProgressTTY
|
|
39
65
|
? (adapter, model, status, durationMs) => console.error(` ${adapter}:${model} ${status} (${(durationMs / 1000).toFixed(1)}s)`)
|
|
40
66
|
: undefined);
|
|
41
67
|
writeDoctor(cwd, health);
|
|
42
68
|
const rows = adapters.map((a) => {
|
|
43
69
|
const h = health[a.id];
|
|
44
70
|
const state = !h.installed ? "not installed" : `${h.version ?? "installed"}${h.note ? ` (${h.note})` : ""}`;
|
|
45
|
-
|
|
71
|
+
const healthy = h.installed && (a.id !== kimi.id || h.authed);
|
|
72
|
+
return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
|
|
46
73
|
});
|
|
47
74
|
// v1.48 T1: advisory sweep for known agent CLIs with no adapter — never written to doctor.json health.
|
|
48
75
|
rows.push(...detectCandidateClis().map(({ binary, version }) => alignedStatusRow("warn", binary, `detected: ${version ?? "version unknown"} (no tickmarkr adapter — not routable)`)));
|
|
@@ -57,6 +57,14 @@ function withModeLine(yaml, mode) {
|
|
|
57
57
|
return yaml.replace(/^routing:$/m, `routing:\n mode: ${mode}`);
|
|
58
58
|
return `routing:\n mode: ${mode}\n${yaml}`;
|
|
59
59
|
}
|
|
60
|
+
function formatFleetSteering(cfg) {
|
|
61
|
+
const blocks = [];
|
|
62
|
+
if (cfg.review.prefer?.length)
|
|
63
|
+
blocks.push(`review:\n prefer: ${JSON.stringify(cfg.review.prefer)}`);
|
|
64
|
+
if (cfg.consult.prefer?.length)
|
|
65
|
+
blocks.push(`consult:\n prefer: ${JSON.stringify(cfg.consult.prefer)}`);
|
|
66
|
+
return blocks.length ? `${blocks.join("\n")}\n` : "";
|
|
67
|
+
}
|
|
60
68
|
// v1.51 T4: one gloss per routing mode on the fleet mode screen — mirrors the preset compiler.
|
|
61
69
|
const MODE_GLOSS = {
|
|
62
70
|
"partner-led": "every shape frontier · explore off",
|
|
@@ -83,7 +91,9 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
|
|
|
83
91
|
const rm = resolveRunMode(cwd, { globalDir });
|
|
84
92
|
const body = formatFleetPrint(cwd, { globalDir });
|
|
85
93
|
const nl = body.indexOf("\n");
|
|
86
|
-
|
|
94
|
+
// Steering comes from the same resolved config snapshot the editor consumes below,
|
|
95
|
+
// not from another parse of either raw overlay.
|
|
96
|
+
return `${body.slice(0, nl)}\n# mode: ${rm.mode.mode} (${rm.source})${body.slice(nl)}${formatFleetSteering(rm.cfg)}`;
|
|
87
97
|
}
|
|
88
98
|
if (!interactive)
|
|
89
99
|
return { out: NON_TTY_MSG, code: 1 };
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -14,6 +14,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
14
14
|
private deliverySerial;
|
|
15
15
|
private dispatchLeases;
|
|
16
16
|
private deliveredPanes;
|
|
17
|
+
private inputBoxes;
|
|
17
18
|
private ws;
|
|
18
19
|
private callerPane;
|
|
19
20
|
private watches;
|
|
@@ -23,6 +24,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
23
24
|
private reserveDispatch;
|
|
24
25
|
private verifyPaneIdentityBinding;
|
|
25
26
|
private deliveryMatches;
|
|
27
|
+
private submissionRegistered;
|
|
26
28
|
static available(): boolean;
|
|
27
29
|
private herdr;
|
|
28
30
|
private namedPaneId;
|
|
@@ -39,6 +41,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
39
41
|
private joinGroup;
|
|
40
42
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
41
43
|
private deliver;
|
|
44
|
+
private submitVerifiedDelivery;
|
|
42
45
|
private settleDeliveryLine;
|
|
43
46
|
private deliveryReadMatches;
|
|
44
47
|
private waitOk;
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { shq } from "../adapters/types.js";
|
|
1
|
+
import { declaredInputBoxForWorkerName, matchesInputBox, shq } from "../adapters/types.js";
|
|
2
2
|
import { PANE_IDENTITY_ENV, paneIdentityLine } from "../brand.js";
|
|
3
3
|
import { createWorktree, sh } from "../run/git.js";
|
|
4
4
|
import { herdrSealShellPrefix } from "./subprocess.js";
|
|
@@ -8,6 +8,7 @@ export const TRAILER_SAFE_FLOOR_COLS = 108;
|
|
|
8
8
|
export const TRAILER_WIDTH_MARGIN = 2; // cols below (floor + margin) refuse a rightward first split
|
|
9
9
|
// OBS-85 verified delivery: bounded type→read-back→enter attempts before failing closed.
|
|
10
10
|
export const DELIVERY_ATTEMPTS = 3;
|
|
11
|
+
const DELIVERY_SUBMIT_ATTEMPTS = 2; // initial Enter + one evidence-backed re-press (OBS-140)
|
|
11
12
|
const DELIVERY_VERIFY_TIMEOUT_MS = 2000; // per attempt — a paste that hasn't rendered in 2s is retyped
|
|
12
13
|
const DELIVERY_READ_LINES = 80;
|
|
13
14
|
const DELIVERY_SETTLE_READ_ATTEMPTS = 6;
|
|
@@ -32,6 +33,7 @@ export class HerdrDriver {
|
|
|
32
33
|
deliverySerial = Promise.resolve();
|
|
33
34
|
dispatchLeases = new WeakMap();
|
|
34
35
|
deliveredPanes = new WeakMap();
|
|
36
|
+
inputBoxes = new WeakMap();
|
|
35
37
|
// VIS-10: the run's workspace id, captured once at construction (the daemon inherits it from the
|
|
36
38
|
// operator's env before the driver is built). Required at slot() time, never in the constructor —
|
|
37
39
|
// pickDriver and its unit test construct HerdrDriver without env, so slot() is the trust gate.
|
|
@@ -94,6 +96,22 @@ export class HerdrDriver {
|
|
|
94
96
|
const needle = norm(cmd);
|
|
95
97
|
return needle.length > 0 && hay.includes(needle);
|
|
96
98
|
}
|
|
99
|
+
// Submission succeeds when the typed prompt disappears, or when it has moved above a fresh
|
|
100
|
+
// adapter-declared input box (the prompt is now transcript, not input). Shell-line delivery uses
|
|
101
|
+
// the same normalized seam: a prompt still at the bottom ends the pane text; output/a fresh prompt
|
|
102
|
+
// after it proves Enter registered. No adapter-specific fingerprint lives in the driver.
|
|
103
|
+
submissionRegistered(transcript, cmd, inputBox) {
|
|
104
|
+
const norm = (s) => s.replace(/\s+/g, "");
|
|
105
|
+
const hay = norm(transcript);
|
|
106
|
+
const needle = norm(cmd);
|
|
107
|
+
const promptAt = hay.lastIndexOf(needle);
|
|
108
|
+
if (needle.length === 0 || promptAt < 0)
|
|
109
|
+
return true;
|
|
110
|
+
if (inputBox && matchesInputBox(transcript, inputBox)) {
|
|
111
|
+
return hay.lastIndexOf(norm(inputBox.fingerprint)) > promptAt;
|
|
112
|
+
}
|
|
113
|
+
return promptAt + needle.length < hay.length;
|
|
114
|
+
}
|
|
97
115
|
static available() {
|
|
98
116
|
return process.env.HERDR_ENV === "1";
|
|
99
117
|
}
|
|
@@ -142,6 +160,7 @@ export class HerdrDriver {
|
|
|
142
160
|
}
|
|
143
161
|
}
|
|
144
162
|
async slot(cwd, name, opts) {
|
|
163
|
+
const inputBox = declaredInputBoxForWorkerName(name);
|
|
145
164
|
// T1 ownership contract: `opts.owned` (T2 call sites) names the pane canonically —
|
|
146
165
|
// tickmarkr:<role>:<taskId>:<attempt>:<runId>. Without it, `name` passes through byte-identical
|
|
147
166
|
// (today's legacy daemon/gates/consult shapes) — canonicalizeLegacyName (types.ts) is what lets
|
|
@@ -154,12 +173,13 @@ export class HerdrDriver {
|
|
|
154
173
|
// Production dispatch names are canonical even when the gate call site supplies the already-
|
|
155
174
|
// formatted name rather than SlotOpts.owned. Hold one lease across slot() → run(); legacy/manual
|
|
156
175
|
// slots retain their existing allocation-only semantics for compatibility.
|
|
157
|
-
|
|
158
|
-
|
|
176
|
+
const slot = parseOwnedName(resolved) ? await this.reserveDispatch(allocate) : await allocate();
|
|
177
|
+
if (inputBox)
|
|
178
|
+
this.inputBoxes.set(slot, inputBox);
|
|
159
179
|
// group wins if both are set (a group tab is already stage-labeled; passing both is a caller bug).
|
|
160
180
|
// label (without group) → dedicated labeled tab via tabSlot's third param: no groups-map entry, no
|
|
161
181
|
// refcount, no groupSerial, no degrade latch — dedicated tabs have no shared state to guard (SUP-01).
|
|
162
|
-
return
|
|
182
|
+
return slot; // label undefined → defaults to name (today's behavior)
|
|
163
183
|
}
|
|
164
184
|
// today's per-slot tab path, plus the VIS-04 orphan reap
|
|
165
185
|
// label defaults to the slot name; group tabs pass the STAGE name instead — a first-member label
|
|
@@ -397,25 +417,26 @@ export class HerdrDriver {
|
|
|
397
417
|
// line only after two consecutive pane reads agree; an already-stable frame returns on the
|
|
398
418
|
// first fresh read without a timer. A changing pane is bounded and preserves OBS-85's
|
|
399
419
|
// fail-closed error instead of guessing from an adapter fingerprint.
|
|
400
|
-
const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript);
|
|
420
|
+
const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript, this.inputBoxes.get(slot));
|
|
401
421
|
transcript = settled.transcript;
|
|
402
422
|
if (!settled.ok) {
|
|
403
423
|
throw new Error(`herdr delivery clear failed — refusing to retype onto a corrupted line (OBS-85); pane transcript:\n${transcript}`);
|
|
404
424
|
}
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
425
|
+
if (!settled.recognizedInputBox) {
|
|
426
|
+
// Clear the corrupted shell input line before retyping; a failed clear must NOT be retyped
|
|
427
|
+
// onto — corrupt-prefix + clean-retype would concatenate and false-verify by containment.
|
|
428
|
+
// A stable adapter-declared input box is already an empty legitimate delivery target.
|
|
429
|
+
const cleared = await this.herdr(`pane send-keys ${shq(pane)} C-u`, slot.cwd);
|
|
430
|
+
if (cleared.code !== 0)
|
|
431
|
+
throw new Error(`herdr delivery clear failed — refusing to retype onto a corrupted line (OBS-85); pane transcript:\n${transcript}`);
|
|
432
|
+
}
|
|
410
433
|
}
|
|
411
434
|
const typed = await this.herdr(`pane send-text ${shq(pane)} ${shq(cmd)}`, slot.cwd);
|
|
412
435
|
if (typed.code !== 0)
|
|
413
436
|
throw new Error(`herdr pane send-text failed: ${typed.stderr || typed.stdout}`);
|
|
414
437
|
const back = await this.herdr(`pane wait-output ${shq(pane)} --match ${shq(cmd)} --timeout ${DELIVERY_VERIFY_TIMEOUT_MS}`, slot.cwd, DELIVERY_VERIFY_TIMEOUT_MS + 15_000);
|
|
415
438
|
if (this.waitOk(back.code, back.stdout) || await this.deliveryReadMatches(pane, cmd, slot.cwd)) {
|
|
416
|
-
|
|
417
|
-
if (enter.code !== 0)
|
|
418
|
-
throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
|
|
439
|
+
await this.submitVerifiedDelivery(slot, cmd, pane);
|
|
419
440
|
return;
|
|
420
441
|
}
|
|
421
442
|
// capture the corrupted delivery BEFORE clearing it — the OBS-85 byte-level evidence
|
|
@@ -423,20 +444,52 @@ export class HerdrDriver {
|
|
|
423
444
|
}
|
|
424
445
|
throw new Error(`herdr delivery corrupted after ${DELIVERY_ATTEMPTS} attempts — enter never pressed (OBS-85); pane transcript:\n${transcript}`);
|
|
425
446
|
}
|
|
426
|
-
async
|
|
447
|
+
async submitVerifiedDelivery(slot, cmd, pane) {
|
|
448
|
+
let transcript = "";
|
|
449
|
+
const inputBox = this.inputBoxes.get(slot);
|
|
450
|
+
for (let attempt = 0; attempt < DELIVERY_SUBMIT_ATTEMPTS; attempt++) {
|
|
451
|
+
const enter = await this.herdr(`pane send-keys ${shq(pane)} Enter`, slot.cwd);
|
|
452
|
+
if (enter.code !== 0)
|
|
453
|
+
throw new Error(`herdr pane send-keys Enter failed: ${enter.stderr || enter.stdout}`);
|
|
454
|
+
// Reuse the existing settle-read window. A first-read success returns before any timer; only
|
|
455
|
+
// a prompt that still occupies the delivery target spends the bounded settle window. This
|
|
456
|
+
// verification always completes before a possible re-press, so a slow submit cannot duplicate.
|
|
457
|
+
const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript, inputBox, (candidate) => this.submissionRegistered(candidate, cmd, inputBox));
|
|
458
|
+
transcript = settled.transcript;
|
|
459
|
+
if (settled.ok)
|
|
460
|
+
return;
|
|
461
|
+
if (settled.readFailed) {
|
|
462
|
+
throw new Error(`herdr delivery corrupted — submission verification failed, refusing to re-press Enter (OBS-140); pane transcript:\n${transcript}`);
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
throw new Error(`herdr delivery corrupted after ${DELIVERY_SUBMIT_ATTEMPTS} submit attempts — submission never registered (OBS-140); pane transcript:\n${transcript}`);
|
|
466
|
+
}
|
|
467
|
+
async settleDeliveryLine(pane, cwd, initialTranscript, inputBox, accept) {
|
|
427
468
|
let transcript = initialTranscript;
|
|
428
469
|
for (let readAttempt = 0; readAttempt < DELIVERY_SETTLE_READ_ATTEMPTS; readAttempt++) {
|
|
429
470
|
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
|
|
430
471
|
if (read.code !== 0)
|
|
431
|
-
return { ok: false, transcript: read.stdout || transcript };
|
|
432
|
-
if (read.stdout
|
|
433
|
-
return {
|
|
472
|
+
return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false, readFailed: true };
|
|
473
|
+
if (accept?.(read.stdout)) {
|
|
474
|
+
return {
|
|
475
|
+
ok: true,
|
|
476
|
+
transcript: read.stdout,
|
|
477
|
+
recognizedInputBox: inputBox !== undefined && matchesInputBox(read.stdout, inputBox),
|
|
478
|
+
};
|
|
479
|
+
}
|
|
480
|
+
if (accept === undefined && read.stdout === transcript) {
|
|
481
|
+
return {
|
|
482
|
+
ok: true,
|
|
483
|
+
transcript,
|
|
484
|
+
recognizedInputBox: inputBox !== undefined && matchesInputBox(transcript, inputBox),
|
|
485
|
+
};
|
|
486
|
+
}
|
|
434
487
|
transcript = read.stdout;
|
|
435
488
|
if (readAttempt < DELIVERY_SETTLE_READ_ATTEMPTS - 1) {
|
|
436
489
|
await new Promise((resolve) => setTimeout(resolve, DELIVERY_SETTLE_POLL_MS));
|
|
437
490
|
}
|
|
438
491
|
}
|
|
439
|
-
return { ok: false, transcript };
|
|
492
|
+
return { ok: false, transcript, recognizedInputBox: false };
|
|
440
493
|
}
|
|
441
494
|
async deliveryReadMatches(pane, cmd, cwd) {
|
|
442
495
|
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
|
package/dist/run/daemon.js
CHANGED
|
@@ -24,7 +24,7 @@ import { acquireRunLock, releaseRunLock } from "./lock.js";
|
|
|
24
24
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
25
25
|
import { nextChannel, route } from "../route/router.js";
|
|
26
26
|
import { desiredPanes } from "./reconcile.js";
|
|
27
|
-
import {
|
|
27
|
+
import { StallProgressTracker } from "./stall.js";
|
|
28
28
|
const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
|
|
29
29
|
// An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
|
|
30
30
|
// carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
|
|
@@ -795,13 +795,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
795
795
|
// single site, so a test can reason about it; keep Date.now() out of profile.ts (still pure) and
|
|
796
796
|
// out of adapter module scope (the cursor is a parameter, threaded from the daemon).
|
|
797
797
|
const attemptStart = Date.now();
|
|
798
|
-
// v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing
|
|
799
|
-
//
|
|
798
|
+
// v1.23 T2: once-per-attempt latch for context threshold crossing. Sample ONLY at existing worker
|
|
799
|
+
// wait slices — never a new timer loop. null/unknown usage fails OPEN
|
|
800
800
|
// (never treated as over-threshold). Journal + notify fire at most once while the value stays high.
|
|
801
801
|
let contextWarned = false;
|
|
802
802
|
let contextTokens;
|
|
803
803
|
const sampleContext = async () => {
|
|
804
|
-
if (
|
|
804
|
+
if (!adapter.contextUsage)
|
|
805
805
|
return;
|
|
806
806
|
let usage = null;
|
|
807
807
|
try {
|
|
@@ -814,7 +814,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
814
814
|
if (!usage || typeof usage.tokens !== "number" || !Number.isFinite(usage.tokens))
|
|
815
815
|
return;
|
|
816
816
|
contextTokens = usage.tokens; // last known valid sample, including under-threshold resume candidates
|
|
817
|
-
if (usage.tokens < cfg.contextWarnTokens)
|
|
817
|
+
if (contextWarned || usage.tokens < cfg.contextWarnTokens)
|
|
818
818
|
return;
|
|
819
819
|
contextWarned = true;
|
|
820
820
|
lastContextTokens = usage.tokens;
|
|
@@ -862,15 +862,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
862
862
|
// OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
|
|
863
863
|
// stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
|
|
864
864
|
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
865
|
-
//
|
|
866
|
-
//
|
|
867
|
-
// trailer detection, harvest, paging, and quota checks all read the raw pane.
|
|
865
|
+
// v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
|
|
866
|
+
// the stall clock. Raw pane differences are terminal chrome until proven otherwise.
|
|
868
867
|
let everHadOutput = output.length > 0;
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
868
|
+
const stallProgress = new StallProgressTracker();
|
|
869
|
+
stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
|
|
870
|
+
let lastProgressAt = Date.now();
|
|
871
|
+
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
872
872
|
const sliceStart = Date.now();
|
|
873
|
-
const remaining = stallWindowMs - (sliceStart -
|
|
873
|
+
const remaining = stallWindowMs - (sliceStart - lastProgressAt);
|
|
874
874
|
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
875
875
|
if (!everHadOutput) {
|
|
876
876
|
const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
|
|
@@ -893,11 +893,6 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
893
893
|
const paneText = await driver.read(slot, 1000);
|
|
894
894
|
if (paneText.length > 0)
|
|
895
895
|
everHadOutput = true;
|
|
896
|
-
const currentStallSnapshot = normalizeStallSnapshot(paneText);
|
|
897
|
-
if (currentStallSnapshot !== lastStallSnapshot) {
|
|
898
|
-
lastStallSnapshot = currentStallSnapshot;
|
|
899
|
-
lastOutputAt = Date.now();
|
|
900
|
-
}
|
|
901
896
|
// OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
|
|
902
897
|
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
903
898
|
earlyLaunchDead = true;
|
|
@@ -906,6 +901,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
906
901
|
}
|
|
907
902
|
// v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
|
|
908
903
|
await sampleContext();
|
|
904
|
+
if (stallProgress.observe({ paneText, contextTokens }))
|
|
905
|
+
lastProgressAt = Date.now();
|
|
909
906
|
// page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
|
|
910
907
|
// (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
|
|
911
908
|
// a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
|
|
@@ -943,7 +940,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
943
940
|
}
|
|
944
941
|
if (!finished && exitCode === null) {
|
|
945
942
|
// timed out (or only ever saw false positives): harvest whatever the pane holds now
|
|
946
|
-
timedOut = Date.now() -
|
|
943
|
+
timedOut = Date.now() - lastProgressAt >= stallWindowMs;
|
|
947
944
|
output = await driver.read(slot, 1000);
|
|
948
945
|
finished = new RegExp(trailerPattern(nonce)).test(output);
|
|
949
946
|
const exit = exitRe.exec(output);
|
|
@@ -981,16 +978,16 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
981
978
|
else {
|
|
982
979
|
await driver.run(slot, paneDispatchCommand(dispatchScript));
|
|
983
980
|
// OBS-54: headless workers have the same output-inactivity budget as visible panes.
|
|
984
|
-
//
|
|
985
|
-
// exhaust the budget here too; harvest below still reads the raw pane.
|
|
981
|
+
// v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
|
|
986
982
|
const stallWindowMs = taskTimeoutMinutes * 60_000;
|
|
987
983
|
const initialPane = await driver.read(slot, 500);
|
|
988
984
|
let everHadOutput = initialPane.length > 0;
|
|
989
|
-
|
|
990
|
-
|
|
985
|
+
const stallProgress = new StallProgressTracker();
|
|
986
|
+
stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
|
|
987
|
+
let lastProgressAt = Date.now();
|
|
991
988
|
finished = false;
|
|
992
|
-
while (Date.now() -
|
|
993
|
-
const remaining = stallWindowMs - (Date.now() -
|
|
989
|
+
while (Date.now() - lastProgressAt < stallWindowMs) {
|
|
990
|
+
const remaining = stallWindowMs - (Date.now() - lastProgressAt);
|
|
994
991
|
let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
|
|
995
992
|
if (!everHadOutput) {
|
|
996
993
|
const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
|
|
@@ -1004,19 +1001,17 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1004
1001
|
const paneText = await driver.read(slot, 500);
|
|
1005
1002
|
if (paneText.length > 0)
|
|
1006
1003
|
everHadOutput = true;
|
|
1007
|
-
const currentStallSnapshot = normalizeStallSnapshot(paneText);
|
|
1008
|
-
if (currentStallSnapshot !== lastStallSnapshot) {
|
|
1009
|
-
lastStallSnapshot = currentStallSnapshot;
|
|
1010
|
-
lastOutputAt = Date.now();
|
|
1011
|
-
}
|
|
1012
1004
|
if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
|
|
1013
1005
|
earlyLaunchDead = true;
|
|
1014
1006
|
break;
|
|
1015
1007
|
}
|
|
1008
|
+
await sampleContext();
|
|
1009
|
+
if (stallProgress.observe({ paneText, contextTokens }))
|
|
1010
|
+
lastProgressAt = Date.now();
|
|
1016
1011
|
}
|
|
1017
1012
|
output = await driver.read(slot, 500);
|
|
1018
1013
|
exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
|
|
1019
|
-
timedOut = !finished && Date.now() -
|
|
1014
|
+
timedOut = !finished && Date.now() - lastProgressAt >= stallWindowMs;
|
|
1020
1015
|
}
|
|
1021
1016
|
// SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
|
|
1022
1017
|
// shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
|
package/dist/run/stall.d.ts
CHANGED
|
@@ -1,8 +1,25 @@
|
|
|
1
|
-
/** Normalize
|
|
2
|
-
* waitOutput, and paging read the raw text
|
|
3
|
-
* CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
|
|
4
|
-
* equal are the same frame modulo spinner presentation; any other byte difference is activity. */
|
|
1
|
+
/** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
|
|
2
|
+
* parsing, harvest, waitOutput, and paging always read the raw text. */
|
|
5
3
|
export declare function normalizeStallSnapshot(text: string): string;
|
|
4
|
+
export interface StallProgressSample {
|
|
5
|
+
paneText: string;
|
|
6
|
+
seedSubmitted?: boolean;
|
|
7
|
+
contextTokens?: number;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Monotonic worker-progress measure for the stall watchdog.
|
|
11
|
+
*
|
|
12
|
+
* Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
|
|
13
|
+
* evidence of work. A rendered transcript is only known to have grown when it occupies more
|
|
14
|
+
* non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
|
|
15
|
+
* advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
|
|
16
|
+
*/
|
|
17
|
+
export declare class StallProgressTracker {
|
|
18
|
+
private transcriptRows;
|
|
19
|
+
private seedSubmitted;
|
|
20
|
+
private contextTokens;
|
|
21
|
+
observe(sample: StallProgressSample): boolean;
|
|
22
|
+
}
|
|
6
23
|
/** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
|
|
7
24
|
* seam exists for fault injection in tests only — production callers pass text alone. */
|
|
8
25
|
export declare function filterLlmTranscript(text: string, classify?: (t: string) => string): string;
|
package/dist/run/stall.js
CHANGED
|
@@ -1,12 +1,9 @@
|
|
|
1
|
-
// OBS-82:
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
// design: an allowlist MISS degrades to today's recoverable no-reap behavior, while an over-broad
|
|
8
|
-
// deletion would reap a healthy worker — a new failure class. Grow the allowlist only with
|
|
9
|
-
// captured evidence (tests/fixtures/codex-mcp-spinner/).
|
|
1
|
+
// OBS-82: normalize known presentation tokens before measuring transcript extent or filtering an
|
|
2
|
+
// LLM-bound transcript. This remains a closed allowlist — ANSI/VT escapes, braille-range spinner
|
|
3
|
+
// glyphs, and elapsed-time tokens bound to time-unit suffixes. Every other byte passes through
|
|
4
|
+
// identical. v1.76 deliberately stopped treating arbitrary normalized byte changes as progress:
|
|
5
|
+
// StallProgressTracker below requires monotonic evidence, so an unknown repaint fails closed toward
|
|
6
|
+
// a recoverable consult instead of holding the watchdog silent.
|
|
10
7
|
// CSI (with intermediates), OSC (BEL- or ST-terminated), DCS/SOS/PM/APC strings, single-char
|
|
11
8
|
// escapes, and charset selection — the raw-pty forms; herdr pane reads are already rendered.
|
|
12
9
|
// eslint-disable-next-line no-control-regex
|
|
@@ -16,13 +13,46 @@ const SPINNER_RE = /[⠀-⣿]/g;
|
|
|
16
13
|
// A digit run (optionally decimal) bound directly to a time-unit suffix, standing alone as a
|
|
17
14
|
// word: 9s, 41s, 3m, 1h, 800ms. Never bare digits — "(6/7)" and "5 of 7" stay change-sensitive.
|
|
18
15
|
const ELAPSED_RE = /(?<![\w.])\d+(?:\.\d+)?(?:ms|[hms])(?!\w)/g;
|
|
19
|
-
/** Normalize
|
|
20
|
-
* waitOutput, and paging read the raw text
|
|
21
|
-
* CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
|
|
22
|
-
* equal are the same frame modulo spinner presentation; any other byte difference is activity. */
|
|
16
|
+
/** Normalize presentation tokens for transcript extent and LLM-noise classification. Trailer
|
|
17
|
+
* parsing, harvest, waitOutput, and paging always read the raw text. */
|
|
23
18
|
export function normalizeStallSnapshot(text) {
|
|
24
19
|
return text.replace(ANSI_RE, "").replace(SPINNER_RE, "").replace(ELAPSED_RE, "");
|
|
25
20
|
}
|
|
21
|
+
/**
|
|
22
|
+
* Monotonic worker-progress measure for the stall watchdog.
|
|
23
|
+
*
|
|
24
|
+
* Terminal chrome is allowed to repaint arbitrary bytes in place, so byte differences are not
|
|
25
|
+
* evidence of work. A rendered transcript is only known to have grown when it occupies more
|
|
26
|
+
* non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
|
|
27
|
+
* advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
|
|
28
|
+
*/
|
|
29
|
+
export class StallProgressTracker {
|
|
30
|
+
transcriptRows = 0;
|
|
31
|
+
seedSubmitted = false;
|
|
32
|
+
contextTokens;
|
|
33
|
+
observe(sample) {
|
|
34
|
+
let advanced = false;
|
|
35
|
+
const rows = normalizeStallSnapshot(sample.paneText)
|
|
36
|
+
.split("\n")
|
|
37
|
+
.filter((line) => line.trim().length > 0)
|
|
38
|
+
.length;
|
|
39
|
+
if (rows > this.transcriptRows) {
|
|
40
|
+
this.transcriptRows = rows;
|
|
41
|
+
advanced = true;
|
|
42
|
+
}
|
|
43
|
+
if (sample.seedSubmitted && !this.seedSubmitted) {
|
|
44
|
+
this.seedSubmitted = true;
|
|
45
|
+
advanced = true;
|
|
46
|
+
}
|
|
47
|
+
const tokens = sample.contextTokens;
|
|
48
|
+
if (tokens !== undefined && Number.isFinite(tokens)) {
|
|
49
|
+
if (tokens > (this.contextTokens ?? 0))
|
|
50
|
+
advanced = true;
|
|
51
|
+
this.contextTokens = Math.max(this.contextTokens ?? 0, tokens);
|
|
52
|
+
}
|
|
53
|
+
return advanced;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
26
56
|
// ─── v1.65 T2: LLM-bound transcript filter ──────────────────────────────────────────────────────
|
|
27
57
|
// Consult dossiers and gate prompts pay tokens per transcript byte, so LLM-bound text runs through
|
|
28
58
|
// a per-line classifier: carriage-return overwrite churn keeps only the final paint, lines that are
|
package/package.json
CHANGED
|
@@ -15,8 +15,9 @@ When working in a multi-agent terminal environment, decide your role before star
|
|
|
15
15
|
- **Orchestrator:** your session was started to execute the mission. Rename your own tab/pane `ORCH · <version>` (short labels: ≤20 chars, `ROLE · token`) and run the loop below.
|
|
16
16
|
- **Supervisor with a live orchestrator:** do not start a second run. Relay the mission to the existing orchestrator with a [verified handoff](#verified-handoffs-agent-to-agent-messaging), then supervise it as OVERSEER.
|
|
17
17
|
- **Primary session without an orchestrator:** rename your own tab `OVERSEER · <version>` and your agent `overseer`, spawn one child orchestration session with your host's launch form, label its tab `ORCH · <version>` and name its agent, give it the mission and these rules verbatim, then supervise it. Do not drive a duplicate single-tier run yourself. Before spawning, confirm any PREVIOUS orchestrator has stood down (monitors stopped, input box empty — dim ghost-text suggestions are UI, not queued input; ANSI-verify before alarming) and close its tab — the journal, records, and ledger hold the story; scrollback is disposable.
|
|
18
|
-
- **
|
|
19
|
-
- **
|
|
18
|
+
- **Spawning on current herdr is two-step** — the one-shot `agent start --cwd/--tab/--no-focus` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138). First create the pane: `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` (tab create does not steal focus unless `--focus` is passed; parse `result.root_pane.pane_id` from its JSON), then start the agent in it:
|
|
19
|
+
- **Claude Code:** `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions`
|
|
20
|
+
- **Codex:** `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
|
|
20
21
|
- **Auxiliary agents you spawn (consultants, reviewers, scouts) follow the same forms.** Never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions` (tickmarkr's own adapter uses exactly this for workers, judges, and consults). A read-only codex consultant may use `--sandbox read-only`; any codex session that must touch git needs the unsandboxed flag above.
|
|
21
22
|
|
|
22
23
|
Outside a multi-agent terminal environment, run the loop directly.
|
|
@@ -14,8 +14,9 @@ When working in a multi-agent terminal environment, decide your role before star
|
|
|
14
14
|
- **Orchestrator:** your session was started to execute the mission. Rename your own tab/pane `ORCH · <version>` (short labels: ≤20 chars, `ROLE · token`) and run the loop below.
|
|
15
15
|
- **Supervisor with a live orchestrator:** do not start a second run. Relay the mission to the existing orchestrator with a [verified handoff](#verified-handoffs-agent-to-agent-messaging), then supervise it as OVERSEER.
|
|
16
16
|
- **Primary session without an orchestrator:** rename your own tab `OVERSEER · <version>` and your agent `overseer`, spawn one child orchestration session with your host's launch form, label its tab `ORCH · <version>` and name its agent, give it the mission and these rules verbatim, then supervise it. Do not drive a duplicate single-tier run yourself. Before spawning, confirm any PREVIOUS orchestrator has [stood down](#stand-down-mission-end-and-retirement) and close its tab.
|
|
17
|
-
- **
|
|
18
|
-
- **
|
|
17
|
+
- **Spawning on current herdr is two-step** — the one-shot `agent start --cwd/--tab/--no-focus` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138). First create the pane: `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` (tab create does not steal focus unless `--focus` is passed; parse `result.root_pane.pane_id` from its JSON), then start the agent in it:
|
|
18
|
+
- **Claude Code:** `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions`
|
|
19
|
+
- **Codex:** `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
|
|
19
20
|
- **Auxiliary agents you spawn (consultants, reviewers, scouts) follow the same forms.** Never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions` (tickmarkr's own adapter uses exactly this for workers, judges, and consults). A read-only codex consultant may use `--sandbox read-only`; any codex session that must touch git needs the unsandboxed flag above.
|
|
20
21
|
|
|
21
22
|
Outside a multi-agent terminal environment, run the loop directly.
|
|
@@ -29,7 +29,7 @@ Requires `HERDR_ENV=1`; if unset, say so and stop.
|
|
|
29
29
|
main name plus at most ONE hot-state token. Vocabulary: ORCH carries the milestone and progress
|
|
30
30
|
fraction (`ORCH · v1.19 4/5`, updated on every task-done); WORKERS carries the task token (tickmarkr
|
|
31
31
|
updates it). Never long context strings or ✓-chains.
|
|
32
|
-
2. **Orchestrator**: Launch the orchestrator with your agent host. For Claude Code, use `herdr agent start orchestrator --
|
|
32
|
+
2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions`, and a read-only codex consultant may use `--sandbox read-only`.
|
|
33
33
|
3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
|
|
34
34
|
truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
|
|
35
35
|
(inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then send one line:
|