tickmarkr 1.73.0 → 1.75.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code.d.ts +2 -1
- package/dist/adapters/claude-code.js +10 -6
- package/dist/adapters/codex.d.ts +2 -1
- package/dist/adapters/codex.js +7 -0
- package/dist/adapters/kimi.d.ts +1 -0
- package/dist/adapters/kimi.js +7 -1
- package/dist/adapters/types.d.ts +7 -0
- package/dist/adapters/types.js +12 -0
- package/dist/cli/commands/fleet.js +11 -1
- package/dist/cli/commands/status.js +45 -15
- package/dist/drivers/herdr.d.ts +2 -0
- package/dist/drivers/herdr.js +45 -8
- package/dist/gates/acceptance.js +19 -1
- package/dist/gates/llm.d.ts +4 -0
- package/dist/gates/llm.js +16 -3
- package/dist/gates/run-gates.d.ts +1 -1
- package/dist/gates/run-gates.js +25 -3
- package/dist/run/journal.d.ts +15 -0
- package/dist/run/journal.js +55 -4
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +3 -2
- package/skills/tickmarkr-loop/SKILL.md +3 -2
- package/skills/tickmarkr-overseer/SKILL.md +1 -1
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { type AuthHealth, type WorkerAdapter } from "./types.js";
|
|
1
|
+
import { type AuthHealth, type TrustDialog, type WorkerAdapter } from "./types.js";
|
|
2
|
+
export declare const CLAUDE_TRUST_DIALOG: TrustDialog;
|
|
2
3
|
export declare function claudeSlug(real: string): string;
|
|
3
4
|
export declare function probeVersion(bin: string): AuthHealth;
|
|
4
5
|
export declare const claudeCode: WorkerAdapter;
|
|
@@ -20,6 +20,12 @@ import { channelsFromConfig, shq, TokenUsageSchema } from "./types.js";
|
|
|
20
20
|
// Phase 18's operator-price × tokens derivation, not a CLI claim.
|
|
21
21
|
const MAX_SESSION_FILES = 20; // newest-first; a long-lived project dir can hold many sessions
|
|
22
22
|
const MAX_SESSION_BYTES = 8_000_000; // per-file cap; a runaway JSONL cannot make the read unbounded
|
|
23
|
+
// v1.75 T2 / OBS-137: current Claude Code workspace-trust prompt (2.1.218). The full question
|
|
24
|
+
// distinguishes this startup gate from routine agent text; Enter accepts the selected trust option.
|
|
25
|
+
export const CLAUDE_TRUST_DIALOG = {
|
|
26
|
+
fingerprint: "Quick safety check: Is this a project you created or one you trust?",
|
|
27
|
+
key: "Enter",
|
|
28
|
+
};
|
|
23
29
|
export function claudeSlug(real) {
|
|
24
30
|
return real.replace(/[^A-Za-z0-9]/g, "-");
|
|
25
31
|
}
|
|
@@ -54,13 +60,11 @@ export const claudeCode = {
|
|
|
54
60
|
// and --mcp-config is VARIADIC — a positional after it is eaten as a config-file path, so another
|
|
55
61
|
// flag must always follow the value, never the prompt.
|
|
56
62
|
headlessCommand: (promptFile, model) => `claude -p "$(cat ${shq(promptFile)})" --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text`,
|
|
57
|
-
// HYG-03: the residual first-entry dialog
|
|
58
|
-
//
|
|
59
|
-
//
|
|
60
|
-
// to that file (a seed races claude's own writes, nondeterministically). Amortizes to one operator dismissal
|
|
61
|
-
// per stable worktree path; blocked-pane paging surfaces it. Do NOT change this command to "fix" the dialog —
|
|
62
|
-
// see .planning/REQUIREMENTS.md HYG-03 and 21-02-LIVE-CHECK.md. Revisit if upstream ships a --trust flag.
|
|
63
|
+
// HYG-03 / OBS-137: the residual first-entry dialog is workspace trust, not MCP config loading.
|
|
64
|
+
// Claude's only store is global last-writer-wins ~/.claude.json, so tickmarkr still does not seed it;
|
|
65
|
+
// the daemon safely answers only the exact adapter-declared dialog once per slot.
|
|
63
66
|
interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
67
|
+
trustDialog: CLAUDE_TRUST_DIALOG,
|
|
64
68
|
resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
65
69
|
invoke(task, _cwd, a, ctx) {
|
|
66
70
|
return { command: this.headlessCommand(ctx.promptFile, a.model) };
|
package/dist/adapters/codex.d.ts
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
|
-
import { type TrustVerdict, type WorkerAdapter } from "./types.js";
|
|
1
|
+
import { type TrustDialog, type TrustVerdict, type WorkerAdapter } from "./types.js";
|
|
2
2
|
export declare function readCodexModelsCache(path?: string): {
|
|
3
3
|
models: string[];
|
|
4
4
|
fetchedAt?: string;
|
|
5
5
|
};
|
|
6
|
+
export declare const CODEX_TRUST_DIALOG: TrustDialog;
|
|
6
7
|
export declare function seedCodexTrust(repoRoot: string, configPath?: string): TrustVerdict;
|
|
7
8
|
export declare function hasCodexTrustedProject(text: string, root: string): boolean;
|
|
8
9
|
export declare function codexConfigMcpServerNames(configPath?: string): string[];
|
package/dist/adapters/codex.js
CHANGED
|
@@ -89,6 +89,12 @@ const GITDIR_WRITABLE = `-c "sandbox_workspace_write.writable_roots=[\\"$(git re
|
|
|
89
89
|
// -s/--sandbox workspace-write sandbox (deliberately NOT --dangerously-bypass-approvals-and-sandbox,
|
|
90
90
|
// which would drop the sandbox). Listed by `codex --help` and `codex exec --help` (verified 2026-07-23).
|
|
91
91
|
const CODEX_HOOK_TRUST = "--dangerously-bypass-hook-trust";
|
|
92
|
+
// v1.75 T2 / OBS-137: current Codex workspace-trust prompt (0.144.6). The exact heading
|
|
93
|
+
// is distinct from normal agent output; Enter accepts the selected "Yes, continue" option.
|
|
94
|
+
export const CODEX_TRUST_DIALOG = {
|
|
95
|
+
fingerprint: "Do you trust the contents of this directory?",
|
|
96
|
+
key: "Enter",
|
|
97
|
+
};
|
|
92
98
|
// v1.22 T5 / OBS-16: codex keys trust on absolute path under [projects."<root>"] trust_level="trusted"
|
|
93
99
|
// in ~/.codex/config.toml (CODEX_HOME relocates the dir). Worktrees inherit parent-project trust when
|
|
94
100
|
// the REPO ROOT is trusted — seed the root once, cover every future worktree. Idempotent: a second
|
|
@@ -196,6 +202,7 @@ export const codex = {
|
|
|
196
202
|
// v1.22 T5: seed [projects."<repoRoot>"] trust_level="trusted" so fresh worktrees never stall on
|
|
197
203
|
// "Do you trust this directory?" (OBS-16). doctor-only side effect.
|
|
198
204
|
trust: (repoRoot) => seedCodexTrust(repoRoot),
|
|
205
|
+
trustDialog: CODEX_TRUST_DIALOG,
|
|
199
206
|
// v1.5 MODEL-01: file read only (no `codex models` subcommand exists, verified 2026-07-10).
|
|
200
207
|
// Already fails OPEN to [] internally — advisory detection, unlike gates' fail-closed.
|
|
201
208
|
listModels: async () => readCodexModelsCache().models,
|
package/dist/adapters/kimi.d.ts
CHANGED
|
@@ -20,6 +20,7 @@ export interface KimiInteractiveSeedResult {
|
|
|
20
20
|
seedError?: string;
|
|
21
21
|
sessionId?: string;
|
|
22
22
|
}
|
|
23
|
+
export declare const KIMI_INPUT_BOX: import("./types.js").InputBox;
|
|
23
24
|
export declare function runKimiInteractiveSeed(opts: {
|
|
24
25
|
driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
|
|
25
26
|
slot: Slot;
|
package/dist/adapters/kimi.js
CHANGED
|
@@ -5,7 +5,7 @@ import { probeVersion } from "./claude-code.js";
|
|
|
5
5
|
import { parseWorkerResult } from "./prompt.js";
|
|
6
6
|
import { runInteractiveSeed } from "../run/interactive-seed.js";
|
|
7
7
|
import { sh } from "../run/git.js";
|
|
8
|
-
import { channelsFromConfig, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
|
|
8
|
+
import { channelsFromConfig, declareInputBox, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
|
|
9
9
|
// KIMI-03 → v1.58 T5: the "no harness-readable counter" block (research F-6, 2026-07-17) is
|
|
10
10
|
// LIFTED for collectUsage — kimi 0.27.0 writes a wire journal per agent at
|
|
11
11
|
// ~/.kimi-code/sessions/<wd>/session_<uuid>/agents/<agent>/wire.jsonl, and ~/.kimi-code/
|
|
@@ -118,6 +118,11 @@ export function confirmKimiSeedBanner(banner, assignedModel) {
|
|
|
118
118
|
}
|
|
119
119
|
return { ok: true, sessionId: kimiBannerSessionId(banner) };
|
|
120
120
|
}
|
|
121
|
+
// v1.75 T1 / OBS-136: this readiness line is rendered inside Kimi Code's bordered steady-state
|
|
122
|
+
// input box. The adapter owns the fingerprint; the driver only consults declarations generically.
|
|
123
|
+
export const KIMI_INPUT_BOX = declareInputBox("kimi", {
|
|
124
|
+
fingerprint: "Send /help for help information.",
|
|
125
|
+
});
|
|
121
126
|
// Shared launch-then-seed surface (T6) + banner confirm (T7/T2). One definition so the adapter
|
|
122
127
|
// property and the daemon's generic runInteractiveSeed path cannot drift.
|
|
123
128
|
const KIMI_SEED = {
|
|
@@ -165,6 +170,7 @@ export const kimi = {
|
|
|
165
170
|
// prompt as one user turn. Banner model/session confirmation runs on the daemon's generic
|
|
166
171
|
// runInteractiveSeed path via confirmBanner on KIMI_SEED (not a separate dispatch helper).
|
|
167
172
|
interactiveSeed: KIMI_SEED,
|
|
173
|
+
inputBox: KIMI_INPUT_BOX,
|
|
168
174
|
// v1.53 T3 resume — live-probed 2026-07-18: `-p` + `-S <id>` compose cleanly (no OBS-67-class
|
|
169
175
|
// flag rejection) and the resumed session carries prior conversation state. `-S <id>` is the
|
|
170
176
|
// deterministic form; `-c` rejected as primary — cwd-keyed, nondeterministic under worktree
|
package/dist/adapters/types.d.ts
CHANGED
|
@@ -82,6 +82,12 @@ export interface TrustDialog {
|
|
|
82
82
|
key: string;
|
|
83
83
|
}
|
|
84
84
|
export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): boolean;
|
|
85
|
+
export interface InputBox {
|
|
86
|
+
fingerprint: string;
|
|
87
|
+
}
|
|
88
|
+
export declare function declareInputBox(adapterId: string, inputBox: InputBox): InputBox;
|
|
89
|
+
export declare function declaredInputBoxForWorkerName(workerName: string): InputBox | undefined;
|
|
90
|
+
export declare function matchesInputBox(paneText: string, inputBox: InputBox): boolean;
|
|
85
91
|
export interface WorkerAdapter {
|
|
86
92
|
id: string;
|
|
87
93
|
vendor: string;
|
|
@@ -105,6 +111,7 @@ export interface WorkerAdapter {
|
|
|
105
111
|
contextUsage?(session: SessionRef): ContextUsage | null;
|
|
106
112
|
trust?(repoRoot: string): TrustVerdict;
|
|
107
113
|
trustDialog?: TrustDialog;
|
|
114
|
+
inputBox?: InputBox;
|
|
108
115
|
hardcodedFlags?: {
|
|
109
116
|
binary: string;
|
|
110
117
|
flags: string[];
|
package/dist/adapters/types.js
CHANGED
|
@@ -30,6 +30,18 @@ export function modelAuthed(health, model, allowUnverifiedModels = false) {
|
|
|
30
30
|
export function matchesTrustDialog(paneText, dialog) {
|
|
31
31
|
return paneText.includes(dialog.fingerprint);
|
|
32
32
|
}
|
|
33
|
+
const inputBoxes = new Map();
|
|
34
|
+
export function declareInputBox(adapterId, inputBox) {
|
|
35
|
+
inputBoxes.set(adapterId, inputBox);
|
|
36
|
+
return inputBox;
|
|
37
|
+
}
|
|
38
|
+
export function declaredInputBoxForWorkerName(workerName) {
|
|
39
|
+
const adapterId = /^.+-worker-(.+)-a\d+-.+$/.exec(workerName)?.[1];
|
|
40
|
+
return adapterId === undefined ? undefined : inputBoxes.get(adapterId);
|
|
41
|
+
}
|
|
42
|
+
export function matchesInputBox(paneText, inputBox) {
|
|
43
|
+
return paneText.includes(inputBox.fingerprint);
|
|
44
|
+
}
|
|
33
45
|
export function channelsFromConfig(adapterId, cfg) {
|
|
34
46
|
const e = cfg.tiers[adapterId];
|
|
35
47
|
if (!e)
|
|
@@ -57,6 +57,14 @@ function withModeLine(yaml, mode) {
|
|
|
57
57
|
return yaml.replace(/^routing:$/m, `routing:\n mode: ${mode}`);
|
|
58
58
|
return `routing:\n mode: ${mode}\n${yaml}`;
|
|
59
59
|
}
|
|
60
|
+
function formatFleetSteering(cfg) {
|
|
61
|
+
const blocks = [];
|
|
62
|
+
if (cfg.review.prefer?.length)
|
|
63
|
+
blocks.push(`review:\n prefer: ${JSON.stringify(cfg.review.prefer)}`);
|
|
64
|
+
if (cfg.consult.prefer?.length)
|
|
65
|
+
blocks.push(`consult:\n prefer: ${JSON.stringify(cfg.consult.prefer)}`);
|
|
66
|
+
return blocks.length ? `${blocks.join("\n")}\n` : "";
|
|
67
|
+
}
|
|
60
68
|
// v1.51 T4: one gloss per routing mode on the fleet mode screen — mirrors the preset compiler.
|
|
61
69
|
const MODE_GLOSS = {
|
|
62
70
|
"partner-led": "every shape frontier · explore off",
|
|
@@ -83,7 +91,9 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
|
|
|
83
91
|
const rm = resolveRunMode(cwd, { globalDir });
|
|
84
92
|
const body = formatFleetPrint(cwd, { globalDir });
|
|
85
93
|
const nl = body.indexOf("\n");
|
|
86
|
-
|
|
94
|
+
// Steering comes from the same resolved config snapshot the editor consumes below,
|
|
95
|
+
// not from another parse of either raw overlay.
|
|
96
|
+
return `${body.slice(0, nl)}\n# mode: ${rm.mode.mode} (${rm.source})${body.slice(nl)}${formatFleetSteering(rm.cfg)}`;
|
|
87
97
|
}
|
|
88
98
|
if (!interactive)
|
|
89
99
|
return { out: NON_TTY_MSG, code: 1 };
|
|
@@ -8,7 +8,8 @@ import { Journal, engagementComparable, isQualityFailureParkKind, recordedTaskFa
|
|
|
8
8
|
import { normalizeStallSnapshot } from "../../run/stall.js";
|
|
9
9
|
// ponytail: fixed 2s refresh; promote to config.visibility.* only when an operator asks.
|
|
10
10
|
const REFRESH_MS = 2000;
|
|
11
|
-
const NOT_COMPARABLE_NOTICE = "graph recompiled since this run — task states not comparable;
|
|
11
|
+
const NOT_COMPARABLE_NOTICE = "graph recompiled since this run — task states not comparable; resume with `--graph-changed` to audit this recompile";
|
|
12
|
+
const PRIOR_GRAPH_MARKER = "prior graph";
|
|
12
13
|
// The timer must keep the process ALIVE: an unref'd timer here let the event loop drain after the
|
|
13
14
|
// first frame, so a live `--watch` printed once and exited 0 (OBS-11). Never unref this.
|
|
14
15
|
const defaultSleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
@@ -120,23 +121,28 @@ export const gateBox = (state, unicode) => {
|
|
|
120
121
|
return state === "pass" ? "[x]" : state === "fail" ? "[!]" : state === "skip" ? "." : "[ ]";
|
|
121
122
|
};
|
|
122
123
|
export const defaultGateStates = (task) => GATE_NAMES.map((gate) => task.gates.includes(gate) ? "open" : "skip");
|
|
123
|
-
|
|
124
|
+
const gateSnapshot = (task, events, rehashAt) => {
|
|
124
125
|
const outcomes = new Map();
|
|
125
126
|
const start = attemptStartIdx(events, task.id);
|
|
126
127
|
if (start >= 0) {
|
|
127
|
-
for (
|
|
128
|
+
for (let eventIndex = start; eventIndex < events.length; eventIndex++) {
|
|
129
|
+
const e = events[eventIndex];
|
|
128
130
|
if (e.taskId !== task.id || e.event !== "gate-result" || typeof e.data.gate !== "string")
|
|
129
131
|
continue;
|
|
130
132
|
if (e.data.skipped === true)
|
|
131
|
-
outcomes.set(e.data.gate, "skip");
|
|
133
|
+
outcomes.set(e.data.gate, { state: "skip", eventIndex });
|
|
132
134
|
else if (e.data.pass === true)
|
|
133
|
-
outcomes.set(e.data.gate, "pass");
|
|
135
|
+
outcomes.set(e.data.gate, { state: "pass", eventIndex });
|
|
134
136
|
else if (e.data.pass === false)
|
|
135
|
-
outcomes.set(e.data.gate, "fail");
|
|
137
|
+
outcomes.set(e.data.gate, { state: "fail", eventIndex });
|
|
136
138
|
}
|
|
137
139
|
}
|
|
138
|
-
return
|
|
140
|
+
return {
|
|
141
|
+
states: GATE_NAMES.map((gate) => task.gates.includes(gate) ? outcomes.get(gate)?.state ?? "open" : "skip"),
|
|
142
|
+
priorGraph: rehashAt !== undefined && [...outcomes.values()].some((outcome) => outcome.eventIndex < rehashAt),
|
|
143
|
+
};
|
|
139
144
|
};
|
|
145
|
+
export const gateStates = (task, events) => gateSnapshot(task, events).states;
|
|
140
146
|
// verdict semantics only: pass brand green, fail red, skip/open dim chrome — everything else stays quiet
|
|
141
147
|
const GATE_STATE_TOKEN = { pass: ok, fail, skip: dim, open: dim };
|
|
142
148
|
// TTY cells are bare glyphs in fixed GATE_NAMES order — gate identity lives once in the frame
|
|
@@ -220,6 +226,24 @@ const liveness = (events, now = Date.now()) => {
|
|
|
220
226
|
} // EPERM ⇒ alive
|
|
221
227
|
return `last event ${age} ago · daemon pid ${pid} ${state}${cause ? ` · ${cause}` : ""}`;
|
|
222
228
|
};
|
|
229
|
+
// Resume's shared comparator remains the fail-closed baseline. Status may additionally accept the
|
|
230
|
+
// daemon's audited graph-rehash release, but only when its `from` names that baseline identity (or
|
|
231
|
+
// null for an explicitly released legacy journal) and its `to` names the graph loaded now.
|
|
232
|
+
const statusEngagement = (events, loadedHash) => {
|
|
233
|
+
const baseline = engagementComparable(events, loadedHash);
|
|
234
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
235
|
+
const event = events[i];
|
|
236
|
+
if (event.event !== "graph-rehash")
|
|
237
|
+
continue;
|
|
238
|
+
const auditsBaseline = baseline.comparable || baseline.reason === "mismatch"
|
|
239
|
+
? event.data.from === baseline.recorded
|
|
240
|
+
: event.data.from === null;
|
|
241
|
+
return event.data.to === loadedHash && auditsBaseline
|
|
242
|
+
? { comparable: true, rehashAt: i }
|
|
243
|
+
: { comparable: false };
|
|
244
|
+
}
|
|
245
|
+
return { comparable: baseline.comparable };
|
|
246
|
+
};
|
|
223
247
|
const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness = new Map()) => {
|
|
224
248
|
const g = loadGraph(cwd);
|
|
225
249
|
const runId = Journal.latestRunId(cwd, { withJournal: true });
|
|
@@ -228,16 +252,18 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
228
252
|
let events = [];
|
|
229
253
|
const contexts = new Map();
|
|
230
254
|
let comparable = false;
|
|
255
|
+
let rehashAt;
|
|
231
256
|
let supersededBy; // v1.53 T5: this run is dead — a newer run replaced it
|
|
232
257
|
if (runId) {
|
|
233
258
|
const j = Journal.open(cwd, runId);
|
|
234
259
|
events = j.read();
|
|
235
260
|
const sup = [...events].reverse().find((e) => e.event === "superseded" && typeof e.data.by === "string");
|
|
236
261
|
supersededBy = sup?.data.by;
|
|
237
|
-
//
|
|
238
|
-
//
|
|
239
|
-
|
|
240
|
-
comparable =
|
|
262
|
+
// The resume comparator is the fail-closed baseline; a matching graph-rehash is the daemon's
|
|
263
|
+
// append-only audit that authorizes this status replay after stop-amend-resume.
|
|
264
|
+
const engagement = statusEngagement(events, graphDefinitionHash(g));
|
|
265
|
+
comparable = engagement.comparable;
|
|
266
|
+
rehashAt = engagement.rehashAt;
|
|
241
267
|
if (comparable) {
|
|
242
268
|
replayed = j.replayStatuses();
|
|
243
269
|
for (const e of events) {
|
|
@@ -285,15 +311,18 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
285
311
|
const channel = assignments.get(t.id) ?? "-";
|
|
286
312
|
const ctx = contexts.get(t.id);
|
|
287
313
|
const assignCol = ctx !== undefined ? `${channel}${divider}ctx ${ctx}` : channel;
|
|
288
|
-
|
|
314
|
+
const gates = comparable
|
|
315
|
+
? gateSnapshot(t, events, rehashAt)
|
|
316
|
+
: { states: defaultGateStates(t), priorGraph: false };
|
|
317
|
+
return { t, st, failureKind, redTier, label, assignCol, isStarved, phrase, channel, ctx, livePhase, ...gates };
|
|
289
318
|
});
|
|
290
319
|
if (!unicode) {
|
|
291
320
|
// machine/CI surface — journals without phase-start stay byte-identical; new phase-aware frames
|
|
292
321
|
// use an ASCII spinner so pipes never receive terminal-only braille/ANSI.
|
|
293
|
-
const rows = cells.map(({ t, st, label, assignCol, livePhase, states }) => {
|
|
322
|
+
const rows = cells.map(({ t, st, label, assignCol, livePhase, states, priorGraph }) => {
|
|
294
323
|
const chain = gateChain(states, false);
|
|
295
324
|
const prefix = livePhase ? ` ${ASCII_SPINNER[animationFrame % ASCII_SPINNER.length]} ${t.id} ` : ` ${taskBox(st)} ${t.id} `;
|
|
296
|
-
const suffix = ` ${chain} ${livePhase ? "running" : String(st)}${label} ${assignCol}`;
|
|
325
|
+
const suffix = ` ${chain}${priorGraph ? ` ${PRIOR_GRAPH_MARKER}` : ""} ${livePhase ? "running" : String(st)}${label} ${assignCol}`;
|
|
297
326
|
return `${prefix}${shortGoal(t.goal, Math.max(0, width - prefix.length - suffix.length))}${suffix}`;
|
|
298
327
|
});
|
|
299
328
|
const header = runId
|
|
@@ -348,7 +377,7 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
348
377
|
const goalW = Math.max(8, ...goals.map((s) => s.length));
|
|
349
378
|
const indent = " ".repeat(idW + 5); // line 2 starts under the goal column
|
|
350
379
|
const rows = cells.map((c, i) => {
|
|
351
|
-
const { t, st, failureKind, redTier, states, isStarved, phrase, channel, ctx, livePhase } = c;
|
|
380
|
+
const { t, st, failureKind, redTier, states, priorGraph, isStarved, phrase, channel, ctx, livePhase } = c;
|
|
352
381
|
const word = statusWord(c);
|
|
353
382
|
const staleWorker = livePhase?.phase === "worker"
|
|
354
383
|
&& workerOutputAge(livePhase, now, workerLiveness.get(t.id)) >= 60_000;
|
|
@@ -366,6 +395,7 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
366
395
|
: ` ${statusRow(taskVerdict(c), taskLabel)}`;
|
|
367
396
|
// activity already names its channel for in-flight attempts — never repeat it
|
|
368
397
|
const detail = [
|
|
398
|
+
...(priorGraph ? [PRIOR_GRAPH_MARKER] : []),
|
|
369
399
|
...(phrase ? [phrase, ...(phrase.includes(channel) || channel === "-" ? [] : [channel])] : [channel]),
|
|
370
400
|
...(failureKind && !phrase?.includes(failureKind) ? [failureKind] : []),
|
|
371
401
|
...(ctx !== undefined ? [`ctx ${ctx}`] : []),
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -14,6 +14,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
14
14
|
private deliverySerial;
|
|
15
15
|
private dispatchLeases;
|
|
16
16
|
private deliveredPanes;
|
|
17
|
+
private inputBoxes;
|
|
17
18
|
private ws;
|
|
18
19
|
private callerPane;
|
|
19
20
|
private watches;
|
|
@@ -39,6 +40,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
39
40
|
private joinGroup;
|
|
40
41
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
41
42
|
private deliver;
|
|
43
|
+
private settleDeliveryLine;
|
|
42
44
|
private deliveryReadMatches;
|
|
43
45
|
private waitOk;
|
|
44
46
|
waitOutput(slot: Slot, pattern: string, timeoutMs: number, opts?: {
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { shq } from "../adapters/types.js";
|
|
1
|
+
import { declaredInputBoxForWorkerName, matchesInputBox, shq } from "../adapters/types.js";
|
|
2
2
|
import { PANE_IDENTITY_ENV, paneIdentityLine } from "../brand.js";
|
|
3
3
|
import { createWorktree, sh } from "../run/git.js";
|
|
4
4
|
import { herdrSealShellPrefix } from "./subprocess.js";
|
|
@@ -10,6 +10,8 @@ export const TRAILER_WIDTH_MARGIN = 2; // cols below (floor + margin) refuse a r
|
|
|
10
10
|
export const DELIVERY_ATTEMPTS = 3;
|
|
11
11
|
const DELIVERY_VERIFY_TIMEOUT_MS = 2000; // per attempt — a paste that hasn't rendered in 2s is retyped
|
|
12
12
|
const DELIVERY_READ_LINES = 80;
|
|
13
|
+
const DELIVERY_SETTLE_READ_ATTEMPTS = 6;
|
|
14
|
+
const DELIVERY_SETTLE_POLL_MS = 100;
|
|
13
15
|
/** First-generation join direction from measured trailer-safe floor (43-MEASUREMENT.md). */
|
|
14
16
|
export function workerSplitDirection(paneCols, safeFloor = TRAILER_SAFE_FLOOR_COLS, margin = TRAILER_WIDTH_MARGIN) {
|
|
15
17
|
if (paneCols == null || paneCols <= 0)
|
|
@@ -30,6 +32,7 @@ export class HerdrDriver {
|
|
|
30
32
|
deliverySerial = Promise.resolve();
|
|
31
33
|
dispatchLeases = new WeakMap();
|
|
32
34
|
deliveredPanes = new WeakMap();
|
|
35
|
+
inputBoxes = new WeakMap();
|
|
33
36
|
// VIS-10: the run's workspace id, captured once at construction (the daemon inherits it from the
|
|
34
37
|
// operator's env before the driver is built). Required at slot() time, never in the constructor —
|
|
35
38
|
// pickDriver and its unit test construct HerdrDriver without env, so slot() is the trust gate.
|
|
@@ -140,6 +143,7 @@ export class HerdrDriver {
|
|
|
140
143
|
}
|
|
141
144
|
}
|
|
142
145
|
async slot(cwd, name, opts) {
|
|
146
|
+
const inputBox = declaredInputBoxForWorkerName(name);
|
|
143
147
|
// T1 ownership contract: `opts.owned` (T2 call sites) names the pane canonically —
|
|
144
148
|
// tickmarkr:<role>:<taskId>:<attempt>:<runId>. Without it, `name` passes through byte-identical
|
|
145
149
|
// (today's legacy daemon/gates/consult shapes) — canonicalizeLegacyName (types.ts) is what lets
|
|
@@ -152,12 +156,13 @@ export class HerdrDriver {
|
|
|
152
156
|
// Production dispatch names are canonical even when the gate call site supplies the already-
|
|
153
157
|
// formatted name rather than SlotOpts.owned. Hold one lease across slot() → run(); legacy/manual
|
|
154
158
|
// slots retain their existing allocation-only semantics for compatibility.
|
|
155
|
-
|
|
156
|
-
|
|
159
|
+
const slot = parseOwnedName(resolved) ? await this.reserveDispatch(allocate) : await allocate();
|
|
160
|
+
if (inputBox)
|
|
161
|
+
this.inputBoxes.set(slot, inputBox);
|
|
157
162
|
// group wins if both are set (a group tab is already stage-labeled; passing both is a caller bug).
|
|
158
163
|
// label (without group) → dedicated labeled tab via tabSlot's third param: no groups-map entry, no
|
|
159
164
|
// refcount, no groupSerial, no degrade latch — dedicated tabs have no shared state to guard (SUP-01).
|
|
160
|
-
return
|
|
165
|
+
return slot; // label undefined → defaults to name (today's behavior)
|
|
161
166
|
}
|
|
162
167
|
// today's per-slot tab path, plus the VIS-04 orphan reap
|
|
163
168
|
// label defaults to the slot name; group tabs pass the STAGE name instead — a first-member label
|
|
@@ -391,11 +396,23 @@ export class HerdrDriver {
|
|
|
391
396
|
let transcript = "";
|
|
392
397
|
for (let attempt = 0; attempt < DELIVERY_ATTEMPTS; attempt++) {
|
|
393
398
|
if (attempt > 0) {
|
|
394
|
-
//
|
|
395
|
-
//
|
|
396
|
-
|
|
397
|
-
|
|
399
|
+
// A TUI may still be painting while the failed delivery is captured (OBS-135). Judge the
|
|
400
|
+
// line only after two consecutive pane reads agree; an already-stable frame returns on the
|
|
401
|
+
// first fresh read without a timer. A changing pane is bounded and preserves OBS-85's
|
|
402
|
+
// fail-closed error instead of guessing from an adapter fingerprint.
|
|
403
|
+
const settled = await this.settleDeliveryLine(pane, slot.cwd, transcript, this.inputBoxes.get(slot));
|
|
404
|
+
transcript = settled.transcript;
|
|
405
|
+
if (!settled.ok) {
|
|
398
406
|
throw new Error(`herdr delivery clear failed — refusing to retype onto a corrupted line (OBS-85); pane transcript:\n${transcript}`);
|
|
407
|
+
}
|
|
408
|
+
if (!settled.recognizedInputBox) {
|
|
409
|
+
// Clear the corrupted shell input line before retyping; a failed clear must NOT be retyped
|
|
410
|
+
// onto — corrupt-prefix + clean-retype would concatenate and false-verify by containment.
|
|
411
|
+
// A stable adapter-declared input box is already an empty legitimate delivery target.
|
|
412
|
+
const cleared = await this.herdr(`pane send-keys ${shq(pane)} C-u`, slot.cwd);
|
|
413
|
+
if (cleared.code !== 0)
|
|
414
|
+
throw new Error(`herdr delivery clear failed — refusing to retype onto a corrupted line (OBS-85); pane transcript:\n${transcript}`);
|
|
415
|
+
}
|
|
399
416
|
}
|
|
400
417
|
const typed = await this.herdr(`pane send-text ${shq(pane)} ${shq(cmd)}`, slot.cwd);
|
|
401
418
|
if (typed.code !== 0)
|
|
@@ -412,6 +429,26 @@ export class HerdrDriver {
|
|
|
412
429
|
}
|
|
413
430
|
throw new Error(`herdr delivery corrupted after ${DELIVERY_ATTEMPTS} attempts — enter never pressed (OBS-85); pane transcript:\n${transcript}`);
|
|
414
431
|
}
|
|
432
|
+
async settleDeliveryLine(pane, cwd, initialTranscript, inputBox) {
|
|
433
|
+
let transcript = initialTranscript;
|
|
434
|
+
for (let readAttempt = 0; readAttempt < DELIVERY_SETTLE_READ_ATTEMPTS; readAttempt++) {
|
|
435
|
+
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
|
|
436
|
+
if (read.code !== 0)
|
|
437
|
+
return { ok: false, transcript: read.stdout || transcript, recognizedInputBox: false };
|
|
438
|
+
if (read.stdout === transcript) {
|
|
439
|
+
return {
|
|
440
|
+
ok: true,
|
|
441
|
+
transcript,
|
|
442
|
+
recognizedInputBox: inputBox !== undefined && matchesInputBox(transcript, inputBox),
|
|
443
|
+
};
|
|
444
|
+
}
|
|
445
|
+
transcript = read.stdout;
|
|
446
|
+
if (readAttempt < DELIVERY_SETTLE_READ_ATTEMPTS - 1) {
|
|
447
|
+
await new Promise((resolve) => setTimeout(resolve, DELIVERY_SETTLE_POLL_MS));
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
return { ok: false, transcript, recognizedInputBox: false };
|
|
451
|
+
}
|
|
415
452
|
async deliveryReadMatches(pane, cmd, cwd) {
|
|
416
453
|
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, cwd);
|
|
417
454
|
return read.code === 0 && this.deliveryMatches(read.stdout, cmd);
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -172,6 +172,22 @@ function compressLines(lines) {
|
|
|
172
172
|
}
|
|
173
173
|
return parts.join(", ");
|
|
174
174
|
}
|
|
175
|
+
// OBS-134: whole-file deletions are already fully described by their path. Sending every removed line
|
|
176
|
+
// spends the cap and judge context on content that cannot exist after the change. Added and modified
|
|
177
|
+
// sections pass through byte-for-byte, so their anti-flooding budget is unchanged.
|
|
178
|
+
function judgeRelevantDiff(diff) {
|
|
179
|
+
return diff.split(/(?=^diff --git )/m).map((section) => {
|
|
180
|
+
if (!/^deleted file mode /m.test(section))
|
|
181
|
+
return section;
|
|
182
|
+
const oldPath = /^--- (.+)$/m.exec(section)?.[1]
|
|
183
|
+
?? /^Binary files (.+) and \/dev\/null differ$/m.exec(section)?.[1];
|
|
184
|
+
if (!oldPath || oldPath === "/dev/null")
|
|
185
|
+
return section;
|
|
186
|
+
const unquoted = oldPath.startsWith('"') && oldPath.endsWith('"') ? oldPath.slice(1, -1) : oldPath;
|
|
187
|
+
const path = unquoted.replace(/^a\//, "");
|
|
188
|
+
return `deleted file: ${path}\n`;
|
|
189
|
+
}).join("");
|
|
190
|
+
}
|
|
175
191
|
// A citation is valid evidence iff the cited line falls inside a changed hunk of the cited file (OBS-129:
|
|
176
192
|
// changed-hunk span, not exact-added-line). A legacy free-text quote (string) keeps v1.64's substring
|
|
177
193
|
// check — non-empty and present somewhere in the diff.
|
|
@@ -225,7 +241,9 @@ export async function acceptanceGate(task, worktree, baseRef, judge, via, opts =
|
|
|
225
241
|
const warn = onlyJudge
|
|
226
242
|
? "WARNING: only judge oracles — no deterministic command/test oracle guards this task (deterministic preferred, spec §2).\n"
|
|
227
243
|
: "";
|
|
228
|
-
const
|
|
244
|
+
const fetched = await fetchTaskDiff(worktree, baseRef);
|
|
245
|
+
const diff = judgeRelevantDiff(fetched.full);
|
|
246
|
+
const forCap = judgeRelevantDiff(fetched.forCap);
|
|
229
247
|
const diffCap = opts.diffCap ?? DEFAULT_DIFF_CAP;
|
|
230
248
|
const capFail = checkDiffCap("acceptance", forCap.length, diffCap, warn + detBlock);
|
|
231
249
|
if (capFail)
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -26,6 +26,10 @@ export interface GateVia {
|
|
|
26
26
|
nameFor: (role: "judge" | "review", adapter: string) => string;
|
|
27
27
|
labelFor: (role: "judge" | "review") => string;
|
|
28
28
|
}
|
|
29
|
+
export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
|
|
30
|
+
value: T;
|
|
31
|
+
outputs: string[];
|
|
32
|
+
}>;
|
|
29
33
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
30
34
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
|
31
35
|
export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
|
package/dist/gates/llm.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { AsyncLocalStorage } from "node:async_hooks";
|
|
1
2
|
import { randomBytes } from "node:crypto";
|
|
2
3
|
import { mkdtempSync, writeFileSync } from "node:fs";
|
|
3
4
|
import { tmpdir } from "node:os";
|
|
@@ -82,6 +83,16 @@ export function rolePaneNameFromPrompt(prompt, fallback) {
|
|
|
82
83
|
return gatePaneName("review", id, retry ? "-r1" : "");
|
|
83
84
|
return fallback;
|
|
84
85
|
}
|
|
86
|
+
const llmOutputCapture = new AsyncLocalStorage();
|
|
87
|
+
// OBS-132: acceptance.ts owns verdict parsing and is deliberately byte-untouched. This async-scoped
|
|
88
|
+
// recorder lets run-gates observe the exact output that acceptance parsed without changing runLlm's
|
|
89
|
+
// return value or leaking concurrent tasks into one another. Callers retain output only when the
|
|
90
|
+
// resulting verdict is a production failure; healthy output is discarded in memory.
|
|
91
|
+
export async function captureLlmOutput(run) {
|
|
92
|
+
const outputs = [];
|
|
93
|
+
const value = await llmOutputCapture.run(outputs, run);
|
|
94
|
+
return { value, outputs };
|
|
95
|
+
}
|
|
85
96
|
export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
86
97
|
const pf = join(mkdtempSync(join(tmpdir(), "tickmarkr-llm-")), "prompt.md");
|
|
87
98
|
writeFileSync(pf, prompt);
|
|
@@ -119,10 +130,12 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
119
130
|
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
120
131
|
return out;
|
|
121
132
|
}
|
|
122
|
-
export function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
123
|
-
|
|
133
|
+
export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
134
|
+
const out = await (via
|
|
124
135
|
? runViaDriver(adapter, model, prompt, cwd, via, timeoutMs)
|
|
125
|
-
: runHeadless(adapter, model, prompt, cwd, timeoutMs);
|
|
136
|
+
: runHeadless(adapter, model, prompt, cwd, timeoutMs));
|
|
137
|
+
llmOutputCapture.getStore()?.push(out);
|
|
138
|
+
return out;
|
|
126
139
|
}
|
|
127
140
|
export function extractJson(raw) {
|
|
128
141
|
const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
|
|
@@ -2,7 +2,7 @@ import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerRe
|
|
|
2
2
|
import { type TickmarkrConfig } from "../config/config.js";
|
|
3
3
|
import { type GateName, type Task } from "../graph/schema.js";
|
|
4
4
|
import { type Baseline } from "./baseline.js";
|
|
5
|
-
import type
|
|
5
|
+
import { type GateVia } from "./llm.js";
|
|
6
6
|
import type { GateResult } from "./types.js";
|
|
7
7
|
export type GateEvent = {
|
|
8
8
|
phase: "start";
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -5,9 +5,11 @@ import { GATE_NAMES } from "../graph/schema.js";
|
|
|
5
5
|
import { acceptanceGate } from "./acceptance.js";
|
|
6
6
|
import { compareToBaseline } from "./baseline.js";
|
|
7
7
|
import { evidenceGate } from "./evidence.js";
|
|
8
|
+
import { captureLlmOutput } from "./llm.js";
|
|
8
9
|
import { marginalCostRank } from "../route/router.js";
|
|
9
10
|
import { reviewGate } from "./review.js";
|
|
10
11
|
import { scopeGate } from "./scope.js";
|
|
12
|
+
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
11
13
|
export async function runGates(task, ctx) {
|
|
12
14
|
const results = [];
|
|
13
15
|
let commits = [];
|
|
@@ -63,7 +65,27 @@ export async function runGates(task, ctx) {
|
|
|
63
65
|
: undefined;
|
|
64
66
|
// v1.19 (T2): testCmd threads the detected test runner to the gate so named-test oracles run
|
|
65
67
|
// deterministically (filtered via -t) before any LLM judge dispatch.
|
|
66
|
-
|
|
68
|
+
const invocations = [];
|
|
69
|
+
const invokeJudge = async (adapter, model, via) => {
|
|
70
|
+
const started = Date.now();
|
|
71
|
+
const captured = await captureLlmOutput(() => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter, model }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
|
|
72
|
+
const channel = channelKey({ adapter: adapter.id, model });
|
|
73
|
+
const unparseable = captured.value.meta?.unparseable === true;
|
|
74
|
+
// acceptanceGate has exactly one runLlm call. Keep the map shape so a future deterministic early
|
|
75
|
+
// return (zero outputs) stays telemetry-free instead of manufacturing a judge invocation.
|
|
76
|
+
for (const output of captured.outputs) {
|
|
77
|
+
invocations.push({
|
|
78
|
+
taskId: task.id,
|
|
79
|
+
channel,
|
|
80
|
+
outcome: unparseable ? "failed" : "done",
|
|
81
|
+
judgeOutcome: unparseable ? "unparseable" : "parseable",
|
|
82
|
+
durationMs: Date.now() - started,
|
|
83
|
+
...(unparseable ? { transcript: output } : {}),
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
return captured.value;
|
|
87
|
+
};
|
|
88
|
+
let a = await invokeJudge(judgeAdapter, ctx.cfg.judge.model, jvia);
|
|
67
89
|
// GATE-09: an unparseable judge verdict retries the JUDGE exactly once on a failover channel — never
|
|
68
90
|
// the worker (run-20260711-185020 P43-03 L70-72 billed a judge flake as a worker attempt). The flaked
|
|
69
91
|
// first verdict NEVER enters results (no false gate-result journal event, no operator notify, no stale
|
|
@@ -96,10 +118,10 @@ export async function runGates(task, ctx) {
|
|
|
96
118
|
? { driver: ctx.via.driver, keep: ctx.via.keep, onSlot: ctx.via.onSlot, name: ctx.via.nameFor("judge", retryAdapter.id) + "-r1", label: ctx.via.labelFor("judge") }
|
|
97
119
|
: undefined;
|
|
98
120
|
// the retry IS a second acceptanceGate call: one code path, one parser, zero new parse leniency.
|
|
99
|
-
a = await
|
|
121
|
+
a = await invokeJudge(retryAdapter, retry.model, retryJvia);
|
|
100
122
|
a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
|
|
101
123
|
}
|
|
102
|
-
await record(a);
|
|
124
|
+
await withJudgeInvocationEvidence(invocations, () => record(a));
|
|
103
125
|
if (failed())
|
|
104
126
|
return { results, commits };
|
|
105
127
|
}
|
package/dist/run/journal.d.ts
CHANGED
|
@@ -92,8 +92,22 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
92
92
|
vacuous: "vacuous";
|
|
93
93
|
skipped: "skipped";
|
|
94
94
|
}>>;
|
|
95
|
+
kind: z.ZodOptional<z.ZodLiteral<"judge">>;
|
|
96
|
+
judgeOutcome: z.ZodOptional<z.ZodEnum<{
|
|
97
|
+
parseable: "parseable";
|
|
98
|
+
unparseable: "unparseable";
|
|
99
|
+
}>>;
|
|
95
100
|
}, z.core.$strip>;
|
|
96
101
|
export type TelemetryRow = z.infer<typeof TelemetryRowSchema>;
|
|
102
|
+
export interface JudgeInvocationEvidence {
|
|
103
|
+
taskId: string;
|
|
104
|
+
channel: string;
|
|
105
|
+
outcome: "done" | "failed";
|
|
106
|
+
judgeOutcome: "parseable" | "unparseable";
|
|
107
|
+
durationMs: number;
|
|
108
|
+
transcript?: string;
|
|
109
|
+
}
|
|
110
|
+
export declare function withJudgeInvocationEvidence<T>(invocations: JudgeInvocationEvidence[], persist: () => Promise<T>): Promise<T>;
|
|
97
111
|
export declare const SIGNAL_BASIS: readonly ["proved", "review-agree", "judge-only", "legacy", "vacuous", "skipped"];
|
|
98
112
|
export type SignalBasis = (typeof SIGNAL_BASIS)[number];
|
|
99
113
|
export declare const SIGNAL_QUALITY: Record<SignalBasis, 0 | 0.25 | 0.5 | 0.75 | 1>;
|
|
@@ -154,4 +168,5 @@ export declare class Journal {
|
|
|
154
168
|
replayExcludedChannels(): Set<string>;
|
|
155
169
|
telemetry(row: TelemetryRow): void;
|
|
156
170
|
readTelemetry(): TelemetryRow[];
|
|
171
|
+
readJudgeTelemetry(): TelemetryRow[];
|
|
157
172
|
}
|
package/dist/run/journal.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { AsyncLocalStorage } from "node:async_hooks";
|
|
1
2
|
import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync } from "node:fs";
|
|
2
3
|
import { join } from "node:path";
|
|
3
4
|
import { z } from "zod";
|
|
@@ -141,7 +142,19 @@ export const TelemetryRowSchema = z.object({
|
|
|
141
142
|
// v1.46 additive (T5): gate-signal quality for routing hygiene — OPTIONAL, absent = legacy (0.25 at fold).
|
|
142
143
|
signalQuality: z.union([z.literal(0), z.literal(0.25), z.literal(0.5), z.literal(0.75), z.literal(1)]).optional(),
|
|
143
144
|
signalBasis: z.enum(["proved", "review-agree", "judge-only", "legacy", "vacuous", "skipped"]).optional(),
|
|
145
|
+
// OBS-132 additive judge invocation rows. Absent means the legacy per-task row. Judge rows reuse the
|
|
146
|
+
// required v1.5 envelope so per-run raw readers stay source-compatible, while readAllTelemetry filters
|
|
147
|
+
// them before profile derivation: verdict production must be observable without becoming worker reward.
|
|
148
|
+
kind: z.literal("judge").optional(),
|
|
149
|
+
judgeOutcome: z.enum(["parseable", "unparseable"]).optional(),
|
|
144
150
|
});
|
|
151
|
+
const judgePersistence = new AsyncLocalStorage();
|
|
152
|
+
// run-gates scopes this context around the existing daemon onGate callback. Journal.append therefore
|
|
153
|
+
// remains the sole persistence boundary: it can enrich the existing judge-retry row and write invocation
|
|
154
|
+
// telemetry without requiring a parallel daemon callback or changing healthy gate-result payloads.
|
|
155
|
+
export function withJudgeInvocationEvidence(invocations, persist) {
|
|
156
|
+
return judgePersistence.run({ invocations, written: false }, persist);
|
|
157
|
+
}
|
|
145
158
|
// v1.46 T5 (Sol signal telemetry): gate-result rows carry explicit signalQuality so future defect windows
|
|
146
159
|
// are identifiable without forensics. Basis is the provenance claim; quality is the dyadic h-fold weight.
|
|
147
160
|
export const SIGNAL_BASIS = ["proved", "review-agree", "judge-only", "legacy", "vacuous", "skipped"];
|
|
@@ -259,7 +272,7 @@ export function readAllTelemetry(repoRoot, lastK, opts = {}) {
|
|
|
259
272
|
for (const runId of runIds) {
|
|
260
273
|
for (const raw of readJsonl(join(dir, runId, "telemetry.jsonl"))) {
|
|
261
274
|
const r = TelemetryRowSchema.safeParse(raw);
|
|
262
|
-
if (r.success)
|
|
275
|
+
if (r.success && r.data.kind !== "judge")
|
|
263
276
|
out.push({ ...r.data, runId });
|
|
264
277
|
}
|
|
265
278
|
}
|
|
@@ -358,7 +371,18 @@ export class Journal {
|
|
|
358
371
|
return join(this.dir, "journal.jsonl");
|
|
359
372
|
}
|
|
360
373
|
append(event, taskId, data = {}) {
|
|
361
|
-
const
|
|
374
|
+
const evidence = judgePersistence.getStore();
|
|
375
|
+
const failed = evidence?.invocations.filter((invocation) => invocation.transcript !== undefined) ?? [];
|
|
376
|
+
const persistedData = event === "judge-retry" && failed.length > 0
|
|
377
|
+
? {
|
|
378
|
+
...data,
|
|
379
|
+
transcript: failed[0].transcript,
|
|
380
|
+
...(failed[1] ? { retryTranscript: failed[1].transcript } : {}),
|
|
381
|
+
}
|
|
382
|
+
: data;
|
|
383
|
+
const row = {
|
|
384
|
+
ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data: persistedData,
|
|
385
|
+
};
|
|
362
386
|
// T3 secret redaction: only the persisted bytes are masked — the caller's data stays untouched in
|
|
363
387
|
// memory. The narrator receives the persisted (masked) row so a pane sink never shows a credential.
|
|
364
388
|
const line = redactSecrets(JSON.stringify(row));
|
|
@@ -369,6 +393,26 @@ export class Journal {
|
|
|
369
393
|
catch {
|
|
370
394
|
// narration is observational; a broken sink must not affect the journal or run
|
|
371
395
|
}
|
|
396
|
+
if (evidence && !evidence.written && (event === "judge-retry" || event === "gate-result")) {
|
|
397
|
+
evidence.written = true;
|
|
398
|
+
for (const invocation of evidence.invocations) {
|
|
399
|
+
const colon = invocation.channel.indexOf(":");
|
|
400
|
+
const adapter = colon === -1 ? invocation.channel : invocation.channel.slice(0, colon);
|
|
401
|
+
const model = colon === -1 ? "" : invocation.channel.slice(colon + 1);
|
|
402
|
+
this.telemetry({
|
|
403
|
+
kind: "judge",
|
|
404
|
+
taskId: invocation.taskId,
|
|
405
|
+
shape: "judge",
|
|
406
|
+
adapter,
|
|
407
|
+
model,
|
|
408
|
+
channel: invocation.channel,
|
|
409
|
+
attempts: 1,
|
|
410
|
+
outcome: invocation.outcome,
|
|
411
|
+
durationMs: invocation.durationMs,
|
|
412
|
+
judgeOutcome: invocation.judgeOutcome,
|
|
413
|
+
});
|
|
414
|
+
}
|
|
415
|
+
}
|
|
372
416
|
}
|
|
373
417
|
phaseStart(taskId, phase, data = {}) {
|
|
374
418
|
this.append("phase-start", taskId, { ...data, phase });
|
|
@@ -506,8 +550,15 @@ export class Journal {
|
|
|
506
550
|
// T3 secret redaction: same persistence seam as append — credential-free rows are byte-identical.
|
|
507
551
|
appendFileSync(join(this.dir, "telemetry.jsonl"), redactSecrets(JSON.stringify(row)) + "\n");
|
|
508
552
|
}
|
|
509
|
-
// Per-run
|
|
553
|
+
// Per-run task rows stay byte-compatible for legacy readers (report and task-level test helpers).
|
|
554
|
+
// Judge rows share telemetry.jsonl but have a dedicated reader so their earlier gate-time ordering
|
|
555
|
+
// cannot make a find(taskId) caller mistake invocation evidence for the later terminal task row.
|
|
510
556
|
readTelemetry() {
|
|
511
|
-
return readJsonl(join(this.dir, "telemetry.jsonl"))
|
|
557
|
+
return readJsonl(join(this.dir, "telemetry.jsonl"))
|
|
558
|
+
.filter((row) => !(row && typeof row === "object" && "kind" in row && row.kind === "judge"));
|
|
559
|
+
}
|
|
560
|
+
readJudgeTelemetry() {
|
|
561
|
+
return readJsonl(join(this.dir, "telemetry.jsonl"))
|
|
562
|
+
.filter((row) => row && typeof row === "object" && "kind" in row && row.kind === "judge");
|
|
512
563
|
}
|
|
513
564
|
}
|
package/package.json
CHANGED
|
@@ -15,8 +15,9 @@ When working in a multi-agent terminal environment, decide your role before star
|
|
|
15
15
|
- **Orchestrator:** your session was started to execute the mission. Rename your own tab/pane `ORCH · <version>` (short labels: ≤20 chars, `ROLE · token`) and run the loop below.
|
|
16
16
|
- **Supervisor with a live orchestrator:** do not start a second run. Relay the mission to the existing orchestrator with a [verified handoff](#verified-handoffs-agent-to-agent-messaging), then supervise it as OVERSEER.
|
|
17
17
|
- **Primary session without an orchestrator:** rename your own tab `OVERSEER · <version>` and your agent `overseer`, spawn one child orchestration session with your host's launch form, label its tab `ORCH · <version>` and name its agent, give it the mission and these rules verbatim, then supervise it. Do not drive a duplicate single-tier run yourself. Before spawning, confirm any PREVIOUS orchestrator has stood down (monitors stopped, input box empty — dim ghost-text suggestions are UI, not queued input; ANSI-verify before alarming) and close its tab — the journal, records, and ledger hold the story; scrollback is disposable.
|
|
18
|
-
- **
|
|
19
|
-
- **
|
|
18
|
+
- **Spawning on current herdr is two-step** — the one-shot `agent start --cwd/--tab/--no-focus` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138). First create the pane: `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` (tab create does not steal focus unless `--focus` is passed; parse `result.root_pane.pane_id` from its JSON), then start the agent in it:
|
|
19
|
+
- **Claude Code:** `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions`
|
|
20
|
+
- **Codex:** `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
|
|
20
21
|
- **Auxiliary agents you spawn (consultants, reviewers, scouts) follow the same forms.** Never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions` (tickmarkr's own adapter uses exactly this for workers, judges, and consults). A read-only codex consultant may use `--sandbox read-only`; any codex session that must touch git needs the unsandboxed flag above.
|
|
21
22
|
|
|
22
23
|
Outside a multi-agent terminal environment, run the loop directly.
|
|
@@ -14,8 +14,9 @@ When working in a multi-agent terminal environment, decide your role before star
|
|
|
14
14
|
- **Orchestrator:** your session was started to execute the mission. Rename your own tab/pane `ORCH · <version>` (short labels: ≤20 chars, `ROLE · token`) and run the loop below.
|
|
15
15
|
- **Supervisor with a live orchestrator:** do not start a second run. Relay the mission to the existing orchestrator with a [verified handoff](#verified-handoffs-agent-to-agent-messaging), then supervise it as OVERSEER.
|
|
16
16
|
- **Primary session without an orchestrator:** rename your own tab `OVERSEER · <version>` and your agent `overseer`, spawn one child orchestration session with your host's launch form, label its tab `ORCH · <version>` and name its agent, give it the mission and these rules verbatim, then supervise it. Do not drive a duplicate single-tier run yourself. Before spawning, confirm any PREVIOUS orchestrator has [stood down](#stand-down-mission-end-and-retirement) and close its tab.
|
|
17
|
-
- **
|
|
18
|
-
- **
|
|
17
|
+
- **Spawning on current herdr is two-step** — the one-shot `agent start --cwd/--tab/--no-focus` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138). First create the pane: `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` (tab create does not steal focus unless `--focus` is passed; parse `result.root_pane.pane_id` from its JSON), then start the agent in it:
|
|
18
|
+
- **Claude Code:** `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions`
|
|
19
|
+
- **Codex:** `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` — the unsandboxed flag is REQUIRED, not optional: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation (`git worktree add` cannot lock the ref). Do not downgrade this flag; the herdr pane and repo scope are the containment.
|
|
19
20
|
- **Auxiliary agents you spawn (consultants, reviewers, scouts) follow the same forms.** Never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions` (tickmarkr's own adapter uses exactly this for workers, judges, and consults). A read-only codex consultant may use `--sandbox read-only`; any codex session that must touch git needs the unsandboxed flag above.
|
|
20
21
|
|
|
21
22
|
Outside a multi-agent terminal environment, run the loop directly.
|
|
@@ -29,7 +29,7 @@ Requires `HERDR_ENV=1`; if unset, say so and stop.
|
|
|
29
29
|
main name plus at most ONE hot-state token. Vocabulary: ORCH carries the milestone and progress
|
|
30
30
|
fraction (`ORCH · v1.19 4/5`, updated on every task-done); WORKERS carries the task token (tickmarkr
|
|
31
31
|
updates it). Never long context strings or ✓-chains.
|
|
32
|
-
2. **Orchestrator**: Launch the orchestrator with your agent host. For Claude Code, use `herdr agent start orchestrator --
|
|
32
|
+
2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions`, and a read-only codex consultant may use `--sandbox read-only`.
|
|
33
33
|
3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
|
|
34
34
|
truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
|
|
35
35
|
(inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then send one line:
|