tickmarkr 1.77.0 → 1.79.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code.d.ts +11 -0
- package/dist/adapters/claude-code.js +105 -16
- package/dist/adapters/kimi.d.ts +6 -0
- package/dist/adapters/kimi.js +66 -21
- package/dist/adapters/registry.js +6 -1
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +23 -0
- package/dist/cli/commands/status.d.ts +4 -0
- package/dist/cli/commands/status.js +237 -18
- package/dist/drivers/herdr.d.ts +7 -1
- package/dist/drivers/herdr.js +41 -8
- package/dist/drivers/types.d.ts +22 -1
- package/dist/gates/acceptance.d.ts +5 -0
- package/dist/gates/acceptance.js +5 -3
- package/dist/gates/llm.d.ts +9 -0
- package/dist/gates/llm.js +74 -0
- package/dist/gates/review.d.ts +5 -0
- package/dist/gates/review.js +5 -3
- package/dist/run/journal.d.ts +1 -1
- package/package.json +1 -1
|
@@ -1,5 +1,16 @@
|
|
|
1
1
|
import { type AuthHealth, type TrustDialog, type WorkerAdapter } from "./types.js";
|
|
2
|
+
export declare const CLAUDE_ALIAS_IDENTITY_STAMPS: {
|
|
3
|
+
readonly fable: "claude-fable-5";
|
|
4
|
+
readonly opus: "claude-opus-4-8";
|
|
5
|
+
readonly sonnet: "claude-sonnet-5";
|
|
6
|
+
readonly haiku: "claude-haiku-4-5-20251001";
|
|
7
|
+
};
|
|
8
|
+
export type ClaudeAlias = keyof typeof CLAUDE_ALIAS_IDENTITY_STAMPS;
|
|
9
|
+
export type ClaudeAliasIdentityProbe = (cwd: string, alias: ClaudeAlias) => string | undefined;
|
|
2
10
|
export declare const CLAUDE_TRUST_DIALOG: TrustDialog;
|
|
3
11
|
export declare function claudeSlug(real: string): string;
|
|
12
|
+
export declare function readClaudeAliasIdentity(cwd: string, alias: ClaudeAlias): string | undefined;
|
|
13
|
+
export declare function probeClaudeAliasIdentity(cwd: string, alias: ClaudeAlias): string | undefined;
|
|
14
|
+
export declare function resolveClaudeAliasIdentity(cwd: string, alias: ClaudeAlias, probe?: ClaudeAliasIdentityProbe): string | undefined;
|
|
4
15
|
export declare function probeVersion(bin: string): AuthHealth;
|
|
5
16
|
export declare const claudeCode: WorkerAdapter;
|
|
@@ -3,7 +3,7 @@ import { readdirSync, readFileSync, realpathSync, statSync } from "node:fs";
|
|
|
3
3
|
import { homedir } from "node:os";
|
|
4
4
|
import { join } from "node:path";
|
|
5
5
|
import { parseWorkerResult } from "./prompt.js";
|
|
6
|
-
import { channelsFromConfig, shq, TokenUsageSchema } from "./types.js";
|
|
6
|
+
import { channelsFromConfig, MODEL_ID_RE, shq, TokenUsageSchema } from "./types.js";
|
|
7
7
|
// SPEND-01/SPEND-11: claude writes a per-session JSONL to ~/.claude/projects/<slug>/ where slug is the
|
|
8
8
|
// realpath'd cwd with every non-alphanumeric char replaced by "-" (verified 114/114 — 36-DIAGNOSIS.md).
|
|
9
9
|
// The old `/`-only formula missed the "." in `.tickmarkr/worktrees/…` — ENOENT on every worktree dispatch.
|
|
@@ -20,6 +20,20 @@ import { channelsFromConfig, shq, TokenUsageSchema } from "./types.js";
|
|
|
20
20
|
// Phase 18's operator-price × tokens derivation, not a CLI claim.
|
|
21
21
|
const MAX_SESSION_FILES = 20; // newest-first; a long-lived project dir can hold many sessions
|
|
22
22
|
const MAX_SESSION_BYTES = 8_000_000; // per-file cap; a runaway JSONL cannot make the read unbounded
|
|
23
|
+
// OBS-145: these aliases float at the Claude CLI layer, while their tier/pricing decisions were
|
|
24
|
+
// made for the dated identities below (claude-code seeds stamped 2026-07-09 — doctor's own lint).
|
|
25
|
+
// Updating a stamp is a deliberate benchmark-policy act; doctor only compares against it and never
|
|
26
|
+
// edits tiers, pricing, learned scores, or routing. `opus` is deliberately left at 4-8: the OBS-145
|
|
27
|
+
// drill probed the alias at claude-opus-4-8 matching its stamp, then captured it re-pointing to
|
|
28
|
+
// claude-opus-5 ~30 min later with zero events fired. The 2026-07-24 fit added an EXPLICIT
|
|
29
|
+
// claude-opus-5 channel to the repo overlay — it did not re-date this alias's stamps, so the
|
|
30
|
+
// alias channel still carries 4-8-dated tier/pricing while serving 5. That warning is true.
|
|
31
|
+
export const CLAUDE_ALIAS_IDENTITY_STAMPS = {
|
|
32
|
+
fable: "claude-fable-5",
|
|
33
|
+
opus: "claude-opus-4-8",
|
|
34
|
+
sonnet: "claude-sonnet-5",
|
|
35
|
+
haiku: "claude-haiku-4-5-20251001",
|
|
36
|
+
};
|
|
23
37
|
// v1.75 T2 / OBS-137: current Claude Code workspace-trust prompt (2.1.218). The full question
|
|
24
38
|
// distinguishes this startup gate from routine agent text; Enter accepts the selected trust option.
|
|
25
39
|
export const CLAUDE_TRUST_DIALOG = {
|
|
@@ -29,6 +43,95 @@ export const CLAUDE_TRUST_DIALOG = {
|
|
|
29
43
|
export function claudeSlug(real) {
|
|
30
44
|
return real.replace(/[^A-Za-z0-9]/g, "-");
|
|
31
45
|
}
|
|
46
|
+
// newest-first by mtime, bounded — mtime picks WHICH files to scan, never a record's cursor.
|
|
47
|
+
// Shared by collectUsage (spend) and readClaudeAliasIdentity (OBS-145): same store, same bounds.
|
|
48
|
+
function newestSessionFiles(dir) {
|
|
49
|
+
return readdirSync(dir)
|
|
50
|
+
.filter((f) => f.endsWith(".jsonl"))
|
|
51
|
+
.map((f) => {
|
|
52
|
+
try {
|
|
53
|
+
return { f, m: statSync(join(dir, f)).mtimeMs };
|
|
54
|
+
}
|
|
55
|
+
catch {
|
|
56
|
+
return undefined; // a stat failure just drops that file
|
|
57
|
+
}
|
|
58
|
+
})
|
|
59
|
+
.filter((x) => x !== undefined)
|
|
60
|
+
.sort((a, b) => b.m - a.m)
|
|
61
|
+
.slice(0, MAX_SESSION_FILES)
|
|
62
|
+
.map((x) => x.f);
|
|
63
|
+
}
|
|
64
|
+
const belongsToAliasFamily = (identity, alias) => identity === alias || identity.split(/[^A-Za-z0-9]+/).includes(alias);
|
|
65
|
+
// Zero-token first choice: Claude assistant records state the resolved message.model. The newest
|
|
66
|
+
// matching family record in this cwd wins. Every failure is advisory/unknown, never an exception.
|
|
67
|
+
export function readClaudeAliasIdentity(cwd, alias) {
|
|
68
|
+
try {
|
|
69
|
+
const real = realpathSync(cwd);
|
|
70
|
+
const dir = join(homedir(), ".claude", "projects", claudeSlug(real));
|
|
71
|
+
let newest;
|
|
72
|
+
for (const f of newestSessionFiles(dir)) {
|
|
73
|
+
let text;
|
|
74
|
+
try {
|
|
75
|
+
text = readFileSync(join(dir, f), "utf8").slice(0, MAX_SESSION_BYTES);
|
|
76
|
+
}
|
|
77
|
+
catch {
|
|
78
|
+
continue;
|
|
79
|
+
}
|
|
80
|
+
for (const line of text.split("\n")) {
|
|
81
|
+
if (!line.trim())
|
|
82
|
+
continue;
|
|
83
|
+
let raw;
|
|
84
|
+
try {
|
|
85
|
+
raw = JSON.parse(line);
|
|
86
|
+
}
|
|
87
|
+
catch {
|
|
88
|
+
continue;
|
|
89
|
+
}
|
|
90
|
+
const rec = raw;
|
|
91
|
+
if (rec.cwd !== real || typeof rec.message?.model !== "string")
|
|
92
|
+
continue;
|
|
93
|
+
const identity = rec.message.model;
|
|
94
|
+
const timestamp = Date.parse(String(rec.timestamp));
|
|
95
|
+
if (!MODEL_ID_RE.test(identity) || !belongsToAliasFamily(identity, alias) || !Number.isFinite(timestamp))
|
|
96
|
+
continue;
|
|
97
|
+
if (!newest || timestamp > newest.timestamp)
|
|
98
|
+
newest = { identity, timestamp };
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
return newest?.identity;
|
|
102
|
+
}
|
|
103
|
+
catch {
|
|
104
|
+
return undefined;
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
// Store silence earns one minimal stated-identity turn. The output contract is intentionally strict:
|
|
108
|
+
// one safe model id from the requested alias family, otherwise unknown/fail-open. Vitest never spawns
|
|
109
|
+
// a real agent CLI; tests inject the probe callback at resolveClaudeAliasIdentity instead.
|
|
110
|
+
export function probeClaudeAliasIdentity(cwd, alias) {
|
|
111
|
+
if (process.env.VITEST)
|
|
112
|
+
return undefined;
|
|
113
|
+
try {
|
|
114
|
+
const r = spawnSync("claude", [
|
|
115
|
+
"-p",
|
|
116
|
+
"State the exact model identifier serving this request. Reply with only that identifier.",
|
|
117
|
+
"--model", alias,
|
|
118
|
+
"--permission-mode", "bypassPermissions",
|
|
119
|
+
"--strict-mcp-config",
|
|
120
|
+
"--mcp-config", '{"mcpServers":{}}',
|
|
121
|
+
"--output-format", "text",
|
|
122
|
+
], { cwd, encoding: "utf8", timeout: 60000, stdio: ["ignore", "pipe", "pipe"] });
|
|
123
|
+
if (r.error || r.status !== 0)
|
|
124
|
+
return undefined;
|
|
125
|
+
const identity = (r.stdout || "").trim();
|
|
126
|
+
return MODEL_ID_RE.test(identity) && belongsToAliasFamily(identity, alias) ? identity : undefined;
|
|
127
|
+
}
|
|
128
|
+
catch {
|
|
129
|
+
return undefined;
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
export function resolveClaudeAliasIdentity(cwd, alias, probe = probeClaudeAliasIdentity) {
|
|
133
|
+
return readClaudeAliasIdentity(cwd, alias) ?? probe(cwd, alias);
|
|
134
|
+
}
|
|
32
135
|
export function probeVersion(bin) {
|
|
33
136
|
const r = spawnSync(bin, ["--version"], { encoding: "utf8", timeout: 10000 });
|
|
34
137
|
if (r.error || r.status !== 0)
|
|
@@ -75,24 +178,10 @@ export const claudeCode = {
|
|
|
75
178
|
const real = realpathSync(cwd); // resolve symlinks (darwin /tmp → /private/tmp)
|
|
76
179
|
const slug = claudeSlug(real);
|
|
77
180
|
const dir = join(homedir(), ".claude", "projects", slug);
|
|
78
|
-
// newest-first by mtime, bounded — mtime picks WHICH files to scan, never a record's cursor.
|
|
79
|
-
const files = readdirSync(dir)
|
|
80
|
-
.filter((f) => f.endsWith(".jsonl"))
|
|
81
|
-
.map((f) => {
|
|
82
|
-
try {
|
|
83
|
-
return { f, m: statSync(join(dir, f)).mtimeMs };
|
|
84
|
-
}
|
|
85
|
-
catch {
|
|
86
|
-
return undefined; // a stat failure just drops that file
|
|
87
|
-
}
|
|
88
|
-
})
|
|
89
|
-
.filter((x) => x !== undefined)
|
|
90
|
-
.sort((a, b) => b.m - a.m)
|
|
91
|
-
.slice(0, MAX_SESSION_FILES);
|
|
92
181
|
let input = 0, output = 0, kept = false;
|
|
93
182
|
let cacheRead, cacheWrite;
|
|
94
183
|
const seen = new Set(); // message.id is globally unique across session files in one call
|
|
95
|
-
for (const
|
|
184
|
+
for (const f of newestSessionFiles(dir)) {
|
|
96
185
|
let text;
|
|
97
186
|
try {
|
|
98
187
|
text = readFileSync(join(dir, f), "utf8").slice(0, MAX_SESSION_BYTES);
|
package/dist/adapters/kimi.d.ts
CHANGED
|
@@ -13,9 +13,15 @@ export declare function kimiBannerModel(banner: string): string | undefined;
|
|
|
13
13
|
export declare function kimiBannerSessionId(banner: string): string | undefined;
|
|
14
14
|
export type KimiBannerConfirm = {
|
|
15
15
|
ok: true;
|
|
16
|
+
status: "confirmed";
|
|
17
|
+
sessionId?: string;
|
|
18
|
+
} | {
|
|
19
|
+
status: "unknown";
|
|
20
|
+
printedModel?: string;
|
|
16
21
|
sessionId?: string;
|
|
17
22
|
} | {
|
|
18
23
|
ok: false;
|
|
24
|
+
status: "mismatch";
|
|
19
25
|
error: string;
|
|
20
26
|
};
|
|
21
27
|
export declare function confirmKimiSeedBanner(banner: string, assignedModel: string): KimiBannerConfirm;
|
package/dist/adapters/kimi.js
CHANGED
|
@@ -96,11 +96,26 @@ export async function probeKimiDoctorTurn(cwd) {
|
|
|
96
96
|
// Prefer a banner Session line when present so seed-mode captures the id from the launch banner
|
|
97
97
|
// itself rather than waiting for a completion-time resume trailer (which the TUI may never emit).
|
|
98
98
|
const RESUME_TRAILER_RE = /^\s*To resume this session: kimi -r (session_[0-9a-f-]+)\s*$/;
|
|
99
|
-
const
|
|
99
|
+
const BANNER_SESSION_RE = /^Session:\s*(session_[0-9a-f-]+)$/;
|
|
100
|
+
const BANNER_MODEL_RE = /^Model:\s*(.+)$/;
|
|
101
|
+
function kimiBannerLines(banner) {
|
|
102
|
+
// The live TUI renders banner fields as indented rows inside `│ ... │`. Strip only that
|
|
103
|
+
// per-line margin/chrome, following prompt.ts's pane-text precedent; preserve field content.
|
|
104
|
+
return banner.split("\n")
|
|
105
|
+
.map((line) => line.replace(/^[\s│|]+/, "").replace(/[\s│|]+$/, ""));
|
|
106
|
+
}
|
|
107
|
+
function kimiBannerPrintedModel(banner) {
|
|
108
|
+
for (const line of kimiBannerLines(banner)) {
|
|
109
|
+
const printedModel = BANNER_MODEL_RE.exec(line)?.[1]?.trim();
|
|
110
|
+
if (printedModel)
|
|
111
|
+
return printedModel;
|
|
112
|
+
}
|
|
113
|
+
return undefined;
|
|
114
|
+
}
|
|
100
115
|
export function kimiSessionId(output) {
|
|
101
116
|
let id;
|
|
102
|
-
for (const line of output
|
|
103
|
-
const banner =
|
|
117
|
+
for (const line of kimiBannerLines(output)) {
|
|
118
|
+
const banner = BANNER_SESSION_RE.exec(line);
|
|
104
119
|
if (banner) {
|
|
105
120
|
id = banner[1];
|
|
106
121
|
continue;
|
|
@@ -113,29 +128,48 @@ export function kimiSessionId(output) {
|
|
|
113
128
|
}
|
|
114
129
|
// v1.69 T7: the cold-start banner prints the model alias and session id. Parse them from the
|
|
115
130
|
// banner text already captured for the readiness match — no new probe, no extra dispatch. Kimi
|
|
116
|
-
//
|
|
117
|
-
//
|
|
118
|
-
const
|
|
119
|
-
|
|
131
|
+
// may print a display name, the full config key, or its legacy suffix. Resolve only names with a
|
|
132
|
+
// known channel mapping: inventing `kimi-code/<future display>` would turn unknown into mismatch.
|
|
133
|
+
const KIMI_BANNER_MODEL_CHANNELS = new Map([
|
|
134
|
+
["K3", "kimi-code/k3"],
|
|
135
|
+
["k3", "kimi-code/k3"],
|
|
136
|
+
["kimi-code/k3", "kimi-code/k3"],
|
|
137
|
+
["K2.7 Coding", "kimi-code/kimi-for-coding"],
|
|
138
|
+
["kimi-for-coding", "kimi-code/kimi-for-coding"],
|
|
139
|
+
["kimi-code/kimi-for-coding", "kimi-code/kimi-for-coding"],
|
|
140
|
+
["K2.7 Coding Highspeed", "kimi-code/kimi-for-coding-highspeed"],
|
|
141
|
+
["kimi-for-coding-highspeed", "kimi-code/kimi-for-coding-highspeed"],
|
|
142
|
+
["kimi-code/kimi-for-coding-highspeed", "kimi-code/kimi-for-coding-highspeed"],
|
|
143
|
+
]);
|
|
120
144
|
export function kimiBannerModel(banner) {
|
|
121
|
-
const
|
|
122
|
-
|
|
123
|
-
return undefined;
|
|
124
|
-
const printedModel = m[1].trim();
|
|
125
|
-
if (!printedModel)
|
|
126
|
-
return undefined;
|
|
127
|
-
return printedModel.startsWith("kimi-code/") ? printedModel : `kimi-code/${printedModel}`;
|
|
145
|
+
const printedModel = kimiBannerPrintedModel(banner);
|
|
146
|
+
return printedModel === undefined ? undefined : KIMI_BANNER_MODEL_CHANNELS.get(printedModel);
|
|
128
147
|
}
|
|
129
148
|
export function kimiBannerSessionId(banner) {
|
|
130
|
-
|
|
149
|
+
for (const line of kimiBannerLines(banner)) {
|
|
150
|
+
const sessionId = BANNER_SESSION_RE.exec(line)?.[1];
|
|
151
|
+
if (sessionId)
|
|
152
|
+
return sessionId;
|
|
153
|
+
}
|
|
154
|
+
return undefined;
|
|
131
155
|
}
|
|
132
|
-
// Fail closed on a
|
|
156
|
+
// Fail closed on a mapped-model mismatch. Unknown/missing names remain non-blocking for forward
|
|
157
|
+
// compatibility, but carry an explicit unknown status rather than masquerading as confirmation.
|
|
133
158
|
export function confirmKimiSeedBanner(banner, assignedModel) {
|
|
159
|
+
const sessionId = kimiBannerSessionId(banner);
|
|
160
|
+
const printedModel = kimiBannerPrintedModel(banner);
|
|
134
161
|
const saw = kimiBannerModel(banner);
|
|
135
|
-
if (
|
|
136
|
-
return {
|
|
162
|
+
if (printedModel === undefined || saw === undefined) {
|
|
163
|
+
return {
|
|
164
|
+
status: "unknown",
|
|
165
|
+
...(printedModel === undefined ? {} : { printedModel }),
|
|
166
|
+
...(sessionId === undefined ? {} : { sessionId }),
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
if (saw !== assignedModel) {
|
|
170
|
+
return { ok: false, status: "mismatch", error: `model mismatch: expected ${assignedModel}, saw ${saw}` };
|
|
137
171
|
}
|
|
138
|
-
return { ok: true, sessionId:
|
|
172
|
+
return { ok: true, status: "confirmed", ...(sessionId === undefined ? {} : { sessionId }) };
|
|
139
173
|
}
|
|
140
174
|
const kimiTuiLaunch = (model) => `kimi -y -m ${shq(model)}`;
|
|
141
175
|
const KIMI_ANSI_SGR_RE = /\u001B\[[0-9;]*m/g;
|
|
@@ -144,7 +178,11 @@ const KIMI_EDITOR_INPUT_RE = /^│ > .*│$/;
|
|
|
144
178
|
const KIMI_EDITOR_EMPTY_RE = /^│ > +│$/;
|
|
145
179
|
const KIMI_EDITOR_BOTTOM_RE = /^╰─+╯$/;
|
|
146
180
|
function matchesKimiEditorBox(paneText, empty) {
|
|
147
|
-
|
|
181
|
+
// OBS-152: kimi renders its whole TUI indented one column, so `^`-anchored row matches never fire
|
|
182
|
+
// on a LIVE editor box — probe 6 timed out readiness against a frame whose box was fully painted
|
|
183
|
+
// and interactive. Trim per line: the left margin is chrome, and what actually distinguishes the
|
|
184
|
+
// editor from the welcome panel (also a bordered box) is the three-row adjacency asserted below.
|
|
185
|
+
const lines = paneText.replace(KIMI_ANSI_SGR_RE, "").split("\n").map((line) => line.trim());
|
|
148
186
|
const input = empty ? KIMI_EDITOR_EMPTY_RE : KIMI_EDITOR_INPUT_RE;
|
|
149
187
|
return lines.some((line, index) => input.test(line)
|
|
150
188
|
&& index > 0
|
|
@@ -170,7 +208,14 @@ const KIMI_SEED = {
|
|
|
170
208
|
launch: kimiTuiLaunch,
|
|
171
209
|
readinessMatch: "Send /help for help information.",
|
|
172
210
|
seedLine: (promptFile) => `Read ${promptFile} and do exactly what it says.`,
|
|
173
|
-
confirmBanner:
|
|
211
|
+
confirmBanner: (banner, assignedModel) => {
|
|
212
|
+
const confirmation = confirmKimiSeedBanner(banner, assignedModel);
|
|
213
|
+
// The generic seed contract is binary: unknown display names remain admissible while the
|
|
214
|
+
// Kimi-specific result above truthfully keeps them distinct from a confirmed identity.
|
|
215
|
+
return confirmation.status === "unknown"
|
|
216
|
+
? { ok: true, status: confirmation.status, sessionId: confirmation.sessionId }
|
|
217
|
+
: confirmation;
|
|
218
|
+
},
|
|
174
219
|
};
|
|
175
220
|
// v1.69 T7 / v1.71 T2: thin wrapper over the daemon's generic dispatch path for kimi-only tests.
|
|
176
221
|
export async function runKimiInteractiveSeed(opts) {
|
|
@@ -203,7 +203,12 @@ const MODEL_PROBE_PROMPT = "Reply with exactly OK and nothing else.";
|
|
|
203
203
|
// 60s: a healthy MCP-free codex probe measured 29.9s round-trip (2026-07-15) — 30s had zero headroom.
|
|
204
204
|
const MODEL_PROBE_TIMEOUT_MS = 60000;
|
|
205
205
|
// Auth-words only match when tied to a failure word; bare "auth"/"OAuth"/"authored" never fail (v1.27 T2).
|
|
206
|
-
|
|
206
|
+
// OBS-150: the 4xx alternative must refuse digit-group context. Bare `\b4\d\d\b` matched "485" inside the
|
|
207
|
+
// token footer "tokens used 26,485" and failed a HEALTHY probe closed — a ~1-in-10 dice roll on EVERY model
|
|
208
|
+
// probe of any adapter that prints a comma-grouped token count. Lookarounds reject a group that is part of a
|
|
209
|
+
// larger number (26,485 · 1.400 · 400.5) while keeping real bare codes (401 · "error 429" · "403 Forbidden."
|
|
210
|
+
// — trailing sentence punctuation still matches, only a following DIGIT means "decimal, not status code").
|
|
211
|
+
const AUTH_FAILURE_RE = /\b(?<![\d.,])4\d\d\b(?![.,]\d)|\bauth(?:entication|orization)?\s+(?:error|failed|failure|denied)|unauthori[sz]ed|forbidden|access denied|credit(?:s)?\s+(?:exhausted|error|denied)/i;
|
|
207
212
|
const PROBE_REASON_CAP = 240;
|
|
208
213
|
// v1.55 T3: the tail must open at a word boundary — a mid-word slice ("odel 'grok-…'") reads as
|
|
209
214
|
// corruption in operator-facing diagnostics. Input is already space-normalized, so " " is the only
|
|
@@ -1,7 +1,9 @@
|
|
|
1
|
+
import { type ClaudeAlias } from "../../adapters/claude-code.js";
|
|
1
2
|
import type { WorkerAdapter } from "../../adapters/types.js";
|
|
2
3
|
import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
|
|
3
4
|
export type DoctorOpts = {
|
|
4
5
|
banner?: boolean;
|
|
5
6
|
kimiTurnProbe?: (cwd: string) => Promise<KimiDoctorTurnResult>;
|
|
7
|
+
resolveClaudeAliasIdentity?: (cwd: string, alias: ClaudeAlias) => string | undefined;
|
|
6
8
|
};
|
|
7
9
|
export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { writeFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { allAdapters, binaryShadowWarnings, detectCandidateClis, flagDriftWarnings, modelAliasExclusions, modelAliasLine, probeAll, probeModels, readAutoPrefer, servableExclusions, servabilityLine, writeDoctor } from "../../adapters/registry.js";
|
|
4
|
+
import { CLAUDE_ALIAS_IDENTITY_STAMPS, claudeCode, resolveClaudeAliasIdentity } from "../../adapters/claude-code.js";
|
|
4
5
|
import { BANNER, dim, fail, kvRow, legend, ok, rule, statusRow, title } from "../../brand.js";
|
|
5
6
|
import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
|
|
6
7
|
import { declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
|
|
@@ -120,6 +121,28 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
120
121
|
// v1.65 T3: hardcoded-flag drift — advisory warn rows only. Runs AFTER writeDoctor so the verdicts
|
|
121
122
|
// can never leak into doctor.json, and discoverChannels/routing never read them.
|
|
122
123
|
rows.push(...flagDriftWarnings(adapters, health).map(attentionRow));
|
|
124
|
+
// OBS-145: resolved-identity drift is the same class of display-only doctor warning. Stamps live
|
|
125
|
+
// beside the alias-owning adapter; the comparison runs only for configured floating aliases and
|
|
126
|
+
// never enters health/doctor.json, config, channel discovery, learned profiles, or route().
|
|
127
|
+
const identityResolver = opts.resolveClaudeAliasIdentity
|
|
128
|
+
?? (adapters.includes(claudeCode) ? resolveClaudeAliasIdentity : undefined);
|
|
129
|
+
if (identityResolver && health["claude-code"]?.installed) {
|
|
130
|
+
const configured = cfg.tiers["claude-code"]?.models ?? {};
|
|
131
|
+
for (const [alias, stampedIdentity] of Object.entries(CLAUDE_ALIAS_IDENTITY_STAMPS)) {
|
|
132
|
+
if (!(alias in configured))
|
|
133
|
+
continue;
|
|
134
|
+
let resolvedIdentity;
|
|
135
|
+
try {
|
|
136
|
+
resolvedIdentity = identityResolver(cwd, alias);
|
|
137
|
+
}
|
|
138
|
+
catch {
|
|
139
|
+
continue; // advisory source failure is unknown, never a doctor failure
|
|
140
|
+
}
|
|
141
|
+
if (resolvedIdentity && resolvedIdentity !== stampedIdentity) {
|
|
142
|
+
rows.push(attentionRow(`resolved-identity drift: claude-code:${alias} resolved to ${resolvedIdentity}, stamped identity ${stampedIdentity} — reclassify per benchmark policy (advisory — routing unchanged)`));
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
}
|
|
123
146
|
// MODEL-05/06: print-only drift fragment; advisory, whole-line-commented additions, tickmarkr NEVER applies it.
|
|
124
147
|
// TTY gets a one-line summary + the fragment as a file (the full dump drowned everything else,
|
|
125
148
|
// v1.33.1 onboarding); machine/CI surface keeps the inline dump — layout is pinned by tests.
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { type DecisionEvent, type DecisionWebhookPost } from "../../drivers/types.js";
|
|
1
2
|
import { type Task, type TaskStatus } from "../../graph/schema.js";
|
|
2
3
|
import { type JournalEvent } from "../../run/journal.js";
|
|
3
4
|
export type StatusOpts = {
|
|
@@ -5,6 +6,8 @@ export type StatusOpts = {
|
|
|
5
6
|
sleep?: (ms: number) => Promise<void>;
|
|
6
7
|
now?: () => number;
|
|
7
8
|
readWorkerOutput?: (taskId: string, attempt: number, runId: string) => Promise<string | undefined>;
|
|
9
|
+
webhookUrl?: string;
|
|
10
|
+
postWebhook?: DecisionWebhookPost;
|
|
8
11
|
};
|
|
9
12
|
export declare const GATE_KEYS: {
|
|
10
13
|
readonly build: "B";
|
|
@@ -15,6 +18,7 @@ export declare const GATE_KEYS: {
|
|
|
15
18
|
readonly acceptance: "A";
|
|
16
19
|
readonly review: "R";
|
|
17
20
|
};
|
|
21
|
+
export declare const decisionEventsFromJournal: (events: JournalEvent[], runId: string, stateDir?: string) => DecisionEvent[];
|
|
18
22
|
export declare const taskBox: (status: TaskStatus) => string;
|
|
19
23
|
export declare const gateBox: (state: "open" | "pass" | "fail" | "skip", unicode: boolean) => string;
|
|
20
24
|
export type GateState = "open" | "pass" | "fail" | "skip";
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { BANNER, GLYPHS, dim, fail, legend, ok, rule, statusRow, title, warn } from "../../brand.js";
|
|
2
2
|
import { HerdrDriver } from "../../drivers/herdr.js";
|
|
3
|
-
import { formatOwnedName } from "../../drivers/types.js";
|
|
4
|
-
import { blockedTasks, graphDefinitionHash, loadGraph } from "../../graph/graph.js";
|
|
3
|
+
import { formatOwnedName, } from "../../drivers/types.js";
|
|
4
|
+
import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../../graph/graph.js";
|
|
5
5
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
6
6
|
import { foldActivity } from "../../run/activity.js";
|
|
7
7
|
import { Journal, engagementComparable, isQualityFailureParkKind, recordedTaskFailureKind, runHasEnded, } from "../../run/journal.js";
|
|
@@ -18,6 +18,89 @@ const SPINNER = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇",
|
|
|
18
18
|
const ASCII_SPINNER = ["|", "/", "-", "\\"];
|
|
19
19
|
const SAVE_TERMINAL_TITLE = "\x1b[22;0t";
|
|
20
20
|
const RESTORE_TERMINAL_TITLE = "\x1b[23;0t";
|
|
21
|
+
const decisionEvidence = (stateDir, runId, sequence) => `${stateDir}/runs/${runId}/journal.jsonl#L${sequence}`;
|
|
22
|
+
// v1.79 T4: a deterministic JSONL projection over journal truth. Sequence/evidence come from the
|
|
23
|
+
// append-only line position, timestamps and claims come from the row, and no watcher-local clock or
|
|
24
|
+
// filesystem write participates. Re-reading the same bytes therefore returns the same event bytes.
|
|
25
|
+
export const decisionEventsFromJournal = (events, runId, stateDir = ".tickmarkr") => events.flatMap((event, index) => {
|
|
26
|
+
const sequence = index + 1;
|
|
27
|
+
const base = {
|
|
28
|
+
version: 1,
|
|
29
|
+
sequence,
|
|
30
|
+
ts: event.ts,
|
|
31
|
+
runId,
|
|
32
|
+
evidence: decisionEvidence(stateDir, runId, sequence),
|
|
33
|
+
...(event.taskId ? { taskId: event.taskId } : {}),
|
|
34
|
+
};
|
|
35
|
+
if (event.event === "phase-start" && typeof event.data.phase === "string") {
|
|
36
|
+
return [{ ...base, type: "phase-change", tier: "routine", phase: event.data.phase }];
|
|
37
|
+
}
|
|
38
|
+
if (event.event === "gate-result" && typeof event.data.gate === "string") {
|
|
39
|
+
const verdict = event.data.skipped === true
|
|
40
|
+
? "skipped"
|
|
41
|
+
: event.data.pass === true
|
|
42
|
+
? "passed"
|
|
43
|
+
: event.data.pass === false
|
|
44
|
+
? "failed"
|
|
45
|
+
: "unknown";
|
|
46
|
+
return [{
|
|
47
|
+
...base,
|
|
48
|
+
type: "gate-verdict",
|
|
49
|
+
tier: verdict === "failed" ? "decision" : "routine",
|
|
50
|
+
gate: event.data.gate,
|
|
51
|
+
verdict,
|
|
52
|
+
}];
|
|
53
|
+
}
|
|
54
|
+
if (event.event === "escalation") {
|
|
55
|
+
return [{
|
|
56
|
+
...base,
|
|
57
|
+
type: "escalation",
|
|
58
|
+
tier: "decision",
|
|
59
|
+
...(typeof event.data.step === "string" ? { step: event.data.step } : {}),
|
|
60
|
+
...(typeof event.data.attempt === "number" ? { attempt: event.data.attempt } : {}),
|
|
61
|
+
}];
|
|
62
|
+
}
|
|
63
|
+
if (event.event === "task-human" && event.taskId) {
|
|
64
|
+
return [{
|
|
65
|
+
...base,
|
|
66
|
+
type: "human-decision-required",
|
|
67
|
+
tier: "decision",
|
|
68
|
+
approvalCommand: `tickmarkr approve ${runId} ${event.taskId}`,
|
|
69
|
+
...(typeof event.data.kind === "string" ? { kind: event.data.kind } : {}),
|
|
70
|
+
...(typeof event.data.reason === "string" ? { reason: event.data.reason } : {}),
|
|
71
|
+
}];
|
|
72
|
+
}
|
|
73
|
+
if (event.event === "run-end") {
|
|
74
|
+
return [{
|
|
75
|
+
...base,
|
|
76
|
+
type: "run-end",
|
|
77
|
+
tier: "decision",
|
|
78
|
+
summary: event.data,
|
|
79
|
+
}];
|
|
80
|
+
}
|
|
81
|
+
return [];
|
|
82
|
+
});
|
|
83
|
+
const optionValue = (argv, name) => {
|
|
84
|
+
const inline = argv.find((arg) => arg.startsWith(`${name}=`));
|
|
85
|
+
if (inline)
|
|
86
|
+
return inline.slice(name.length + 1) || undefined;
|
|
87
|
+
const index = argv.indexOf(name);
|
|
88
|
+
const value = index >= 0 ? argv[index + 1] : undefined;
|
|
89
|
+
return value && !value.startsWith("-") ? value : undefined;
|
|
90
|
+
};
|
|
91
|
+
const defaultPostWebhook = (url, event) => fetch(url, {
|
|
92
|
+
method: "POST",
|
|
93
|
+
headers: { "content-type": "application/json" },
|
|
94
|
+
body: JSON.stringify(event),
|
|
95
|
+
}).then(() => undefined);
|
|
96
|
+
const postWebhookFireAndForget = (post, url, event) => {
|
|
97
|
+
try {
|
|
98
|
+
void Promise.resolve(post(url, event)).catch(() => undefined);
|
|
99
|
+
}
|
|
100
|
+
catch {
|
|
101
|
+
// Webhooks are observational. A synchronous bridge failure is as inert as a rejected request.
|
|
102
|
+
}
|
|
103
|
+
};
|
|
21
104
|
const taskPhase = (value) => {
|
|
22
105
|
if (value === "worker" || value === "gates" || value === "judge" || value === "review" || value === "merge")
|
|
23
106
|
return value;
|
|
@@ -204,6 +287,91 @@ const terminalFailureCause = (events) => {
|
|
|
204
287
|
}
|
|
205
288
|
return undefined;
|
|
206
289
|
};
|
|
290
|
+
const tipFailureEvidence = (events, start, end) => {
|
|
291
|
+
const gates = new Set();
|
|
292
|
+
let fingerprints = 0;
|
|
293
|
+
let hasFingerprintCount = false;
|
|
294
|
+
for (let i = start; i <= end; i++) {
|
|
295
|
+
const event = events[i];
|
|
296
|
+
if (event.event !== "tip-verify-failed")
|
|
297
|
+
continue;
|
|
298
|
+
if (typeof event.data.gate === "string" && event.data.gate.trim())
|
|
299
|
+
gates.add(event.data.gate.trim());
|
|
300
|
+
if (Array.isArray(event.data.fingerprints)) {
|
|
301
|
+
fingerprints += event.data.fingerprints.length;
|
|
302
|
+
hasFingerprintCount = true;
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
return { gates: [...gates], ...(hasFingerprintCount ? { fingerprints } : {}) };
|
|
306
|
+
};
|
|
307
|
+
const priorLifecycleIndex = (events, end) => {
|
|
308
|
+
for (let i = end; i >= 0; i--) {
|
|
309
|
+
if (events[i].event === "run-start" || events[i].event === "run-resume")
|
|
310
|
+
return i;
|
|
311
|
+
}
|
|
312
|
+
return 0;
|
|
313
|
+
};
|
|
314
|
+
// OBS-146: tip verification is a run phase, folded exclusively from append-only journal facts.
|
|
315
|
+
// A completed run-end owns passed/failed; a later resume keeps the prior failure visible while
|
|
316
|
+
// recording a re-verification attempt; an active run becomes pending only after a recorded merge
|
|
317
|
+
// and recorded non-empty command set make a tip verification applicable. Unknown evidence stays
|
|
318
|
+
// unknown rather than borrowing optimistic state from graph.json.
|
|
319
|
+
const tipVerifyPhase = (events) => {
|
|
320
|
+
let lastRunEnd = -1;
|
|
321
|
+
let lastLifecycle = -1;
|
|
322
|
+
let attempts = 0;
|
|
323
|
+
let hasCommands = false;
|
|
324
|
+
let hasMerge = false;
|
|
325
|
+
for (let i = 0; i < events.length; i++) {
|
|
326
|
+
const event = events[i];
|
|
327
|
+
if (event.event === "run-start" || event.event === "run-resume") {
|
|
328
|
+
lastLifecycle = i;
|
|
329
|
+
attempts++;
|
|
330
|
+
if (event.event === "run-start" && typeof event.data.commands === "object" && event.data.commands !== null
|
|
331
|
+
&& !Array.isArray(event.data.commands) && Object.keys(event.data.commands).length > 0) {
|
|
332
|
+
hasCommands = true;
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
else if (event.event === "run-end") {
|
|
336
|
+
lastRunEnd = i;
|
|
337
|
+
}
|
|
338
|
+
else if (event.event === "merge") {
|
|
339
|
+
hasMerge = true;
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
if (lastRunEnd >= 0 && events[lastRunEnd].data.tipVerify === "failed") {
|
|
343
|
+
const failureStart = priorLifecycleIndex(events, lastRunEnd);
|
|
344
|
+
const priorFailure = tipFailureEvidence(events, failureStart, lastRunEnd);
|
|
345
|
+
if (lastLifecycle > lastRunEnd) {
|
|
346
|
+
const currentFailure = tipFailureEvidence(events, lastLifecycle, events.length - 1);
|
|
347
|
+
if (currentFailure.gates.length > 0 || currentFailure.fingerprints !== undefined) {
|
|
348
|
+
return { state: "failed", ...currentFailure };
|
|
349
|
+
}
|
|
350
|
+
return { state: "re-verifying", ...priorFailure, attempt: Math.max(2, attempts) };
|
|
351
|
+
}
|
|
352
|
+
return { state: "failed", ...priorFailure };
|
|
353
|
+
}
|
|
354
|
+
if (lastLifecycle > lastRunEnd) {
|
|
355
|
+
const currentFailure = tipFailureEvidence(events, lastLifecycle, events.length - 1);
|
|
356
|
+
if (currentFailure.gates.length > 0 || currentFailure.fingerprints !== undefined) {
|
|
357
|
+
return { state: "failed", ...currentFailure };
|
|
358
|
+
}
|
|
359
|
+
if (hasCommands && hasMerge)
|
|
360
|
+
return { state: "pending", gates: [] };
|
|
361
|
+
}
|
|
362
|
+
return undefined;
|
|
363
|
+
};
|
|
364
|
+
const tipVerifyText = (phase) => {
|
|
365
|
+
if (phase.state === "pending")
|
|
366
|
+
return "tip-verify: pending";
|
|
367
|
+
const gates = phase.gates.length ? phase.gates.join(", ") : "gate unknown";
|
|
368
|
+
const failed = `tip-verify: FAILED (${gates})`;
|
|
369
|
+
const reverify = phase.state === "re-verifying" ? ` → re-verifying (attempt ${phase.attempt})` : "";
|
|
370
|
+
const fingerprints = phase.fingerprints === undefined
|
|
371
|
+
? ""
|
|
372
|
+
: ` · ${phase.fingerprints} fingerprint${phase.fingerprints === 1 ? "" : "s"}`;
|
|
373
|
+
return `${failed}${reverify}${fingerprints}`;
|
|
374
|
+
};
|
|
207
375
|
const liveness = (events, now = Date.now()) => {
|
|
208
376
|
const last = events.at(-1);
|
|
209
377
|
if (!last)
|
|
@@ -289,6 +457,7 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
289
457
|
const width = process.stdout.columns ?? 120;
|
|
290
458
|
const done = effective.tasks.filter((t) => t.status === "done").length;
|
|
291
459
|
const ended = comparable && runHasEnded(events);
|
|
460
|
+
const tipPhase = comparable ? tipVerifyPhase(events) : undefined;
|
|
292
461
|
const taskIds = new Set(g.tasks.map((task) => task.id));
|
|
293
462
|
const phases = comparable ? livePhases(events) : new Map();
|
|
294
463
|
for (const taskId of phases.keys())
|
|
@@ -326,7 +495,7 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
326
495
|
return `${prefix}${shortGoal(t.goal, Math.max(0, width - prefix.length - suffix.length))}${suffix}`;
|
|
327
496
|
});
|
|
328
497
|
const header = runId
|
|
329
|
-
? `tickmarkr status${divider}run ${runId}${supersededBy ? `${divider}superseded by ${supersededBy}` : ""}${!comparable ? `${divider}${NOT_COMPARABLE_NOTICE}` : ""}${divider}${liveness(events, now).replaceAll(" · ", divider)}${divider}${done}/${g.tasks.length} done`
|
|
498
|
+
? `tickmarkr status${divider}run ${runId}${supersededBy ? `${divider}superseded by ${supersededBy}` : ""}${!comparable ? `${divider}${NOT_COMPARABLE_NOTICE}` : ""}${divider}${liveness(events, now).replaceAll(" · ", divider)}${tipPhase ? `${divider}${tipVerifyText(tipPhase)}` : ""}${divider}${tipPhase ? `${done}/${g.tasks.length} tasks done${divider}run not verified` : `${done}/${g.tasks.length} done`}`
|
|
330
499
|
: `tickmarkr status${divider}no runs yet${divider}${done}/${g.tasks.length} done`;
|
|
331
500
|
const legendLine = ` gates: ${GATE_NAMES.map((gate) => `${GATE_KEYS[gate]} ${gate}`).join(divider)}`;
|
|
332
501
|
return {
|
|
@@ -344,18 +513,25 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
344
513
|
const anyFailed = cells.some((c) => c.redTier);
|
|
345
514
|
const gaugeCells = 10;
|
|
346
515
|
const fill = g.tasks.length ? Math.round((done / g.tasks.length) * gaugeCells) : 0;
|
|
347
|
-
const
|
|
516
|
+
const tipFailed = tipPhase?.state === "failed";
|
|
517
|
+
const progressTone = anyFailed || tipFailed ? fail : tipPhase ? warn : ok;
|
|
518
|
+
const gauge = (fill ? progressTone("█".repeat(fill)) : "") + (fill < gaugeCells ? dim("░".repeat(gaugeCells - fill)) : "");
|
|
348
519
|
const live = liveness(events, now)
|
|
349
520
|
.replace(/\bdead\b/, fail("dead"))
|
|
350
521
|
.replace(/\bfinished\b/, dim("finished"))
|
|
351
522
|
.replace(/\balive\b/, ok("alive"))
|
|
352
523
|
.replaceAll(" · ", dot);
|
|
353
|
-
const tally =
|
|
524
|
+
const tally = tipPhase
|
|
525
|
+
? `${progressTone(`${done}/${g.tasks.length} tasks done`)}${dot}${progressTone("run not verified")}`
|
|
526
|
+
: `${done}/${g.tasks.length} done`;
|
|
527
|
+
const tipStatus = tipPhase
|
|
528
|
+
? `${tipFailed ? fail(tipVerifyText(tipPhase)) : warn(tipVerifyText(tipPhase))}${dot}`
|
|
529
|
+
: "";
|
|
354
530
|
const header = ` ${title(runId ? `run ${runId}` : "tickmarkr")}${dot}` +
|
|
355
531
|
(runId
|
|
356
|
-
? `${supersededBy ? `${warn(`superseded by ${supersededBy}`)}${dot}` : ""}${!comparable ? `${warn(NOT_COMPARABLE_NOTICE)}${dot}` : ""}${live}${dot}`
|
|
532
|
+
? `${supersededBy ? `${warn(`superseded by ${supersededBy}`)}${dot}` : ""}${!comparable ? `${warn(NOT_COMPARABLE_NOTICE)}${dot}` : ""}${live}${dot}${tipStatus}`
|
|
357
533
|
: `no runs yet${dot}`) +
|
|
358
|
-
`${gauge} ${done === g.tasks.length && g.tasks.length > 0 ? ok(tally) : tally}`;
|
|
534
|
+
`${gauge} ${tipPhase ? tally : done === g.tasks.length && g.tasks.length > 0 ? ok(tally) : tally}`;
|
|
359
535
|
const hr = rule(Math.min(width, 100));
|
|
360
536
|
// OBS-104 run-level now line: names the most recent journal event. TTY frame only — the non-TTY
|
|
361
537
|
// machine surface is byte-pinned (status-brand golden) and must not drift. Rendered BELOW the task
|
|
@@ -417,15 +593,21 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
417
593
|
const { content } = renderFrame(cwd, opts.now?.() ?? Date.now());
|
|
418
594
|
return visual() ? BANNER + content : content;
|
|
419
595
|
}
|
|
596
|
+
const eventStream = argv.some((arg) => arg === "--events" || arg === "--jsonl" || arg === "--decision-events");
|
|
597
|
+
const webhookUrl = opts.webhookUrl
|
|
598
|
+
?? optionValue(argv, "--webhook")
|
|
599
|
+
?? process.env.TICKMARKR_DECISION_WEBHOOK;
|
|
600
|
+
const postWebhook = opts.postWebhook ?? defaultPostWebhook;
|
|
420
601
|
const iterations = opts.iterations ?? Infinity;
|
|
421
602
|
const sleep = opts.sleep ?? defaultSleep;
|
|
422
603
|
const now = opts.now ?? Date.now;
|
|
423
604
|
const bounded = Number.isFinite(iterations);
|
|
424
605
|
const frames = [];
|
|
606
|
+
const eventLines = [];
|
|
425
607
|
const sep = "\n---\n";
|
|
426
|
-
const tty = visual();
|
|
608
|
+
const tty = !eventStream && visual();
|
|
427
609
|
const workerLiveness = new Map();
|
|
428
|
-
const herdr = HerdrDriver.available() ? new HerdrDriver() : undefined;
|
|
610
|
+
const herdr = !eventStream && HerdrDriver.available() ? new HerdrDriver() : undefined;
|
|
429
611
|
const readWorkerOutput = opts.readWorkerOutput ?? (herdr
|
|
430
612
|
? async (taskId, attempt, runId) => {
|
|
431
613
|
const name = formatOwnedName({ role: "worker", taskId, attempt, runId });
|
|
@@ -467,6 +649,30 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
467
649
|
prior.snapshot = snapshot;
|
|
468
650
|
}));
|
|
469
651
|
};
|
|
652
|
+
let decisionRunId;
|
|
653
|
+
let journalCursor = 0;
|
|
654
|
+
const consumeDecisionEvents = () => {
|
|
655
|
+
const runId = Journal.latestRunId(cwd, { withJournal: true });
|
|
656
|
+
if (!runId)
|
|
657
|
+
return [];
|
|
658
|
+
if (decisionRunId !== runId) {
|
|
659
|
+
decisionRunId = runId;
|
|
660
|
+
journalCursor = 0;
|
|
661
|
+
}
|
|
662
|
+
const journalEvents = Journal.open(cwd, runId).read();
|
|
663
|
+
if (journalEvents.length < journalCursor)
|
|
664
|
+
journalCursor = 0;
|
|
665
|
+
const fresh = decisionEventsFromJournal(journalEvents, runId, stateDirName(cwd))
|
|
666
|
+
.filter((event) => event.sequence > journalCursor);
|
|
667
|
+
journalCursor = journalEvents.length;
|
|
668
|
+
if (webhookUrl) {
|
|
669
|
+
for (const event of fresh) {
|
|
670
|
+
if (event.tier === "decision")
|
|
671
|
+
postWebhookFireAndForget(postWebhook, webhookUrl, event);
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
return fresh;
|
|
675
|
+
};
|
|
470
676
|
let titleSaved = false;
|
|
471
677
|
const restoreTitle = () => {
|
|
472
678
|
if (!titleSaved)
|
|
@@ -492,18 +698,31 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
492
698
|
try {
|
|
493
699
|
for (let i = 0; i < iterations; i++) {
|
|
494
700
|
const nowMs = now();
|
|
495
|
-
const
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
701
|
+
const decisionEvents = eventStream || webhookUrl ? consumeDecisionEvents() : [];
|
|
702
|
+
let frame;
|
|
703
|
+
if (eventStream) {
|
|
704
|
+
for (const event of decisionEvents) {
|
|
705
|
+
const line = JSON.stringify(event);
|
|
706
|
+
process.stdout.write(line + "\n");
|
|
707
|
+
if (bounded)
|
|
708
|
+
eventLines.push(line);
|
|
709
|
+
}
|
|
499
710
|
}
|
|
500
711
|
else {
|
|
501
|
-
|
|
712
|
+
frame = renderFrame(cwd, nowMs, i, workerLiveness);
|
|
713
|
+
if (tty) {
|
|
714
|
+
updateTitle(frame.hotPhase, nowMs);
|
|
715
|
+
process.stdout.write(`\x1b[2J\x1b[H${BANNER}${frame.content}\n${legend(` watching · refresh ${REFRESH_MS / 1000}s · ^C to quit`)}`);
|
|
716
|
+
}
|
|
717
|
+
else {
|
|
718
|
+
process.stdout.write(frame.content + sep);
|
|
719
|
+
}
|
|
720
|
+
if (bounded)
|
|
721
|
+
frames.push(frame.content);
|
|
502
722
|
}
|
|
503
|
-
if (bounded)
|
|
504
|
-
frames.push(frame.content);
|
|
505
723
|
if (i + 1 < iterations) {
|
|
506
|
-
|
|
724
|
+
if (frame)
|
|
725
|
+
await observeWorkerOutput(frame.workerPhases, frame.runId, nowMs);
|
|
507
726
|
await sleep(REFRESH_MS);
|
|
508
727
|
}
|
|
509
728
|
}
|
|
@@ -514,5 +733,5 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
514
733
|
restoreTitle();
|
|
515
734
|
}
|
|
516
735
|
}
|
|
517
|
-
return bounded ? frames.join(sep) : "";
|
|
736
|
+
return bounded ? eventStream ? eventLines.join("\n") : frames.join(sep) : "";
|
|
518
737
|
}
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -2,6 +2,10 @@ import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "
|
|
|
2
2
|
export declare const TRAILER_SAFE_FLOOR_COLS = 108;
|
|
3
3
|
export declare const TRAILER_WIDTH_MARGIN = 2;
|
|
4
4
|
export declare const DELIVERY_ATTEMPTS = 3;
|
|
5
|
+
export interface HerdrTimeSource {
|
|
6
|
+
now: () => number;
|
|
7
|
+
sleep: (ms: number) => Promise<void>;
|
|
8
|
+
}
|
|
5
9
|
export declare class DeliveryReadinessError extends Error {
|
|
6
10
|
readonly waitedMs: number;
|
|
7
11
|
readonly transcript: string;
|
|
@@ -13,6 +17,7 @@ export declare function workerSplitDirection(paneCols: number | null, safeFloor?
|
|
|
13
17
|
export declare class HerdrDriver implements ExecutorDriver {
|
|
14
18
|
private bin;
|
|
15
19
|
private workersPerTab;
|
|
20
|
+
private time;
|
|
16
21
|
id: string;
|
|
17
22
|
interactive: boolean;
|
|
18
23
|
private groups;
|
|
@@ -24,12 +29,13 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
24
29
|
private ws;
|
|
25
30
|
private callerPane;
|
|
26
31
|
private watches;
|
|
27
|
-
constructor(bin?: string, workersPerTab?: number);
|
|
32
|
+
constructor(bin?: string, workersPerTab?: number, time?: HerdrTimeSource);
|
|
28
33
|
private serial;
|
|
29
34
|
private deliveryQueue;
|
|
30
35
|
private reserveDispatch;
|
|
31
36
|
private verifyPaneIdentityBinding;
|
|
32
37
|
private deliveryMatches;
|
|
38
|
+
private shellExecutionEchoed;
|
|
33
39
|
private submissionRegistered;
|
|
34
40
|
static available(): boolean;
|
|
35
41
|
private herdr;
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -14,6 +14,10 @@ const DELIVERY_READ_LINES = 80;
|
|
|
14
14
|
const DELIVERY_SETTLE_READ_ATTEMPTS = 6;
|
|
15
15
|
const DELIVERY_SETTLE_POLL_MS = 100;
|
|
16
16
|
const DELIVERY_READINESS_TIMEOUT_MS = 1_000;
|
|
17
|
+
const SYSTEM_TIME = {
|
|
18
|
+
now: () => Date.now(),
|
|
19
|
+
sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
|
|
20
|
+
};
|
|
17
21
|
// OBS-142: a typed identity lets the daemon's next task distinguish cold-start variance from
|
|
18
22
|
// structural driver faults without parsing prose. The message also carries the bounded wait and
|
|
19
23
|
// final pane evidence so a terminal failure is diagnosable on its own.
|
|
@@ -37,6 +41,7 @@ export function workerSplitDirection(paneCols, safeFloor = TRAILER_SAFE_FLOOR_CO
|
|
|
37
41
|
export class HerdrDriver {
|
|
38
42
|
bin;
|
|
39
43
|
workersPerTab;
|
|
44
|
+
time;
|
|
40
45
|
id = "herdr";
|
|
41
46
|
interactive = true;
|
|
42
47
|
groups = new Map();
|
|
@@ -55,9 +60,10 @@ export class HerdrDriver {
|
|
|
55
60
|
ws = process.env.HERDR_WORKSPACE_ID;
|
|
56
61
|
callerPane = process.env.HERDR_PANE_ID;
|
|
57
62
|
watches = new Map();
|
|
58
|
-
constructor(bin = "herdr", workersPerTab = 3) {
|
|
63
|
+
constructor(bin = "herdr", workersPerTab = 3, time = SYSTEM_TIME) {
|
|
59
64
|
this.bin = bin;
|
|
60
65
|
this.workersPerTab = workersPerTab;
|
|
66
|
+
this.time = time;
|
|
61
67
|
}
|
|
62
68
|
serial(fn) {
|
|
63
69
|
const p = this.groupSerial.then(fn, fn);
|
|
@@ -106,14 +112,39 @@ export class HerdrDriver {
|
|
|
106
112
|
}
|
|
107
113
|
// ponytail: narrow panes hard-wrap the input line — collapse whitespace before comparing.
|
|
108
114
|
deliveryMatches(transcript, cmd) {
|
|
109
|
-
|
|
115
|
+
// OBS-154: a TUI editor re-wraps a long delivery across its own bordered rows, so the pane text
|
|
116
|
+
// carries `│` between fragments we typed as ONE line. Stripping whitespace alone left the needle
|
|
117
|
+
// uncontainable, so the read-back could not recognize its own SUCCESSFUL delivery and the OBS-85
|
|
118
|
+
// guard then refused to retype onto what it had been told was corruption (probe 6b).
|
|
119
|
+
// This is the opposite question to shellExecutionEchoed, which deliberately REJECTS box rows: a
|
|
120
|
+
// shell echo must come from a shell prompt, whereas here we only ask whether our text landed —
|
|
121
|
+
// and inside the editor box is exactly where it is supposed to land.
|
|
122
|
+
const norm = (s) => s.replace(/[│┃|]/g, "").replace(/\s+/g, "");
|
|
110
123
|
const hay = norm(transcript);
|
|
111
124
|
const needle = norm(cmd);
|
|
112
125
|
return needle.length > 0 && hay.includes(needle);
|
|
113
126
|
}
|
|
127
|
+
shellExecutionEchoed(transcript, cmd) {
|
|
128
|
+
const norm = (s) => s.replace(/\u001B\[[0-9;?]*[ -/]*[@-~]/g, "").replace(/\s+/g, " ").trim();
|
|
129
|
+
const needle = norm(cmd);
|
|
130
|
+
if (!needle)
|
|
131
|
+
return false;
|
|
132
|
+
return transcript.split("\n").some((rawLine) => {
|
|
133
|
+
const line = norm(rawLine);
|
|
134
|
+
if (line === needle || !line.endsWith(needle))
|
|
135
|
+
return false;
|
|
136
|
+
const prefix = line.slice(0, -needle.length).trimEnd();
|
|
137
|
+
if (/[│┃╭╰┌└]/.test(prefix))
|
|
138
|
+
return false;
|
|
139
|
+
return /^(?:[$%#>]|[➜❯❱›»λ])(?:\s|$)/.test(prefix)
|
|
140
|
+
|| /[$%#>✗❯❱›»λ]$/.test(prefix);
|
|
141
|
+
});
|
|
142
|
+
}
|
|
114
143
|
// Submission requires positive post-Enter evidence. A prompt echoed above a fresh input target
|
|
115
144
|
// proves registration; a declared box that is visibly empty after the verified paste also proves
|
|
116
|
-
// it consumed the prompt.
|
|
145
|
+
// it consumed the prompt. At the launch stage, a structurally prompt-prefixed shell execution echo
|
|
146
|
+
// is stronger evidence than waiting for first paint from the launched interface (OBS-144).
|
|
147
|
+
// Prompt absence alone is never success (OBS-142).
|
|
117
148
|
submissionRegistered(transcript, cmd, inputBox) {
|
|
118
149
|
const norm = (s) => s.replace(/\s+/g, "");
|
|
119
150
|
const hay = norm(transcript);
|
|
@@ -121,6 +152,8 @@ export class HerdrDriver {
|
|
|
121
152
|
const promptAt = hay.lastIndexOf(needle);
|
|
122
153
|
if (needle.length === 0)
|
|
123
154
|
return false;
|
|
155
|
+
if (inputBox?.launchCommand?.(cmd) === true && this.shellExecutionEchoed(transcript, cmd))
|
|
156
|
+
return true;
|
|
124
157
|
if (promptAt < 0)
|
|
125
158
|
return inputBox !== undefined && matchesEmptyInputBox(transcript, inputBox);
|
|
126
159
|
if (inputBox && matchesInputBox(transcript, inputBox)) {
|
|
@@ -527,7 +560,7 @@ export class HerdrDriver {
|
|
|
527
560
|
}
|
|
528
561
|
transcript = read.stdout;
|
|
529
562
|
if (readAttempt < readAttempts - 1) {
|
|
530
|
-
await
|
|
563
|
+
await this.time.sleep(DELIVERY_SETTLE_POLL_MS);
|
|
531
564
|
}
|
|
532
565
|
}
|
|
533
566
|
return { ok: false, transcript, recognizedInputBox: false };
|
|
@@ -536,19 +569,19 @@ export class HerdrDriver {
|
|
|
536
569
|
const inputBox = this.inputBoxes.get(slot);
|
|
537
570
|
const requireInputBox = inputBox !== undefined && inputBox.launchCommand?.(cmd) !== true;
|
|
538
571
|
const timeoutMs = inputBox?.readinessTimeoutMs ?? DELIVERY_READINESS_TIMEOUT_MS;
|
|
539
|
-
const started =
|
|
572
|
+
const started = this.time.now();
|
|
540
573
|
let previous;
|
|
541
574
|
let transcript = "";
|
|
542
575
|
let reads = 0;
|
|
543
576
|
while (true) {
|
|
544
|
-
const elapsedBeforeRead =
|
|
577
|
+
const elapsedBeforeRead = this.time.now() - started;
|
|
545
578
|
const remainingBeforeRead = timeoutMs - elapsedBeforeRead;
|
|
546
579
|
if (remainingBeforeRead <= 0)
|
|
547
580
|
throw new DeliveryReadinessError(elapsedBeforeRead, transcript);
|
|
548
581
|
const read = await this.herdr(`pane read ${shq(pane)} --source recent-unwrapped --lines ${DELIVERY_READ_LINES}`, slot.cwd, Math.max(1, Math.ceil(remainingBeforeRead)));
|
|
549
582
|
reads++;
|
|
550
583
|
transcript = read.stdout || transcript;
|
|
551
|
-
const waitedMs =
|
|
584
|
+
const waitedMs = this.time.now() - started;
|
|
552
585
|
if (read.code !== 0) {
|
|
553
586
|
// Once at least one valid frame has been observed, a read that consumes the remainder of
|
|
554
587
|
// the readiness budget is the bounded window expiring, not a new submission/protocol
|
|
@@ -574,7 +607,7 @@ export class HerdrDriver {
|
|
|
574
607
|
// The second read is immediate: an already-painted stable target proves readiness with no
|
|
575
608
|
// added delay. Only observed change spends a poll interval, and every path remains bounded.
|
|
576
609
|
if (reads > 1) {
|
|
577
|
-
await
|
|
610
|
+
await this.time.sleep(Math.min(DELIVERY_SETTLE_POLL_MS, remaining));
|
|
578
611
|
}
|
|
579
612
|
}
|
|
580
613
|
}
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -5,11 +5,32 @@ export interface Slot {
|
|
|
5
5
|
tabId?: string;
|
|
6
6
|
group?: string;
|
|
7
7
|
}
|
|
8
|
-
export type NotifyTier = "routine" | "attention";
|
|
8
|
+
export type NotifyTier = "routine" | "attention" | "decision";
|
|
9
9
|
export interface NotifyOpts {
|
|
10
10
|
tier?: NotifyTier;
|
|
11
11
|
sound?: "none" | "done" | "request";
|
|
12
12
|
}
|
|
13
|
+
export type DecisionEventType = "phase-change" | "gate-verdict" | "escalation" | "human-decision-required" | "run-end";
|
|
14
|
+
export interface DecisionEvent {
|
|
15
|
+
version: 1;
|
|
16
|
+
sequence: number;
|
|
17
|
+
type: DecisionEventType;
|
|
18
|
+
tier: "routine" | "decision";
|
|
19
|
+
ts: string;
|
|
20
|
+
runId: string;
|
|
21
|
+
taskId?: string;
|
|
22
|
+
evidence: string;
|
|
23
|
+
phase?: string;
|
|
24
|
+
gate?: string;
|
|
25
|
+
verdict?: "passed" | "failed" | "skipped" | "unknown";
|
|
26
|
+
step?: string;
|
|
27
|
+
attempt?: number;
|
|
28
|
+
kind?: string;
|
|
29
|
+
reason?: string;
|
|
30
|
+
approvalCommand?: string;
|
|
31
|
+
summary?: Record<string, unknown>;
|
|
32
|
+
}
|
|
33
|
+
export type DecisionWebhookPost = (url: string, event: DecisionEvent) => void | Promise<unknown>;
|
|
13
34
|
export interface SlotOpts {
|
|
14
35
|
group?: string;
|
|
15
36
|
label?: string;
|
|
@@ -14,6 +14,11 @@ export interface JudgeVerdict {
|
|
|
14
14
|
reason: string;
|
|
15
15
|
evidence: string | EvidenceCitation;
|
|
16
16
|
}>;
|
|
17
|
+
comments?: Array<{
|
|
18
|
+
path: string;
|
|
19
|
+
line: number;
|
|
20
|
+
body: string;
|
|
21
|
+
}>;
|
|
17
22
|
}
|
|
18
23
|
export declare function judgeCriterionId(index: number): string;
|
|
19
24
|
export declare function testFiltered(testCmd: string, name: string): string;
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -4,7 +4,7 @@ import { DEFAULT_DIFF_CAP } from "../config/config.js";
|
|
|
4
4
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
5
5
|
import { sh } from "../run/git.js";
|
|
6
6
|
import { checkDiffCap, fetchTaskDiff } from "./review.js";
|
|
7
|
-
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
7
|
+
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
8
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
9
9
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
10
10
|
const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
|
|
@@ -278,9 +278,10 @@ ${citable || "(the diff changes no lines)"}
|
|
|
278
278
|
${verdictNonceLine(nonce)}
|
|
279
279
|
|
|
280
280
|
Respond with ONLY this JSON (no prose before or after):
|
|
281
|
-
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": {"path": "path/to/file", "line": 42}}]}
|
|
281
|
+
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": {"path": "path/to/file", "line": 42}}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
|
|
282
282
|
Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) exactly once.
|
|
283
283
|
Each criteria[].evidence MUST be a structured citation {"path", "line"} whose "path" and "line" appear in the "Citable evidence lines" list above (a new-file line number inside a changed hunk); a citation outside every changed hunk or to an untouched file voids the whole verdict.
|
|
284
|
+
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
284
285
|
`;
|
|
285
286
|
const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
|
|
286
287
|
const extracted = extractVerdictJson(raw, nonce);
|
|
@@ -309,5 +310,6 @@ Each criteria[].evidence MUST be a structured citation {"path", "line"} whose "p
|
|
|
309
310
|
if (!v.pass)
|
|
310
311
|
lines.push("judge verdict pass=false");
|
|
311
312
|
lines.push(...inconsistencies);
|
|
312
|
-
|
|
313
|
+
const prose = warn + detBlock + (lines.join("\n") || "judge passed");
|
|
314
|
+
return { gate: "acceptance", pass, details: appendAnchoredReview(prose, extracted) };
|
|
313
315
|
}
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -4,6 +4,14 @@ export declare const GATE_PANE_SEP = " \u00B7 ";
|
|
|
4
4
|
export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
|
|
5
5
|
/** Fable F3: per-call nonce echoed in verdict JSON and gate exit markers. */
|
|
6
6
|
export declare function generateVerdictNonce(): string;
|
|
7
|
+
export interface AnchoredComment {
|
|
8
|
+
path: string;
|
|
9
|
+
line: number;
|
|
10
|
+
body: string;
|
|
11
|
+
}
|
|
12
|
+
export declare function parseAnchoredComments(verdict: unknown): AnchoredComment[];
|
|
13
|
+
export declare function renderAnchoredReview(verdict: unknown): string;
|
|
14
|
+
export declare function appendAnchoredReview(prose: string, verdict: unknown): string;
|
|
7
15
|
export declare function verdictNonceLine(nonce: string): string;
|
|
8
16
|
export declare function extractPromptNonce(prompt: string): string | null;
|
|
9
17
|
export declare function gateExitTrailer(nonce: string): string;
|
|
@@ -32,6 +40,7 @@ export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
|
|
|
32
40
|
}>;
|
|
33
41
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
34
42
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
|
43
|
+
export declare function dewrapPaneVerdict(out: string, nonce: string): string;
|
|
35
44
|
export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
|
|
36
45
|
export declare function extractJson<T>(raw: string): T | null;
|
|
37
46
|
/** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
|
package/dist/gates/llm.js
CHANGED
|
@@ -27,6 +27,39 @@ When a criterion fails, the verdict MUST name which shortcut above it matches, o
|
|
|
27
27
|
export function generateVerdictNonce() {
|
|
28
28
|
return randomBytes(4).toString("hex");
|
|
29
29
|
}
|
|
30
|
+
// v1.79 T5: comments are an optional side channel on an otherwise authoritative verdict. The whole
|
|
31
|
+
// block is accepted only when every row names one actionable path:line; absent or malformed input
|
|
32
|
+
// becomes no comments and therefore preserves the pre-comments prose bytes and verdict semantics.
|
|
33
|
+
export function parseAnchoredComments(verdict) {
|
|
34
|
+
if (!verdict || typeof verdict !== "object")
|
|
35
|
+
return [];
|
|
36
|
+
const raw = verdict.comments;
|
|
37
|
+
if (!Array.isArray(raw) || raw.length === 0)
|
|
38
|
+
return [];
|
|
39
|
+
const comments = [];
|
|
40
|
+
for (const row of raw) {
|
|
41
|
+
if (!row || typeof row !== "object")
|
|
42
|
+
return [];
|
|
43
|
+
const { path, line, body } = row;
|
|
44
|
+
const cleanPath = typeof path === "string" ? path.trim() : "";
|
|
45
|
+
const cleanBody = typeof body === "string" ? body.trim() : "";
|
|
46
|
+
if (!cleanPath || /[\r\n]/.test(cleanPath) || !Number.isInteger(line) || line < 1 || !cleanBody) {
|
|
47
|
+
return [];
|
|
48
|
+
}
|
|
49
|
+
comments.push({ path: cleanPath, line: line, body: cleanBody });
|
|
50
|
+
}
|
|
51
|
+
return comments;
|
|
52
|
+
}
|
|
53
|
+
export function renderAnchoredReview(verdict) {
|
|
54
|
+
const comments = parseAnchoredComments(verdict);
|
|
55
|
+
if (comments.length === 0)
|
|
56
|
+
return "";
|
|
57
|
+
return `## Anchored review\n${comments.map((c) => `- ${c.path}:${c.line} — ${c.body}`).join("\n")}`;
|
|
58
|
+
}
|
|
59
|
+
export function appendAnchoredReview(prose, verdict) {
|
|
60
|
+
const block = renderAnchoredReview(verdict);
|
|
61
|
+
return block ? `${prose}\n\n${block}` : prose;
|
|
62
|
+
}
|
|
30
63
|
export function verdictNonceLine(nonce) {
|
|
31
64
|
return `VERDICT_NONCE: ${nonce}`;
|
|
32
65
|
}
|
|
@@ -128,6 +161,47 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
128
161
|
if (!via.keep)
|
|
129
162
|
await via.driver.close(slot);
|
|
130
163
|
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
164
|
+
return dewrapPaneVerdict(out, nonce);
|
|
165
|
+
}
|
|
166
|
+
// OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
|
|
167
|
+
// continuation indent, splitting words mid-token — so literal newlines land inside JSON string
|
|
168
|
+
// literals and ZERO lines begin with `{`. `--source recent-unwrapped` cannot undo it: the wrap is
|
|
169
|
+
// the TUI's own rendering, not a terminal soft wrap (driver.read already requests that source, and
|
|
170
|
+
// the captured k3 transcript arrived wrapped anyway). A perfect verdict was therefore scored a
|
|
171
|
+
// flake and rescued by a retry on another channel — 70 such judge-retries are on record across
|
|
172
|
+
// k3, fable AND sol, so this is a width-dependent flake generator under every pane-mode gate.
|
|
173
|
+
//
|
|
174
|
+
// Fail-closed by construction, per the overseer's constraints: reconstruction is bounded to the
|
|
175
|
+
// brace-delimited region, the rejoined text must PARSE, and it must carry THIS call's nonce.
|
|
176
|
+
// Anything else returns the bytes untouched, so an unparseable verdict stays a failure. The
|
|
177
|
+
// reconstruction is APPENDED, never substituted: extractJson takes the last balanced object, so
|
|
178
|
+
// the good copy wins while the original rendering survives verbatim for the evidence capture.
|
|
179
|
+
export function dewrapPaneVerdict(out, nonce) {
|
|
180
|
+
if (!out.includes(nonce))
|
|
181
|
+
return out;
|
|
182
|
+
const lines = out.split("\n");
|
|
183
|
+
const start = lines.findIndex((line) => /^\s*(?:[•*-]\s+)?\{/.test(line));
|
|
184
|
+
if (start < 0)
|
|
185
|
+
return out;
|
|
186
|
+
for (let end = start; end < lines.length; end++) {
|
|
187
|
+
const joined = lines
|
|
188
|
+
.slice(start, end + 1)
|
|
189
|
+
.map((line, i) => (i === 0 ? line.replace(/^\s*(?:[•*-]\s+)?/, "") : line.replace(/^\s+/, "")))
|
|
190
|
+
.join("")
|
|
191
|
+
.trimEnd();
|
|
192
|
+
if (!joined.endsWith("}"))
|
|
193
|
+
continue;
|
|
194
|
+
try {
|
|
195
|
+
const parsed = JSON.parse(joined);
|
|
196
|
+
// the nonce IS the acceptance test — never reconstruct a verdict this call did not ask for
|
|
197
|
+
if (parsed && typeof parsed === "object" && parsed.nonce === nonce) {
|
|
198
|
+
return `${out}\n${joined}`;
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
catch {
|
|
202
|
+
/* not yet a complete object — keep extending within the bounded region */
|
|
203
|
+
}
|
|
204
|
+
}
|
|
131
205
|
return out;
|
|
132
206
|
}
|
|
133
207
|
export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -14,6 +14,11 @@ export interface ReviewVerdict {
|
|
|
14
14
|
approve?: boolean;
|
|
15
15
|
issues?: string[];
|
|
16
16
|
findings?: ReviewFinding[];
|
|
17
|
+
comments?: Array<{
|
|
18
|
+
path: string;
|
|
19
|
+
line: number;
|
|
20
|
+
body: string;
|
|
21
|
+
}>;
|
|
17
22
|
}
|
|
18
23
|
export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
|
|
19
24
|
full: string;
|
package/dist/gates/review.js
CHANGED
|
@@ -4,7 +4,7 @@ import { renderAcceptanceItem } from "../graph/schema.js";
|
|
|
4
4
|
import { getAdapter } from "../adapters/registry.js";
|
|
5
5
|
import { shOk } from "../run/git.js";
|
|
6
6
|
import { marginalCostRank } from "../route/router.js";
|
|
7
|
-
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
7
|
+
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
8
|
// legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
|
|
9
9
|
function classifyReviewIssues(approve, issues) {
|
|
10
10
|
const inconsistencies = [];
|
|
@@ -158,8 +158,9 @@ block approval. For a minor concern you have decided not to block on, set "defer
|
|
|
158
158
|
one-line "rationale" — it is recorded in the review, never dropped.
|
|
159
159
|
|
|
160
160
|
Respond with ONLY this JSON:
|
|
161
|
-
{"nonce": "${nonce}", "approve": true|false, "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}]}
|
|
161
|
+
{"nonce": "${nonce}", "approve": true|false, "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
|
|
162
162
|
Approve iff no material finding remains; an empty findings list is a clean approval.
|
|
163
|
+
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
163
164
|
`;
|
|
164
165
|
const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("review", reviewer.adapter), label: via.labelFor("review") } : undefined,
|
|
165
166
|
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
@@ -182,10 +183,11 @@ Approve iff no material finding remains; an empty findings list is a clean appro
|
|
|
182
183
|
const decided = findings !== null
|
|
183
184
|
? classifyReviewFindings(findings)
|
|
184
185
|
: classifyReviewIssues(v.approve, v.issues);
|
|
186
|
+
const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (${reviewer.vendor}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
|
|
185
187
|
return {
|
|
186
188
|
gate: "review",
|
|
187
189
|
pass: decided.pass,
|
|
188
|
-
details:
|
|
190
|
+
details: appendAnchoredReview(prose, v),
|
|
189
191
|
meta: { reviewer: channelKey(reviewer) },
|
|
190
192
|
};
|
|
191
193
|
}
|
package/dist/run/journal.d.ts
CHANGED
|
@@ -85,12 +85,12 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
85
85
|
}>>;
|
|
86
86
|
signalQuality: z.ZodOptional<z.ZodUnion<readonly [z.ZodLiteral<0>, z.ZodLiteral<0.25>, z.ZodLiteral<0.5>, z.ZodLiteral<0.75>, z.ZodLiteral<1>]>>;
|
|
87
87
|
signalBasis: z.ZodOptional<z.ZodEnum<{
|
|
88
|
+
skipped: "skipped";
|
|
88
89
|
proved: "proved";
|
|
89
90
|
"review-agree": "review-agree";
|
|
90
91
|
"judge-only": "judge-only";
|
|
91
92
|
legacy: "legacy";
|
|
92
93
|
vacuous: "vacuous";
|
|
93
|
-
skipped: "skipped";
|
|
94
94
|
}>>;
|
|
95
95
|
kind: z.ZodOptional<z.ZodLiteral<"judge">>;
|
|
96
96
|
judgeOutcome: z.ZodOptional<z.ZodEnum<{
|