tickmarkr 1.63.0 → 1.65.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code.js +3 -0
- package/dist/adapters/codex.js +3 -0
- package/dist/adapters/cursor-agent.js +2 -0
- package/dist/adapters/grok.js +2 -0
- package/dist/adapters/kimi.js +3 -0
- package/dist/adapters/opencode.js +3 -0
- package/dist/adapters/pi.js +2 -0
- package/dist/adapters/prompt.d.ts +4 -0
- package/dist/adapters/prompt.js +28 -2
- package/dist/adapters/registry.d.ts +3 -0
- package/dist/adapters/registry.js +38 -0
- package/dist/adapters/types.d.ts +4 -0
- package/dist/cli/commands/doctor.js +4 -1
- package/dist/cli/commands/status.js +12 -4
- package/dist/config/config.js +4 -1
- package/dist/gates/acceptance.d.ts +1 -0
- package/dist/gates/acceptance.js +17 -2
- package/dist/gates/baseline.d.ts +11 -0
- package/dist/gates/baseline.js +26 -0
- package/dist/gates/llm.d.ts +2 -1
- package/dist/gates/llm.js +32 -4
- package/dist/gates/review.js +3 -1
- package/dist/run/activity.d.ts +15 -0
- package/dist/run/activity.js +96 -0
- package/dist/run/consult.js +8 -2
- package/dist/run/daemon.js +62 -4
- package/dist/run/journal.js +8 -3
- package/dist/run/reconcile.js +3 -1
- package/dist/run/redact.d.ts +2 -0
- package/dist/run/redact.js +53 -0
- package/dist/run/stall.d.ts +7 -3
- package/dist/run/stall.js +71 -3
- package/package.json +1 -1
|
@@ -41,6 +41,9 @@ export const claudeCode = {
|
|
|
41
41
|
probeCwd: "neutral",
|
|
42
42
|
probe: async () => probeVersion("claude"),
|
|
43
43
|
channels: (cfg) => channelsFromConfig("claude-code", cfg),
|
|
44
|
+
// v1.65 T3: every flag the command builders below hardcode — doctor checks `claude --help` still
|
|
45
|
+
// lists each (all present on claude 2.x, verified 2026-07-22). Advisory only, never routing.
|
|
46
|
+
hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r"] },
|
|
44
47
|
// --strict-mcp-config --mcp-config '{"mcpServers":{}}': pin the MCP surface to empty so fresh-worktree
|
|
45
48
|
// workers/gates don't load project .mcp.json servers (herdr scrapes dialogs as idle — v1.4 incident,
|
|
46
49
|
// memory tickmarkr-worker-mcp-dialog-stall). Live-verified 2026-07-10 on claude 2.1.205 (operator check):
|
package/dist/adapters/codex.js
CHANGED
|
@@ -169,6 +169,9 @@ export const codex = {
|
|
|
169
169
|
probeConcurrency: 1,
|
|
170
170
|
probe: async () => probeVersion("codex"),
|
|
171
171
|
channels: (cfg) => channelsFromConfig("codex", cfg),
|
|
172
|
+
// v1.65 T3: every flag the command builders below hardcode (incl. codexMcpSuppressionFlags' -c/
|
|
173
|
+
// --disable and GITDIR_WRITABLE's -c) — all listed by top-level `codex --help`, verified 2026-07-22.
|
|
174
|
+
hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-a", "-s", "-c", "--disable"] },
|
|
172
175
|
// --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
|
|
173
176
|
// MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
|
|
174
177
|
headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
|
|
@@ -36,6 +36,8 @@ export const cursorAgent = {
|
|
|
36
36
|
vendor: "cursor",
|
|
37
37
|
probe: async () => probeVersion("cursor-agent"),
|
|
38
38
|
channels: (cfg) => channelsFromConfig("cursor-agent", cfg),
|
|
39
|
+
// v1.65 T3: every flag the command builders below hardcode — verified in `cursor-agent --help` 2026-07-22.
|
|
40
|
+
hardcodedFlags: { binary: "cursor-agent", flags: ["-p", "--model", "--force", "--output-format"] },
|
|
39
41
|
headlessCommand: (promptFile, model) => `cursor-agent -p "$(cat ${shq(promptFile)})" --model ${shq(model)} --force --output-format text`,
|
|
40
42
|
// NO --trust here: cursor rejects it outside --print ("--trust can only be used with --print/headless
|
|
41
43
|
// mode", exit 1 — v1.4 phase-1 incident). Fresh worktrees therefore show the trust dialog; v1.22 T5
|
package/dist/adapters/grok.js
CHANGED
|
@@ -98,6 +98,8 @@ export const grok = {
|
|
|
98
98
|
: { ...h, authed: false, note: "no valid ~/.grok/auth.json entry (grok login to fix)" };
|
|
99
99
|
},
|
|
100
100
|
channels: (cfg) => channelsFromConfig("grok", cfg),
|
|
101
|
+
// v1.65 T3: every flag the command builders below hardcode — verified in `grok --help` 2026-07-22.
|
|
102
|
+
hardcodedFlags: { binary: "grok", flags: ["-p", "--model", "--permission-mode", "--output-format"] },
|
|
101
103
|
// GROK-02 headless. --output-format plain is the default but pin it explicitly; live-verified
|
|
102
104
|
// 2026-07-11 (40-RESEARCH F-3): trailer intact + unwrapped, exit 0. "$(cat file)" matches every
|
|
103
105
|
// other adapter and is already quoting-proven. --permission-mode bypassPermissions is the
|
package/dist/adapters/kimi.js
CHANGED
|
@@ -96,6 +96,9 @@ export const kimi = {
|
|
|
96
96
|
: { ...h, authed: false, note: "no valid ~/.kimi-code/credentials/kimi-code.json (kimi login to fix)" };
|
|
97
97
|
},
|
|
98
98
|
channels: (cfg) => channelsFromConfig("kimi", cfg),
|
|
99
|
+
// v1.65 T3: every flag the command builders below hardcode (-S is resumeCommand's) — verified in
|
|
100
|
+
// `kimi --help` 2026-07-22.
|
|
101
|
+
hardcodedFlags: { binary: "kimi", flags: ["-p", "--model", "--output-format", "-S"] },
|
|
99
102
|
// KIMI-02 headless. -p prompt mode + explicit model, NO permission flag: kimi 0.26.0 rejects
|
|
100
103
|
// -p combined with -y/--auto at argument parse time ("Cannot combine --prompt with --yolo",
|
|
101
104
|
// OBS-67 — doctor probes all failed on it), and prompt mode is already non-interactive with
|
|
@@ -19,6 +19,9 @@ export const opencode = {
|
|
|
19
19
|
probeCwd: "neutral",
|
|
20
20
|
probe: async () => probeVersion("opencode"),
|
|
21
21
|
channels: (cfg) => channelsFromConfig("opencode", cfg),
|
|
22
|
+
// v1.65 T3: every flag the command builders below hardcode (-m on run, --prompt on the TUI) —
|
|
23
|
+
// per the 2026-07-10 live verification, opencode 1.17.15.
|
|
24
|
+
hardcodedFlags: { binary: "opencode", flags: ["-m", "--prompt"] },
|
|
22
25
|
headlessCommand: (promptFile, model) => `opencode run -m ${shq(model)} "$(cat ${shq(promptFile)})"`,
|
|
23
26
|
interactiveCommand: (promptFile, model) => `opencode -m ${shq(model)} --prompt "$(cat ${shq(promptFile)})"`,
|
|
24
27
|
invoke(task, _cwd, a, ctx) {
|
package/dist/adapters/pi.js
CHANGED
|
@@ -48,6 +48,8 @@ export const pi = {
|
|
|
48
48
|
return { ...h, servable: parsePiModels(r.stdout || ""), note: "auth verified via pi --list-models (free; auth-filtered by pi)" };
|
|
49
49
|
},
|
|
50
50
|
channels: (cfg) => channelsFromConfig("pi", cfg),
|
|
51
|
+
// v1.65 T3: every flag the command builders below hardcode — verified in `pi --help` 2026-07-22.
|
|
52
|
+
hardcodedFlags: { binary: "pi", flags: ["-p", "--approve", "--model"] },
|
|
51
53
|
// --approve: pi's per-directory trust prompt would stall fresh worktrees (herdr scrapes the dialog
|
|
52
54
|
// as idle — cursor/claude incident class, milestone PITFALLS #2). Global option, legal in BOTH modes
|
|
53
55
|
// per pi --help v0.80.3 (2026-07-10) — NOT print-only like cursor's --trust. Chosen over the more
|
|
@@ -3,4 +3,8 @@ import type { WorkerResult } from "./types.js";
|
|
|
3
3
|
export declare function buildTaskPrompt(task: Task, feedback?: string, nonce?: string): string;
|
|
4
4
|
export declare function writePrompt(dir: string, task: Task, attempt: number, feedback?: string, nonce?: string): string;
|
|
5
5
|
export declare function trailerPattern(nonce: string): string;
|
|
6
|
+
export declare const NO_TRAILER_SUMMARY = "worker produced no TICKMARKR_RESULT trailer";
|
|
7
|
+
export declare const UNPARSEABLE_TRAILER_SUMMARY = "unparseable TICKMARKR_RESULT trailer";
|
|
6
8
|
export declare function parseWorkerResult(raw: string, nonce: string): WorkerResult;
|
|
9
|
+
export type DeadChannelReason = "auth-required" | "setup-required" | "provider-outage" | "timeout";
|
|
10
|
+
export declare function classifyDeadChannel(result: WorkerResult): DeadChannelReason | undefined;
|
package/dist/adapters/prompt.js
CHANGED
|
@@ -41,11 +41,15 @@ function trailerTokenPositions(raw, nonce) {
|
|
|
41
41
|
positions.push(i);
|
|
42
42
|
return positions;
|
|
43
43
|
}
|
|
44
|
+
// v1.65 T1: the parse boundary's own no-trailer sentinel summaries. classifyDeadChannel keys on
|
|
45
|
+
// these — a result carrying any other summary is a PARSED trailer, i.e. the worker speaking.
|
|
46
|
+
export const NO_TRAILER_SUMMARY = "worker produced no TICKMARKR_RESULT trailer";
|
|
47
|
+
export const UNPARSEABLE_TRAILER_SUMMARY = "unparseable TICKMARKR_RESULT trailer";
|
|
44
48
|
export function parseWorkerResult(raw, nonce) {
|
|
45
49
|
const fail = (summary) => ({ ok: false, summary, deviations: [], raw });
|
|
46
50
|
const positions = trailerTokenPositions(raw, nonce);
|
|
47
51
|
if (positions.length === 0)
|
|
48
|
-
return fail(
|
|
52
|
+
return fail(NO_TRAILER_SUMMARY);
|
|
49
53
|
// TUIs echo the prompt template, redraw lines, and HARD-wrap the JSON with per-line margins
|
|
50
54
|
// (cursor does; recent-unwrapped can't rejoin hard newlines). Scan occurrences backward — last
|
|
51
55
|
// parseable wins — joining wrapped lines, stripping margin/box chrome, and growing the candidate
|
|
@@ -75,5 +79,27 @@ export function parseWorkerResult(raw, nonce) {
|
|
|
75
79
|
}
|
|
76
80
|
}
|
|
77
81
|
}
|
|
78
|
-
return fail(
|
|
82
|
+
return fail(UNPARSEABLE_TRAILER_SUMMARY);
|
|
83
|
+
}
|
|
84
|
+
// Signatures anchor distinctive CLI-error phrasing — never bare fragments or bare status-code
|
|
85
|
+
// numbers (the QUOTA_RE Pitfall-3 lesson): these run only over no-trailer output, but a stalled
|
|
86
|
+
// worker's harvested pane can still contain ordinary work text.
|
|
87
|
+
const AUTH_RE = /not logged in|please (?:log ?in|sign in)|please run [^\n]{0,30}log ?in|authentication[ _](?:required|failed|error)|invalid (?:api key|credentials)|api key (?:is )?(?:not set|missing|invalid|required)|credentials? (?:have )?expired|401 unauthorized/i;
|
|
88
|
+
const SETUP_RE = /command not found|not recognized as an internal or external command|spawn \S+ ENOENT|missing required config|workspace trust (?:required|not granted)/i;
|
|
89
|
+
const OUTAGE_RE = /unable to reach the model provider|cannot reach the model provider|model provider.{0,40}(?:unavailable|unreachable)|service (?:is )?temporarily unavailable|upstream connect error|overloaded_error/i;
|
|
90
|
+
const TIMEOUT_RE = /\bETIMEDOUT\b|request timed out|connection timed out|deadline exceeded|timed out waiting for/i;
|
|
91
|
+
export function classifyDeadChannel(result) {
|
|
92
|
+
// A parsed trailer — ok:true OR ok:false — is the worker speaking: genuine work outcomes walk
|
|
93
|
+
// the normal gate/ladder path even when their transcript mentions auth/outage/timeout text.
|
|
94
|
+
if (result.ok || (result.summary !== NO_TRAILER_SUMMARY && result.summary !== UNPARSEABLE_TRAILER_SUMMARY))
|
|
95
|
+
return undefined;
|
|
96
|
+
if (AUTH_RE.test(result.raw))
|
|
97
|
+
return "auth-required";
|
|
98
|
+
if (SETUP_RE.test(result.raw))
|
|
99
|
+
return "setup-required";
|
|
100
|
+
if (OUTAGE_RE.test(result.raw))
|
|
101
|
+
return "provider-outage";
|
|
102
|
+
if (TIMEOUT_RE.test(result.raw))
|
|
103
|
+
return "timeout";
|
|
104
|
+
return undefined;
|
|
79
105
|
}
|
|
@@ -12,6 +12,9 @@ export declare function detectCandidateClis(opts?: {
|
|
|
12
12
|
pathEnv?: string;
|
|
13
13
|
}): CandidateCliDetection[];
|
|
14
14
|
export declare function formatCandidateCliRow({ binary, version }: CandidateCliDetection): string;
|
|
15
|
+
export declare function probeHelpText(bin: string): string | undefined;
|
|
16
|
+
export declare function missingDeclaredFlags(helpText: string, flags: string[]): string[];
|
|
17
|
+
export declare function flagDriftWarnings(adapters: WorkerAdapter[], health: Record<string, AuthHealth>, helpOf?: (bin: string) => string | undefined): string[];
|
|
15
18
|
export declare function allAdapters(opts?: {
|
|
16
19
|
fakeScriptPath?: string;
|
|
17
20
|
}): WorkerAdapter[];
|
|
@@ -53,6 +53,44 @@ export function formatCandidateCliRow({ binary, version }) {
|
|
|
53
53
|
const ver = version ?? "version unknown";
|
|
54
54
|
return ` ! ${binary.padEnd(14)} detected: ${ver} (no tickmarkr adapter — not routable)`;
|
|
55
55
|
}
|
|
56
|
+
// v1.65 T3: same guarded spawn pattern as probeCandidateVersion — bounded, error/status-checked,
|
|
57
|
+
// fail-open. undefined = binary unavailable/broken ⇒ no drift verdict (the existing auth reporting
|
|
58
|
+
// already names an uninstalled CLI; drift must never pile on).
|
|
59
|
+
export function probeHelpText(bin) {
|
|
60
|
+
try {
|
|
61
|
+
const r = spawnSync(bin, ["--help"], { encoding: "utf8", timeout: 10000 });
|
|
62
|
+
if (r.error || r.status !== 0)
|
|
63
|
+
return undefined;
|
|
64
|
+
return `${r.stdout || ""}\n${r.stderr || ""}`;
|
|
65
|
+
}
|
|
66
|
+
catch {
|
|
67
|
+
return undefined;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
// Whole-token match only: "-p" must never count as present because "--print" contains it, and
|
|
71
|
+
// "--output" must not match a listed "--output-format". Splitting on non-flag chars keeps each
|
|
72
|
+
// help-listed flag intact as one token ("-p, --print <fmt>" → "-p", "--print", "fmt").
|
|
73
|
+
export function missingDeclaredFlags(helpText, flags) {
|
|
74
|
+
const tokens = new Set(helpText.split(/[^A-Za-z0-9_-]+/));
|
|
75
|
+
return flags.filter((f) => !tokens.has(f));
|
|
76
|
+
}
|
|
77
|
+
// v1.65 T3: advisory hardcoded-flag drift check. Reads help, mutates nothing — health, doctor.json,
|
|
78
|
+
// discoverChannels, and routing never see these strings; they are doctor display rows only.
|
|
79
|
+
export function flagDriftWarnings(adapters, health, helpOf = probeHelpText) {
|
|
80
|
+
const out = [];
|
|
81
|
+
for (const a of adapters) {
|
|
82
|
+
const decl = a.hardcodedFlags;
|
|
83
|
+
if (!decl || !health[a.id]?.installed)
|
|
84
|
+
continue;
|
|
85
|
+
const help = helpOf(decl.binary);
|
|
86
|
+
if (help === undefined)
|
|
87
|
+
continue;
|
|
88
|
+
for (const flag of missingDeclaredFlags(help, decl.flags)) {
|
|
89
|
+
out.push(`flag drift: ${a.id} hardcodes ${flag} but ${decl.binary} --help no longer lists it (advisory — routing unchanged)`);
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
return out;
|
|
93
|
+
}
|
|
56
94
|
export function allAdapters(opts = {}) {
|
|
57
95
|
// pi + grok + kimi appended LAST: same-tier ties resolve by discovery order (Phase 6 D2), so
|
|
58
96
|
// appending keeps the Phase 6 shape→channel matrix byte-identical; inserting anywhere else
|
package/dist/adapters/types.d.ts
CHANGED
|
@@ -91,6 +91,10 @@ export interface WorkerAdapter {
|
|
|
91
91
|
contextUsage?(session: SessionRef): ContextUsage | null;
|
|
92
92
|
trust?(repoRoot: string): TrustVerdict;
|
|
93
93
|
trustDialog?: TrustDialog;
|
|
94
|
+
hardcodedFlags?: {
|
|
95
|
+
binary: string;
|
|
96
|
+
flags: string[];
|
|
97
|
+
};
|
|
94
98
|
}
|
|
95
99
|
export declare function channelsFromConfig(adapterId: string, cfg: TickmarkrConfig): BillingChannel[];
|
|
96
100
|
export declare function channelKey(c: {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { writeFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
|
-
import { allAdapters, detectCandidateClis, probeAll, probeModels, readAutoPrefer, servableExclusions, servabilityLine, writeDoctor } from "../../adapters/registry.js";
|
|
3
|
+
import { allAdapters, detectCandidateClis, flagDriftWarnings, probeAll, probeModels, readAutoPrefer, servableExclusions, servabilityLine, writeDoctor } from "../../adapters/registry.js";
|
|
4
4
|
import { BANNER, dim, fail, kvRow, legend, ok, rule, statusRow, title } from "../../brand.js";
|
|
5
5
|
import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
|
|
6
6
|
import { declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
|
|
@@ -85,6 +85,9 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
85
85
|
const servable = servableExclusions(cfg, adapters, health);
|
|
86
86
|
if (servable.length)
|
|
87
87
|
rows.push(attentionRow(servabilityLine(servable)));
|
|
88
|
+
// v1.65 T3: hardcoded-flag drift — advisory warn rows only. Runs AFTER writeDoctor so the verdicts
|
|
89
|
+
// can never leak into doctor.json, and discoverChannels/routing never read them.
|
|
90
|
+
rows.push(...flagDriftWarnings(adapters, health).map(attentionRow));
|
|
88
91
|
// MODEL-05/06: print-only drift fragment; advisory, whole-line-commented additions, tickmarkr NEVER applies it.
|
|
89
92
|
// TTY gets a one-line summary + the fragment as a file (the full dump drowned everything else,
|
|
90
93
|
// v1.33.1 onboarding); machine/CI surface keeps the inline dump — layout is pinned by tests.
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { BANNER, GLYPHS, dim, fail, legend, ok, rule, statusRow, title, warn } from "../../brand.js";
|
|
2
|
-
import { blockedTasks, graphDefinitionHash, loadGraph
|
|
2
|
+
import { blockedTasks, graphDefinitionHash, loadGraph } from "../../graph/graph.js";
|
|
3
3
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
4
|
+
import { foldActivity } from "../../run/activity.js";
|
|
4
5
|
import { Journal, engagementComparable } from "../../run/journal.js";
|
|
5
6
|
// ponytail: fixed 2s refresh; promote to config.visibility.* only when an operator asks.
|
|
6
7
|
const REFRESH_MS = 2000;
|
|
@@ -171,14 +172,17 @@ const renderFrame = (cwd) => {
|
|
|
171
172
|
}
|
|
172
173
|
const effective = { ...g, tasks: g.tasks.map((t) => ({ ...t, status: replayed?.get(t.id) ?? t.status })) };
|
|
173
174
|
const starved = new Set(blockedTasks(effective).map((t) => t.id));
|
|
174
|
-
|
|
175
|
+
// OBS-104: ONE activity fold feeds both surfaces — never re-derived here. Comparable events only
|
|
176
|
+
// (a recompiled graph's journal must not animate the wrong tasks); with no or stale journal the
|
|
177
|
+
// dep-waiting cells still derive from the effective graph statuses.
|
|
178
|
+
const activity = foldActivity(comparable ? events : [], effective.tasks);
|
|
175
179
|
const unicode = visual();
|
|
176
180
|
const divider = unicode ? " · " : " / ";
|
|
177
181
|
const width = process.stdout.columns ?? 120;
|
|
178
182
|
const done = effective.tasks.filter((t) => t.status === "done").length;
|
|
179
183
|
const cells = g.tasks.map((t) => {
|
|
180
184
|
const st = replayed?.get(t.id) ?? t.status;
|
|
181
|
-
const label = starved.has(t.id) ? " starved" :
|
|
185
|
+
const label = starved.has(t.id) ? " starved" : activity.cells.has(t.id) ? ` ${activity.cells.get(t.id)}` : "";
|
|
182
186
|
const channel = assignments.get(t.id) ?? "-";
|
|
183
187
|
const assignCol = contexts.has(t.id) ? `${channel}${divider}ctx ${contexts.get(t.id)}` : channel;
|
|
184
188
|
return { t, st, label, assignCol, states: comparable ? gateStates(t, events) : defaultGateStates(t) };
|
|
@@ -218,6 +222,10 @@ const renderFrame = (cwd) => {
|
|
|
218
222
|
: `no runs yet${dot}`) +
|
|
219
223
|
`${gauge} ${done === g.tasks.length && g.tasks.length > 0 ? ok(tally) : tally}`;
|
|
220
224
|
const hr = rule(Math.min(width, 100));
|
|
225
|
+
// OBS-104 run-level now line: names the most recent journal event. TTY frame only — the non-TTY
|
|
226
|
+
// machine surface is byte-pinned (status-brand golden) and must not drift. Rendered BELOW the task
|
|
227
|
+
// rows: the line names task ids, and rows must stay the first id-bearing lines for grep consumers.
|
|
228
|
+
const nowLine = activity.now ? [legend(` now: ${activity.now}`)] : [];
|
|
221
229
|
const gatesLegend = legend(` gates: ${GATE_NAMES.join(" · ")}`);
|
|
222
230
|
const taskVerdict = (st) => st === "done" ? "pass" : st === "failed" ? "fail" : st === "human" ? "warn" : "neutral";
|
|
223
231
|
const idW = Math.max(...cells.map((c) => c.t.id.length), 2);
|
|
@@ -238,7 +246,7 @@ const renderFrame = (cwd) => {
|
|
|
238
246
|
" ".repeat(stW - (String(st) + label + failedSuffix(states) + human).length);
|
|
239
247
|
return ` ${statusRow(taskVerdict(st), `${t.id.padEnd(idW)} ${goal} ${gateChain(states, true)} ${statusCell} ${dim(assignCol)}`)}`;
|
|
240
248
|
});
|
|
241
|
-
return [header, hr, gatesLegend, ...rows].join("\n");
|
|
249
|
+
return [header, hr, gatesLegend, ...rows, ...nowLine].join("\n");
|
|
242
250
|
};
|
|
243
251
|
export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
244
252
|
// cockpit surface: banner + frame on a TTY (doctor's pattern); pipes get the bare frame
|
package/dist/config/config.js
CHANGED
|
@@ -468,7 +468,10 @@ export function configTemplate(overlay) {
|
|
|
468
468
|
# explore: { mode: on, excludeShapes: [], excludeComplexityAtOrAbove: null, cap: 5 } # optional; absent ⇒ byte-identical
|
|
469
469
|
# sla: { implement: 15 } # optional per-shape minutes — advisory plan lint only; absent ⇒ no lint
|
|
470
470
|
# allow: { adapters: [claude-code, codex] } # optional fleet allowlist; presence activates even if empty (fail-closed)
|
|
471
|
-
# deny:
|
|
471
|
+
# deny: # optional fleet denylist; deny beats allow on conflict
|
|
472
|
+
# models:
|
|
473
|
+
# - pi:zai/glm-5.2 # OBS-57: pi passes run-start probe but hangs at finish without TICKMARKR_RESULT — remove after no-trailer demotion ships (v1.46 provider-outage taxonomy)
|
|
474
|
+
# # incident-born deny/pin entries MUST name OBS id + root cause + removal condition (see docs/codebase/CONVENTIONS.md)
|
|
472
475
|
# # entry grammar: adapter id | model id | adapter:model (entries in either list accept all three forms)
|
|
473
476
|
# # a hint pinning a denied channel FAILS at plan time (RoutingError) — never silent reroute
|
|
474
477
|
# # tombstone: deny: null in a repo overlay removes a global deny (arrays replace wholesale, never merge)
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -4,13 +4,16 @@ import { DEFAULT_DIFF_CAP } from "../config/config.js";
|
|
|
4
4
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
5
5
|
import { sh } from "../run/git.js";
|
|
6
6
|
import { checkDiffCap, fetchTaskDiff } from "./review.js";
|
|
7
|
-
import { extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
7
|
+
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
8
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
9
9
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
10
10
|
const JudgeVerdictRowSchema = z.object({
|
|
11
11
|
criterion: z.string(),
|
|
12
12
|
met: z.boolean(),
|
|
13
13
|
reason: z.string(),
|
|
14
|
+
// v1.64: a verbatim quote from the judged diff grounding the ruling — required; a verdict
|
|
15
|
+
// omitting it is malformed and fails closed like any other shape violation.
|
|
16
|
+
evidence: z.string(),
|
|
14
17
|
});
|
|
15
18
|
const JudgeVerdictSchema = z.object({
|
|
16
19
|
pass: z.boolean(),
|
|
@@ -157,6 +160,8 @@ You are a strict acceptance judge. Decide whether the diff satisfies EVERY accep
|
|
|
157
160
|
Judge only what the diff proves — plausible-but-wrong must fail. Do not award partial credit.
|
|
158
161
|
Deterministic command/test oracles have already passed mechanically; judge ONLY the rubric items below.
|
|
159
162
|
|
|
163
|
+
${COMPLETION_FAKING_CHECKLIST}
|
|
164
|
+
|
|
160
165
|
## Task ${task.id}: ${task.title}
|
|
161
166
|
Goal: ${task.goal}
|
|
162
167
|
|
|
@@ -171,8 +176,9 @@ ${diff}
|
|
|
171
176
|
${verdictNonceLine(nonce)}
|
|
172
177
|
|
|
173
178
|
Respond with ONLY this JSON (no prose before or after):
|
|
174
|
-
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "..."}]}
|
|
179
|
+
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": "..."}]}
|
|
175
180
|
Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) exactly once.
|
|
181
|
+
Each criteria[].evidence MUST be a short verbatim quote copied from the diff above that grounds the ruling; a quote not found in the diff voids the whole verdict.
|
|
176
182
|
`;
|
|
177
183
|
const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
|
|
178
184
|
const extracted = extractVerdictJson(raw, nonce);
|
|
@@ -185,6 +191,15 @@ Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) ex
|
|
|
185
191
|
meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
|
|
186
192
|
}
|
|
187
193
|
const { verdict: v, inconsistencies } = checkJudgeVerdict(extracted, expectedIds);
|
|
194
|
+
// v1.64: quoted evidence must appear verbatim in `diff` — the exact string embedded in the prompt
|
|
195
|
+
// above, never the worktree or any other artifact. A quote the diff doesn't contain is a
|
|
196
|
+
// hallucinated verdict: treated as unparseable so GATE-09 retries the judge on a failover channel.
|
|
197
|
+
const fabricated = v.criteria.filter((row) => !row.evidence.trim() || !diff.includes(row.evidence));
|
|
198
|
+
if (fabricated.length) {
|
|
199
|
+
return { gate: "acceptance", pass: false,
|
|
200
|
+
details: warn + detBlock + `judge verdict quotes evidence absent from the judged diff (${fabricated.map((row) => row.criterion).join(", ")}) — treating as unparseable, failing closed`,
|
|
201
|
+
meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
|
|
202
|
+
}
|
|
188
203
|
const pass = v.pass === true && inconsistencies.length === 0 && v.criteria.every((row) => row.met);
|
|
189
204
|
const lines = v.criteria.map((row) => `${row.met ? "✓" : "✗"} ${row.criterion}: ${row.reason}`);
|
|
190
205
|
if (!v.pass)
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
2
|
+
import type { AcceptanceItem } from "../graph/schema.js";
|
|
2
3
|
import type { GateResult } from "./types.js";
|
|
3
4
|
export interface Baseline {
|
|
4
5
|
commands: Record<string, {
|
|
@@ -16,4 +17,14 @@ export interface BaselineWarning {
|
|
|
16
17
|
export declare function fingerprint(output: string): string[];
|
|
17
18
|
export declare function detectGateCommands(repoRoot: string, cfg: TickmarkrConfig): Record<string, string>;
|
|
18
19
|
export declare function captureBaseline(cwd: string, commands: Record<string, string>): Promise<Baseline>;
|
|
20
|
+
export interface VacuousOracleWarning {
|
|
21
|
+
kind: "vacuous-oracle";
|
|
22
|
+
taskId: string;
|
|
23
|
+
oracles: string[];
|
|
24
|
+
reason: string;
|
|
25
|
+
}
|
|
26
|
+
export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
|
|
27
|
+
id: string;
|
|
28
|
+
acceptance: AcceptanceItem[];
|
|
29
|
+
}>): Promise<VacuousOracleWarning[]>;
|
|
19
30
|
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[]): Promise<GateResult[]>;
|
package/dist/gates/baseline.js
CHANGED
|
@@ -83,6 +83,32 @@ export async function captureBaseline(cwd, commands) {
|
|
|
83
83
|
}
|
|
84
84
|
return base;
|
|
85
85
|
}
|
|
86
|
+
// Tier A #3 (2026-07-21 repo-scan reconciliation): a command oracle that already exits 0 before any
|
|
87
|
+
// work exists cannot falsify the work — surface it at baseline capture. Observational only: journaled
|
|
88
|
+
// warning, never a gate input, and an oracle that fails at baseline changes nothing. Judge oracles
|
|
89
|
+
// (including plain-string compat judges) are never executed; test oracles stay gate-only (they need
|
|
90
|
+
// the detected runner and the worker's diff to mean anything).
|
|
91
|
+
export async function detectVacuousOracles(cwd, tasks) {
|
|
92
|
+
const out = [];
|
|
93
|
+
for (const t of tasks) {
|
|
94
|
+
const vacuous = [];
|
|
95
|
+
for (const a of t.acceptance) {
|
|
96
|
+
if (typeof a !== "object" || a.oracle !== "command")
|
|
97
|
+
continue;
|
|
98
|
+
if ((await sh(a.command, cwd)).code === 0)
|
|
99
|
+
vacuous.push(a.command);
|
|
100
|
+
}
|
|
101
|
+
if (vacuous.length) {
|
|
102
|
+
out.push({
|
|
103
|
+
kind: "vacuous-oracle",
|
|
104
|
+
taskId: t.id,
|
|
105
|
+
oracles: vacuous,
|
|
106
|
+
reason: `vacuous acceptance oracle on ${t.id}: already passes before any work exists — ${vacuous.map((c) => `$ ${c}`).join("; ")}`,
|
|
107
|
+
});
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
return out;
|
|
111
|
+
}
|
|
86
112
|
// HYG-08 (D-01): headline the runner's own failure naming; demote the fingerprint diff to a secondary
|
|
87
113
|
// section. Extracts from `raw` — the SAME cwd-stripped, per-line ANSI_RE-stripped string that was
|
|
88
114
|
// fingerprinted, digits UN-normalized (Pitfall 2: normalization mangles test names, and the diff set could
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
import type { WorkerAdapter } from "../adapters/types.js";
|
|
2
2
|
import { type ExecutorDriver, type Slot } from "../drivers/types.js";
|
|
3
3
|
export declare const GATE_PANE_SEP = " \u00B7 ";
|
|
4
|
+
export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
|
|
4
5
|
/** Fable F3: per-call nonce echoed in verdict JSON and gate exit markers. */
|
|
5
6
|
export declare function generateVerdictNonce(): string;
|
|
6
7
|
export declare function verdictNonceLine(nonce: string): string;
|
|
7
8
|
export declare function extractPromptNonce(prompt: string): string | null;
|
|
8
9
|
export declare function gateExitTrailer(nonce: string): string;
|
|
9
|
-
export declare function augmentFakeVerdictOutput(adapter: WorkerAdapter, out: string, nonce: string): string;
|
|
10
|
+
export declare function augmentFakeVerdictOutput(adapter: WorkerAdapter, out: string, nonce: string, prompt?: string): string;
|
|
10
11
|
export type GatePaneRole = "judge" | "review" | "consult";
|
|
11
12
|
/** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
|
|
12
13
|
export declare function gatePaneName(role: GatePaneRole, taskId: string, suffix?: string): string;
|
package/dist/gates/llm.js
CHANGED
|
@@ -6,6 +6,22 @@ import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
|
|
|
6
6
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
7
7
|
import { sh } from "../run/git.js";
|
|
8
8
|
export const GATE_PANE_SEP = " · ";
|
|
9
|
+
// v1.64 gate-integrity (repo-scan Tier A·1): the concrete completion-faking shortcuts every
|
|
10
|
+
// judge/review verdict must hunt for. Shared verbatim by the acceptance judge and review prompts.
|
|
11
|
+
export const COMPLETION_FAKING_CHECKLIST = `## Completion-faking checklist
|
|
12
|
+
Hunt for these concrete completion-faking shortcuts before ruling on any criterion:
|
|
13
|
+
- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic
|
|
14
|
+
- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green
|
|
15
|
+
- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)
|
|
16
|
+
- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior
|
|
17
|
+
- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself
|
|
18
|
+
- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be
|
|
19
|
+
- error-swallowing: catch or fallback that hides failures instead of handling them
|
|
20
|
+
- self-mocking: the code under test mocked or faked so the test exercises the mock
|
|
21
|
+
- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green
|
|
22
|
+
- rename-as-work: code moved or renamed and presented as the requested change
|
|
23
|
+
- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched
|
|
24
|
+
When a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.`;
|
|
9
25
|
/** Fable F3: per-call nonce echoed in verdict JSON and gate exit markers. */
|
|
10
26
|
export function generateVerdictNonce() {
|
|
11
27
|
return randomBytes(4).toString("hex");
|
|
@@ -19,8 +35,20 @@ export function extractPromptNonce(prompt) {
|
|
|
19
35
|
export function gateExitTrailer(nonce) {
|
|
20
36
|
return `printf '\\nTICKMARKR_''EXIT_${nonce}:%s\\n' $?`;
|
|
21
37
|
}
|
|
38
|
+
// v1.64: scripted fake judge verdicts predate the required per-criterion evidence field — quote the
|
|
39
|
+
// first line of the prompt's own diff block into rows lacking one so zero-token fixtures keep their
|
|
40
|
+
// outcomes. Rows scripting an explicit evidence value pass through verbatim (tests exercise both paths).
|
|
41
|
+
function injectFakeEvidence(obj, prompt) {
|
|
42
|
+
if (!prompt.startsWith("TICKMARKR-JUDGE") || !Array.isArray(obj.criteria))
|
|
43
|
+
return obj;
|
|
44
|
+
const line = /```diff\n([\s\S]*?)```/.exec(prompt)?.[1].split("\n").find((l) => l.trim());
|
|
45
|
+
if (!line)
|
|
46
|
+
return obj;
|
|
47
|
+
const criteria = obj.criteria.map((row) => row && typeof row === "object" && !("evidence" in row) ? { ...row, evidence: line } : row);
|
|
48
|
+
return { ...obj, criteria };
|
|
49
|
+
}
|
|
22
50
|
// ponytail: fake adapter serves static verdict JSON without nonce; append a bound copy for zero-token tests.
|
|
23
|
-
export function augmentFakeVerdictOutput(adapter, out, nonce) {
|
|
51
|
+
export function augmentFakeVerdictOutput(adapter, out, nonce, prompt = "") {
|
|
24
52
|
if (adapter.id !== "fake")
|
|
25
53
|
return out;
|
|
26
54
|
const obj = extractJson(out);
|
|
@@ -28,7 +56,7 @@ export function augmentFakeVerdictOutput(adapter, out, nonce) {
|
|
|
28
56
|
return out;
|
|
29
57
|
if (typeof obj.nonce === "string")
|
|
30
58
|
return out;
|
|
31
|
-
return `${out}\n${JSON.stringify({ ...obj, nonce })}`;
|
|
59
|
+
return `${out}\n${JSON.stringify(injectFakeEvidence({ ...obj, nonce }, prompt))}`;
|
|
32
60
|
}
|
|
33
61
|
/** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
|
|
34
62
|
export function gatePaneName(role, taskId, suffix = "") {
|
|
@@ -61,7 +89,7 @@ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 30000
|
|
|
61
89
|
const nonce = extractPromptNonce(prompt);
|
|
62
90
|
let out = r.stdout + "\n" + r.stderr;
|
|
63
91
|
if (nonce)
|
|
64
|
-
out = augmentFakeVerdictOutput(adapter, out, nonce);
|
|
92
|
+
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
65
93
|
return out;
|
|
66
94
|
}
|
|
67
95
|
// v1.1 default path: the same headless CLI call, but dispatched through the driver
|
|
@@ -88,7 +116,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
88
116
|
let out = await via.driver.read(slot, 400);
|
|
89
117
|
if (!via.keep)
|
|
90
118
|
await via.driver.close(slot);
|
|
91
|
-
out = augmentFakeVerdictOutput(adapter, out, nonce);
|
|
119
|
+
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
92
120
|
return out;
|
|
93
121
|
}
|
|
94
122
|
export function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
package/dist/gates/review.js
CHANGED
|
@@ -4,7 +4,7 @@ import { renderAcceptanceItem } from "../graph/schema.js";
|
|
|
4
4
|
import { getAdapter } from "../adapters/registry.js";
|
|
5
5
|
import { shOk } from "../run/git.js";
|
|
6
6
|
import { marginalCostRank } from "../route/router.js";
|
|
7
|
-
import { extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
7
|
+
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
8
|
// OBS-48: cap on zero-context diff bytes (git diff -U0), not context-padded full diff — scattered
|
|
9
9
|
// one-line hunks no longer trip at ~370 diff-bytes per changed line. Full diff still goes to the judge.
|
|
10
10
|
const DIFF_CAP_REMEDY = "split the task, or raise gates.diffCap";
|
|
@@ -82,6 +82,8 @@ export async function reviewGate(task, worktree, baseRef, author, channels, adap
|
|
|
82
82
|
You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
|
|
83
83
|
Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
|
|
84
84
|
|
|
85
|
+
${COMPLETION_FAKING_CHECKLIST}
|
|
86
|
+
|
|
85
87
|
## Task ${task.id}: ${task.title} (complexity ${task.complexity})
|
|
86
88
|
## Acceptance criteria
|
|
87
89
|
${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import { type JournalEvent } from "./journal.js";
|
|
2
|
+
export interface ActivityTask {
|
|
3
|
+
id: string;
|
|
4
|
+
gates: readonly string[];
|
|
5
|
+
deps: readonly string[];
|
|
6
|
+
/** the surface's effective status for the task (replayed ?? graph) */
|
|
7
|
+
status: string;
|
|
8
|
+
}
|
|
9
|
+
export interface ActivitySnapshot {
|
|
10
|
+
/** run-level now line naming the most recent journal event; absent when there are no events */
|
|
11
|
+
now?: string;
|
|
12
|
+
/** taskId → current-activity phrase; absent = idle (terminal, or queued with met deps) */
|
|
13
|
+
cells: Map<string, string>;
|
|
14
|
+
}
|
|
15
|
+
export declare function foldActivity(events: JournalEvent[], tasks: readonly ActivityTask[]): ActivitySnapshot;
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
import { formatJournalNarration } from "./journal.js";
|
|
2
|
+
const channelOf = (assignment) => {
|
|
3
|
+
const a = assignment;
|
|
4
|
+
return typeof a?.adapter === "string" && typeof a.model === "string" ? `${a.adapter}:${a.model}` : "unknown channel";
|
|
5
|
+
};
|
|
6
|
+
const cellText = (st, task) => {
|
|
7
|
+
switch (st.kind) {
|
|
8
|
+
case "worker":
|
|
9
|
+
// since is the dispatch event's own ISO ts — sliced, never re-clocked (purity)
|
|
10
|
+
return `attempt ${st.attempt} in flight on ${st.channel} since ${st.since.slice(11, 19)}`;
|
|
11
|
+
case "gates": {
|
|
12
|
+
const next = task.gates.find((g) => !st.results.has(g));
|
|
13
|
+
if (next)
|
|
14
|
+
return `gate ${next} running`;
|
|
15
|
+
// every declared gate has a result: all pass ⇒ the daemon is merging; any fail ⇒ a retry decision is next
|
|
16
|
+
return [...st.results.values()].every(Boolean) ? "merging" : "retrying";
|
|
17
|
+
}
|
|
18
|
+
case "retrying":
|
|
19
|
+
return "retrying";
|
|
20
|
+
case "parked":
|
|
21
|
+
return st.note ? `parked (${st.note})` : "parked";
|
|
22
|
+
}
|
|
23
|
+
};
|
|
24
|
+
export function foldActivity(events, tasks) {
|
|
25
|
+
const live = new Map();
|
|
26
|
+
// a daemon (re)start or run-end means no attempt/gate is in flight — mirror reconcile.ts; parks persist
|
|
27
|
+
const clearTransient = () => {
|
|
28
|
+
// deleting the current entry mid-iteration is well-defined for Map
|
|
29
|
+
for (const [id, st] of live)
|
|
30
|
+
if (st.kind !== "parked")
|
|
31
|
+
live.delete(id);
|
|
32
|
+
};
|
|
33
|
+
for (const e of events) {
|
|
34
|
+
if (e.event === "run-start" || e.event === "run-resume" || e.event === "run-end") {
|
|
35
|
+
clearTransient();
|
|
36
|
+
continue;
|
|
37
|
+
}
|
|
38
|
+
const id = e.taskId;
|
|
39
|
+
if (!id)
|
|
40
|
+
continue;
|
|
41
|
+
switch (e.event) {
|
|
42
|
+
case "task-dispatch":
|
|
43
|
+
live.set(id, {
|
|
44
|
+
kind: "worker",
|
|
45
|
+
attempt: (Number.isInteger(e.data.attempt) ? e.data.attempt : 0) + 1,
|
|
46
|
+
channel: channelOf(e.data.assignment),
|
|
47
|
+
since: e.ts,
|
|
48
|
+
});
|
|
49
|
+
break;
|
|
50
|
+
case "worker-result":
|
|
51
|
+
// a clean trailer moves the task into gating; anything else is heading for a retry decision
|
|
52
|
+
live.set(id, e.data.ok === true && e.data.finished === true ? { kind: "gates", results: new Map() } : { kind: "retrying" });
|
|
53
|
+
break;
|
|
54
|
+
case "gate-result": {
|
|
55
|
+
const prev = live.get(id);
|
|
56
|
+
const results = prev?.kind === "gates" ? prev.results : new Map();
|
|
57
|
+
if (typeof e.data.gate === "string")
|
|
58
|
+
results.set(e.data.gate, e.data.pass === true || e.data.skipped === true);
|
|
59
|
+
live.set(id, { kind: "gates", results });
|
|
60
|
+
break;
|
|
61
|
+
}
|
|
62
|
+
case "escalation":
|
|
63
|
+
case "consult-verdict":
|
|
64
|
+
case "quota-failover":
|
|
65
|
+
case "provider-death-requeue":
|
|
66
|
+
case "merge-conflict":
|
|
67
|
+
live.set(id, { kind: "retrying" });
|
|
68
|
+
break;
|
|
69
|
+
case "task-human":
|
|
70
|
+
live.set(id, { kind: "parked", ...(typeof e.data.kind === "string" ? { note: e.data.kind } : {}) });
|
|
71
|
+
break;
|
|
72
|
+
case "task-done":
|
|
73
|
+
case "task-failed":
|
|
74
|
+
case "task-approved":
|
|
75
|
+
live.delete(id);
|
|
76
|
+
break;
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
const status = new Map(tasks.map((t) => [t.id, t.status]));
|
|
80
|
+
const cells = new Map();
|
|
81
|
+
for (const t of tasks) {
|
|
82
|
+
const st = live.get(t.id);
|
|
83
|
+
if (st) {
|
|
84
|
+
cells.set(t.id, cellText(st, t));
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
if (t.status !== "pending")
|
|
88
|
+
continue;
|
|
89
|
+
// dep-waiting is reserved for genuinely unmet deps — a bare pending task gets no cell (OBS-104 fix 1)
|
|
90
|
+
const unmet = t.deps.filter((d) => status.get(d) !== "done");
|
|
91
|
+
if (unmet.length)
|
|
92
|
+
cells.set(t.id, `dep-waiting on ${unmet.join(", ")}`);
|
|
93
|
+
}
|
|
94
|
+
const last = events.at(-1);
|
|
95
|
+
return { ...(last ? { now: formatJournalNarration(last) } : {}), cells };
|
|
96
|
+
}
|
package/dist/run/consult.js
CHANGED
|
@@ -4,6 +4,8 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
4
4
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
5
5
|
import { augmentFakeVerdictOutput, extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
|
|
6
6
|
import { sh } from "./git.js";
|
|
7
|
+
import { redactSecrets } from "./redact.js";
|
|
8
|
+
import { filterLlmTranscript } from "./stall.js";
|
|
7
9
|
const MAX_RETRY_GUIDANCE_LINES = 10;
|
|
8
10
|
function guidanceParts(text) {
|
|
9
11
|
return text.split(/\n+/).flatMap((line) => line.split(/(?<=[.!?])\s+/)).map((s) => s.trim()).filter(Boolean);
|
|
@@ -55,6 +57,8 @@ export function augmentRetryBrief(feedback, opts) {
|
|
|
55
57
|
return parts.join("\n\n");
|
|
56
58
|
}
|
|
57
59
|
const ACTIONS = ["retry", "reroute", "decompose", "human"];
|
|
60
|
+
// v1.65 T2: transcript noise (spinner repaints, CR churn, pass-run spam) is squashed at build time —
|
|
61
|
+
// the filtered form is what persists to the consults/ artifact AND what the model reads; fail-open.
|
|
58
62
|
export function buildDossierPrompt(d, nonce) {
|
|
59
63
|
return `TICKMARKR-CONSULT
|
|
60
64
|
You are a senior engineering consult for the tickmarkr orchestrator. A worker task hit trouble.
|
|
@@ -69,7 +73,7 @@ ${d.gates.map((g) => `- [${g.pass ? "pass" : "FAIL"}] ${g.gate}: ${g.details}`).
|
|
|
69
73
|
${d.diff || "(none)"}
|
|
70
74
|
|
|
71
75
|
## Worker transcript (tail)
|
|
72
|
-
${d.transcript || "(none)"}
|
|
76
|
+
${filterLlmTranscript(d.transcript) || "(none)"}
|
|
73
77
|
|
|
74
78
|
## Journal (recent events)
|
|
75
79
|
${d.journalTail}
|
|
@@ -100,7 +104,9 @@ opts = {}) {
|
|
|
100
104
|
const dir = join(runDir, "consults");
|
|
101
105
|
mkdirSync(dir, { recursive: true });
|
|
102
106
|
const promptFile = join(dir, `${d.taskId}-${n}.md`);
|
|
103
|
-
|
|
107
|
+
// T3 secret redaction: the persisted dossier artifact (transcript/diff/journal tail — also what the
|
|
108
|
+
// consult model reads) is masked at this seam; the in-memory Dossier stays untouched.
|
|
109
|
+
writeFileSync(promptFile, redactSecrets(buildDossierPrompt(d, nonce)));
|
|
104
110
|
// One seat = the WHOLE invoke-and-parse unit, both visibility branches (OBS-69 class: a headless-only
|
|
105
111
|
// failover would leave the production pane path hard-failing on seat one). null = no parseable verdict.
|
|
106
112
|
const invokeSeat = async (seatAdapter, seatModel, seatIdx) => {
|
package/dist/run/daemon.js
CHANGED
|
@@ -4,14 +4,14 @@ import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync
|
|
|
4
4
|
import { tmpdir } from "node:os";
|
|
5
5
|
import { join } from "node:path";
|
|
6
6
|
import { stringify } from "yaml";
|
|
7
|
-
import { trailerPattern, writePrompt } from "../adapters/prompt.js";
|
|
7
|
+
import { classifyDeadChannel, trailerPattern, writePrompt } from "../adapters/prompt.js";
|
|
8
8
|
import { allAdapters, discoverChannels, getAdapter, probeAll, readDoctor } from "../adapters/registry.js";
|
|
9
9
|
import { addUsage, channelKey, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
|
|
10
10
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
11
11
|
import { globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
|
|
12
12
|
import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
|
|
13
13
|
import { formatOwnedName } from "../drivers/types.js";
|
|
14
|
-
import { captureBaseline, detectGateCommands } from "../gates/baseline.js";
|
|
14
|
+
import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
|
|
15
15
|
import { runGates } from "../gates/run-gates.js";
|
|
16
16
|
import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus } from "../graph/graph.js";
|
|
17
17
|
import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
@@ -259,12 +259,21 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
259
259
|
prior?.append("superseded", undefined, { by: runId });
|
|
260
260
|
for (const warning of baseline.warnings ?? [])
|
|
261
261
|
journal.append("baseline-warning", undefined, { ...warning });
|
|
262
|
+
// Tier A #3: run each task's command-typed acceptance oracles against the pristine baseline —
|
|
263
|
+
// one that already exits 0 verifies nothing. Warning only, taskId-stamped; never a gate input.
|
|
264
|
+
for (const w of await detectVacuousOracles(repoRoot, graph.tasks))
|
|
265
|
+
journal.append("baseline-warning", w.taskId, { ...w });
|
|
262
266
|
}
|
|
263
267
|
// T6: open the narrator AFTER run-start/run-resume is journaled so the watch surface has a run to
|
|
264
268
|
// show. driver.narrator is undefined on subprocess → no-op (subprocess spawns nothing). Swallowed:
|
|
265
269
|
// a failed-to-open or later-dead watch pane never affects the run.
|
|
270
|
+
// OBS-103: hold the returned slot — narrator() adopts an already-open watch by its owned name
|
|
271
|
+
// (a prior daemon instance's, after a stop→resume cycle), so the run-end sweep below can retire
|
|
272
|
+
// it regardless of which instance split the pane.
|
|
273
|
+
const watchName = formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId });
|
|
274
|
+
let watchSlot;
|
|
266
275
|
try {
|
|
267
|
-
await driver.narrator?.(repoRoot, "tickmarkr status --watch", runId);
|
|
276
|
+
watchSlot = await driver.narrator?.(repoRoot, "tickmarkr status --watch", runId);
|
|
268
277
|
}
|
|
269
278
|
catch {
|
|
270
279
|
/* cosmetic-only — the run proceeds without a live surface */
|
|
@@ -293,7 +302,22 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
293
302
|
if (keepForever)
|
|
294
303
|
return;
|
|
295
304
|
try {
|
|
296
|
-
|
|
305
|
+
const desired = desiredPanes(journal.read(), runId);
|
|
306
|
+
// The watch pane is never the DRIVER sweep's candidate (panesToClose spares role "watch":
|
|
307
|
+
// herdr's watches bookkeeping lives in close(), and a raw pane-close in the sweep would
|
|
308
|
+
// leave narrator() a stale cache) — the driver always sees it as desired; its lifecycle is
|
|
309
|
+
// decided here from the fold alone.
|
|
310
|
+
await driver.reconcile?.(new Set([...desired, watchName]), runId, opts);
|
|
311
|
+
// OBS-103: when the fold retires the watch (run-end boundary), close the narrator. The
|
|
312
|
+
// decision keys on the run identity in the pane name — narrator() adopts a prior daemon
|
|
313
|
+
// instance's pane under the same owned name, so a stop→resume cycle's leftover narrator
|
|
314
|
+
// closes exactly like one this instance opened. A narrator carrying a non-canonical name
|
|
315
|
+
// (no run identity) is never this run's to sweep.
|
|
316
|
+
if (watchSlot && watchSlot.name === watchName && !desired.has(watchName)) {
|
|
317
|
+
const w = watchSlot;
|
|
318
|
+
watchSlot = undefined;
|
|
319
|
+
await driver.close(w);
|
|
320
|
+
}
|
|
297
321
|
}
|
|
298
322
|
catch {
|
|
299
323
|
/* cosmetic — visibility is never a gate */
|
|
@@ -835,6 +859,40 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
835
859
|
await park(t, "quota exhausted on every eligible channel", "quota", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
836
860
|
return;
|
|
837
861
|
}
|
|
862
|
+
// v1.65 T1: typed dead-channel failure — the parse boundary classified this no-trailer result
|
|
863
|
+
// as auth-required / setup-required / provider-outage / timeout (classifyDeadChannel; the
|
|
864
|
+
// daemon consumes the type, never re-derives it from raw text). Same free failover as quota —
|
|
865
|
+
// no escalation-ladder step — plus run-wide exclusion via demotedChannels: unlike a quota
|
|
866
|
+
// window that may reset, a dead channel stays dead for this run (OBS-57 class). Strictly
|
|
867
|
+
// AFTER the quota check so quota behavior stays byte-identical (a quota hit returns/continues
|
|
868
|
+
// before reaching here); provider-outage lands here only once the v1.46 same-channel requeue
|
|
869
|
+
// cap above is spent, so a transient blip still recovers in place.
|
|
870
|
+
const dead = classifyDeadChannel(result);
|
|
871
|
+
if (dead) {
|
|
872
|
+
const from = channelKey(assignment);
|
|
873
|
+
demotedChannels.add(from); // excluded for later attempts AND later tasks in this run
|
|
874
|
+
const next = failover("dead-channel");
|
|
875
|
+
journal.append("dead-channel-failover", t.id, { reason: dead, from, to: next ? channelKey(next) : null });
|
|
876
|
+
if (next) {
|
|
877
|
+
await driver.notify(`tickmarkr ${runId}: ${t.id} dead channel (${dead}) failover`, { tier: "attention" });
|
|
878
|
+
// OBS-17 T2 (quota parity): the superseded slot holds a dead-end, not failure context.
|
|
879
|
+
if (!keepForever) {
|
|
880
|
+
const idx = keptSlots.indexOf(slot);
|
|
881
|
+
if (idx >= 0) {
|
|
882
|
+
keptSlots.splice(idx, 1);
|
|
883
|
+
try {
|
|
884
|
+
await closeSlot(slot);
|
|
885
|
+
}
|
|
886
|
+
catch { /* cosmetic — reconcile is the backstop */ }
|
|
887
|
+
}
|
|
888
|
+
}
|
|
889
|
+
assignment = next;
|
|
890
|
+
tried.push(channelKey(next));
|
|
891
|
+
continue;
|
|
892
|
+
}
|
|
893
|
+
await park(t, `dead channel (${dead}) and no eligible channel remains`, "reroute-exhausted", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
894
|
+
return;
|
|
895
|
+
}
|
|
838
896
|
if (!finished) {
|
|
839
897
|
// ROUTE-18 (OBS-04): the channel burned a window without emitting a trailer (no-trailer timeout
|
|
840
898
|
// OR trailer-less crash-exit — both finished:false). durationMs:0 marks a FACT row, not a timed
|
package/dist/run/journal.js
CHANGED
|
@@ -5,6 +5,7 @@ import { channelKey, TokenUsageSchema } from "../adapters/types.js";
|
|
|
5
5
|
import { stateDirName, tickmarkrDir } from "../graph/graph.js";
|
|
6
6
|
import { TIERS } from "../graph/schema.js";
|
|
7
7
|
import { buildProfile } from "../route/profile.js";
|
|
8
|
+
import { redactSecrets } from "./redact.js";
|
|
8
9
|
export function formatJournalNarration({ event, taskId, data }) {
|
|
9
10
|
const assignment = data.assignment;
|
|
10
11
|
const direct = [data.summary, data.reason, data.error, data.step, data.action, data.lint, data.branch, data.from]
|
|
@@ -311,9 +312,12 @@ export class Journal {
|
|
|
311
312
|
}
|
|
312
313
|
append(event, taskId, data = {}) {
|
|
313
314
|
const row = { ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data };
|
|
314
|
-
|
|
315
|
+
// T3 secret redaction: only the persisted bytes are masked — the caller's data stays untouched in
|
|
316
|
+
// memory. The narrator receives the persisted (masked) row so a pane sink never shows a credential.
|
|
317
|
+
const line = redactSecrets(JSON.stringify(row));
|
|
318
|
+
appendFileSync(this.journalPath, line + "\n");
|
|
315
319
|
try {
|
|
316
|
-
this.narrate?.(
|
|
320
|
+
this.narrate?.(JSON.parse(line));
|
|
317
321
|
}
|
|
318
322
|
catch {
|
|
319
323
|
// narration is observational; a broken sink must not affect the journal or run
|
|
@@ -421,7 +425,8 @@ export class Journal {
|
|
|
421
425
|
return m;
|
|
422
426
|
}
|
|
423
427
|
telemetry(row) {
|
|
424
|
-
|
|
428
|
+
// T3 secret redaction: same persistence seam as append — credential-free rows are byte-identical.
|
|
429
|
+
appendFileSync(join(this.dir, "telemetry.jsonl"), redactSecrets(JSON.stringify(row)) + "\n");
|
|
425
430
|
}
|
|
426
431
|
// Per-run, raw (NOT schema-validated) — report.ts reads v1.5 core fields; stays byte-compatible.
|
|
427
432
|
readTelemetry() {
|
package/dist/run/reconcile.js
CHANGED
|
@@ -87,8 +87,10 @@ export function desiredPanes(rows, runId) {
|
|
|
87
87
|
clearTask(row.taskId);
|
|
88
88
|
break;
|
|
89
89
|
case "run-end":
|
|
90
|
+
// OBS-103: run-end retires EVERY run-tagged pane, the watch narrator included. The fold
|
|
91
|
+
// keys on the run identity in the pane name, so a narrator a prior daemon instance opened
|
|
92
|
+
// (stop→resume cycle) retires all the same; a later run-resume re-desires it above.
|
|
90
93
|
desired.clear();
|
|
91
|
-
desired.add(formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId }));
|
|
92
94
|
break;
|
|
93
95
|
}
|
|
94
96
|
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
// T3 (repo-scan Tier A #2): the shared secret-redaction pass for captured agent text at the
|
|
2
|
+
// persistence seams. Every seam that persists captured agent text routes its bytes through
|
|
3
|
+
// redactSecrets before hitting disk — journal event payloads and telemetry rows (journal.ts) and
|
|
4
|
+
// consult dossier artifacts (consult.ts). In-memory originals are never touched: redaction applies
|
|
5
|
+
// to the serialized bytes only, at the write site.
|
|
6
|
+
//
|
|
7
|
+
// Two shape families:
|
|
8
|
+
// - vendor keys: recognizable prefix + high-entropy body. The prefix survives masking so the
|
|
9
|
+
// credential class stays identifiable; the body never persists.
|
|
10
|
+
// - secret assignments: KEY=value / "key": "value" where the key NAME announces a secret. The key
|
|
11
|
+
// survives; the value is masked.
|
|
12
|
+
// Text with no credential shapes passes through byte-identical. Masks contain no quote or backslash
|
|
13
|
+
// and value charsets never cross a JSON string boundary, so redacting a serialized JSON line always
|
|
14
|
+
// yields a line that still parses.
|
|
15
|
+
export const MASK = "[REDACTED]";
|
|
16
|
+
// Order matters: more specific prefixes (sk-ant-, sk-proj-) before the generic sk- form.
|
|
17
|
+
const VENDOR_KEY_RES = [
|
|
18
|
+
/\b(sk-ant-)[A-Za-z0-9_-]{8,}/g, // Anthropic
|
|
19
|
+
/\b(sk-proj-)[A-Za-z0-9_-]{8,}/g, // OpenAI project key
|
|
20
|
+
/\b(sk-)[A-Za-z0-9_-]{16,}/g, // OpenAI classic / sk-prefixed secret keys
|
|
21
|
+
/\b(github_pat_)[A-Za-z0-9_]{8,}/g, // GitHub fine-grained PAT
|
|
22
|
+
/\b(gh[pousr]_)[A-Za-z0-9]{16,}/g, // GitHub token family (ghp_/gho_/ghu_/ghs_/ghr_)
|
|
23
|
+
/\b(glpat-)[A-Za-z0-9_-]{8,}/g, // GitLab PAT
|
|
24
|
+
/\b(xox[baprs]-)[A-Za-z0-9-]{8,}/g, // Slack token family
|
|
25
|
+
/\b(npm_)[A-Za-z0-9]{16,}/g, // npm token
|
|
26
|
+
/\b(AKIA)[0-9A-Z]{16}\b/g, // AWS access key id
|
|
27
|
+
/\b(AIza)[0-9A-Za-z_-]{35}\b/g, // Google API key
|
|
28
|
+
];
|
|
29
|
+
// Key names that announce a secret. The optional quote after the key closes a JSON key ("apiKey": "…").
|
|
30
|
+
// The match starts AT the keyword — chars before it (MY_ in MY_API_KEY=) are simply left in place by
|
|
31
|
+
// replace, so anchoring on the keyword yields identical output while keeping the scan linear (a
|
|
32
|
+
// leading [A-Za-z0-9_.-]* here backtracks quadratically on long tokens — seconds on a 50K blob, and
|
|
33
|
+
// journal payloads carry parked diffs up to the diff cap).
|
|
34
|
+
// The value charset excludes quotes/backslash (never crosses a JSON string boundary) and square
|
|
35
|
+
// brackets (an already-masked value never re-matches ⇒ idempotent, and a vendor prefix kept by the
|
|
36
|
+
// pass above survives). The letter lookahead skips purely numeric values ("maxTokens":30000000 is a
|
|
37
|
+
// count, not a credential — and masking a bare JSON number would break the line's parse).
|
|
38
|
+
// Quotes may arrive backslash-escaped (a transcript already serialized inside a JSON string), so the
|
|
39
|
+
// optional opening/closing quote around key and value accepts \" as well as ".
|
|
40
|
+
const ASSIGNMENT_RE = /((?:api[_-]?key|apikey|secret|token|passwd|password|credential|access[_-]?key)[A-Za-z0-9_.-]*(?:\\?["'])?\s*[=:]\s*(?:\\?["'])?)(?=[^\s"'`\\,;[\]]*[A-Za-z])([^\s"'`\\,;[\]]{8,})/gi;
|
|
41
|
+
// Authorization headers are the most common credential shape in captured transcripts.
|
|
42
|
+
const BEARER_RE = /\b(Bearer\s+)[A-Za-z0-9._~+/=-]{16,}/g;
|
|
43
|
+
// Cheap literal prescan: every pattern above needs one of these substrings, and most persisted text
|
|
44
|
+
// (diffs, prose, telemetry) has none — those payloads skip all 12 regex passes in one linear scan.
|
|
45
|
+
const HINT_RE = /sk-|github_pat_|gh[pousr]_|glpat-|xox[baprs]-|npm_|AKIA|AIza|Bearer|api[_-]?key|apikey|secret|token|passwd|password|credential|access[_-]?key/i;
|
|
46
|
+
export function redactSecrets(text) {
|
|
47
|
+
if (!HINT_RE.test(text))
|
|
48
|
+
return text;
|
|
49
|
+
let out = text;
|
|
50
|
+
for (const re of VENDOR_KEY_RES)
|
|
51
|
+
out = out.replace(re, `$1${MASK}`);
|
|
52
|
+
return out.replace(ASSIGNMENT_RE, `$1${MASK}`).replace(BEARER_RE, `$1${MASK}`);
|
|
53
|
+
}
|
package/dist/run/stall.d.ts
CHANGED
|
@@ -1,4 +1,8 @@
|
|
|
1
|
-
/** Normalize one pane snapshot for the stall-inactivity compare (
|
|
2
|
-
*
|
|
3
|
-
*
|
|
1
|
+
/** Normalize one pane snapshot for the stall-inactivity compare (trailer parsing, harvest,
|
|
2
|
+
* waitOutput, and paging read the raw text; the LLM transcript filter below reuses this to
|
|
3
|
+
* CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
|
|
4
|
+
* equal are the same frame modulo spinner presentation; any other byte difference is activity. */
|
|
4
5
|
export declare function normalizeStallSnapshot(text: string): string;
|
|
6
|
+
/** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
|
|
7
|
+
* seam exists for fault injection in tests only — production callers pass text alone. */
|
|
8
|
+
export declare function filterLlmTranscript(text: string, classify?: (t: string) => string): string;
|
package/dist/run/stall.js
CHANGED
|
@@ -16,9 +16,77 @@ const SPINNER_RE = /[⠀-⣿]/g;
|
|
|
16
16
|
// A digit run (optionally decimal) bound directly to a time-unit suffix, standing alone as a
|
|
17
17
|
// word: 9s, 41s, 3m, 1h, 800ms. Never bare digits — "(6/7)" and "5 of 7" stay change-sensitive.
|
|
18
18
|
const ELAPSED_RE = /(?<![\w.])\d+(?:\.\d+)?(?:ms|[hms])(?!\w)/g;
|
|
19
|
-
/** Normalize one pane snapshot for the stall-inactivity compare (
|
|
20
|
-
*
|
|
21
|
-
*
|
|
19
|
+
/** Normalize one pane snapshot for the stall-inactivity compare (trailer parsing, harvest,
|
|
20
|
+
* waitOutput, and paging read the raw text; the LLM transcript filter below reuses this to
|
|
21
|
+
* CLASSIFY presentation-only lines, never to rewrite kept bytes). Two snapshots that normalize
|
|
22
|
+
* equal are the same frame modulo spinner presentation; any other byte difference is activity. */
|
|
22
23
|
export function normalizeStallSnapshot(text) {
|
|
23
24
|
return text.replace(ANSI_RE, "").replace(SPINNER_RE, "").replace(ELAPSED_RE, "");
|
|
24
25
|
}
|
|
26
|
+
// ─── v1.65 T2: LLM-bound transcript filter ──────────────────────────────────────────────────────
|
|
27
|
+
// Consult dossiers and gate prompts pay tokens per transcript byte, so LLM-bound text runs through
|
|
28
|
+
// a per-line classifier: carriage-return overwrite churn keeps only the final paint, lines that are
|
|
29
|
+
// pure presentation (spinner/ANSI/elapsed only — classified via normalizeStallSnapshot above, never
|
|
30
|
+
// a parallel normalizer) drop, consecutive repaint frames that normalize equal squash to the last,
|
|
31
|
+
// and runs of passing-test lines collapse to a count line. Failure lines, exit codes, and summary
|
|
32
|
+
// lines always survive verbatim. Fail-open by contract: any internal error — or trivial savings —
|
|
33
|
+
// returns the original text; the filter may only ever cost noise, never evidence.
|
|
34
|
+
// Signal that must never drop: failure markers, exit codes, run summaries. Substring matches on
|
|
35
|
+
// purpose (AssertionError, FAILED) — over-keeping is the safe miss, same asymmetry as the allowlist.
|
|
36
|
+
const KEEP_RE = /[✗✖]|fail|error|exception|fatal|panic|exit\s*code|exit(?:ed)?\s+with|non-?zero|traceback|^\s*(?:tests?\b|test\s+(?:files|suites)|suites?\b|snapshots?\b|duration\b|summary\b)/i;
|
|
37
|
+
// Passing-test line shapes (vitest/jest/tap/go/pytest). Only KEEP-negative lines reach this class.
|
|
38
|
+
const PASS_RE = /^\s*(?:[✓✔√]\s|ok\s+\d|PASS\b|---\s*PASS:)|\bPASSED\b/;
|
|
39
|
+
const COLLAPSE_MIN = 3; // a 1–2 line run costs less than the count line that would replace it
|
|
40
|
+
const MIN_SAVINGS_RATIO = 0.1; // below 10% shrink the rewrite is not worth its risk — pass through
|
|
41
|
+
function classifyTranscript(text) {
|
|
42
|
+
const out = [];
|
|
43
|
+
let run = [];
|
|
44
|
+
let prevNorm = null;
|
|
45
|
+
const flush = () => {
|
|
46
|
+
if (run.length >= COLLAPSE_MIN)
|
|
47
|
+
out.push(`[${run.length} passing-test lines collapsed]`);
|
|
48
|
+
else
|
|
49
|
+
out.push(...run);
|
|
50
|
+
run = [];
|
|
51
|
+
};
|
|
52
|
+
for (const raw of text.split("\n")) {
|
|
53
|
+
// CR overwrite churn: the final paint wins; earlier paints carrying must-keep signal survive too.
|
|
54
|
+
const segs = raw.split("\r");
|
|
55
|
+
for (const line of segs.filter((s, i) => i === segs.length - 1 || KEEP_RE.test(s))) {
|
|
56
|
+
if (KEEP_RE.test(line)) {
|
|
57
|
+
flush();
|
|
58
|
+
out.push(line);
|
|
59
|
+
prevNorm = null;
|
|
60
|
+
continue;
|
|
61
|
+
}
|
|
62
|
+
if (PASS_RE.test(line)) {
|
|
63
|
+
run.push(line);
|
|
64
|
+
prevNorm = null;
|
|
65
|
+
continue;
|
|
66
|
+
}
|
|
67
|
+
flush();
|
|
68
|
+
const norm = normalizeStallSnapshot(line);
|
|
69
|
+
if (line.trim() !== "" && norm.trim() === "")
|
|
70
|
+
continue; // pure spinner/ANSI/elapsed frame
|
|
71
|
+
if (norm !== "" && norm === prevNorm) {
|
|
72
|
+
out[out.length - 1] = line;
|
|
73
|
+
continue;
|
|
74
|
+
} // repaint of the prior line — latest wins
|
|
75
|
+
out.push(line);
|
|
76
|
+
prevNorm = norm;
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
flush();
|
|
80
|
+
return out.join("\n");
|
|
81
|
+
}
|
|
82
|
+
/** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
|
|
83
|
+
* seam exists for fault injection in tests only — production callers pass text alone. */
|
|
84
|
+
export function filterLlmTranscript(text, classify = classifyTranscript) {
|
|
85
|
+
try {
|
|
86
|
+
const filtered = classify(text);
|
|
87
|
+
return text.length - filtered.length < text.length * MIN_SAVINGS_RATIO ? text : filtered;
|
|
88
|
+
}
|
|
89
|
+
catch {
|
|
90
|
+
return text; // fail open — a filter defect must never cost the consult its evidence
|
|
91
|
+
}
|
|
92
|
+
}
|