tickmarkr 1.78.0 → 1.80.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/adapters/claude-code.d.ts +11 -0
  2. package/dist/adapters/claude-code.js +105 -16
  3. package/dist/adapters/kimi.d.ts +6 -0
  4. package/dist/adapters/kimi.js +88 -26
  5. package/dist/adapters/registry.js +6 -1
  6. package/dist/brand.d.ts +4 -0
  7. package/dist/brand.js +4 -0
  8. package/dist/cli/commands/doctor.d.ts +2 -0
  9. package/dist/cli/commands/doctor.js +23 -0
  10. package/dist/cli/commands/status.d.ts +4 -0
  11. package/dist/cli/commands/status.js +237 -18
  12. package/dist/cli/commands/ui.d.ts +1 -1
  13. package/dist/cli/commands/ui.js +11 -1
  14. package/dist/drivers/herdr.js +36 -7
  15. package/dist/drivers/types.d.ts +22 -1
  16. package/dist/gates/acceptance.d.ts +5 -0
  17. package/dist/gates/acceptance.js +5 -3
  18. package/dist/gates/llm.d.ts +9 -0
  19. package/dist/gates/llm.js +74 -0
  20. package/dist/gates/review.d.ts +5 -0
  21. package/dist/gates/review.js +5 -3
  22. package/dist/run/journal.d.ts +1 -1
  23. package/dist/tui/cockpit/capture.d.ts +117 -0
  24. package/dist/tui/cockpit/capture.js +329 -0
  25. package/dist/tui/cockpit/components.d.ts +179 -0
  26. package/dist/tui/cockpit/components.js +280 -0
  27. package/dist/tui/cockpit/demo.d.ts +41 -0
  28. package/dist/tui/cockpit/demo.js +144 -0
  29. package/dist/tui/cockpit/derive.d.ts +39 -0
  30. package/dist/tui/cockpit/derive.js +390 -0
  31. package/dist/tui/cockpit/layout.d.ts +76 -0
  32. package/dist/tui/cockpit/layout.js +84 -0
  33. package/dist/tui/cockpit/run-cockpit.d.ts +19 -0
  34. package/dist/tui/cockpit/run-cockpit.js +194 -0
  35. package/dist/tui/cockpit/setup-cockpit.d.ts +41 -0
  36. package/dist/tui/cockpit/setup-cockpit.js +488 -0
  37. package/dist/tui/cockpit/theme.d.ts +161 -0
  38. package/dist/tui/cockpit/theme.js +170 -0
  39. package/package.json +2 -1
@@ -1,7 +1,7 @@
1
1
  import { BANNER, GLYPHS, dim, fail, legend, ok, rule, statusRow, title, warn } from "../../brand.js";
2
2
  import { HerdrDriver } from "../../drivers/herdr.js";
3
- import { formatOwnedName } from "../../drivers/types.js";
4
- import { blockedTasks, graphDefinitionHash, loadGraph } from "../../graph/graph.js";
3
+ import { formatOwnedName, } from "../../drivers/types.js";
4
+ import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../../graph/graph.js";
5
5
  import { GATE_NAMES } from "../../graph/schema.js";
6
6
  import { foldActivity } from "../../run/activity.js";
7
7
  import { Journal, engagementComparable, isQualityFailureParkKind, recordedTaskFailureKind, runHasEnded, } from "../../run/journal.js";
@@ -18,6 +18,89 @@ const SPINNER = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇",
18
18
  const ASCII_SPINNER = ["|", "/", "-", "\\"];
19
19
  const SAVE_TERMINAL_TITLE = "\x1b[22;0t";
20
20
  const RESTORE_TERMINAL_TITLE = "\x1b[23;0t";
21
+ const decisionEvidence = (stateDir, runId, sequence) => `${stateDir}/runs/${runId}/journal.jsonl#L${sequence}`;
22
+ // v1.79 T4: a deterministic JSONL projection over journal truth. Sequence/evidence come from the
23
+ // append-only line position, timestamps and claims come from the row, and no watcher-local clock or
24
+ // filesystem write participates. Re-reading the same bytes therefore returns the same event bytes.
25
+ export const decisionEventsFromJournal = (events, runId, stateDir = ".tickmarkr") => events.flatMap((event, index) => {
26
+ const sequence = index + 1;
27
+ const base = {
28
+ version: 1,
29
+ sequence,
30
+ ts: event.ts,
31
+ runId,
32
+ evidence: decisionEvidence(stateDir, runId, sequence),
33
+ ...(event.taskId ? { taskId: event.taskId } : {}),
34
+ };
35
+ if (event.event === "phase-start" && typeof event.data.phase === "string") {
36
+ return [{ ...base, type: "phase-change", tier: "routine", phase: event.data.phase }];
37
+ }
38
+ if (event.event === "gate-result" && typeof event.data.gate === "string") {
39
+ const verdict = event.data.skipped === true
40
+ ? "skipped"
41
+ : event.data.pass === true
42
+ ? "passed"
43
+ : event.data.pass === false
44
+ ? "failed"
45
+ : "unknown";
46
+ return [{
47
+ ...base,
48
+ type: "gate-verdict",
49
+ tier: verdict === "failed" ? "decision" : "routine",
50
+ gate: event.data.gate,
51
+ verdict,
52
+ }];
53
+ }
54
+ if (event.event === "escalation") {
55
+ return [{
56
+ ...base,
57
+ type: "escalation",
58
+ tier: "decision",
59
+ ...(typeof event.data.step === "string" ? { step: event.data.step } : {}),
60
+ ...(typeof event.data.attempt === "number" ? { attempt: event.data.attempt } : {}),
61
+ }];
62
+ }
63
+ if (event.event === "task-human" && event.taskId) {
64
+ return [{
65
+ ...base,
66
+ type: "human-decision-required",
67
+ tier: "decision",
68
+ approvalCommand: `tickmarkr approve ${runId} ${event.taskId}`,
69
+ ...(typeof event.data.kind === "string" ? { kind: event.data.kind } : {}),
70
+ ...(typeof event.data.reason === "string" ? { reason: event.data.reason } : {}),
71
+ }];
72
+ }
73
+ if (event.event === "run-end") {
74
+ return [{
75
+ ...base,
76
+ type: "run-end",
77
+ tier: "decision",
78
+ summary: event.data,
79
+ }];
80
+ }
81
+ return [];
82
+ });
83
+ const optionValue = (argv, name) => {
84
+ const inline = argv.find((arg) => arg.startsWith(`${name}=`));
85
+ if (inline)
86
+ return inline.slice(name.length + 1) || undefined;
87
+ const index = argv.indexOf(name);
88
+ const value = index >= 0 ? argv[index + 1] : undefined;
89
+ return value && !value.startsWith("-") ? value : undefined;
90
+ };
91
+ const defaultPostWebhook = (url, event) => fetch(url, {
92
+ method: "POST",
93
+ headers: { "content-type": "application/json" },
94
+ body: JSON.stringify(event),
95
+ }).then(() => undefined);
96
+ const postWebhookFireAndForget = (post, url, event) => {
97
+ try {
98
+ void Promise.resolve(post(url, event)).catch(() => undefined);
99
+ }
100
+ catch {
101
+ // Webhooks are observational. A synchronous bridge failure is as inert as a rejected request.
102
+ }
103
+ };
21
104
  const taskPhase = (value) => {
22
105
  if (value === "worker" || value === "gates" || value === "judge" || value === "review" || value === "merge")
23
106
  return value;
@@ -204,6 +287,91 @@ const terminalFailureCause = (events) => {
204
287
  }
205
288
  return undefined;
206
289
  };
290
+ const tipFailureEvidence = (events, start, end) => {
291
+ const gates = new Set();
292
+ let fingerprints = 0;
293
+ let hasFingerprintCount = false;
294
+ for (let i = start; i <= end; i++) {
295
+ const event = events[i];
296
+ if (event.event !== "tip-verify-failed")
297
+ continue;
298
+ if (typeof event.data.gate === "string" && event.data.gate.trim())
299
+ gates.add(event.data.gate.trim());
300
+ if (Array.isArray(event.data.fingerprints)) {
301
+ fingerprints += event.data.fingerprints.length;
302
+ hasFingerprintCount = true;
303
+ }
304
+ }
305
+ return { gates: [...gates], ...(hasFingerprintCount ? { fingerprints } : {}) };
306
+ };
307
+ const priorLifecycleIndex = (events, end) => {
308
+ for (let i = end; i >= 0; i--) {
309
+ if (events[i].event === "run-start" || events[i].event === "run-resume")
310
+ return i;
311
+ }
312
+ return 0;
313
+ };
314
+ // OBS-146: tip verification is a run phase, folded exclusively from append-only journal facts.
315
+ // A completed run-end owns passed/failed; a later resume keeps the prior failure visible while
316
+ // recording a re-verification attempt; an active run becomes pending only after a recorded merge
317
+ // and recorded non-empty command set make a tip verification applicable. Unknown evidence stays
318
+ // unknown rather than borrowing optimistic state from graph.json.
319
+ const tipVerifyPhase = (events) => {
320
+ let lastRunEnd = -1;
321
+ let lastLifecycle = -1;
322
+ let attempts = 0;
323
+ let hasCommands = false;
324
+ let hasMerge = false;
325
+ for (let i = 0; i < events.length; i++) {
326
+ const event = events[i];
327
+ if (event.event === "run-start" || event.event === "run-resume") {
328
+ lastLifecycle = i;
329
+ attempts++;
330
+ if (event.event === "run-start" && typeof event.data.commands === "object" && event.data.commands !== null
331
+ && !Array.isArray(event.data.commands) && Object.keys(event.data.commands).length > 0) {
332
+ hasCommands = true;
333
+ }
334
+ }
335
+ else if (event.event === "run-end") {
336
+ lastRunEnd = i;
337
+ }
338
+ else if (event.event === "merge") {
339
+ hasMerge = true;
340
+ }
341
+ }
342
+ if (lastRunEnd >= 0 && events[lastRunEnd].data.tipVerify === "failed") {
343
+ const failureStart = priorLifecycleIndex(events, lastRunEnd);
344
+ const priorFailure = tipFailureEvidence(events, failureStart, lastRunEnd);
345
+ if (lastLifecycle > lastRunEnd) {
346
+ const currentFailure = tipFailureEvidence(events, lastLifecycle, events.length - 1);
347
+ if (currentFailure.gates.length > 0 || currentFailure.fingerprints !== undefined) {
348
+ return { state: "failed", ...currentFailure };
349
+ }
350
+ return { state: "re-verifying", ...priorFailure, attempt: Math.max(2, attempts) };
351
+ }
352
+ return { state: "failed", ...priorFailure };
353
+ }
354
+ if (lastLifecycle > lastRunEnd) {
355
+ const currentFailure = tipFailureEvidence(events, lastLifecycle, events.length - 1);
356
+ if (currentFailure.gates.length > 0 || currentFailure.fingerprints !== undefined) {
357
+ return { state: "failed", ...currentFailure };
358
+ }
359
+ if (hasCommands && hasMerge)
360
+ return { state: "pending", gates: [] };
361
+ }
362
+ return undefined;
363
+ };
364
+ const tipVerifyText = (phase) => {
365
+ if (phase.state === "pending")
366
+ return "tip-verify: pending";
367
+ const gates = phase.gates.length ? phase.gates.join(", ") : "gate unknown";
368
+ const failed = `tip-verify: FAILED (${gates})`;
369
+ const reverify = phase.state === "re-verifying" ? ` → re-verifying (attempt ${phase.attempt})` : "";
370
+ const fingerprints = phase.fingerprints === undefined
371
+ ? ""
372
+ : ` · ${phase.fingerprints} fingerprint${phase.fingerprints === 1 ? "" : "s"}`;
373
+ return `${failed}${reverify}${fingerprints}`;
374
+ };
207
375
  const liveness = (events, now = Date.now()) => {
208
376
  const last = events.at(-1);
209
377
  if (!last)
@@ -289,6 +457,7 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
289
457
  const width = process.stdout.columns ?? 120;
290
458
  const done = effective.tasks.filter((t) => t.status === "done").length;
291
459
  const ended = comparable && runHasEnded(events);
460
+ const tipPhase = comparable ? tipVerifyPhase(events) : undefined;
292
461
  const taskIds = new Set(g.tasks.map((task) => task.id));
293
462
  const phases = comparable ? livePhases(events) : new Map();
294
463
  for (const taskId of phases.keys())
@@ -326,7 +495,7 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
326
495
  return `${prefix}${shortGoal(t.goal, Math.max(0, width - prefix.length - suffix.length))}${suffix}`;
327
496
  });
328
497
  const header = runId
329
- ? `tickmarkr status${divider}run ${runId}${supersededBy ? `${divider}superseded by ${supersededBy}` : ""}${!comparable ? `${divider}${NOT_COMPARABLE_NOTICE}` : ""}${divider}${liveness(events, now).replaceAll(" · ", divider)}${divider}${done}/${g.tasks.length} done`
498
+ ? `tickmarkr status${divider}run ${runId}${supersededBy ? `${divider}superseded by ${supersededBy}` : ""}${!comparable ? `${divider}${NOT_COMPARABLE_NOTICE}` : ""}${divider}${liveness(events, now).replaceAll(" · ", divider)}${tipPhase ? `${divider}${tipVerifyText(tipPhase)}` : ""}${divider}${tipPhase ? `${done}/${g.tasks.length} tasks done${divider}run not verified` : `${done}/${g.tasks.length} done`}`
330
499
  : `tickmarkr status${divider}no runs yet${divider}${done}/${g.tasks.length} done`;
331
500
  const legendLine = ` gates: ${GATE_NAMES.map((gate) => `${GATE_KEYS[gate]} ${gate}`).join(divider)}`;
332
501
  return {
@@ -344,18 +513,25 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
344
513
  const anyFailed = cells.some((c) => c.redTier);
345
514
  const gaugeCells = 10;
346
515
  const fill = g.tasks.length ? Math.round((done / g.tasks.length) * gaugeCells) : 0;
347
- const gauge = (fill ? (anyFailed ? fail : ok)("█".repeat(fill)) : "") + (fill < gaugeCells ? dim("░".repeat(gaugeCells - fill)) : "");
516
+ const tipFailed = tipPhase?.state === "failed";
517
+ const progressTone = anyFailed || tipFailed ? fail : tipPhase ? warn : ok;
518
+ const gauge = (fill ? progressTone("█".repeat(fill)) : "") + (fill < gaugeCells ? dim("░".repeat(gaugeCells - fill)) : "");
348
519
  const live = liveness(events, now)
349
520
  .replace(/\bdead\b/, fail("dead"))
350
521
  .replace(/\bfinished\b/, dim("finished"))
351
522
  .replace(/\balive\b/, ok("alive"))
352
523
  .replaceAll(" · ", dot);
353
- const tally = `${done}/${g.tasks.length} done`;
524
+ const tally = tipPhase
525
+ ? `${progressTone(`${done}/${g.tasks.length} tasks done`)}${dot}${progressTone("run not verified")}`
526
+ : `${done}/${g.tasks.length} done`;
527
+ const tipStatus = tipPhase
528
+ ? `${tipFailed ? fail(tipVerifyText(tipPhase)) : warn(tipVerifyText(tipPhase))}${dot}`
529
+ : "";
354
530
  const header = ` ${title(runId ? `run ${runId}` : "tickmarkr")}${dot}` +
355
531
  (runId
356
- ? `${supersededBy ? `${warn(`superseded by ${supersededBy}`)}${dot}` : ""}${!comparable ? `${warn(NOT_COMPARABLE_NOTICE)}${dot}` : ""}${live}${dot}`
532
+ ? `${supersededBy ? `${warn(`superseded by ${supersededBy}`)}${dot}` : ""}${!comparable ? `${warn(NOT_COMPARABLE_NOTICE)}${dot}` : ""}${live}${dot}${tipStatus}`
357
533
  : `no runs yet${dot}`) +
358
- `${gauge} ${done === g.tasks.length && g.tasks.length > 0 ? ok(tally) : tally}`;
534
+ `${gauge} ${tipPhase ? tally : done === g.tasks.length && g.tasks.length > 0 ? ok(tally) : tally}`;
359
535
  const hr = rule(Math.min(width, 100));
360
536
  // OBS-104 run-level now line: names the most recent journal event. TTY frame only — the non-TTY
361
537
  // machine surface is byte-pinned (status-brand golden) and must not drift. Rendered BELOW the task
@@ -417,15 +593,21 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
417
593
  const { content } = renderFrame(cwd, opts.now?.() ?? Date.now());
418
594
  return visual() ? BANNER + content : content;
419
595
  }
596
+ const eventStream = argv.some((arg) => arg === "--events" || arg === "--jsonl" || arg === "--decision-events");
597
+ const webhookUrl = opts.webhookUrl
598
+ ?? optionValue(argv, "--webhook")
599
+ ?? process.env.TICKMARKR_DECISION_WEBHOOK;
600
+ const postWebhook = opts.postWebhook ?? defaultPostWebhook;
420
601
  const iterations = opts.iterations ?? Infinity;
421
602
  const sleep = opts.sleep ?? defaultSleep;
422
603
  const now = opts.now ?? Date.now;
423
604
  const bounded = Number.isFinite(iterations);
424
605
  const frames = [];
606
+ const eventLines = [];
425
607
  const sep = "\n---\n";
426
- const tty = visual();
608
+ const tty = !eventStream && visual();
427
609
  const workerLiveness = new Map();
428
- const herdr = HerdrDriver.available() ? new HerdrDriver() : undefined;
610
+ const herdr = !eventStream && HerdrDriver.available() ? new HerdrDriver() : undefined;
429
611
  const readWorkerOutput = opts.readWorkerOutput ?? (herdr
430
612
  ? async (taskId, attempt, runId) => {
431
613
  const name = formatOwnedName({ role: "worker", taskId, attempt, runId });
@@ -467,6 +649,30 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
467
649
  prior.snapshot = snapshot;
468
650
  }));
469
651
  };
652
+ let decisionRunId;
653
+ let journalCursor = 0;
654
+ const consumeDecisionEvents = () => {
655
+ const runId = Journal.latestRunId(cwd, { withJournal: true });
656
+ if (!runId)
657
+ return [];
658
+ if (decisionRunId !== runId) {
659
+ decisionRunId = runId;
660
+ journalCursor = 0;
661
+ }
662
+ const journalEvents = Journal.open(cwd, runId).read();
663
+ if (journalEvents.length < journalCursor)
664
+ journalCursor = 0;
665
+ const fresh = decisionEventsFromJournal(journalEvents, runId, stateDirName(cwd))
666
+ .filter((event) => event.sequence > journalCursor);
667
+ journalCursor = journalEvents.length;
668
+ if (webhookUrl) {
669
+ for (const event of fresh) {
670
+ if (event.tier === "decision")
671
+ postWebhookFireAndForget(postWebhook, webhookUrl, event);
672
+ }
673
+ }
674
+ return fresh;
675
+ };
470
676
  let titleSaved = false;
471
677
  const restoreTitle = () => {
472
678
  if (!titleSaved)
@@ -492,18 +698,31 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
492
698
  try {
493
699
  for (let i = 0; i < iterations; i++) {
494
700
  const nowMs = now();
495
- const frame = renderFrame(cwd, nowMs, i, workerLiveness);
496
- if (tty) {
497
- updateTitle(frame.hotPhase, nowMs);
498
- process.stdout.write(`\x1b[2J\x1b[H${BANNER}${frame.content}\n${legend(` watching · refresh ${REFRESH_MS / 1000}s · ^C to quit`)}`);
701
+ const decisionEvents = eventStream || webhookUrl ? consumeDecisionEvents() : [];
702
+ let frame;
703
+ if (eventStream) {
704
+ for (const event of decisionEvents) {
705
+ const line = JSON.stringify(event);
706
+ process.stdout.write(line + "\n");
707
+ if (bounded)
708
+ eventLines.push(line);
709
+ }
499
710
  }
500
711
  else {
501
- process.stdout.write(frame.content + sep);
712
+ frame = renderFrame(cwd, nowMs, i, workerLiveness);
713
+ if (tty) {
714
+ updateTitle(frame.hotPhase, nowMs);
715
+ process.stdout.write(`\x1b[2J\x1b[H${BANNER}${frame.content}\n${legend(` watching · refresh ${REFRESH_MS / 1000}s · ^C to quit`)}`);
716
+ }
717
+ else {
718
+ process.stdout.write(frame.content + sep);
719
+ }
720
+ if (bounded)
721
+ frames.push(frame.content);
502
722
  }
503
- if (bounded)
504
- frames.push(frame.content);
505
723
  if (i + 1 < iterations) {
506
- await observeWorkerOutput(frame.workerPhases, frame.runId, nowMs);
724
+ if (frame)
725
+ await observeWorkerOutput(frame.workerPhases, frame.runId, nowMs);
507
726
  await sleep(REFRESH_MS);
508
727
  }
509
728
  }
@@ -514,5 +733,5 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
514
733
  restoreTitle();
515
734
  }
516
735
  }
517
- return bounded ? frames.join(sep) : "";
736
+ return bounded ? eventStream ? eventLines.join("\n") : frames.join(sep) : "";
518
737
  }
@@ -2,7 +2,7 @@ type StudioIO = {
2
2
  input: NodeJS.ReadStream;
3
3
  output: NodeJS.WriteStream;
4
4
  };
5
- export declare function ui(_argv: string[], io?: Partial<StudioIO>): Promise<string | {
5
+ export declare function ui(argv: string[], io?: Partial<StudioIO>): Promise<string | {
6
6
  out: string;
7
7
  code: number;
8
8
  }>;
@@ -1,10 +1,20 @@
1
1
  const NON_TTY_MSG = "tickmarkr ui: studio requires a TTY — use `tickmarkr fleet --print` or `tickmarkr status --watch` for line-mode output";
2
- export async function ui(_argv, io = {}) {
2
+ export async function ui(argv, io = {}) {
3
3
  const input = io.input ?? process.stdin;
4
4
  const output = io.output ?? process.stdout;
5
5
  if (input.isTTY !== true || output.isTTY !== true) {
6
6
  return { out: NON_TTY_MSG, code: 1 };
7
7
  }
8
+ if (argv.includes("--demo")) {
9
+ const { runCockpitDemo } = await import("../../tui/cockpit/demo.js");
10
+ const { version } = await import("./version.js");
11
+ await runCockpitDemo({
12
+ input,
13
+ output,
14
+ binaryVersion: await version(),
15
+ });
16
+ return "ui: closed";
17
+ }
8
18
  const { runStudioInk } = await import("../../tui/ink/studio-app.js");
9
19
  await runStudioInk({ input, output });
10
20
  return "ui: closed";
@@ -112,7 +112,14 @@ export class HerdrDriver {
112
112
  }
113
113
  // ponytail: narrow panes hard-wrap the input line — collapse whitespace before comparing.
114
114
  deliveryMatches(transcript, cmd) {
115
- const norm = (s) => s.replace(/\s+/g, "");
115
+ // OBS-154: a TUI editor re-wraps a long delivery across its own bordered rows, so the pane text
116
+ // carries `│` between fragments we typed as ONE line. Stripping whitespace alone left the needle
117
+ // uncontainable, so the read-back could not recognize its own SUCCESSFUL delivery and the OBS-85
118
+ // guard then refused to retype onto what it had been told was corruption (probe 6b).
119
+ // This is the opposite question to shellExecutionEchoed, which deliberately REJECTS box rows: a
120
+ // shell echo must come from a shell prompt, whereas here we only ask whether our text landed —
121
+ // and inside the editor box is exactly where it is supposed to land.
122
+ const norm = (s) => s.replace(/[│┃|]/g, "").replace(/\s+/g, "");
116
123
  const hay = norm(transcript);
117
124
  const needle = norm(cmd);
118
125
  return needle.length > 0 && hay.includes(needle);
@@ -149,9 +156,29 @@ export class HerdrDriver {
149
156
  return true;
150
157
  if (promptAt < 0)
151
158
  return inputBox !== undefined && matchesEmptyInputBox(transcript, inputBox);
152
- if (inputBox && matchesInputBox(transcript, inputBox)) {
153
- return hay.lastIndexOf(norm(inputBox.fingerprint)) > promptAt;
159
+ if (inputBox) {
160
+ // OBS-181: an EMPTY declared input box is positive evidence of submission even while the
161
+ // prompt is still visible above as the echoed user turn — which is exactly how every TUI
162
+ // renders the moment after Enter lands. Test it FIRST: `match` means "the box is painted" and
163
+ // is deliberately true for an empty box, so asking `match` first cannot decide submission.
164
+ if (matchesEmptyInputBox(transcript, inputBox))
165
+ return true;
166
+ // The box is painted and is NOT empty: the prompt is still sitting in it. Only a fingerprint
167
+ // appearing AFTER the prompt — a fresh input line below the submitted text — counts.
168
+ //
169
+ // This is the branch a wedged kimi pane belongs in, and could not reach: its occupied editor
170
+ // renders a blank continuation row, the matcher demanded the bottom border immediately below
171
+ // the prompt, so NEITHER matcher fired and the positional fallback below answered "delivered"
172
+ // for a pane that had processed nothing. The matcher fix (adapters/kimi.ts) is what routes it
173
+ // here; keeping the fallback unreachable for a painted box is what stops the next one.
174
+ if (matchesInputBox(transcript, inputBox)) {
175
+ return hay.lastIndexOf(norm(inputBox.fingerprint)) > promptAt;
176
+ }
154
177
  }
178
+ // No declared box, or the box is not on screen at all (it may have scrolled out of the read
179
+ // window). Positional evidence only — weaker, and unsound for any surface that paints chrome
180
+ // below its prompt, which is why a declared box must be able to recognise its own OCCUPIED
181
+ // state. Every adapter declaring an inputBox needs a captured occupied frame in its tests.
155
182
  return promptAt + needle.length < hay.length;
156
183
  }
157
184
  static available() {
@@ -576,10 +603,12 @@ export class HerdrDriver {
576
603
  transcript = read.stdout || transcript;
577
604
  const waitedMs = this.time.now() - started;
578
605
  if (read.code !== 0) {
579
- // Once at least one valid frame has been observed, a read that consumes the remainder of
580
- // the readiness budget is the bounded window expiring, not a new submission/protocol
581
- // identity. A first-read timeout and every non-timeout read failure remain structural.
582
- if (read.timedOut && previous !== undefined && waitedMs >= timeoutMs) {
606
+ // Once at least one valid frame has been observed, a timed-out read IS the bounded window
607
+ // expiring: the read was given exactly the remaining readiness budget as its real timeout,
608
+ // so real time genuinely elapsed even when an injected clock has not advanced (an injected
609
+ // clock never moves during a read, so a wall-clock comparison here can never hold under
610
+ // one). A first-read timeout and every non-timeout read failure remain structural.
611
+ if (read.timedOut && previous !== undefined) {
583
612
  throw new DeliveryReadinessError(waitedMs, transcript);
584
613
  }
585
614
  const detail = read.stderr || read.stdout || `exit ${read.code}`;
@@ -5,11 +5,32 @@ export interface Slot {
5
5
  tabId?: string;
6
6
  group?: string;
7
7
  }
8
- export type NotifyTier = "routine" | "attention";
8
+ export type NotifyTier = "routine" | "attention" | "decision";
9
9
  export interface NotifyOpts {
10
10
  tier?: NotifyTier;
11
11
  sound?: "none" | "done" | "request";
12
12
  }
13
+ export type DecisionEventType = "phase-change" | "gate-verdict" | "escalation" | "human-decision-required" | "run-end";
14
+ export interface DecisionEvent {
15
+ version: 1;
16
+ sequence: number;
17
+ type: DecisionEventType;
18
+ tier: "routine" | "decision";
19
+ ts: string;
20
+ runId: string;
21
+ taskId?: string;
22
+ evidence: string;
23
+ phase?: string;
24
+ gate?: string;
25
+ verdict?: "passed" | "failed" | "skipped" | "unknown";
26
+ step?: string;
27
+ attempt?: number;
28
+ kind?: string;
29
+ reason?: string;
30
+ approvalCommand?: string;
31
+ summary?: Record<string, unknown>;
32
+ }
33
+ export type DecisionWebhookPost = (url: string, event: DecisionEvent) => void | Promise<unknown>;
13
34
  export interface SlotOpts {
14
35
  group?: string;
15
36
  label?: string;
@@ -14,6 +14,11 @@ export interface JudgeVerdict {
14
14
  reason: string;
15
15
  evidence: string | EvidenceCitation;
16
16
  }>;
17
+ comments?: Array<{
18
+ path: string;
19
+ line: number;
20
+ body: string;
21
+ }>;
17
22
  }
18
23
  export declare function judgeCriterionId(index: number): string;
19
24
  export declare function testFiltered(testCmd: string, name: string): string;
@@ -4,7 +4,7 @@ import { DEFAULT_DIFF_CAP } from "../config/config.js";
4
4
  import { renderAcceptanceItem } from "../graph/schema.js";
5
5
  import { sh } from "../run/git.js";
6
6
  import { checkDiffCap, fetchTaskDiff } from "./review.js";
7
- import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
7
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
8
8
  // Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
9
9
  const JUDGE_TIMEOUT_MS = 900_000;
10
10
  const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
@@ -278,9 +278,10 @@ ${citable || "(the diff changes no lines)"}
278
278
  ${verdictNonceLine(nonce)}
279
279
 
280
280
  Respond with ONLY this JSON (no prose before or after):
281
- {"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": {"path": "path/to/file", "line": 42}}]}
281
+ {"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": {"path": "path/to/file", "line": 42}}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
282
282
  Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) exactly once.
283
283
  Each criteria[].evidence MUST be a structured citation {"path", "line"} whose "path" and "line" appear in the "Citable evidence lines" list above (a new-file line number inside a changed hunk); a citation outside every changed hunk or to an untouched file voids the whole verdict.
284
+ The top-level comments array is optional. Use it only for actionable line-anchored feedback.
284
285
  `;
285
286
  const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
286
287
  const extracted = extractVerdictJson(raw, nonce);
@@ -309,5 +310,6 @@ Each criteria[].evidence MUST be a structured citation {"path", "line"} whose "p
309
310
  if (!v.pass)
310
311
  lines.push("judge verdict pass=false");
311
312
  lines.push(...inconsistencies);
312
- return { gate: "acceptance", pass, details: warn + detBlock + (lines.join("\n") || "judge passed") };
313
+ const prose = warn + detBlock + (lines.join("\n") || "judge passed");
314
+ return { gate: "acceptance", pass, details: appendAnchoredReview(prose, extracted) };
313
315
  }
@@ -4,6 +4,14 @@ export declare const GATE_PANE_SEP = " \u00B7 ";
4
4
  export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
5
5
  /** Fable F3: per-call nonce echoed in verdict JSON and gate exit markers. */
6
6
  export declare function generateVerdictNonce(): string;
7
+ export interface AnchoredComment {
8
+ path: string;
9
+ line: number;
10
+ body: string;
11
+ }
12
+ export declare function parseAnchoredComments(verdict: unknown): AnchoredComment[];
13
+ export declare function renderAnchoredReview(verdict: unknown): string;
14
+ export declare function appendAnchoredReview(prose: string, verdict: unknown): string;
7
15
  export declare function verdictNonceLine(nonce: string): string;
8
16
  export declare function extractPromptNonce(prompt: string): string | null;
9
17
  export declare function gateExitTrailer(nonce: string): string;
@@ -32,6 +40,7 @@ export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
32
40
  }>;
33
41
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
34
42
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
43
+ export declare function dewrapPaneVerdict(out: string, nonce: string): string;
35
44
  export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
36
45
  export declare function extractJson<T>(raw: string): T | null;
37
46
  /** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
package/dist/gates/llm.js CHANGED
@@ -27,6 +27,39 @@ When a criterion fails, the verdict MUST name which shortcut above it matches, o
27
27
  export function generateVerdictNonce() {
28
28
  return randomBytes(4).toString("hex");
29
29
  }
30
+ // v1.79 T5: comments are an optional side channel on an otherwise authoritative verdict. The whole
31
+ // block is accepted only when every row names one actionable path:line; absent or malformed input
32
+ // becomes no comments and therefore preserves the pre-comments prose bytes and verdict semantics.
33
+ export function parseAnchoredComments(verdict) {
34
+ if (!verdict || typeof verdict !== "object")
35
+ return [];
36
+ const raw = verdict.comments;
37
+ if (!Array.isArray(raw) || raw.length === 0)
38
+ return [];
39
+ const comments = [];
40
+ for (const row of raw) {
41
+ if (!row || typeof row !== "object")
42
+ return [];
43
+ const { path, line, body } = row;
44
+ const cleanPath = typeof path === "string" ? path.trim() : "";
45
+ const cleanBody = typeof body === "string" ? body.trim() : "";
46
+ if (!cleanPath || /[\r\n]/.test(cleanPath) || !Number.isInteger(line) || line < 1 || !cleanBody) {
47
+ return [];
48
+ }
49
+ comments.push({ path: cleanPath, line: line, body: cleanBody });
50
+ }
51
+ return comments;
52
+ }
53
+ export function renderAnchoredReview(verdict) {
54
+ const comments = parseAnchoredComments(verdict);
55
+ if (comments.length === 0)
56
+ return "";
57
+ return `## Anchored review\n${comments.map((c) => `- ${c.path}:${c.line} — ${c.body}`).join("\n")}`;
58
+ }
59
+ export function appendAnchoredReview(prose, verdict) {
60
+ const block = renderAnchoredReview(verdict);
61
+ return block ? `${prose}\n\n${block}` : prose;
62
+ }
30
63
  export function verdictNonceLine(nonce) {
31
64
  return `VERDICT_NONCE: ${nonce}`;
32
65
  }
@@ -128,6 +161,47 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
128
161
  if (!via.keep)
129
162
  await via.driver.close(slot);
130
163
  out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
164
+ return dewrapPaneVerdict(out, nonce);
165
+ }
166
+ // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
167
+ // continuation indent, splitting words mid-token — so literal newlines land inside JSON string
168
+ // literals and ZERO lines begin with `{`. `--source recent-unwrapped` cannot undo it: the wrap is
169
+ // the TUI's own rendering, not a terminal soft wrap (driver.read already requests that source, and
170
+ // the captured k3 transcript arrived wrapped anyway). A perfect verdict was therefore scored a
171
+ // flake and rescued by a retry on another channel — 70 such judge-retries are on record across
172
+ // k3, fable AND sol, so this is a width-dependent flake generator under every pane-mode gate.
173
+ //
174
+ // Fail-closed by construction, per the overseer's constraints: reconstruction is bounded to the
175
+ // brace-delimited region, the rejoined text must PARSE, and it must carry THIS call's nonce.
176
+ // Anything else returns the bytes untouched, so an unparseable verdict stays a failure. The
177
+ // reconstruction is APPENDED, never substituted: extractJson takes the last balanced object, so
178
+ // the good copy wins while the original rendering survives verbatim for the evidence capture.
179
+ export function dewrapPaneVerdict(out, nonce) {
180
+ if (!out.includes(nonce))
181
+ return out;
182
+ const lines = out.split("\n");
183
+ const start = lines.findIndex((line) => /^\s*(?:[•*-]\s+)?\{/.test(line));
184
+ if (start < 0)
185
+ return out;
186
+ for (let end = start; end < lines.length; end++) {
187
+ const joined = lines
188
+ .slice(start, end + 1)
189
+ .map((line, i) => (i === 0 ? line.replace(/^\s*(?:[•*-]\s+)?/, "") : line.replace(/^\s+/, "")))
190
+ .join("")
191
+ .trimEnd();
192
+ if (!joined.endsWith("}"))
193
+ continue;
194
+ try {
195
+ const parsed = JSON.parse(joined);
196
+ // the nonce IS the acceptance test — never reconstruct a verdict this call did not ask for
197
+ if (parsed && typeof parsed === "object" && parsed.nonce === nonce) {
198
+ return `${out}\n${joined}`;
199
+ }
200
+ }
201
+ catch {
202
+ /* not yet a complete object — keep extending within the bounded region */
203
+ }
204
+ }
131
205
  return out;
132
206
  }
133
207
  export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
@@ -14,6 +14,11 @@ export interface ReviewVerdict {
14
14
  approve?: boolean;
15
15
  issues?: string[];
16
16
  findings?: ReviewFinding[];
17
+ comments?: Array<{
18
+ path: string;
19
+ line: number;
20
+ body: string;
21
+ }>;
17
22
  }
18
23
  export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
19
24
  full: string;