tickmarkr 1.78.0 → 1.80.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code.d.ts +11 -0
- package/dist/adapters/claude-code.js +105 -16
- package/dist/adapters/kimi.d.ts +6 -0
- package/dist/adapters/kimi.js +88 -26
- package/dist/adapters/registry.js +6 -1
- package/dist/brand.d.ts +4 -0
- package/dist/brand.js +4 -0
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +23 -0
- package/dist/cli/commands/status.d.ts +4 -0
- package/dist/cli/commands/status.js +237 -18
- package/dist/cli/commands/ui.d.ts +1 -1
- package/dist/cli/commands/ui.js +11 -1
- package/dist/drivers/herdr.js +36 -7
- package/dist/drivers/types.d.ts +22 -1
- package/dist/gates/acceptance.d.ts +5 -0
- package/dist/gates/acceptance.js +5 -3
- package/dist/gates/llm.d.ts +9 -0
- package/dist/gates/llm.js +74 -0
- package/dist/gates/review.d.ts +5 -0
- package/dist/gates/review.js +5 -3
- package/dist/run/journal.d.ts +1 -1
- package/dist/tui/cockpit/capture.d.ts +117 -0
- package/dist/tui/cockpit/capture.js +329 -0
- package/dist/tui/cockpit/components.d.ts +179 -0
- package/dist/tui/cockpit/components.js +280 -0
- package/dist/tui/cockpit/demo.d.ts +41 -0
- package/dist/tui/cockpit/demo.js +144 -0
- package/dist/tui/cockpit/derive.d.ts +39 -0
- package/dist/tui/cockpit/derive.js +390 -0
- package/dist/tui/cockpit/layout.d.ts +76 -0
- package/dist/tui/cockpit/layout.js +84 -0
- package/dist/tui/cockpit/run-cockpit.d.ts +19 -0
- package/dist/tui/cockpit/run-cockpit.js +194 -0
- package/dist/tui/cockpit/setup-cockpit.d.ts +41 -0
- package/dist/tui/cockpit/setup-cockpit.js +488 -0
- package/dist/tui/cockpit/theme.d.ts +161 -0
- package/dist/tui/cockpit/theme.js +170 -0
- package/package.json +2 -1
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { BANNER, GLYPHS, dim, fail, legend, ok, rule, statusRow, title, warn } from "../../brand.js";
|
|
2
2
|
import { HerdrDriver } from "../../drivers/herdr.js";
|
|
3
|
-
import { formatOwnedName } from "../../drivers/types.js";
|
|
4
|
-
import { blockedTasks, graphDefinitionHash, loadGraph } from "../../graph/graph.js";
|
|
3
|
+
import { formatOwnedName, } from "../../drivers/types.js";
|
|
4
|
+
import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../../graph/graph.js";
|
|
5
5
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
6
6
|
import { foldActivity } from "../../run/activity.js";
|
|
7
7
|
import { Journal, engagementComparable, isQualityFailureParkKind, recordedTaskFailureKind, runHasEnded, } from "../../run/journal.js";
|
|
@@ -18,6 +18,89 @@ const SPINNER = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇",
|
|
|
18
18
|
const ASCII_SPINNER = ["|", "/", "-", "\\"];
|
|
19
19
|
const SAVE_TERMINAL_TITLE = "\x1b[22;0t";
|
|
20
20
|
const RESTORE_TERMINAL_TITLE = "\x1b[23;0t";
|
|
21
|
+
const decisionEvidence = (stateDir, runId, sequence) => `${stateDir}/runs/${runId}/journal.jsonl#L${sequence}`;
|
|
22
|
+
// v1.79 T4: a deterministic JSONL projection over journal truth. Sequence/evidence come from the
|
|
23
|
+
// append-only line position, timestamps and claims come from the row, and no watcher-local clock or
|
|
24
|
+
// filesystem write participates. Re-reading the same bytes therefore returns the same event bytes.
|
|
25
|
+
export const decisionEventsFromJournal = (events, runId, stateDir = ".tickmarkr") => events.flatMap((event, index) => {
|
|
26
|
+
const sequence = index + 1;
|
|
27
|
+
const base = {
|
|
28
|
+
version: 1,
|
|
29
|
+
sequence,
|
|
30
|
+
ts: event.ts,
|
|
31
|
+
runId,
|
|
32
|
+
evidence: decisionEvidence(stateDir, runId, sequence),
|
|
33
|
+
...(event.taskId ? { taskId: event.taskId } : {}),
|
|
34
|
+
};
|
|
35
|
+
if (event.event === "phase-start" && typeof event.data.phase === "string") {
|
|
36
|
+
return [{ ...base, type: "phase-change", tier: "routine", phase: event.data.phase }];
|
|
37
|
+
}
|
|
38
|
+
if (event.event === "gate-result" && typeof event.data.gate === "string") {
|
|
39
|
+
const verdict = event.data.skipped === true
|
|
40
|
+
? "skipped"
|
|
41
|
+
: event.data.pass === true
|
|
42
|
+
? "passed"
|
|
43
|
+
: event.data.pass === false
|
|
44
|
+
? "failed"
|
|
45
|
+
: "unknown";
|
|
46
|
+
return [{
|
|
47
|
+
...base,
|
|
48
|
+
type: "gate-verdict",
|
|
49
|
+
tier: verdict === "failed" ? "decision" : "routine",
|
|
50
|
+
gate: event.data.gate,
|
|
51
|
+
verdict,
|
|
52
|
+
}];
|
|
53
|
+
}
|
|
54
|
+
if (event.event === "escalation") {
|
|
55
|
+
return [{
|
|
56
|
+
...base,
|
|
57
|
+
type: "escalation",
|
|
58
|
+
tier: "decision",
|
|
59
|
+
...(typeof event.data.step === "string" ? { step: event.data.step } : {}),
|
|
60
|
+
...(typeof event.data.attempt === "number" ? { attempt: event.data.attempt } : {}),
|
|
61
|
+
}];
|
|
62
|
+
}
|
|
63
|
+
if (event.event === "task-human" && event.taskId) {
|
|
64
|
+
return [{
|
|
65
|
+
...base,
|
|
66
|
+
type: "human-decision-required",
|
|
67
|
+
tier: "decision",
|
|
68
|
+
approvalCommand: `tickmarkr approve ${runId} ${event.taskId}`,
|
|
69
|
+
...(typeof event.data.kind === "string" ? { kind: event.data.kind } : {}),
|
|
70
|
+
...(typeof event.data.reason === "string" ? { reason: event.data.reason } : {}),
|
|
71
|
+
}];
|
|
72
|
+
}
|
|
73
|
+
if (event.event === "run-end") {
|
|
74
|
+
return [{
|
|
75
|
+
...base,
|
|
76
|
+
type: "run-end",
|
|
77
|
+
tier: "decision",
|
|
78
|
+
summary: event.data,
|
|
79
|
+
}];
|
|
80
|
+
}
|
|
81
|
+
return [];
|
|
82
|
+
});
|
|
83
|
+
const optionValue = (argv, name) => {
|
|
84
|
+
const inline = argv.find((arg) => arg.startsWith(`${name}=`));
|
|
85
|
+
if (inline)
|
|
86
|
+
return inline.slice(name.length + 1) || undefined;
|
|
87
|
+
const index = argv.indexOf(name);
|
|
88
|
+
const value = index >= 0 ? argv[index + 1] : undefined;
|
|
89
|
+
return value && !value.startsWith("-") ? value : undefined;
|
|
90
|
+
};
|
|
91
|
+
const defaultPostWebhook = (url, event) => fetch(url, {
|
|
92
|
+
method: "POST",
|
|
93
|
+
headers: { "content-type": "application/json" },
|
|
94
|
+
body: JSON.stringify(event),
|
|
95
|
+
}).then(() => undefined);
|
|
96
|
+
const postWebhookFireAndForget = (post, url, event) => {
|
|
97
|
+
try {
|
|
98
|
+
void Promise.resolve(post(url, event)).catch(() => undefined);
|
|
99
|
+
}
|
|
100
|
+
catch {
|
|
101
|
+
// Webhooks are observational. A synchronous bridge failure is as inert as a rejected request.
|
|
102
|
+
}
|
|
103
|
+
};
|
|
21
104
|
const taskPhase = (value) => {
|
|
22
105
|
if (value === "worker" || value === "gates" || value === "judge" || value === "review" || value === "merge")
|
|
23
106
|
return value;
|
|
@@ -204,6 +287,91 @@ const terminalFailureCause = (events) => {
|
|
|
204
287
|
}
|
|
205
288
|
return undefined;
|
|
206
289
|
};
|
|
290
|
+
const tipFailureEvidence = (events, start, end) => {
|
|
291
|
+
const gates = new Set();
|
|
292
|
+
let fingerprints = 0;
|
|
293
|
+
let hasFingerprintCount = false;
|
|
294
|
+
for (let i = start; i <= end; i++) {
|
|
295
|
+
const event = events[i];
|
|
296
|
+
if (event.event !== "tip-verify-failed")
|
|
297
|
+
continue;
|
|
298
|
+
if (typeof event.data.gate === "string" && event.data.gate.trim())
|
|
299
|
+
gates.add(event.data.gate.trim());
|
|
300
|
+
if (Array.isArray(event.data.fingerprints)) {
|
|
301
|
+
fingerprints += event.data.fingerprints.length;
|
|
302
|
+
hasFingerprintCount = true;
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
return { gates: [...gates], ...(hasFingerprintCount ? { fingerprints } : {}) };
|
|
306
|
+
};
|
|
307
|
+
const priorLifecycleIndex = (events, end) => {
|
|
308
|
+
for (let i = end; i >= 0; i--) {
|
|
309
|
+
if (events[i].event === "run-start" || events[i].event === "run-resume")
|
|
310
|
+
return i;
|
|
311
|
+
}
|
|
312
|
+
return 0;
|
|
313
|
+
};
|
|
314
|
+
// OBS-146: tip verification is a run phase, folded exclusively from append-only journal facts.
|
|
315
|
+
// A completed run-end owns passed/failed; a later resume keeps the prior failure visible while
|
|
316
|
+
// recording a re-verification attempt; an active run becomes pending only after a recorded merge
|
|
317
|
+
// and recorded non-empty command set make a tip verification applicable. Unknown evidence stays
|
|
318
|
+
// unknown rather than borrowing optimistic state from graph.json.
|
|
319
|
+
const tipVerifyPhase = (events) => {
|
|
320
|
+
let lastRunEnd = -1;
|
|
321
|
+
let lastLifecycle = -1;
|
|
322
|
+
let attempts = 0;
|
|
323
|
+
let hasCommands = false;
|
|
324
|
+
let hasMerge = false;
|
|
325
|
+
for (let i = 0; i < events.length; i++) {
|
|
326
|
+
const event = events[i];
|
|
327
|
+
if (event.event === "run-start" || event.event === "run-resume") {
|
|
328
|
+
lastLifecycle = i;
|
|
329
|
+
attempts++;
|
|
330
|
+
if (event.event === "run-start" && typeof event.data.commands === "object" && event.data.commands !== null
|
|
331
|
+
&& !Array.isArray(event.data.commands) && Object.keys(event.data.commands).length > 0) {
|
|
332
|
+
hasCommands = true;
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
else if (event.event === "run-end") {
|
|
336
|
+
lastRunEnd = i;
|
|
337
|
+
}
|
|
338
|
+
else if (event.event === "merge") {
|
|
339
|
+
hasMerge = true;
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
if (lastRunEnd >= 0 && events[lastRunEnd].data.tipVerify === "failed") {
|
|
343
|
+
const failureStart = priorLifecycleIndex(events, lastRunEnd);
|
|
344
|
+
const priorFailure = tipFailureEvidence(events, failureStart, lastRunEnd);
|
|
345
|
+
if (lastLifecycle > lastRunEnd) {
|
|
346
|
+
const currentFailure = tipFailureEvidence(events, lastLifecycle, events.length - 1);
|
|
347
|
+
if (currentFailure.gates.length > 0 || currentFailure.fingerprints !== undefined) {
|
|
348
|
+
return { state: "failed", ...currentFailure };
|
|
349
|
+
}
|
|
350
|
+
return { state: "re-verifying", ...priorFailure, attempt: Math.max(2, attempts) };
|
|
351
|
+
}
|
|
352
|
+
return { state: "failed", ...priorFailure };
|
|
353
|
+
}
|
|
354
|
+
if (lastLifecycle > lastRunEnd) {
|
|
355
|
+
const currentFailure = tipFailureEvidence(events, lastLifecycle, events.length - 1);
|
|
356
|
+
if (currentFailure.gates.length > 0 || currentFailure.fingerprints !== undefined) {
|
|
357
|
+
return { state: "failed", ...currentFailure };
|
|
358
|
+
}
|
|
359
|
+
if (hasCommands && hasMerge)
|
|
360
|
+
return { state: "pending", gates: [] };
|
|
361
|
+
}
|
|
362
|
+
return undefined;
|
|
363
|
+
};
|
|
364
|
+
const tipVerifyText = (phase) => {
|
|
365
|
+
if (phase.state === "pending")
|
|
366
|
+
return "tip-verify: pending";
|
|
367
|
+
const gates = phase.gates.length ? phase.gates.join(", ") : "gate unknown";
|
|
368
|
+
const failed = `tip-verify: FAILED (${gates})`;
|
|
369
|
+
const reverify = phase.state === "re-verifying" ? ` → re-verifying (attempt ${phase.attempt})` : "";
|
|
370
|
+
const fingerprints = phase.fingerprints === undefined
|
|
371
|
+
? ""
|
|
372
|
+
: ` · ${phase.fingerprints} fingerprint${phase.fingerprints === 1 ? "" : "s"}`;
|
|
373
|
+
return `${failed}${reverify}${fingerprints}`;
|
|
374
|
+
};
|
|
207
375
|
const liveness = (events, now = Date.now()) => {
|
|
208
376
|
const last = events.at(-1);
|
|
209
377
|
if (!last)
|
|
@@ -289,6 +457,7 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
289
457
|
const width = process.stdout.columns ?? 120;
|
|
290
458
|
const done = effective.tasks.filter((t) => t.status === "done").length;
|
|
291
459
|
const ended = comparable && runHasEnded(events);
|
|
460
|
+
const tipPhase = comparable ? tipVerifyPhase(events) : undefined;
|
|
292
461
|
const taskIds = new Set(g.tasks.map((task) => task.id));
|
|
293
462
|
const phases = comparable ? livePhases(events) : new Map();
|
|
294
463
|
for (const taskId of phases.keys())
|
|
@@ -326,7 +495,7 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
326
495
|
return `${prefix}${shortGoal(t.goal, Math.max(0, width - prefix.length - suffix.length))}${suffix}`;
|
|
327
496
|
});
|
|
328
497
|
const header = runId
|
|
329
|
-
? `tickmarkr status${divider}run ${runId}${supersededBy ? `${divider}superseded by ${supersededBy}` : ""}${!comparable ? `${divider}${NOT_COMPARABLE_NOTICE}` : ""}${divider}${liveness(events, now).replaceAll(" · ", divider)}${divider}${done}/${g.tasks.length} done`
|
|
498
|
+
? `tickmarkr status${divider}run ${runId}${supersededBy ? `${divider}superseded by ${supersededBy}` : ""}${!comparable ? `${divider}${NOT_COMPARABLE_NOTICE}` : ""}${divider}${liveness(events, now).replaceAll(" · ", divider)}${tipPhase ? `${divider}${tipVerifyText(tipPhase)}` : ""}${divider}${tipPhase ? `${done}/${g.tasks.length} tasks done${divider}run not verified` : `${done}/${g.tasks.length} done`}`
|
|
330
499
|
: `tickmarkr status${divider}no runs yet${divider}${done}/${g.tasks.length} done`;
|
|
331
500
|
const legendLine = ` gates: ${GATE_NAMES.map((gate) => `${GATE_KEYS[gate]} ${gate}`).join(divider)}`;
|
|
332
501
|
return {
|
|
@@ -344,18 +513,25 @@ const renderFrame = (cwd, now = Date.now(), animationFrame = 0, workerLiveness =
|
|
|
344
513
|
const anyFailed = cells.some((c) => c.redTier);
|
|
345
514
|
const gaugeCells = 10;
|
|
346
515
|
const fill = g.tasks.length ? Math.round((done / g.tasks.length) * gaugeCells) : 0;
|
|
347
|
-
const
|
|
516
|
+
const tipFailed = tipPhase?.state === "failed";
|
|
517
|
+
const progressTone = anyFailed || tipFailed ? fail : tipPhase ? warn : ok;
|
|
518
|
+
const gauge = (fill ? progressTone("█".repeat(fill)) : "") + (fill < gaugeCells ? dim("░".repeat(gaugeCells - fill)) : "");
|
|
348
519
|
const live = liveness(events, now)
|
|
349
520
|
.replace(/\bdead\b/, fail("dead"))
|
|
350
521
|
.replace(/\bfinished\b/, dim("finished"))
|
|
351
522
|
.replace(/\balive\b/, ok("alive"))
|
|
352
523
|
.replaceAll(" · ", dot);
|
|
353
|
-
const tally =
|
|
524
|
+
const tally = tipPhase
|
|
525
|
+
? `${progressTone(`${done}/${g.tasks.length} tasks done`)}${dot}${progressTone("run not verified")}`
|
|
526
|
+
: `${done}/${g.tasks.length} done`;
|
|
527
|
+
const tipStatus = tipPhase
|
|
528
|
+
? `${tipFailed ? fail(tipVerifyText(tipPhase)) : warn(tipVerifyText(tipPhase))}${dot}`
|
|
529
|
+
: "";
|
|
354
530
|
const header = ` ${title(runId ? `run ${runId}` : "tickmarkr")}${dot}` +
|
|
355
531
|
(runId
|
|
356
|
-
? `${supersededBy ? `${warn(`superseded by ${supersededBy}`)}${dot}` : ""}${!comparable ? `${warn(NOT_COMPARABLE_NOTICE)}${dot}` : ""}${live}${dot}`
|
|
532
|
+
? `${supersededBy ? `${warn(`superseded by ${supersededBy}`)}${dot}` : ""}${!comparable ? `${warn(NOT_COMPARABLE_NOTICE)}${dot}` : ""}${live}${dot}${tipStatus}`
|
|
357
533
|
: `no runs yet${dot}`) +
|
|
358
|
-
`${gauge} ${done === g.tasks.length && g.tasks.length > 0 ? ok(tally) : tally}`;
|
|
534
|
+
`${gauge} ${tipPhase ? tally : done === g.tasks.length && g.tasks.length > 0 ? ok(tally) : tally}`;
|
|
359
535
|
const hr = rule(Math.min(width, 100));
|
|
360
536
|
// OBS-104 run-level now line: names the most recent journal event. TTY frame only — the non-TTY
|
|
361
537
|
// machine surface is byte-pinned (status-brand golden) and must not drift. Rendered BELOW the task
|
|
@@ -417,15 +593,21 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
417
593
|
const { content } = renderFrame(cwd, opts.now?.() ?? Date.now());
|
|
418
594
|
return visual() ? BANNER + content : content;
|
|
419
595
|
}
|
|
596
|
+
const eventStream = argv.some((arg) => arg === "--events" || arg === "--jsonl" || arg === "--decision-events");
|
|
597
|
+
const webhookUrl = opts.webhookUrl
|
|
598
|
+
?? optionValue(argv, "--webhook")
|
|
599
|
+
?? process.env.TICKMARKR_DECISION_WEBHOOK;
|
|
600
|
+
const postWebhook = opts.postWebhook ?? defaultPostWebhook;
|
|
420
601
|
const iterations = opts.iterations ?? Infinity;
|
|
421
602
|
const sleep = opts.sleep ?? defaultSleep;
|
|
422
603
|
const now = opts.now ?? Date.now;
|
|
423
604
|
const bounded = Number.isFinite(iterations);
|
|
424
605
|
const frames = [];
|
|
606
|
+
const eventLines = [];
|
|
425
607
|
const sep = "\n---\n";
|
|
426
|
-
const tty = visual();
|
|
608
|
+
const tty = !eventStream && visual();
|
|
427
609
|
const workerLiveness = new Map();
|
|
428
|
-
const herdr = HerdrDriver.available() ? new HerdrDriver() : undefined;
|
|
610
|
+
const herdr = !eventStream && HerdrDriver.available() ? new HerdrDriver() : undefined;
|
|
429
611
|
const readWorkerOutput = opts.readWorkerOutput ?? (herdr
|
|
430
612
|
? async (taskId, attempt, runId) => {
|
|
431
613
|
const name = formatOwnedName({ role: "worker", taskId, attempt, runId });
|
|
@@ -467,6 +649,30 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
467
649
|
prior.snapshot = snapshot;
|
|
468
650
|
}));
|
|
469
651
|
};
|
|
652
|
+
let decisionRunId;
|
|
653
|
+
let journalCursor = 0;
|
|
654
|
+
const consumeDecisionEvents = () => {
|
|
655
|
+
const runId = Journal.latestRunId(cwd, { withJournal: true });
|
|
656
|
+
if (!runId)
|
|
657
|
+
return [];
|
|
658
|
+
if (decisionRunId !== runId) {
|
|
659
|
+
decisionRunId = runId;
|
|
660
|
+
journalCursor = 0;
|
|
661
|
+
}
|
|
662
|
+
const journalEvents = Journal.open(cwd, runId).read();
|
|
663
|
+
if (journalEvents.length < journalCursor)
|
|
664
|
+
journalCursor = 0;
|
|
665
|
+
const fresh = decisionEventsFromJournal(journalEvents, runId, stateDirName(cwd))
|
|
666
|
+
.filter((event) => event.sequence > journalCursor);
|
|
667
|
+
journalCursor = journalEvents.length;
|
|
668
|
+
if (webhookUrl) {
|
|
669
|
+
for (const event of fresh) {
|
|
670
|
+
if (event.tier === "decision")
|
|
671
|
+
postWebhookFireAndForget(postWebhook, webhookUrl, event);
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
return fresh;
|
|
675
|
+
};
|
|
470
676
|
let titleSaved = false;
|
|
471
677
|
const restoreTitle = () => {
|
|
472
678
|
if (!titleSaved)
|
|
@@ -492,18 +698,31 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
492
698
|
try {
|
|
493
699
|
for (let i = 0; i < iterations; i++) {
|
|
494
700
|
const nowMs = now();
|
|
495
|
-
const
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
701
|
+
const decisionEvents = eventStream || webhookUrl ? consumeDecisionEvents() : [];
|
|
702
|
+
let frame;
|
|
703
|
+
if (eventStream) {
|
|
704
|
+
for (const event of decisionEvents) {
|
|
705
|
+
const line = JSON.stringify(event);
|
|
706
|
+
process.stdout.write(line + "\n");
|
|
707
|
+
if (bounded)
|
|
708
|
+
eventLines.push(line);
|
|
709
|
+
}
|
|
499
710
|
}
|
|
500
711
|
else {
|
|
501
|
-
|
|
712
|
+
frame = renderFrame(cwd, nowMs, i, workerLiveness);
|
|
713
|
+
if (tty) {
|
|
714
|
+
updateTitle(frame.hotPhase, nowMs);
|
|
715
|
+
process.stdout.write(`\x1b[2J\x1b[H${BANNER}${frame.content}\n${legend(` watching · refresh ${REFRESH_MS / 1000}s · ^C to quit`)}`);
|
|
716
|
+
}
|
|
717
|
+
else {
|
|
718
|
+
process.stdout.write(frame.content + sep);
|
|
719
|
+
}
|
|
720
|
+
if (bounded)
|
|
721
|
+
frames.push(frame.content);
|
|
502
722
|
}
|
|
503
|
-
if (bounded)
|
|
504
|
-
frames.push(frame.content);
|
|
505
723
|
if (i + 1 < iterations) {
|
|
506
|
-
|
|
724
|
+
if (frame)
|
|
725
|
+
await observeWorkerOutput(frame.workerPhases, frame.runId, nowMs);
|
|
507
726
|
await sleep(REFRESH_MS);
|
|
508
727
|
}
|
|
509
728
|
}
|
|
@@ -514,5 +733,5 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
514
733
|
restoreTitle();
|
|
515
734
|
}
|
|
516
735
|
}
|
|
517
|
-
return bounded ? frames.join(sep) : "";
|
|
736
|
+
return bounded ? eventStream ? eventLines.join("\n") : frames.join(sep) : "";
|
|
518
737
|
}
|
|
@@ -2,7 +2,7 @@ type StudioIO = {
|
|
|
2
2
|
input: NodeJS.ReadStream;
|
|
3
3
|
output: NodeJS.WriteStream;
|
|
4
4
|
};
|
|
5
|
-
export declare function ui(
|
|
5
|
+
export declare function ui(argv: string[], io?: Partial<StudioIO>): Promise<string | {
|
|
6
6
|
out: string;
|
|
7
7
|
code: number;
|
|
8
8
|
}>;
|
package/dist/cli/commands/ui.js
CHANGED
|
@@ -1,10 +1,20 @@
|
|
|
1
1
|
const NON_TTY_MSG = "tickmarkr ui: studio requires a TTY — use `tickmarkr fleet --print` or `tickmarkr status --watch` for line-mode output";
|
|
2
|
-
export async function ui(
|
|
2
|
+
export async function ui(argv, io = {}) {
|
|
3
3
|
const input = io.input ?? process.stdin;
|
|
4
4
|
const output = io.output ?? process.stdout;
|
|
5
5
|
if (input.isTTY !== true || output.isTTY !== true) {
|
|
6
6
|
return { out: NON_TTY_MSG, code: 1 };
|
|
7
7
|
}
|
|
8
|
+
if (argv.includes("--demo")) {
|
|
9
|
+
const { runCockpitDemo } = await import("../../tui/cockpit/demo.js");
|
|
10
|
+
const { version } = await import("./version.js");
|
|
11
|
+
await runCockpitDemo({
|
|
12
|
+
input,
|
|
13
|
+
output,
|
|
14
|
+
binaryVersion: await version(),
|
|
15
|
+
});
|
|
16
|
+
return "ui: closed";
|
|
17
|
+
}
|
|
8
18
|
const { runStudioInk } = await import("../../tui/ink/studio-app.js");
|
|
9
19
|
await runStudioInk({ input, output });
|
|
10
20
|
return "ui: closed";
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -112,7 +112,14 @@ export class HerdrDriver {
|
|
|
112
112
|
}
|
|
113
113
|
// ponytail: narrow panes hard-wrap the input line — collapse whitespace before comparing.
|
|
114
114
|
deliveryMatches(transcript, cmd) {
|
|
115
|
-
|
|
115
|
+
// OBS-154: a TUI editor re-wraps a long delivery across its own bordered rows, so the pane text
|
|
116
|
+
// carries `│` between fragments we typed as ONE line. Stripping whitespace alone left the needle
|
|
117
|
+
// uncontainable, so the read-back could not recognize its own SUCCESSFUL delivery and the OBS-85
|
|
118
|
+
// guard then refused to retype onto what it had been told was corruption (probe 6b).
|
|
119
|
+
// This is the opposite question to shellExecutionEchoed, which deliberately REJECTS box rows: a
|
|
120
|
+
// shell echo must come from a shell prompt, whereas here we only ask whether our text landed —
|
|
121
|
+
// and inside the editor box is exactly where it is supposed to land.
|
|
122
|
+
const norm = (s) => s.replace(/[│┃|]/g, "").replace(/\s+/g, "");
|
|
116
123
|
const hay = norm(transcript);
|
|
117
124
|
const needle = norm(cmd);
|
|
118
125
|
return needle.length > 0 && hay.includes(needle);
|
|
@@ -149,9 +156,29 @@ export class HerdrDriver {
|
|
|
149
156
|
return true;
|
|
150
157
|
if (promptAt < 0)
|
|
151
158
|
return inputBox !== undefined && matchesEmptyInputBox(transcript, inputBox);
|
|
152
|
-
if (inputBox
|
|
153
|
-
|
|
159
|
+
if (inputBox) {
|
|
160
|
+
// OBS-181: an EMPTY declared input box is positive evidence of submission even while the
|
|
161
|
+
// prompt is still visible above as the echoed user turn — which is exactly how every TUI
|
|
162
|
+
// renders the moment after Enter lands. Test it FIRST: `match` means "the box is painted" and
|
|
163
|
+
// is deliberately true for an empty box, so asking `match` first cannot decide submission.
|
|
164
|
+
if (matchesEmptyInputBox(transcript, inputBox))
|
|
165
|
+
return true;
|
|
166
|
+
// The box is painted and is NOT empty: the prompt is still sitting in it. Only a fingerprint
|
|
167
|
+
// appearing AFTER the prompt — a fresh input line below the submitted text — counts.
|
|
168
|
+
//
|
|
169
|
+
// This is the branch a wedged kimi pane belongs in, and could not reach: its occupied editor
|
|
170
|
+
// renders a blank continuation row, the matcher demanded the bottom border immediately below
|
|
171
|
+
// the prompt, so NEITHER matcher fired and the positional fallback below answered "delivered"
|
|
172
|
+
// for a pane that had processed nothing. The matcher fix (adapters/kimi.ts) is what routes it
|
|
173
|
+
// here; keeping the fallback unreachable for a painted box is what stops the next one.
|
|
174
|
+
if (matchesInputBox(transcript, inputBox)) {
|
|
175
|
+
return hay.lastIndexOf(norm(inputBox.fingerprint)) > promptAt;
|
|
176
|
+
}
|
|
154
177
|
}
|
|
178
|
+
// No declared box, or the box is not on screen at all (it may have scrolled out of the read
|
|
179
|
+
// window). Positional evidence only — weaker, and unsound for any surface that paints chrome
|
|
180
|
+
// below its prompt, which is why a declared box must be able to recognise its own OCCUPIED
|
|
181
|
+
// state. Every adapter declaring an inputBox needs a captured occupied frame in its tests.
|
|
155
182
|
return promptAt + needle.length < hay.length;
|
|
156
183
|
}
|
|
157
184
|
static available() {
|
|
@@ -576,10 +603,12 @@ export class HerdrDriver {
|
|
|
576
603
|
transcript = read.stdout || transcript;
|
|
577
604
|
const waitedMs = this.time.now() - started;
|
|
578
605
|
if (read.code !== 0) {
|
|
579
|
-
// Once at least one valid frame has been observed, a read
|
|
580
|
-
// the
|
|
581
|
-
//
|
|
582
|
-
|
|
606
|
+
// Once at least one valid frame has been observed, a timed-out read IS the bounded window
|
|
607
|
+
// expiring: the read was given exactly the remaining readiness budget as its real timeout,
|
|
608
|
+
// so real time genuinely elapsed even when an injected clock has not advanced (an injected
|
|
609
|
+
// clock never moves during a read, so a wall-clock comparison here can never hold under
|
|
610
|
+
// one). A first-read timeout and every non-timeout read failure remain structural.
|
|
611
|
+
if (read.timedOut && previous !== undefined) {
|
|
583
612
|
throw new DeliveryReadinessError(waitedMs, transcript);
|
|
584
613
|
}
|
|
585
614
|
const detail = read.stderr || read.stdout || `exit ${read.code}`;
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -5,11 +5,32 @@ export interface Slot {
|
|
|
5
5
|
tabId?: string;
|
|
6
6
|
group?: string;
|
|
7
7
|
}
|
|
8
|
-
export type NotifyTier = "routine" | "attention";
|
|
8
|
+
export type NotifyTier = "routine" | "attention" | "decision";
|
|
9
9
|
export interface NotifyOpts {
|
|
10
10
|
tier?: NotifyTier;
|
|
11
11
|
sound?: "none" | "done" | "request";
|
|
12
12
|
}
|
|
13
|
+
export type DecisionEventType = "phase-change" | "gate-verdict" | "escalation" | "human-decision-required" | "run-end";
|
|
14
|
+
export interface DecisionEvent {
|
|
15
|
+
version: 1;
|
|
16
|
+
sequence: number;
|
|
17
|
+
type: DecisionEventType;
|
|
18
|
+
tier: "routine" | "decision";
|
|
19
|
+
ts: string;
|
|
20
|
+
runId: string;
|
|
21
|
+
taskId?: string;
|
|
22
|
+
evidence: string;
|
|
23
|
+
phase?: string;
|
|
24
|
+
gate?: string;
|
|
25
|
+
verdict?: "passed" | "failed" | "skipped" | "unknown";
|
|
26
|
+
step?: string;
|
|
27
|
+
attempt?: number;
|
|
28
|
+
kind?: string;
|
|
29
|
+
reason?: string;
|
|
30
|
+
approvalCommand?: string;
|
|
31
|
+
summary?: Record<string, unknown>;
|
|
32
|
+
}
|
|
33
|
+
export type DecisionWebhookPost = (url: string, event: DecisionEvent) => void | Promise<unknown>;
|
|
13
34
|
export interface SlotOpts {
|
|
14
35
|
group?: string;
|
|
15
36
|
label?: string;
|
|
@@ -14,6 +14,11 @@ export interface JudgeVerdict {
|
|
|
14
14
|
reason: string;
|
|
15
15
|
evidence: string | EvidenceCitation;
|
|
16
16
|
}>;
|
|
17
|
+
comments?: Array<{
|
|
18
|
+
path: string;
|
|
19
|
+
line: number;
|
|
20
|
+
body: string;
|
|
21
|
+
}>;
|
|
17
22
|
}
|
|
18
23
|
export declare function judgeCriterionId(index: number): string;
|
|
19
24
|
export declare function testFiltered(testCmd: string, name: string): string;
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -4,7 +4,7 @@ import { DEFAULT_DIFF_CAP } from "../config/config.js";
|
|
|
4
4
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
5
5
|
import { sh } from "../run/git.js";
|
|
6
6
|
import { checkDiffCap, fetchTaskDiff } from "./review.js";
|
|
7
|
-
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
7
|
+
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
8
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
9
9
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
10
10
|
const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
|
|
@@ -278,9 +278,10 @@ ${citable || "(the diff changes no lines)"}
|
|
|
278
278
|
${verdictNonceLine(nonce)}
|
|
279
279
|
|
|
280
280
|
Respond with ONLY this JSON (no prose before or after):
|
|
281
|
-
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": {"path": "path/to/file", "line": 42}}]}
|
|
281
|
+
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": {"path": "path/to/file", "line": 42}}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
|
|
282
282
|
Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) exactly once.
|
|
283
283
|
Each criteria[].evidence MUST be a structured citation {"path", "line"} whose "path" and "line" appear in the "Citable evidence lines" list above (a new-file line number inside a changed hunk); a citation outside every changed hunk or to an untouched file voids the whole verdict.
|
|
284
|
+
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
284
285
|
`;
|
|
285
286
|
const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
|
|
286
287
|
const extracted = extractVerdictJson(raw, nonce);
|
|
@@ -309,5 +310,6 @@ Each criteria[].evidence MUST be a structured citation {"path", "line"} whose "p
|
|
|
309
310
|
if (!v.pass)
|
|
310
311
|
lines.push("judge verdict pass=false");
|
|
311
312
|
lines.push(...inconsistencies);
|
|
312
|
-
|
|
313
|
+
const prose = warn + detBlock + (lines.join("\n") || "judge passed");
|
|
314
|
+
return { gate: "acceptance", pass, details: appendAnchoredReview(prose, extracted) };
|
|
313
315
|
}
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -4,6 +4,14 @@ export declare const GATE_PANE_SEP = " \u00B7 ";
|
|
|
4
4
|
export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
|
|
5
5
|
/** Fable F3: per-call nonce echoed in verdict JSON and gate exit markers. */
|
|
6
6
|
export declare function generateVerdictNonce(): string;
|
|
7
|
+
export interface AnchoredComment {
|
|
8
|
+
path: string;
|
|
9
|
+
line: number;
|
|
10
|
+
body: string;
|
|
11
|
+
}
|
|
12
|
+
export declare function parseAnchoredComments(verdict: unknown): AnchoredComment[];
|
|
13
|
+
export declare function renderAnchoredReview(verdict: unknown): string;
|
|
14
|
+
export declare function appendAnchoredReview(prose: string, verdict: unknown): string;
|
|
7
15
|
export declare function verdictNonceLine(nonce: string): string;
|
|
8
16
|
export declare function extractPromptNonce(prompt: string): string | null;
|
|
9
17
|
export declare function gateExitTrailer(nonce: string): string;
|
|
@@ -32,6 +40,7 @@ export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
|
|
|
32
40
|
}>;
|
|
33
41
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
34
42
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
|
43
|
+
export declare function dewrapPaneVerdict(out: string, nonce: string): string;
|
|
35
44
|
export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
|
|
36
45
|
export declare function extractJson<T>(raw: string): T | null;
|
|
37
46
|
/** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
|
package/dist/gates/llm.js
CHANGED
|
@@ -27,6 +27,39 @@ When a criterion fails, the verdict MUST name which shortcut above it matches, o
|
|
|
27
27
|
export function generateVerdictNonce() {
|
|
28
28
|
return randomBytes(4).toString("hex");
|
|
29
29
|
}
|
|
30
|
+
// v1.79 T5: comments are an optional side channel on an otherwise authoritative verdict. The whole
|
|
31
|
+
// block is accepted only when every row names one actionable path:line; absent or malformed input
|
|
32
|
+
// becomes no comments and therefore preserves the pre-comments prose bytes and verdict semantics.
|
|
33
|
+
export function parseAnchoredComments(verdict) {
|
|
34
|
+
if (!verdict || typeof verdict !== "object")
|
|
35
|
+
return [];
|
|
36
|
+
const raw = verdict.comments;
|
|
37
|
+
if (!Array.isArray(raw) || raw.length === 0)
|
|
38
|
+
return [];
|
|
39
|
+
const comments = [];
|
|
40
|
+
for (const row of raw) {
|
|
41
|
+
if (!row || typeof row !== "object")
|
|
42
|
+
return [];
|
|
43
|
+
const { path, line, body } = row;
|
|
44
|
+
const cleanPath = typeof path === "string" ? path.trim() : "";
|
|
45
|
+
const cleanBody = typeof body === "string" ? body.trim() : "";
|
|
46
|
+
if (!cleanPath || /[\r\n]/.test(cleanPath) || !Number.isInteger(line) || line < 1 || !cleanBody) {
|
|
47
|
+
return [];
|
|
48
|
+
}
|
|
49
|
+
comments.push({ path: cleanPath, line: line, body: cleanBody });
|
|
50
|
+
}
|
|
51
|
+
return comments;
|
|
52
|
+
}
|
|
53
|
+
export function renderAnchoredReview(verdict) {
|
|
54
|
+
const comments = parseAnchoredComments(verdict);
|
|
55
|
+
if (comments.length === 0)
|
|
56
|
+
return "";
|
|
57
|
+
return `## Anchored review\n${comments.map((c) => `- ${c.path}:${c.line} — ${c.body}`).join("\n")}`;
|
|
58
|
+
}
|
|
59
|
+
export function appendAnchoredReview(prose, verdict) {
|
|
60
|
+
const block = renderAnchoredReview(verdict);
|
|
61
|
+
return block ? `${prose}\n\n${block}` : prose;
|
|
62
|
+
}
|
|
30
63
|
export function verdictNonceLine(nonce) {
|
|
31
64
|
return `VERDICT_NONCE: ${nonce}`;
|
|
32
65
|
}
|
|
@@ -128,6 +161,47 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
128
161
|
if (!via.keep)
|
|
129
162
|
await via.driver.close(slot);
|
|
130
163
|
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
164
|
+
return dewrapPaneVerdict(out, nonce);
|
|
165
|
+
}
|
|
166
|
+
// OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
|
|
167
|
+
// continuation indent, splitting words mid-token — so literal newlines land inside JSON string
|
|
168
|
+
// literals and ZERO lines begin with `{`. `--source recent-unwrapped` cannot undo it: the wrap is
|
|
169
|
+
// the TUI's own rendering, not a terminal soft wrap (driver.read already requests that source, and
|
|
170
|
+
// the captured k3 transcript arrived wrapped anyway). A perfect verdict was therefore scored a
|
|
171
|
+
// flake and rescued by a retry on another channel — 70 such judge-retries are on record across
|
|
172
|
+
// k3, fable AND sol, so this is a width-dependent flake generator under every pane-mode gate.
|
|
173
|
+
//
|
|
174
|
+
// Fail-closed by construction, per the overseer's constraints: reconstruction is bounded to the
|
|
175
|
+
// brace-delimited region, the rejoined text must PARSE, and it must carry THIS call's nonce.
|
|
176
|
+
// Anything else returns the bytes untouched, so an unparseable verdict stays a failure. The
|
|
177
|
+
// reconstruction is APPENDED, never substituted: extractJson takes the last balanced object, so
|
|
178
|
+
// the good copy wins while the original rendering survives verbatim for the evidence capture.
|
|
179
|
+
export function dewrapPaneVerdict(out, nonce) {
|
|
180
|
+
if (!out.includes(nonce))
|
|
181
|
+
return out;
|
|
182
|
+
const lines = out.split("\n");
|
|
183
|
+
const start = lines.findIndex((line) => /^\s*(?:[•*-]\s+)?\{/.test(line));
|
|
184
|
+
if (start < 0)
|
|
185
|
+
return out;
|
|
186
|
+
for (let end = start; end < lines.length; end++) {
|
|
187
|
+
const joined = lines
|
|
188
|
+
.slice(start, end + 1)
|
|
189
|
+
.map((line, i) => (i === 0 ? line.replace(/^\s*(?:[•*-]\s+)?/, "") : line.replace(/^\s+/, "")))
|
|
190
|
+
.join("")
|
|
191
|
+
.trimEnd();
|
|
192
|
+
if (!joined.endsWith("}"))
|
|
193
|
+
continue;
|
|
194
|
+
try {
|
|
195
|
+
const parsed = JSON.parse(joined);
|
|
196
|
+
// the nonce IS the acceptance test — never reconstruct a verdict this call did not ask for
|
|
197
|
+
if (parsed && typeof parsed === "object" && parsed.nonce === nonce) {
|
|
198
|
+
return `${out}\n${joined}`;
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
catch {
|
|
202
|
+
/* not yet a complete object — keep extending within the bounded region */
|
|
203
|
+
}
|
|
204
|
+
}
|
|
131
205
|
return out;
|
|
132
206
|
}
|
|
133
207
|
export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -14,6 +14,11 @@ export interface ReviewVerdict {
|
|
|
14
14
|
approve?: boolean;
|
|
15
15
|
issues?: string[];
|
|
16
16
|
findings?: ReviewFinding[];
|
|
17
|
+
comments?: Array<{
|
|
18
|
+
path: string;
|
|
19
|
+
line: number;
|
|
20
|
+
body: string;
|
|
21
|
+
}>;
|
|
17
22
|
}
|
|
18
23
|
export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
|
|
19
24
|
full: string;
|