tickmarkr 1.67.0 → 1.69.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/dist/adapters/kimi.d.ts +25 -1
  2. package/dist/adapters/kimi.js +80 -0
  3. package/dist/adapters/types.d.ts +6 -0
  4. package/dist/cli/commands/eval.d.ts +4 -0
  5. package/dist/cli/commands/eval.js +26 -0
  6. package/dist/cli/commands/status.d.ts +20 -0
  7. package/dist/cli/commands/status.js +9 -9
  8. package/dist/cli/index.d.ts +1 -1
  9. package/dist/cli/index.js +3 -1
  10. package/dist/eval/canary.d.ts +46 -0
  11. package/dist/eval/canary.js +113 -0
  12. package/dist/eval/dispatch.d.ts +31 -0
  13. package/dist/eval/dispatch.js +207 -0
  14. package/dist/eval/fixtures.d.ts +22 -0
  15. package/dist/eval/fixtures.js +85 -0
  16. package/dist/eval/report.d.ts +35 -0
  17. package/dist/eval/report.js +82 -0
  18. package/dist/eval/selfcheck.d.ts +22 -0
  19. package/dist/eval/selfcheck.js +177 -0
  20. package/dist/run/daemon.js +130 -83
  21. package/dist/run/git.d.ts +2 -0
  22. package/dist/run/git.js +9 -0
  23. package/dist/run/interactive-seed.d.ts +15 -0
  24. package/dist/run/interactive-seed.js +25 -0
  25. package/dist/tui/app.d.ts +3 -0
  26. package/dist/tui/app.js +14 -2
  27. package/dist/tui/views/consult-dossier.d.ts +49 -0
  28. package/dist/tui/views/consult-dossier.js +169 -0
  29. package/dist/tui/views/runs-view.d.ts +70 -0
  30. package/dist/tui/views/runs-view.js +387 -0
  31. package/fixtures/eval/canary/solution/a.txt +1 -0
  32. package/fixtures/eval/canary/spec.md +8 -0
  33. package/fixtures/eval/canary/start/a.txt +1 -0
  34. package/fixtures/eval/sample/solution/a.txt +1 -0
  35. package/fixtures/eval/sample/spec.md +8 -0
  36. package/fixtures/eval/sample/start/a.txt +1 -0
  37. package/fixtures/gsd-sample/07-live-check/07-01-PLAN.md +42 -0
  38. package/fixtures/gsd-sample/07-live-check/07-02-PLAN.md +21 -0
  39. package/fixtures/gsd-sample/07-live-check/07-03-PLAN.md +18 -0
  40. package/fixtures/gsd-sample/07-live-check/07-03-SUMMARY.md +1 -0
  41. package/fixtures/missing-mandatory-gate.native.md +10 -0
  42. package/fixtures/sample-pin.prd.md +18 -0
  43. package/fixtures/sample.native.md +35 -0
  44. package/fixtures/sample.prd.md +22 -0
  45. package/fixtures/speckit-sample/tasks.md +20 -0
  46. package/package.json +3 -2
@@ -4,7 +4,7 @@ import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync
4
4
  import { tmpdir } from "node:os";
5
5
  import { join } from "node:path";
6
6
  import { stringify } from "yaml";
7
- import { classifyDeadChannel, trailerPattern, writePrompt } from "../adapters/prompt.js";
7
+ import { classifyDeadChannel, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TRAILER_SUMMARY, writePrompt } from "../adapters/prompt.js";
8
8
  import { allAdapters, discoverChannels, getAdapter, probeAll, readDoctor } from "../adapters/registry.js";
9
9
  import { addUsage, channelKey, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
10
10
  import { bannerShell, paneDispatchCommand } from "../brand.js";
@@ -16,6 +16,7 @@ import { runGates } from "../gates/run-gates.js";
16
16
  import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus } from "../graph/graph.js";
17
17
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
18
18
  import { cleanupRunWorktrees, gitHead, linkNodeModules, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
19
+ import { runInteractiveSeed } from "./interactive-seed.js";
19
20
  import { classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId } from "./journal.js";
20
21
  import { acquireRunLock, releaseRunLock } from "./lock.js";
21
22
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
@@ -601,17 +602,22 @@ export async function runDaemon(repoRoot, opts = {}) {
601
602
  : cfg.visibility.worker === "interactive" && driver.interactive
602
603
  ? adapter.interactiveCommand(promptFile, assignment.model)
603
604
  : null;
604
- if (cfg.visibility.worker === "interactive" && icmd === null && !modeFallbackNoted) {
605
+ // v1.69 T6: adapters that declare interactiveSeed launch the real TUI and inject the prompt as a
606
+ // user turn; they do NOT need the argv-seeding surface that interactiveCommand represents.
607
+ const hasSeed = retryMode !== "resume" && cfg.visibility.worker === "interactive" && driver.interactive && !!adapter.interactiveSeed;
608
+ if (cfg.visibility.worker === "interactive" && icmd === null && !hasSeed && !modeFallbackNoted) {
605
609
  modeFallbackNoted = true;
606
610
  journal.append("worker-mode-fallback", t.id, { reason: driver.interactive ? "adapter" : "driver" });
607
611
  }
608
- const interactive = icmd !== null;
612
+ const interactive = icmd !== null || hasSeed;
609
613
  // OBS-85 (v1.62 T1): both dispatch branches deliver ONE short script invocation — banner,
610
614
  // adapter command, and nonce exit marker live in a per-attempt script beside the prompt
611
615
  // artifact (the same paneDispatchCommand pattern judge/review/consult dispatches use). The
612
616
  // delivered pane line carries no command substitution and no trailing shell text, so paste
613
617
  // timing can never interleave a `$(…)` with what follows it (the codex corruption class).
614
- const workerCmd = interactive ? icmd : adapter.invoke(t, wt, assignment, { promptFile }).command;
618
+ const workerCmd = interactive
619
+ ? (hasSeed ? ":" : icmd)
620
+ : adapter.invoke(t, wt, assignment, { promptFile }).command;
615
621
  const dispatchScript = promptFile.replace(/\.md$/, ".sh");
616
622
  writeFileSync(dispatchScript, [
617
623
  "export BASH_SILENCE_DEPRECATION_WARNING=1",
@@ -658,95 +664,136 @@ export async function runDaemon(repoRoot, opts = {}) {
658
664
  let output;
659
665
  let exitCode;
660
666
  let timedOut = false;
667
+ let settleParsed;
668
+ let seedResult;
661
669
  if (interactive) {
662
670
  // v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
663
671
  // The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
664
- await driver.run(slot, paneDispatchCommand(dispatchScript));
665
- let paged = false;
666
- // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
667
- // Any other blocked/idle dialog still pages the operator (paged latch below).
668
- let trustAnswered = false;
669
672
  finished = false;
670
673
  exitCode = null;
671
- output = await driver.read(slot, 1000);
672
- // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
673
- // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
674
- const stallWindowMs = taskTimeoutMinutes * 60_000;
675
- // OBS-82: the stall clock compares NORMALIZED snapshots so a spinner glyph/elapsed-time
676
- // repaint is silence, not activity. ONLY this inactivity compare sees normalized text —
677
- // trailer detection, harvest, paging, and quota checks all read the raw pane.
678
- let lastStallSnapshot = normalizeStallSnapshot(output);
679
- let lastOutputAt = Date.now();
680
- while (Date.now() - lastOutputAt < stallWindowMs) {
681
- const sliceStart = Date.now();
682
- const remaining = stallWindowMs - (sliceStart - lastOutputAt);
683
- const slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
684
- if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
685
- // verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
686
- // own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
687
- // parseable trailer or a digit-suffixed exit marker in the harvest is completion.
688
- output = await driver.read(slot, 1000); // TUI transcripts carry chrome — read deeper than print's 500
674
+ if (adapter.interactiveSeed) {
675
+ // v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
676
+ // then fall through to the normal trailer harvest. A failed seed is recorded as a finished
677
+ // failure rather than allowed to race the trailer wait.
678
+ seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
679
+ output = seedResult.output;
680
+ }
681
+ else {
682
+ await driver.run(slot, paneDispatchCommand(dispatchScript));
683
+ output = await driver.read(slot, 1000);
684
+ }
685
+ if (seedResult?.seedFailed) {
686
+ finished = false;
687
+ }
688
+ else {
689
+ let paged = false;
690
+ // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
691
+ // Any other blocked/idle dialog still pages the operator (paged latch below).
692
+ let trustAnswered = false;
693
+ finished = false;
694
+ exitCode = null;
695
+ // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
696
+ // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
697
+ const stallWindowMs = taskTimeoutMinutes * 60_000;
698
+ // OBS-82: the stall clock compares NORMALIZED snapshots so a spinner glyph/elapsed-time
699
+ // repaint is silence, not activity. ONLY this inactivity compare sees normalized text —
700
+ // trailer detection, harvest, paging, and quota checks all read the raw pane.
701
+ let lastStallSnapshot = normalizeStallSnapshot(output);
702
+ let lastOutputAt = Date.now();
703
+ while (Date.now() - lastOutputAt < stallWindowMs) {
704
+ const sliceStart = Date.now();
705
+ const remaining = stallWindowMs - (sliceStart - lastOutputAt);
706
+ const slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
707
+ if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
708
+ // verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
709
+ // own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
710
+ // parseable trailer or a digit-suffixed exit marker in the harvest is completion.
711
+ output = await driver.read(slot, 1000); // TUI transcripts carry chrome — read deeper than print's 500
712
+ finished = new RegExp(trailerPattern(nonce)).test(output);
713
+ const exit = exitRe.exec(output);
714
+ if (finished || exit) {
715
+ exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
716
+ await sampleContext(); // final poll-seam sample before leaving the wait
717
+ break;
718
+ }
719
+ }
720
+ const currentStallSnapshot = normalizeStallSnapshot(await driver.read(slot, 1000));
721
+ if (currentStallSnapshot !== lastStallSnapshot) {
722
+ lastStallSnapshot = currentStallSnapshot;
723
+ lastOutputAt = Date.now();
724
+ }
725
+ // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
726
+ await sampleContext();
727
+ // page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
728
+ // (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
729
+ // a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
730
+ const st = paged ? "" : await driver.status(slot);
731
+ if (!paged && (st === "blocked" || st === "idle")) {
732
+ // T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
733
+ // text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
734
+ if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
735
+ try {
736
+ const paneText = await driver.read(slot, 80);
737
+ if (matchesTrustDialog(paneText, adapter.trustDialog)) {
738
+ trustAnswered = true;
739
+ // v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
740
+ // Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
741
+ journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
742
+ await driver.sendKey(slot, adapter.trustDialog.key);
743
+ const spent = Date.now() - sliceStart;
744
+ if (spent < slice)
745
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
746
+ continue; // do not page — keep waiting for the trailer
747
+ }
748
+ }
749
+ catch {
750
+ /* read/send failed — fall through to page the operator */
751
+ }
752
+ }
753
+ paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
754
+ const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
755
+ await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
756
+ }
757
+ // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
758
+ const spent = Date.now() - sliceStart;
759
+ if (spent < slice)
760
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
761
+ }
762
+ if (!finished && exitCode === null) {
763
+ // timed out (or only ever saw false positives): harvest whatever the pane holds now
764
+ timedOut = Date.now() - lastOutputAt >= stallWindowMs;
765
+ output = await driver.read(slot, 1000);
689
766
  finished = new RegExp(trailerPattern(nonce)).test(output);
690
767
  const exit = exitRe.exec(output);
691
- if (finished || exit) {
692
- exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
693
- await sampleContext(); // final poll-seam sample before leaving the wait
694
- break;
695
- }
768
+ exitCode = exit ? Number(exit[1]) : null;
696
769
  }
697
- const currentStallSnapshot = normalizeStallSnapshot(await driver.read(slot, 1000));
698
- if (currentStallSnapshot !== lastStallSnapshot) {
699
- lastStallSnapshot = currentStallSnapshot;
700
- lastOutputAt = Date.now();
770
+ if (finished) {
771
+ await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
772
+ output = await driver.read(slot, 1000);
701
773
  }
702
- // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
703
- await sampleContext();
704
- // page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
705
- // (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
706
- // a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
707
- const st = paged ? "" : await driver.status(slot);
708
- if (!paged && (st === "blocked" || st === "idle")) {
709
- // T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
710
- // text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
711
- if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
712
- try {
713
- const paneText = await driver.read(slot, 80);
714
- if (matchesTrustDialog(paneText, adapter.trustDialog)) {
715
- trustAnswered = true;
716
- // v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
717
- // Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
718
- journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
719
- await driver.sendKey(slot, adapter.trustDialog.key);
720
- const spent = Date.now() - sliceStart;
721
- if (spent < slice)
722
- await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
723
- continue; // do not page — keep waiting for the trailer
724
- }
725
- }
726
- catch {
727
- /* read/send failed — fall through to page the operator */
728
- }
774
+ // T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
775
+ // contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
776
+ // the existing pane-read seam once or twice before recording a malformed-trailer cause.
777
+ if (interactive) {
778
+ const stallWindowMs = taskTimeoutMinutes * 60_000;
779
+ const settleDeadline = attemptStart + stallWindowMs;
780
+ const settleDelayMs = 1_000;
781
+ const maxSettleRetries = 2;
782
+ let settleTries = 0;
783
+ settleParsed = adapter.parse(output, nonce);
784
+ while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
785
+ const remaining = settleDeadline - Date.now();
786
+ if (remaining <= 0)
787
+ break;
788
+ await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
789
+ output = await driver.read(slot, 1000);
790
+ settleParsed = adapter.parse(output, nonce);
791
+ settleTries++;
792
+ }
793
+ if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
794
+ finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
729
795
  }
730
- paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
731
- const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
732
- await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
733
796
  }
734
- // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
735
- const spent = Date.now() - sliceStart;
736
- if (spent < slice)
737
- await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
738
- }
739
- if (!finished && exitCode === null) {
740
- // timed out (or only ever saw false positives): harvest whatever the pane holds now
741
- timedOut = Date.now() - lastOutputAt >= stallWindowMs;
742
- output = await driver.read(slot, 1000);
743
- finished = new RegExp(trailerPattern(nonce)).test(output);
744
- const exit = exitRe.exec(output);
745
- exitCode = exit ? Number(exit[1]) : null;
746
- }
747
- if (finished) {
748
- await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
749
- output = await driver.read(slot, 1000);
750
797
  }
751
798
  }
752
799
  else {
@@ -800,7 +847,7 @@ export async function runDaemon(repoRoot, opts = {}) {
800
847
  tokens = addUsage(tokens, attemptUsage);
801
848
  metered++;
802
849
  }
803
- const result = adapter.parse(output, nonce);
850
+ const result = settleParsed ?? adapter.parse(output, nonce);
804
851
  const cause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut });
805
852
  journal.append("worker-result", t.id, {
806
853
  ok: result.ok, summary: result.summary, deviations: result.deviations, finished, exitCode,
package/dist/run/git.d.ts CHANGED
@@ -1,5 +1,7 @@
1
1
  import { ROUTING_ENV_SEAMS } from "../route/router.js";
2
2
  export { ROUTING_ENV_SEAMS };
3
+ export declare const FORK_CAP_ENV = "VITEST_MAX_FORKS";
4
+ export declare const DEFAULT_FORK_CAP = "6";
3
5
  export interface ShResult {
4
6
  code: number;
5
7
  stdout: string;
package/dist/run/git.js CHANGED
@@ -5,6 +5,12 @@ import { shq } from "../adapters/types.js";
5
5
  import { tickmarkrDir } from "../graph/graph.js";
6
6
  import { ROUTING_ENV_SEAMS } from "../route/router.js";
7
7
  export { ROUTING_ENV_SEAMS };
8
+ // OBS-110: gate/baseline/tip-verify children are vitest suites; without a fork cap, concurrent
9
+ // full-suite gate runs fork a worker-per-core pool per worktree and saturate the operator box.
10
+ // Default to a modest cap through the environment so vitest honors it natively; never pass it
11
+ // as argv (OBS-55) so child test oracles stay intact. The operator's own export wins.
12
+ export const FORK_CAP_ENV = "VITEST_MAX_FORKS";
13
+ export const DEFAULT_FORK_CAP = "6";
8
14
  // stdin "ignore": same class as HARD-05 / SubprocessDriver — never leave an open pipe a child can block on
9
15
  // (pi -p / codex exec wait for stdin EOF). timedOut distinguishes SIGKILL-timeout from a real nonzero exit.
10
16
  function shell(cmd, cwd, timeoutMs, login) {
@@ -15,6 +21,9 @@ function shell(cmd, cwd, timeoutMs, login) {
15
21
  const env = { ...process.env };
16
22
  for (const k of ROUTING_ENV_SEAMS)
17
23
  delete env[k];
24
+ // OBS-110: apply the default fork cap only when the operator has not already set one.
25
+ if (!(FORK_CAP_ENV in env))
26
+ env[FORK_CAP_ENV] = DEFAULT_FORK_CAP;
18
27
  return new Promise((resolve) => {
19
28
  // detached: bash gets its own process group so a timeout can kill the whole tree —
20
29
  // SIGKILLing bash alone orphans grandchildren (codex/pi) that hold the stdio pipes
@@ -0,0 +1,15 @@
1
+ import type { Assignment, WorkerAdapter } from "../adapters/types.js";
2
+ import type { ExecutorDriver, Slot } from "../drivers/types.js";
3
+ export interface InteractiveSeedResult {
4
+ output: string;
5
+ seedFailed: boolean;
6
+ seedError?: string;
7
+ }
8
+ export declare function runInteractiveSeed(opts: {
9
+ driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
10
+ slot: Slot;
11
+ adapter: WorkerAdapter;
12
+ assignment: Assignment;
13
+ promptFile: string;
14
+ taskTimeoutMinutes: number;
15
+ }): Promise<InteractiveSeedResult>;
@@ -0,0 +1,25 @@
1
+ // v1.69 T6: launch-then-seed handoff for adapters whose real TUI cannot be argv-seeded.
2
+ // Both the launch command and the seed line are delivered through the driver's existing `run`
3
+ // primitive (pane-run on herdr). After the seed line is injected we read the pane back and
4
+ // treat a seed that is still sitting in the input box as a hard failure (OBS-105 discipline).
5
+ export async function runInteractiveSeed(opts) {
6
+ const seed = opts.adapter.interactiveSeed;
7
+ await opts.driver.run(opts.slot, seed.launch(opts.assignment.model));
8
+ const ready = await opts.driver.waitOutput(opts.slot, seed.readinessMatch, opts.taskTimeoutMinutes * 60_000);
9
+ if (!ready) {
10
+ const output = await opts.driver.read(opts.slot, 1000);
11
+ return { output, seedFailed: true, seedError: `readiness pattern not seen: ${seed.readinessMatch}` };
12
+ }
13
+ const seedText = seed.seedLine(opts.promptFile);
14
+ await opts.driver.run(opts.slot, seedText);
15
+ let output = "";
16
+ for (let attempt = 0; attempt < 5; attempt++) {
17
+ await new Promise((r) => setTimeout(r, 200));
18
+ output = await opts.driver.read(opts.slot, 1000);
19
+ const bottom = output.trimEnd().split("\n").pop() ?? "";
20
+ if (!bottom.includes(seedText)) {
21
+ return { output, seedFailed: false };
22
+ }
23
+ }
24
+ return { output, seedFailed: true, seedError: "seed line never left the input box" };
25
+ }
package/dist/tui/app.d.ts CHANGED
@@ -9,6 +9,8 @@ export type View = {
9
9
  cols: number;
10
10
  rows: number;
11
11
  }): string[];
12
+ /** Optional per-view key handler (e.g. up/down cursor movement). */
13
+ key?(name: string): void;
12
14
  };
13
15
  export type StudioOptions = {
14
16
  input: InputStream;
@@ -59,6 +61,7 @@ export declare class StudioApp {
59
61
  private closeSave;
60
62
  private handleRevert;
61
63
  private setView;
64
+ private forwardViewKey;
62
65
  private showNotice;
63
66
  private paint;
64
67
  private buildFrame;
package/dist/tui/app.js CHANGED
@@ -3,6 +3,7 @@ import { createFleetView } from "./views/fleet-view.js";
3
3
  import { createPreviewView } from "./views/preview-view.js";
4
4
  import { createProfileView } from "./views/profile-view.js";
5
5
  import { createRoutingView } from "./views/routing-view.js";
6
+ import { createRunsView } from "./views/runs-view.js";
6
7
  import { renderDiffModal } from "./views/diff-modal.js";
7
8
  import { buildSaveProposal, confirmSave } from "./save.js";
8
9
  import { FleetStaging } from "./staging.js";
@@ -42,7 +43,7 @@ export class StudioApp {
42
43
  this.repoRoot = opts.repoRoot ?? process.cwd();
43
44
  this.globalDir = opts.globalDir;
44
45
  this.staging = new FleetStaging(opts.loaded ?? emptyFleetEditable());
45
- this.views = [createFleetView(), createRoutingView(), createPreviewView(), createProfileView()];
46
+ this.views = [createFleetView(), createRoutingView(), createPreviewView(), createProfileView(), createRunsView()];
46
47
  this.exited = new Promise((resolve) => {
47
48
  this.resolveExit = resolve;
48
49
  });
@@ -91,6 +92,13 @@ export class StudioApp {
91
92
  this.setView(idx);
92
93
  });
93
94
  }
95
+ // per-view navigation
96
+ for (const name of ["up", "down"]) {
97
+ this.engine.key(name, () => {
98
+ if (!this.isModalOpen)
99
+ this.forwardViewKey(name);
100
+ });
101
+ }
94
102
  // help overlay
95
103
  this.engine.key("?", () => {
96
104
  if (!this.confirmingQuit && !this.showingSave) {
@@ -250,6 +258,10 @@ export class StudioApp {
250
258
  this.notice = null;
251
259
  this.paint();
252
260
  }
261
+ forwardViewKey(name) {
262
+ this.views[this.active]?.key?.(name);
263
+ this.paint();
264
+ }
253
265
  showNotice() {
254
266
  this.notice = READONLY_NOTICE;
255
267
  this.paint();
@@ -305,7 +317,7 @@ export class StudioApp {
305
317
  renderHelp() {
306
318
  return [
307
319
  "Key bindings",
308
- "1-4 switch view",
320
+ "1-5 switch view",
309
321
  "tab cycle views",
310
322
  "? show this help",
311
323
  "esc / q quit",
@@ -0,0 +1,49 @@
1
+ import type { ConsultVerdict } from "../../run/consult.js";
2
+ import type { JournalEvent } from "../../run/journal.js";
3
+ /** Placeholder marker surfaced in the Runs detail panel until T2 fills the viewer. */
4
+ export declare const DOSSIER_PLACEHOLDER = "consult dossier viewer \u2014 stub, filled by T2";
5
+ /** Placeholder renderer — returns one dim line. */
6
+ export declare function renderDossierPlaceholder(): string;
7
+ export type ConsultAction = ConsultVerdict["action"];
8
+ /** One recorded consult verdict, with its persisted prompt content injected by the caller. */
9
+ export interface DossierVerdict {
10
+ action: ConsultAction;
11
+ reason?: string;
12
+ notes?: string;
13
+ guidance?: string;
14
+ /** Persisted prompt content (the consults/<task>-<n>.md artifact), loaded by the caller. */
15
+ prompt?: string;
16
+ }
17
+ /** Everything the dossier panel renders — loaded by the caller, never by the render path. */
18
+ export interface ConsultDossierData {
19
+ taskId: string;
20
+ /** Verdicts in dispatch order (journal order). */
21
+ verdicts: DossierVerdict[];
22
+ }
23
+ export interface ConsultDossierPanel {
24
+ /** Change the selected task. Any selection change invalidates every cached prompt render. */
25
+ select(data: ConsultDossierData | undefined): void;
26
+ /** "up"/"down" move the verdict cursor; "enter" toggles the selected verdict's dossier. */
27
+ key(name: string): void;
28
+ /** Render the panel body lines (chrome/divider lines are the host view's job). */
29
+ render(): string[];
30
+ /** Index of the expanded verdict, or null when collapsed. */
31
+ readonly expandedIndex: number | null;
32
+ }
33
+ /** Fold one task's consult-verdict journal events into dossier verdicts, in dispatch order.
34
+ * Pure function over injected events. opts.prompts carries the persisted prompt contents in
35
+ * the same dispatch order (the caller reads the consults/ artifacts); a verdict whose artifact
36
+ * was not loaded simply has no prompt. Malformed events are skipped, never fatal. */
37
+ export declare function foldConsultVerdicts(events: JournalEvent[], taskId: string, opts?: {
38
+ prompts?: string[];
39
+ }): DossierVerdict[];
40
+ /** Translate persisted prompt content into styled terminal lines. Pure. */
41
+ export declare function translatePromptContent(content: string): string[];
42
+ /** A verdict's guidance field is "imperative steps for the worker (newline-separated ok)" —
43
+ * already list-shaped, so it renders as a bullet list, never a wall of prose. Pure. */
44
+ export declare function renderGuidanceSteps(guidance: string): string[];
45
+ /** The consult dossier panel. Stateful over injected data: a verdict cursor, at most one
46
+ * expanded dossier, and a per-verdict render cache cleared only when the selection changes. */
47
+ export declare function createConsultDossierPanel(initial?: ConsultDossierData, opts?: {
48
+ translate?: (content: string) => string[];
49
+ }): ConsultDossierPanel;
@@ -0,0 +1,169 @@
1
+ // T2 (v1.68): consult dossier viewer — the Runs cockpit's detail panel.
2
+ // Lists every consult verdict recorded for the selected task (action + reason, dispatch order)
3
+ // and expands one verdict's persisted prompt content on request. Pure render over INJECTED data:
4
+ // this module never touches the filesystem — the caller folds journal events and loads the
5
+ // consults/ artifacts, then hands both in. The prompt-content translator is a hand-written
6
+ // line-oriented pass over a known narrow subset (headings, blockquotes, bullets, fenced blocks),
7
+ // never a markdown-AST dependency. Rendered prompt lines are cached per verdict and invalidated
8
+ // only on selection change — never re-parsed on repaint (opencode's opentui regression lesson).
9
+ // Read-only by construction: there is no edit or delete path for a dossier or a verdict.
10
+ import { GLYPHS, bold, dim, warn } from "../../brand.js";
11
+ /** Placeholder marker surfaced in the Runs detail panel until T2 fills the viewer. */
12
+ export const DOSSIER_PLACEHOLDER = "consult dossier viewer — stub, filled by T2";
13
+ /** Placeholder renderer — returns one dim line. */
14
+ export function renderDossierPlaceholder() {
15
+ return ` ${DOSSIER_PLACEHOLDER}`;
16
+ }
17
+ const ACTIONS = ["retry", "reroute", "decompose", "human"];
18
+ /** Fold one task's consult-verdict journal events into dossier verdicts, in dispatch order.
19
+ * Pure function over injected events. opts.prompts carries the persisted prompt contents in
20
+ * the same dispatch order (the caller reads the consults/ artifacts); a verdict whose artifact
21
+ * was not loaded simply has no prompt. Malformed events are skipped, never fatal. */
22
+ export function foldConsultVerdicts(events, taskId, opts = {}) {
23
+ const verdicts = [];
24
+ for (const e of events) {
25
+ if (e.taskId !== taskId || e.event !== "consult-verdict")
26
+ continue;
27
+ const d = e.data;
28
+ if (typeof d.action !== "string" || !ACTIONS.includes(d.action))
29
+ continue;
30
+ verdicts.push({
31
+ action: d.action,
32
+ ...(typeof d.reason === "string" ? { reason: d.reason } : {}),
33
+ ...(typeof d.notes === "string" ? { notes: d.notes } : {}),
34
+ ...(typeof d.guidance === "string" ? { guidance: d.guidance } : {}),
35
+ });
36
+ }
37
+ if (opts.prompts) {
38
+ for (const [i, prompt] of opts.prompts.entries()) {
39
+ if (verdicts[i])
40
+ verdicts[i].prompt = prompt;
41
+ }
42
+ }
43
+ return verdicts;
44
+ }
45
+ // ── prompt-content translator ────────────────────────────────────────────────
46
+ // Hand-written, line-oriented, pure. Known subset only: ATX headings (# = bold, deeper = dim),
47
+ // `> ` blockquotes (dim `│ ` bar prefix), `- `/`* ` bullets (• marker), ``` fenced blocks
48
+ // (dim, indented), everything else passes through plain. No markdown-AST dependency.
49
+ const HEADING_RE = /^(#{1,6})[ \t]+(.*)$/;
50
+ const QUOTE_RE = /^>[ \t]?(.*)$/;
51
+ const BULLET_RE = /^[-*][ \t]+(.*)$/;
52
+ const FENCE_RE = /^\s*```/;
53
+ /** Translate persisted prompt content into styled terminal lines. Pure. */
54
+ export function translatePromptContent(content) {
55
+ const out = [];
56
+ let fenced = false;
57
+ for (const raw of content.split("\n")) {
58
+ const line = raw.replace(/\s+$/, "");
59
+ if (FENCE_RE.test(line)) {
60
+ fenced = !fenced;
61
+ continue;
62
+ }
63
+ if (fenced) {
64
+ out.push(line.trim() ? dim(` ${line}`) : "");
65
+ continue;
66
+ }
67
+ const heading = HEADING_RE.exec(line);
68
+ if (heading) {
69
+ out.push(heading[1].length === 1 ? bold(heading[2]) : dim(heading[2]));
70
+ continue;
71
+ }
72
+ const quote = QUOTE_RE.exec(line);
73
+ if (quote) {
74
+ out.push(dim(`│ ${quote[1]}`));
75
+ continue;
76
+ }
77
+ const bullet = BULLET_RE.exec(line);
78
+ if (bullet) {
79
+ out.push(`• ${bullet[1]}`);
80
+ continue;
81
+ }
82
+ out.push(line);
83
+ }
84
+ return out;
85
+ }
86
+ /** A verdict's guidance field is "imperative steps for the worker (newline-separated ok)" —
87
+ * already list-shaped, so it renders as a bullet list, never a wall of prose. Pure. */
88
+ export function renderGuidanceSteps(guidance) {
89
+ return guidance
90
+ .split(/\n+/)
91
+ .map((s) => s.trim())
92
+ .filter(Boolean)
93
+ .map((s) => `• ${s}`);
94
+ }
95
+ /** The consult dossier panel. Stateful over injected data: a verdict cursor, at most one
96
+ * expanded dossier, and a per-verdict render cache cleared only when the selection changes. */
97
+ export function createConsultDossierPanel(initial, opts = {}) {
98
+ const translate = opts.translate ?? translatePromptContent;
99
+ let data = initial;
100
+ let cursor = 0;
101
+ let expanded = null;
102
+ // Per-verdict cached render of the expanded dossier block — parsed once, reused on every
103
+ // later expand of the same verdict within the same selection; never re-parsed per repaint.
104
+ const cache = new Map();
105
+ const dossierBlock = (verdict, index) => {
106
+ const hit = cache.get(index);
107
+ if (hit)
108
+ return hit;
109
+ const lines = [];
110
+ lines.push(dim(`── persisted prompt · consult · ${data?.taskId ?? "?"} · verdict ${index + 1} ──`));
111
+ if (verdict.prompt !== undefined)
112
+ lines.push(...translate(verdict.prompt));
113
+ else
114
+ lines.push(dim("(no persisted prompt content recorded for this verdict)"));
115
+ lines.push(`action: ${verdict.action}`);
116
+ const steps = verdict.guidance ? renderGuidanceSteps(verdict.guidance) : [];
117
+ if (steps.length > 0) {
118
+ lines.push("guidance:");
119
+ for (const s of steps)
120
+ lines.push(` ${s}`);
121
+ }
122
+ cache.set(index, lines);
123
+ return lines;
124
+ };
125
+ return {
126
+ get expandedIndex() {
127
+ return expanded;
128
+ },
129
+ select(next) {
130
+ data = next;
131
+ cursor = 0;
132
+ expanded = null;
133
+ cache.clear();
134
+ },
135
+ key(name) {
136
+ if (!data || data.verdicts.length === 0)
137
+ return;
138
+ if (name === "up")
139
+ cursor = Math.max(cursor - 1, 0);
140
+ else if (name === "down")
141
+ cursor = Math.min(cursor + 1, data.verdicts.length - 1);
142
+ else if (name === "enter")
143
+ expanded = expanded === cursor ? null : cursor;
144
+ },
145
+ render() {
146
+ if (!data)
147
+ return [];
148
+ if (data.verdicts.length === 0) {
149
+ return [
150
+ ` no consult verdicts recorded for ${data.taskId} — every attempt finished without needing a consult.`,
151
+ ];
152
+ }
153
+ const lines = [];
154
+ data.verdicts.forEach((verdict, i) => {
155
+ const pointer = i === cursor ? `${GLYPHS.pointer} ` : " ";
156
+ // Consult actions are not pass/fail: human → warn (needs the operator), the rest plain.
157
+ const actionWord = verdict.action === "human" ? warn(verdict.action) : verdict.action;
158
+ const reason = verdict.reason ?? verdict.notes;
159
+ const hint = dim(i === expanded ? "[enter: collapse]" : "[enter: expand]");
160
+ lines.push(` ${pointer}${i + 1}. ${actionWord}${reason ? ` — ${reason}` : ""} ${hint}`);
161
+ if (i === expanded) {
162
+ for (const l of dossierBlock(verdict, i))
163
+ lines.push(` ${l}`);
164
+ }
165
+ });
166
+ return lines;
167
+ },
168
+ };
169
+ }