immune-brain 3.6.7 → 3.6.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -55,6 +55,8 @@ export interface TaskRailView {
55
55
  recovery?: string;
56
56
  /** Latest per-descriptor QA fact; rendered only while present. */
57
57
  acceptance_progress?: TaskRailAcceptanceProgress;
58
+ /** Optional pipeline milestone progress indicator. */
59
+ pipeline?: boolean;
58
60
  }
59
61
 
60
62
  export interface TaskOverviewEntry {
@@ -144,7 +146,7 @@ export async function requestAuthorityDialog<T extends string, R = T | undefined
144
146
  finish = done;
145
147
  if (settled || options.signal?.aborted) done(undefined);
146
148
  let expanded = false;
147
- const detailText = new Text(theme.fg("muted", "Details collapsed; press d to expand."), 1, 0);
149
+ const detailText = new Text(theme.fg("muted", "▸ Details collapsed; press d to expand."), 1, 0);
148
150
  const selectList = new SelectList(
149
151
  options.actions.map((action): SelectItem => ({ ...action })),
150
152
  options.actions.length,
@@ -158,13 +160,27 @@ export async function requestAuthorityDialog<T extends string, R = T | undefined
158
160
  );
159
161
  selectList.onSelect = (item) => complete(item.value as T);
160
162
  selectList.onCancel = () => complete(undefined);
163
+
164
+ const affirmativeAction = options.actions.find((a) =>
165
+ ["confirm", "authorize", "yes", "accept"].includes(a.value.toLowerCase()),
166
+ );
167
+ const negativeAction = options.actions.find((a) =>
168
+ ["cancel", "decline", "no", "reject"].includes(a.value.toLowerCase()),
169
+ );
170
+
171
+ const affKey = affirmativeAction ? "y/enter" : "enter";
172
+ const negKey = negativeAction ? "n/esc" : "esc";
173
+ const affLabel = affirmativeAction ? affirmativeAction.value : "choose";
174
+ const negLabel = negativeAction ? negativeAction.value : "cancel";
175
+ const hintLine = `d: details | ${affKey}: ${affLabel} | ${negKey}: ${negLabel}`;
176
+
161
177
  const container = new Container();
162
178
  container.addChild(new DynamicBorder((text: string) => theme.fg("accent", text)));
163
179
  container.addChild(new Text(theme.fg("accent", theme.bold(options.title)), 1, 0));
164
- container.addChild(new Text(options.summary, 1, 0));
180
+ container.addChild(new Text(formatDialogSummary(options.summary, theme), 1, 0));
165
181
  container.addChild(detailText);
166
182
  container.addChild(selectList);
167
- container.addChild(new Text(theme.fg("dim", "d: toggle details | enter: choose | esc: cancel"), 1, 0));
183
+ container.addChild(new Text(theme.fg("dim", hintLine), 1, 0));
168
184
  container.addChild(new DynamicBorder((text: string) => theme.fg("accent", text)));
169
185
  return {
170
186
  render: (width) => container.render(width),
@@ -172,10 +188,20 @@ export async function requestAuthorityDialog<T extends string, R = T | undefined
172
188
  handleInput: (data) => {
173
189
  if (data === "d" || data === "D") {
174
190
  expanded = !expanded;
175
- detailText.setText(expanded ? options.details : theme.fg("muted", "Details collapsed; press d to expand."));
191
+ detailText.setText(expanded
192
+ ? `${theme.fg("muted", "▾ Details:")}\n${options.details}`
193
+ : theme.fg("muted", "▸ Details collapsed; press d to expand."));
176
194
  tui.requestRender();
177
195
  return;
178
196
  }
197
+ if ((data === "y" || data === "Y") && affirmativeAction) {
198
+ complete(affirmativeAction.value as T);
199
+ return;
200
+ }
201
+ if ((data === "n" || data === "N") && negativeAction) {
202
+ complete(negativeAction.value as T);
203
+ return;
204
+ }
179
205
  selectList.handleInput(data);
180
206
  tui.requestRender();
181
207
  },
@@ -372,6 +398,13 @@ export function renderStructuredResult(
372
398
  ];
373
399
  const recovery = recoveryHint(details);
374
400
  if (recovery) lines.push(`${theme.fg("muted", "Recovery:")} ${theme.fg("dim", recovery)}`);
401
+ const facts = record(details.facts) ?? record(details.execution_facts);
402
+ if (facts) {
403
+ const factParts = Object.entries(facts)
404
+ .map(([k, v]) => `${k}=${String(v)}`)
405
+ .join(" · ");
406
+ if (factParts) lines.push(`${theme.fg("muted", "Facts:")} ${theme.fg("dim", factParts)}`);
407
+ }
375
408
  if (terminal && taskState) lines.push(...renderFinalLines(taskState, theme));
376
409
  return new Text(lines.join("\n"), 0, 0);
377
410
  }
@@ -423,6 +456,9 @@ function renderTaskRail(view: TaskRailView, width = 120, theme?: Theme): string[
423
456
  const lines = [
424
457
  `Task ${boundedMiddle(view.task_id, taskIdWidth)} · ${stateFormatted}`,
425
458
  ];
459
+ if (view.pipeline) {
460
+ lines.push(`${label("Pipeline:")} ${renderPipelineMilestones(view.state, theme)}`);
461
+ }
426
462
  if (view.phase) {
427
463
  lines.push(`${label("Phase:")} ${body(bounded(view.phase, availableContentWidth))}`);
428
464
  }
@@ -446,6 +482,69 @@ function renderTaskRail(view: TaskRailView, width = 120, theme?: Theme): string[
446
482
  return lines;
447
483
  }
448
484
 
485
+ export function renderPipelineMilestones(state: TaskRailState, theme?: Theme): string {
486
+ const milestones = [
487
+ { name: "Plan", key: "plan" },
488
+ { name: "Exec", key: "exec" },
489
+ { name: "QA", key: "qa" },
490
+ { name: "Review", key: "review" },
491
+ ] as const;
492
+
493
+ type StepStatus = "done" | "active" | "pending" | "blocked" | "stopped";
494
+
495
+ let statuses: [StepStatus, StepStatus, StepStatus, StepStatus];
496
+ switch (state) {
497
+ case "Planning":
498
+ statuses = ["active", "pending", "pending", "pending"];
499
+ break;
500
+ case "Approval required":
501
+ case "Working":
502
+ statuses = ["done", "active", "pending", "pending"];
503
+ break;
504
+ case "Verifying":
505
+ statuses = ["done", "done", "active", "pending"];
506
+ break;
507
+ case "Reviewing":
508
+ statuses = ["done", "done", "done", "active"];
509
+ break;
510
+ case "Completed":
511
+ statuses = ["done", "done", "done", "done"];
512
+ break;
513
+ case "Blocked":
514
+ statuses = ["done", "blocked", "pending", "pending"];
515
+ break;
516
+ case "Stopped":
517
+ statuses = ["done", "stopped", "pending", "pending"];
518
+ break;
519
+ default:
520
+ statuses = ["pending", "pending", "pending", "pending"];
521
+ }
522
+
523
+ const renderStep = (name: string, status: StepStatus, index: number): string => {
524
+ const stepNum = index + 1;
525
+ if (!theme) {
526
+ const sym = status === "done" ? "✔" : status === "active" ? "●" : status === "blocked" ? "⚠" : status === "stopped" ? "■" : "○";
527
+ return `[${stepNum}.${name} ${sym}]`;
528
+ }
529
+ switch (status) {
530
+ case "done":
531
+ return `[${stepNum}.${name} ${theme.fg("success", "✔")}]`;
532
+ case "active":
533
+ return `[${stepNum}.${name} ${theme.fg("accent", "●")}]`;
534
+ case "blocked":
535
+ return `[${stepNum}.${name} ${theme.fg("warning", "⚠")}]`;
536
+ case "stopped":
537
+ return `[${stepNum}.${name} ${theme.fg("muted", "■")}]`;
538
+ case "pending":
539
+ default:
540
+ return `[${stepNum}.${name} ${theme.fg("dim", "○")}]`;
541
+ }
542
+ };
543
+
544
+ const sep = theme ? ` ${theme.fg("dim", "─")} ` : " ─ ";
545
+ return milestones.map((m, i) => renderStep(m.name, statuses[i], i)).join(sep);
546
+ }
547
+
449
548
  export function renderTaskOverview(view: TaskOverviewView, width = 120, theme?: Theme): string[] {
450
549
  const label = (text: string) => (theme ? theme.fg("muted", text) : text);
451
550
  const head = (text: string) => (theme ? theme.fg("accent", theme.bold(text)) : text);
@@ -456,6 +555,7 @@ export function renderTaskOverview(view: TaskOverviewView, width = 120, theme?:
456
555
  }
457
556
  if (view.active) {
458
557
  lines.push(`${label("Active:")} ${formatTaskRailState(view.active.state, theme)} ${bounded(view.active.task_id, 60)}`);
558
+ lines.push(`${label(" Pipeline:")} ${renderPipelineMilestones(view.active.state, theme)}`);
459
559
  lines.push(`${label(" Result:")} ${bounded(view.active.result, 100)}`);
460
560
  lines.push(`${label(" Next:")} ${bounded(view.active.next, 100)}`);
461
561
  } else {
@@ -501,15 +601,60 @@ function renderFinalLines(taskState: Record<string, unknown>, theme: Theme): str
501
601
  + strings(taskState.unresolved_user_decision_ids).length
502
602
  + strings(taskState.replan_required_ids).length;
503
603
  const diffHash = string(taskState.diff_hash);
604
+ const divider = theme.fg("dim", "────────────────────────────────────────");
504
605
  return [
606
+ divider,
607
+ theme.fg("accent", theme.bold("Final Settlement Summary")),
505
608
  `${theme.fg("muted", "Acceptance:")} ${theme.fg(missing === 0 ? "success" : "warning", `${fresh}/${fresh + missing} fresh`)}`,
506
609
  `${theme.fg("muted", "QA / Review:")} ${theme.fg("dim", approvals.length > 0 ? approvals.join(", ") : "not recorded")}`,
507
610
  `${theme.fg("muted", "Residual blockers:")} ${theme.fg(blockers === 0 ? "dim" : "warning", String(blockers))}`,
508
611
  `${theme.fg("muted", "Repository health:")} ${theme.fg("dim", "not assessed")}`,
509
612
  `${theme.fg("muted", "Git:")} ${theme.fg("dim", diffHash ? `task diff ${diffHash.slice(0, 15)}` : "not reported")}`,
613
+ divider,
510
614
  ];
511
615
  }
512
616
 
617
+ function formatDialogSummary(summary: string, theme: Theme): string {
618
+ const lines = summary.split("\n");
619
+ return lines.map((line) => {
620
+ const colonIdx = line.indexOf(":");
621
+ if (colonIdx === -1) return theme.fg("dim", line);
622
+ const key = line.slice(0, colonIdx).trim();
623
+ const value = line.slice(colonIdx + 1).trim();
624
+ let formattedValue = theme.fg("dim", value);
625
+ const lowerKey = key.toLowerCase();
626
+ if (lowerKey === "risk") {
627
+ const isHigh = /high|material|critical/i.test(value);
628
+ formattedValue = isHigh ? theme.fg("warning", theme.bold(value)) : theme.fg("accent", value);
629
+ } else if (lowerKey === "goal") {
630
+ formattedValue = theme.fg("accent", value);
631
+ } else if (lowerKey === "acceptance") {
632
+ formattedValue = theme.fg("success", value);
633
+ }
634
+ return `${theme.fg("muted", `${key}:`)} ${formattedValue}`;
635
+ }).join("\n");
636
+ }
637
+
638
+ export function boundedPath(path: string, max: number): string {
639
+ const width = visibleWidth(path);
640
+ if (width <= max) return path;
641
+ if (!path.includes("/")) return bounded(path, max);
642
+ const parts = path.split("/");
643
+ const fileName = parts.pop() ?? "";
644
+ const fileWidth = visibleWidth(fileName);
645
+ if (fileWidth + 2 >= max) {
646
+ return bounded(fileName, max);
647
+ }
648
+ const remaining = max - fileWidth - 3; // "…/"
649
+ let prefix = "";
650
+ for (const part of parts) {
651
+ const next = prefix ? `${prefix}/${part}` : part;
652
+ if (visibleWidth(next) > remaining) break;
653
+ prefix = next;
654
+ }
655
+ return prefix ? `${prefix}/…/${fileName}` : `…/${fileName}`;
656
+ }
657
+
513
658
  function record(value: unknown): Record<string, unknown> | undefined {
514
659
  return typeof value === "object" && value !== null && !Array.isArray(value)
515
660
  ? value as Record<string, unknown>
@@ -42,7 +42,7 @@ function probeHost(env = process.env, platform = process.platform, hostVersion)
42
42
  }
43
43
 
44
44
  // plugins/immune-brain/runtime/plugin_version.ts
45
- var PLUGIN_VERSION = "3.6.7";
45
+ var PLUGIN_VERSION = "3.6.9";
46
46
 
47
47
  // plugins/immune-brain/runtime/claude/interaction.ts
48
48
  import { createHash, randomUUID } from "node:crypto";
@@ -7126,7 +7126,8 @@ function parseIssues(raw) {
7126
7126
  title: item.title,
7127
7127
  body: typeof item.body === "string" ? item.body : "",
7128
7128
  state: item.state,
7129
- state_reason: typeof item.state_reason === "string" ? item.state_reason.toLowerCase() : null
7129
+ state_reason: typeof item.state_reason === "string" ? item.state_reason.toLowerCase() : null,
7130
+ labels: Array.isArray(item.labels) ? item.labels.map((label) => typeof label === "string" ? label : label?.name).filter((name) => typeof name === "string") : []
7130
7131
  };
7131
7132
  });
7132
7133
  }
@@ -90,6 +90,9 @@ with its `initiative_slug` parameter. That parameter is the opt-in: absent the c
90
90
  `imm-loop` behavior is byte-identical to per-task Enrollment, and no batch state,
91
91
  branch, or Batch Authorization exists. The Standalone Hosts expose the same tool
92
92
  name and the same single parameter; it is never a batch of tasks the Host chose.
93
+ When an Initiative is referenced by its tracker Issue (e.g. `github #<number>`),
94
+ extract `initiative_slug` from the Issue body `<!-- immune-brain:initiative-id=<slug> -->`
95
+ marker or title prefix before invoking the tool.
93
96
 
94
97
  Invoking it authorizes only a user-confirmed batch of already-planned child
95
98
  TaskIntents. The Host projects the batch plan from the Initiative's published
@@ -172,8 +172,13 @@ with `valid: true` and `enrollment_ready: true`. Resolve `../bin/imm-tracker` fr
172
172
  Initiative slug and goal, Parent projection, and every Child's `slice_id`,
173
173
  canonical TaskIntent path, bounded public `acceptance` summaries, and public
174
174
  projection. The Parent projection requires
175
- `problem`, `result`, and `design`, and may include `decisions`,
176
- `testing_strategy`, and `out_of_scope`. `design` records Initiative-level
175
+ `short_name`, `title`, `problem`, `result`, and `design`, and may include
176
+ `source_issue`, `decisions`,
177
+ `testing_strategy`, and `out_of_scope`. `short_name` (1-32 characters) is the
178
+ stable short Initiative name used in every Issue title; `title` (1-60
179
+ characters) is the short Initiative display title; `source_issue` is the
180
+ originating feature Issue number, rendered as a Provenance link. `design`
181
+ records Initiative-level
177
182
  invariants, Slice boundaries and ordering, shared interfaces or state flow, and
178
183
  material compatibility decisions. Every Parent Slice must correspond to one
179
184
  published Child; future checklist-only Slices are not allowed in the batch.
@@ -181,15 +186,24 @@ published Child; future checklist-only Slices are not allowed in the batch.
181
186
  Each Child must provide public `acceptance` entries with `id` and a 1-500
182
187
  character `summary`. Their IDs must match every canonical TaskIntent acceptance
183
188
  ID exactly once. Canonical assertion prose is authority evidence and must never
184
- be copied into public GitHub projection. Each Child projection may contain
189
+ be copied into public GitHub projection. Each Child projection requires
190
+ `title` (1-60 characters), the short Slice display title, and may contain
185
191
  `result`, `current_behavior`,
186
192
  `desired_behavior`, `key_interfaces`, `verification`, `blocked_by` Task IDs,
187
- `out_of_scope`, and `agent_handoff`. The tracker rereads every canonical
193
+ `out_of_scope`, and `agent_handoff`. The tracker composes Issue titles from
194
+ these display names only — the Parent as `[<short_name>] <title>` and each
195
+ Child as `[<short_name>] S<n> <title>` with `n` the declared Slice position —
196
+ and fails the whole batch closed before any remote write when a display name
197
+ is missing or the composed title exceeds 80 characters; it never falls back to
198
+ goal prose and never truncates a title. The tracker rereads every canonical
188
199
  TaskIntent for identity, risk, and acceptance IDs; projection fields and public
189
200
  summaries never widen TaskIntent scope or authority. It validates the complete dependency graph before
190
201
  remote writes, creates the Parent once, creates all Children, attaches every
191
202
  Child as a native Sub-issue, creates native `blocked_by` relations, and rereads
192
- the complete topology. The Child Agent Brief includes a direct Parent Issue link.
203
+ the complete topology. Every Child carries `ready-for-agent`, blocked Children
204
+ additionally carry `blocked`, and the Parent carries neither; the tracker never
205
+ creates labels, so a repository missing a required label fails the batch closed
206
+ before any remote write. The Child Agent Brief includes a direct Parent Issue link.
193
207
  Internal role prompts, tool policies, review gates, model reservations, and
194
208
  prompt digests never belong in this external handoff. If
195
209
  `docs/initiatives/<slug>.md` exists, publication fails with a carrier conflict;
@@ -0,0 +1,123 @@
1
+ ---
2
+ name: imm-review-retro
3
+ description: Use when the user explicitly requests Immune-Brain ranking of models by cross-model review load or a project usage retro.
4
+ ---
5
+
6
+ # Immune-Brain: Review Retro
7
+
8
+ Rank models by how much code review their own edits triggered, and report
9
+ basic project usage over a look-back window the user supplies in days. This
10
+ is a standalone host-native analysis entry, not a Managed Path continuation
11
+ and not an `imm-loop` internal-role dispatch. It reviews no diff — a diff
12
+ review is `code-review`.
13
+
14
+ ## Boundary
15
+
16
+ Allowed: read pi session JSONL under `~/.pi/agent/sessions` (or `--root`),
17
+ run the bundled analyzer, and write a stdout report.
18
+
19
+ Blocked: code, test, Spec, Plan, or `.imm/` edits; session-log writes;
20
+ Kernel, TaskIntent, or TaskRecord mutation; Compounder or scheduled runs;
21
+ `.imm/audit/` lifecycle statistics.
22
+
23
+ An already active Managed task remains owned by `imm-loop`. This Skill does
24
+ not create or resume Managed authority.
25
+
26
+ ## Invocation
27
+
28
+ Requires explicit invocation: `imm-review-retro` or `/imm-review-retro`.
29
+ Ordinary questions such as "which model is worse" stay host-native and do
30
+ not enter this Skill.
31
+
32
+ The look-back window in days is required input. If the user named one, use
33
+ it. If not, ask before running, because the ranking moves with the window.
34
+
35
+ Default scan is the user's full session-log tree. Pass `--project <substr>`
36
+ when the user wants one repo or worktree. Do not invent a project filter.
37
+
38
+ No daemon, no cron, no CI, no automatic commit.
39
+
40
+ ## Counting rules
41
+
42
+ These rules keep numbers comparable across runs. Read the analyzer header
43
+ aloud in the report so the 口径 stays visible.
44
+
45
+ - `review` = an `Agent` tool call with `subagent_type` equal to `Review`.
46
+ - Attribution = the model behind the most recent `edit`, `write`, or
47
+ `multiedit` in that session. If none, the row is `no-edit (review-only)`.
48
+ - `uniq` counts distinct (session, description+prompt prefix) pairs. A wide
49
+ gap versus `reviews` is the same review re-run on the same code.
50
+ - `rev/100ed` is `100 * reviews / devEdits`. Rank on both absolute `reviews`
51
+ and this intensity. A model can lead one axis and sit mid-pack on the
52
+ other.
53
+ - `avgSc` / `pass%` parse `[SCORE: …]` and `[VERDICT: …]` tags from the
54
+ matching Review `toolResult`. Untagged reviews show `-`.
55
+ - `registr` counts `imm_kernel_canary` `submit_review`. It is the
56
+ registration of the same review and is never added into `reviews`.
57
+ - `rounds/task` is registrations per distinct `(cwd, task_id)`. High values
58
+ can be canary/QA harness re-registration, not human-visible rework.
59
+ - Findings are `record_finding` calls, deduped per session. Summaries that
60
+ match `recorded cleanly`, `receipt recorded`, `round recorded`, or
61
+ `no finding(s)` are `bookkeep` / `noisy`, excluded from `block`/`advis`.
62
+
63
+ Usage counters on the same pass: sessions with activity, assistant turns,
64
+ edit counts, a tool-call name histogram, and the project × author table.
65
+
66
+ ## CLI
67
+
68
+ Run the bundled analyzer. Prefer `bun`; `node` (≥23.6, type stripping) is
69
+ an allowed equivalent. The script is erasable TypeScript with `node:` APIs
70
+ only.
71
+
72
+ ```
73
+ bun "<path-to-skill>/scripts/review_retro.ts" <days> [--root <sessions-dir>] [--project <substr>] [--top N]
74
+ ```
75
+
76
+ - `<days>` must be `> 0`.
77
+ - `--root` defaults to `~/.pi/agent/sessions`.
78
+ - `--project` keeps sessions whose `cwd` contains the substring.
79
+ - `--top` is the project-table row cap (default 15).
80
+ - Malformed JSONL lines are skipped. `days <= 0` is a hard error.
81
+
82
+ Do not scan live `.imm/` directories. Tests use committed fixtures under
83
+ `tests/fixtures/review-retro/`.
84
+
85
+ ## Report
86
+
87
+ Write-up order:
88
+
89
+ 1. Window and 口径 in one line (copy the analyzer header).
90
+ 2. Ranked model table, including scores.
91
+ 3. Usage section: sessions, turns, edits, tool mix.
92
+ 4. Quality and score findings.
93
+ 5. Three to five bullets of what the table means (volume versus intensity,
94
+ quality versus rework, where it concentrated).
95
+ 6. Caveats last.
96
+
97
+ Rank on both axes, never one. Name the axis you are ranking by, and call
98
+ out models that flip order between `reviews` and `rev/100ed`.
99
+
100
+ Separate one-pass from rework: compare `uniq` to `reviews`, and read
101
+ `rounds/task` on the kernel path.
102
+
103
+ Evaluate quality: high intensity plus high score is frequent review of
104
+ mostly minor issues; low intensity plus low score is rare review of severe
105
+ defects. Call out REJECT or highRisk ratings.
106
+
107
+ Ground each model in its projects. Cite the two or three worktrees where
108
+ that model's reviews concentrated.
109
+
110
+ ## Caveats
111
+
112
+ Anything the script splits out as `bookkeep` stays visible next to the
113
+ column it contaminates. Flag any finding count you cannot trace to a real
114
+ defect.
115
+
116
+ High `rounds/task` can be canary/QA harness re-registration, not
117
+ human-visible rework.
118
+
119
+ This Skill does not persist snapshots or compute week-over-week diffs.
120
+ Re-run with a new window when the user wants a later period.
121
+
122
+ The personal python prototype under `~/.pi/agent/skills/review-retro/` is
123
+ not this Skill and is not modified by it.
@@ -56,3 +56,12 @@ skills:
56
56
  output_artifacts: [maintain_report]
57
57
  next_actions: []
58
58
  boundary: Minimize tracked agent-instruction context after explicit manifest approval; no Managed authority mutation, contract installation, or reference-document creation.
59
+ - name: imm-review-retro
60
+ path: skills/imm-review-retro/SKILL.md
61
+ role: execute
62
+ title: Review Retro
63
+ role_class: discovery
64
+ canonical: true
65
+ output_artifacts: [retro_report]
66
+ next_actions: []
67
+ boundary: Rank models by review load and report project usage from pi session logs. Read-only. No Managed Path mutation.