immune-brain 3.6.8 → 3.6.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -4
- package/README.zh-CN.md +10 -3
- package/package.json +1 -1
- package/plugins/immune-brain/.claude-plugin/plugin.json +1 -1
- package/plugins/immune-brain/dist/claude/mcp-server.mjs +3 -2
- package/plugins/immune-brain/dist/imm-planner.md +19 -5
- package/plugins/immune-brain/dist/imm-review-retro.md +123 -0
- package/plugins/immune-brain/dist/registry.yaml +9 -0
- package/plugins/immune-brain/runtime/github_issue_tracker.ts +253 -28
- package/plugins/immune-brain/runtime/plugin_version.ts +1 -1
- package/plugins/immune-brain/skills/imm-review-retro/SKILL.md +23 -0
- package/plugins/immune-brain/skills/imm-review-retro/scripts/review_retro.ts +355 -0
- package/plugins/immune-brain/skills/registry.yaml +9 -0
package/README.md
CHANGED
|
@@ -25,7 +25,7 @@ Pi and Claude Code are the supported hosts. Undeclared adapters remain unsupport
|
|
|
25
25
|
- [Installation](#installation)
|
|
26
26
|
- [Quick Start](#quick-start)
|
|
27
27
|
- [How to Use](#how-to-use)
|
|
28
|
-
- [The
|
|
28
|
+
- [The 7 Skills](#the-7-skills)
|
|
29
29
|
- [Lifecycle](#lifecycle)
|
|
30
30
|
- [Unattended Batch Runs](#unattended-batch-runs)
|
|
31
31
|
- [Configuration](#configuration)
|
|
@@ -119,6 +119,7 @@ Immune-Brain provides two clean modes: **Host-native** for daily coding, and **M
|
|
|
119
119
|
| PR has review comments or failing CI | `/imm-pr-fix` on that PR | → Standalone repair: minimal scoped fix in place, no managed task created |
|
|
120
120
|
| Project docs out of date | `/imm-doc-prune` | → Read-only audit; deletes only user-approved stale docs from manifest |
|
|
121
121
|
| Agent instructions bloated | `/imm-agent-doc-maintain` | → Minimizes tracked `AGENTS.md` / `CLAUDE.md` to essential non-discoverable rules |
|
|
122
|
+
| Which model's edits keep coming back for review | `/imm-review-retro` | → Ranks models by review load and reports project usage from session logs |
|
|
122
123
|
|
|
123
124
|
> **Core Principle: Skill-Explicit Entry**
|
|
124
125
|
> - **Ordinary input stays host-native**: Natural language queries never automatically start planning or task enrollment. You choose when to turn on engineering rigor.
|
|
@@ -126,7 +127,7 @@ Immune-Brain provides two clean modes: **Host-native** for daily coding, and **M
|
|
|
126
127
|
|
|
127
128
|
---
|
|
128
129
|
|
|
129
|
-
## The
|
|
130
|
+
## The 7 Skills
|
|
130
131
|
|
|
131
132
|
| Skill | Type | When to use | What it does |
|
|
132
133
|
|---|---|---|---|
|
|
@@ -136,10 +137,11 @@ Immune-Brain provides two clean modes: **Host-native** for daily coding, and **M
|
|
|
136
137
|
| `imm-pr-fix` | Standalone | CI failed / review comments on a PR | Repairs one PR in place, no managed authority |
|
|
137
138
|
| `imm-doc-prune` | Standalone | Stale current docs | Deletes only the hash-approved manifest entries |
|
|
138
139
|
| `imm-agent-doc-maintain` | Standalone | Bloated agent instructions | Minimizes tracked AGENTS/CLAUDE/GEMINI.md to necessary context |
|
|
140
|
+
| `imm-review-retro` | Standalone | Compare models by review load | Ranks authors of reviewed code and reports project usage |
|
|
139
141
|
|
|
140
142
|
Internal roles (Executor, QA, Review, Compounder) are dispatched by `imm-loop` — you never invoke them directly.
|
|
141
143
|
|
|
142
|
-
All
|
|
144
|
+
All 7 skills are invoked explicitly. For new features, start with `imm-brainstorm` (if requirements are uncertain) or `imm-planner` (if requirements are clear), then proceed to `imm-loop` once enrolled.
|
|
143
145
|
|
|
144
146
|
### Managed Path entries (brainstorm → planner → loop)
|
|
145
147
|
|
|
@@ -187,6 +189,11 @@ The three repair/maintenance skills are host-native: they never create a managed
|
|
|
187
189
|
- **Trigger:** explicit request to minimize tracked `AGENTS.md` / `CLAUDE.md` / `GEMINI.md`.
|
|
188
190
|
- **What it does:** keeps only the necessary non-discoverable rules in agent instruction files, under the same read-only-audit + hash-bound-manifest-approval model as `imm-doc-prune`.
|
|
189
191
|
|
|
192
|
+
#### `imm-review-retro` — review load and project usage
|
|
193
|
+
|
|
194
|
+
- **Trigger:** explicit request for a cross-model review retro or project usage look-back.
|
|
195
|
+
- **What it does:** ranks models by how much review their own edits triggered, plus sessions/turns/edits/tool mix, from pi session logs. Read-only. Not a diff review.
|
|
196
|
+
|
|
190
197
|
---
|
|
191
198
|
|
|
192
199
|
## Lifecycle
|
|
@@ -265,7 +272,7 @@ See [`docs/reference/immune-brain-config.md`](docs/reference/immune-brain-config
|
|
|
265
272
|
package.json # Pi package manifest (skills + extensions)
|
|
266
273
|
plugins/immune-brain/
|
|
267
274
|
├── .pi-extension/ # Pi TUI + Kernel authority extension
|
|
268
|
-
├── skills/ #
|
|
275
|
+
├── skills/ # 7 public Skills (trigger shims)
|
|
269
276
|
├── dist/ # Built skill contracts & references
|
|
270
277
|
├── runtime/ # Bun + TypeScript runtime & Kernel
|
|
271
278
|
└── bin/ # CLI wrappers (→ runtime/v4_runtime.ts)
|
package/README.zh-CN.md
CHANGED
|
@@ -25,7 +25,7 @@ Pi 与 Claude Code 是支持的宿主。未声明的适配器仍不受支持。C
|
|
|
25
25
|
- [安装](#安装)
|
|
26
26
|
- [快速开始](#快速开始)
|
|
27
27
|
- [如何使用](#如何使用)
|
|
28
|
-
- [
|
|
28
|
+
- [7 个 Skills](#7-个-skills)
|
|
29
29
|
- [生命周期](#生命周期)
|
|
30
30
|
- [无人值守批次运行](#无人值守批次运行)
|
|
31
31
|
- [配置](#配置)
|
|
@@ -119,6 +119,7 @@ Immune-Brain 提供两种清晰的工作模式:日常轻量编码走 **Host-na
|
|
|
119
119
|
| PR 被评论 / CI 挂了 | 对该 PR 使用 `/imm-pr-fix` | → 独立修复:在当前 PR 内针对性修复,不创建新 managed 任务 |
|
|
120
120
|
| 文档过时需要清理 | `/imm-doc-prune` | → 只读审计过时文档,仅删除经哈希审批的条目 |
|
|
121
121
|
| Agent 指令文件膨胀 | `/imm-agent-doc-maintain` | → 将 tracked `AGENTS.md` / `CLAUDE.md` 压到最小必要上下文 |
|
|
122
|
+
| 想知道哪个模型的改动总被审查 | `/imm-review-retro` | → 按模型排名审查负载,并从 session logs 汇报项目使用量 |
|
|
122
123
|
|
|
123
124
|
> **核心原则:Skill 显式调用**
|
|
124
125
|
> - **普通输入保持 Host-native**:自然语言提问绝不自动绑架流程或发起 Enrollment。你完全自主决定何时开启严格工程保障。
|
|
@@ -126,7 +127,7 @@ Immune-Brain 提供两种清晰的工作模式:日常轻量编码走 **Host-na
|
|
|
126
127
|
|
|
127
128
|
---
|
|
128
129
|
|
|
129
|
-
##
|
|
130
|
+
## 7 个 Skills
|
|
130
131
|
|
|
131
132
|
| Skill | 类型 | 何时使用 | 职责 |
|
|
132
133
|
|---|---|---|---|
|
|
@@ -136,10 +137,11 @@ Immune-Brain 提供两种清晰的工作模式:日常轻量编码走 **Host-na
|
|
|
136
137
|
| `imm-pr-fix` | 独立 | PR 需修复 | 原地修复单个 PR,不触及 managed authority |
|
|
137
138
|
| `imm-doc-prune` | 独立 | 清理过时文档 | 仅删除哈希绑定的 manifest 条目 |
|
|
138
139
|
| `imm-agent-doc-maintain` | 独立 | Agent instruction 膨胀 | 将 tracked AGENTS/CLAUDE/GEMINI.md 压到最小必要上下文 |
|
|
140
|
+
| `imm-review-retro` | 独立 | 比较模型的审查负载 | 排名被审查代码的作者并汇报项目使用量 |
|
|
139
141
|
|
|
140
142
|
Executor、QA、Review、Compounder 等为 `imm-loop` 内部调度的角色,无需手动调用。
|
|
141
143
|
|
|
142
|
-
所有
|
|
144
|
+
所有 7 个 Skill 均显式调用。新需求开发时:若需求含糊先调 `imm-brainstorm`,目标清晰直接调 `imm-planner`,完成确认后调 `imm-loop` 推进闭环。
|
|
143
145
|
|
|
144
146
|
### Managed Path 入口(brainstorm → planner → loop)
|
|
145
147
|
|
|
@@ -187,6 +189,11 @@ Executor、QA、Review、Compounder 等为 `imm-loop` 内部调度的角色,
|
|
|
187
189
|
- **触发方式:** 显式要求精简版本控制下的 `AGENTS.md` / `CLAUDE.md` / `GEMINI.md`。
|
|
188
190
|
- **职责:** 遵循与 `imm-doc-prune` 相同的「只读审计 + 哈希清单审批」模式,仅保留无法直接推导的必要规则。
|
|
189
191
|
|
|
192
|
+
#### `imm-review-retro` — 审查负载与项目使用回顾
|
|
193
|
+
|
|
194
|
+
- **触发方式:** 显式要求跨模型审查复盘或项目使用量回顾。
|
|
195
|
+
- **职责:** 从 pi session logs 按模型排名被审查代码的作者,并汇报 sessions/turns/编辑量/工具分布。只读;不审查 diff。
|
|
196
|
+
|
|
190
197
|
---
|
|
191
198
|
|
|
192
199
|
## 生命周期
|
package/package.json
CHANGED
|
@@ -42,7 +42,7 @@ function probeHost(env = process.env, platform = process.platform, hostVersion)
|
|
|
42
42
|
}
|
|
43
43
|
|
|
44
44
|
// plugins/immune-brain/runtime/plugin_version.ts
|
|
45
|
-
var PLUGIN_VERSION = "3.6.
|
|
45
|
+
var PLUGIN_VERSION = "3.6.9";
|
|
46
46
|
|
|
47
47
|
// plugins/immune-brain/runtime/claude/interaction.ts
|
|
48
48
|
import { createHash, randomUUID } from "node:crypto";
|
|
@@ -7126,7 +7126,8 @@ function parseIssues(raw) {
|
|
|
7126
7126
|
title: item.title,
|
|
7127
7127
|
body: typeof item.body === "string" ? item.body : "",
|
|
7128
7128
|
state: item.state,
|
|
7129
|
-
state_reason: typeof item.state_reason === "string" ? item.state_reason.toLowerCase() : null
|
|
7129
|
+
state_reason: typeof item.state_reason === "string" ? item.state_reason.toLowerCase() : null,
|
|
7130
|
+
labels: Array.isArray(item.labels) ? item.labels.map((label) => typeof label === "string" ? label : label?.name).filter((name) => typeof name === "string") : []
|
|
7130
7131
|
};
|
|
7131
7132
|
});
|
|
7132
7133
|
}
|
|
@@ -172,8 +172,13 @@ with `valid: true` and `enrollment_ready: true`. Resolve `../bin/imm-tracker` fr
|
|
|
172
172
|
Initiative slug and goal, Parent projection, and every Child's `slice_id`,
|
|
173
173
|
canonical TaskIntent path, bounded public `acceptance` summaries, and public
|
|
174
174
|
projection. The Parent projection requires
|
|
175
|
-
`problem`, `result`, and `design`, and may include
|
|
176
|
-
`
|
|
175
|
+
`short_name`, `title`, `problem`, `result`, and `design`, and may include
|
|
176
|
+
`source_issue`, `decisions`,
|
|
177
|
+
`testing_strategy`, and `out_of_scope`. `short_name` (1-32 characters) is the
|
|
178
|
+
stable short Initiative name used in every Issue title; `title` (1-60
|
|
179
|
+
characters) is the short Initiative display title; `source_issue` is the
|
|
180
|
+
originating feature Issue number, rendered as a Provenance link. `design`
|
|
181
|
+
records Initiative-level
|
|
177
182
|
invariants, Slice boundaries and ordering, shared interfaces or state flow, and
|
|
178
183
|
material compatibility decisions. Every Parent Slice must correspond to one
|
|
179
184
|
published Child; future checklist-only Slices are not allowed in the batch.
|
|
@@ -181,15 +186,24 @@ published Child; future checklist-only Slices are not allowed in the batch.
|
|
|
181
186
|
Each Child must provide public `acceptance` entries with `id` and a 1-500
|
|
182
187
|
character `summary`. Their IDs must match every canonical TaskIntent acceptance
|
|
183
188
|
ID exactly once. Canonical assertion prose is authority evidence and must never
|
|
184
|
-
be copied into public GitHub projection. Each Child projection
|
|
189
|
+
be copied into public GitHub projection. Each Child projection requires
|
|
190
|
+
`title` (1-60 characters), the short Slice display title, and may contain
|
|
185
191
|
`result`, `current_behavior`,
|
|
186
192
|
`desired_behavior`, `key_interfaces`, `verification`, `blocked_by` Task IDs,
|
|
187
|
-
`out_of_scope`, and `agent_handoff`. The tracker
|
|
193
|
+
`out_of_scope`, and `agent_handoff`. The tracker composes Issue titles from
|
|
194
|
+
these display names only — the Parent as `[<short_name>] <title>` and each
|
|
195
|
+
Child as `[<short_name>] S<n> <title>` with `n` the declared Slice position —
|
|
196
|
+
and fails the whole batch closed before any remote write when a display name
|
|
197
|
+
is missing or the composed title exceeds 80 characters; it never falls back to
|
|
198
|
+
goal prose and never truncates a title. The tracker rereads every canonical
|
|
188
199
|
TaskIntent for identity, risk, and acceptance IDs; projection fields and public
|
|
189
200
|
summaries never widen TaskIntent scope or authority. It validates the complete dependency graph before
|
|
190
201
|
remote writes, creates the Parent once, creates all Children, attaches every
|
|
191
202
|
Child as a native Sub-issue, creates native `blocked_by` relations, and rereads
|
|
192
|
-
the complete topology.
|
|
203
|
+
the complete topology. Every Child carries `ready-for-agent`, blocked Children
|
|
204
|
+
additionally carry `blocked`, and the Parent carries neither; the tracker never
|
|
205
|
+
creates labels, so a repository missing a required label fails the batch closed
|
|
206
|
+
before any remote write. The Child Agent Brief includes a direct Parent Issue link.
|
|
193
207
|
Internal role prompts, tool policies, review gates, model reservations, and
|
|
194
208
|
prompt digests never belong in this external handoff. If
|
|
195
209
|
`docs/initiatives/<slug>.md` exists, publication fails with a carrier conflict;
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: imm-review-retro
|
|
3
|
+
description: Use when the user explicitly requests Immune-Brain ranking of models by cross-model review load or a project usage retro.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Immune-Brain: Review Retro
|
|
7
|
+
|
|
8
|
+
Rank models by how much code review their own edits triggered, and report
|
|
9
|
+
basic project usage over a look-back window the user supplies in days. This
|
|
10
|
+
is a standalone host-native analysis entry, not a Managed Path continuation
|
|
11
|
+
and not an `imm-loop` internal-role dispatch. It reviews no diff — a diff
|
|
12
|
+
review is `code-review`.
|
|
13
|
+
|
|
14
|
+
## Boundary
|
|
15
|
+
|
|
16
|
+
Allowed: read pi session JSONL under `~/.pi/agent/sessions` (or `--root`),
|
|
17
|
+
run the bundled analyzer, and write a stdout report.
|
|
18
|
+
|
|
19
|
+
Blocked: code, test, Spec, Plan, or `.imm/` edits; session-log writes;
|
|
20
|
+
Kernel, TaskIntent, or TaskRecord mutation; Compounder or scheduled runs;
|
|
21
|
+
`.imm/audit/` lifecycle statistics.
|
|
22
|
+
|
|
23
|
+
An already active Managed task remains owned by `imm-loop`. This Skill does
|
|
24
|
+
not create or resume Managed authority.
|
|
25
|
+
|
|
26
|
+
## Invocation
|
|
27
|
+
|
|
28
|
+
Requires explicit invocation: `imm-review-retro` or `/imm-review-retro`.
|
|
29
|
+
Ordinary questions such as "which model is worse" stay host-native and do
|
|
30
|
+
not enter this Skill.
|
|
31
|
+
|
|
32
|
+
The look-back window in days is required input. If the user named one, use
|
|
33
|
+
it. If not, ask before running, because the ranking moves with the window.
|
|
34
|
+
|
|
35
|
+
Default scan is the user's full session-log tree. Pass `--project <substr>`
|
|
36
|
+
when the user wants one repo or worktree. Do not invent a project filter.
|
|
37
|
+
|
|
38
|
+
No daemon, no cron, no CI, no automatic commit.
|
|
39
|
+
|
|
40
|
+
## Counting rules
|
|
41
|
+
|
|
42
|
+
These rules keep numbers comparable across runs. Read the analyzer header
|
|
43
|
+
aloud in the report so the 口径 stays visible.
|
|
44
|
+
|
|
45
|
+
- `review` = an `Agent` tool call with `subagent_type` equal to `Review`.
|
|
46
|
+
- Attribution = the model behind the most recent `edit`, `write`, or
|
|
47
|
+
`multiedit` in that session. If none, the row is `no-edit (review-only)`.
|
|
48
|
+
- `uniq` counts distinct (session, description+prompt prefix) pairs. A wide
|
|
49
|
+
gap versus `reviews` is the same review re-run on the same code.
|
|
50
|
+
- `rev/100ed` is `100 * reviews / devEdits`. Rank on both absolute `reviews`
|
|
51
|
+
and this intensity. A model can lead one axis and sit mid-pack on the
|
|
52
|
+
other.
|
|
53
|
+
- `avgSc` / `pass%` parse `[SCORE: …]` and `[VERDICT: …]` tags from the
|
|
54
|
+
matching Review `toolResult`. Untagged reviews show `-`.
|
|
55
|
+
- `registr` counts `imm_kernel_canary` `submit_review`. It is the
|
|
56
|
+
registration of the same review and is never added into `reviews`.
|
|
57
|
+
- `rounds/task` is registrations per distinct `(cwd, task_id)`. High values
|
|
58
|
+
can be canary/QA harness re-registration, not human-visible rework.
|
|
59
|
+
- Findings are `record_finding` calls, deduped per session. Summaries that
|
|
60
|
+
match `recorded cleanly`, `receipt recorded`, `round recorded`, or
|
|
61
|
+
`no finding(s)` are `bookkeep` / `noisy`, excluded from `block`/`advis`.
|
|
62
|
+
|
|
63
|
+
Usage counters on the same pass: sessions with activity, assistant turns,
|
|
64
|
+
edit counts, a tool-call name histogram, and the project × author table.
|
|
65
|
+
|
|
66
|
+
## CLI
|
|
67
|
+
|
|
68
|
+
Run the bundled analyzer. Prefer `bun`; `node` (≥23.6, type stripping) is
|
|
69
|
+
an allowed equivalent. The script is erasable TypeScript with `node:` APIs
|
|
70
|
+
only.
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
bun "<path-to-skill>/scripts/review_retro.ts" <days> [--root <sessions-dir>] [--project <substr>] [--top N]
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
- `<days>` must be `> 0`.
|
|
77
|
+
- `--root` defaults to `~/.pi/agent/sessions`.
|
|
78
|
+
- `--project` keeps sessions whose `cwd` contains the substring.
|
|
79
|
+
- `--top` is the project-table row cap (default 15).
|
|
80
|
+
- Malformed JSONL lines are skipped. `days <= 0` is a hard error.
|
|
81
|
+
|
|
82
|
+
Do not scan live `.imm/` directories. Tests use committed fixtures under
|
|
83
|
+
`tests/fixtures/review-retro/`.
|
|
84
|
+
|
|
85
|
+
## Report
|
|
86
|
+
|
|
87
|
+
Write-up order:
|
|
88
|
+
|
|
89
|
+
1. Window and 口径 in one line (copy the analyzer header).
|
|
90
|
+
2. Ranked model table, including scores.
|
|
91
|
+
3. Usage section: sessions, turns, edits, tool mix.
|
|
92
|
+
4. Quality and score findings.
|
|
93
|
+
5. Three to five bullets of what the table means (volume versus intensity,
|
|
94
|
+
quality versus rework, where it concentrated).
|
|
95
|
+
6. Caveats last.
|
|
96
|
+
|
|
97
|
+
Rank on both axes, never one. Name the axis you are ranking by, and call
|
|
98
|
+
out models that flip order between `reviews` and `rev/100ed`.
|
|
99
|
+
|
|
100
|
+
Separate one-pass from rework: compare `uniq` to `reviews`, and read
|
|
101
|
+
`rounds/task` on the kernel path.
|
|
102
|
+
|
|
103
|
+
Evaluate quality: high intensity plus high score is frequent review of
|
|
104
|
+
mostly minor issues; low intensity plus low score is rare review of severe
|
|
105
|
+
defects. Call out REJECT or highRisk ratings.
|
|
106
|
+
|
|
107
|
+
Ground each model in its projects. Cite the two or three worktrees where
|
|
108
|
+
that model's reviews concentrated.
|
|
109
|
+
|
|
110
|
+
## Caveats
|
|
111
|
+
|
|
112
|
+
Anything the script splits out as `bookkeep` stays visible next to the
|
|
113
|
+
column it contaminates. Flag any finding count you cannot trace to a real
|
|
114
|
+
defect.
|
|
115
|
+
|
|
116
|
+
High `rounds/task` can be canary/QA harness re-registration, not
|
|
117
|
+
human-visible rework.
|
|
118
|
+
|
|
119
|
+
This Skill does not persist snapshots or compute week-over-week diffs.
|
|
120
|
+
Re-run with a new window when the user wants a later period.
|
|
121
|
+
|
|
122
|
+
The personal python prototype under `~/.pi/agent/skills/review-retro/` is
|
|
123
|
+
not this Skill and is not modified by it.
|
|
@@ -56,3 +56,12 @@ skills:
|
|
|
56
56
|
output_artifacts: [maintain_report]
|
|
57
57
|
next_actions: []
|
|
58
58
|
boundary: Minimize tracked agent-instruction context after explicit manifest approval; no Managed authority mutation, contract installation, or reference-document creation.
|
|
59
|
+
- name: imm-review-retro
|
|
60
|
+
path: skills/imm-review-retro/SKILL.md
|
|
61
|
+
role: execute
|
|
62
|
+
title: Review Retro
|
|
63
|
+
role_class: discovery
|
|
64
|
+
canonical: true
|
|
65
|
+
output_artifacts: [retro_report]
|
|
66
|
+
next_actions: []
|
|
67
|
+
boundary: Rank models by review load and report project usage from pi session logs. Read-only. No Managed Path mutation.
|
|
@@ -14,6 +14,15 @@ const GH_TIMEOUT_MS = 20_000;
|
|
|
14
14
|
const MAX_SNAPSHOT_PAGES = 100;
|
|
15
15
|
const GITHUB_ISSUE_BODY_LIMIT = 65_536;
|
|
16
16
|
const MAX_TERMINAL_EVENT_ID = 500;
|
|
17
|
+
const MAX_TITLE_LENGTH = 80;
|
|
18
|
+
const MAX_DISPLAY_SHORT_NAME = 32;
|
|
19
|
+
const MAX_DISPLAY_TITLE = 60;
|
|
20
|
+
const READY_FOR_AGENT_LABEL = "ready-for-agent";
|
|
21
|
+
const BLOCKED_LABEL = "blocked";
|
|
22
|
+
const MANAGED_ISSUE_LABELS: readonly string[] = [READY_FOR_AGENT_LABEL, BLOCKED_LABEL];
|
|
23
|
+
|
|
24
|
+
/** Single compressed Activity/Authority footer shared by Parent and Task Issues. */
|
|
25
|
+
const ISSUE_FOOTER = "---\n\n_Outbound visibility only: GitHub state never authorizes or settles work — Kernel TaskIntent, TaskRecord, QA, Review, and Assurance remain the execution authority. An Open Issue only means the Task still needs attention; only a claimless terminal projection closes it (`done` → Completed, `stopped` → Not planned)._";
|
|
17
26
|
|
|
18
27
|
type TrackerStatus =
|
|
19
28
|
| "created"
|
|
@@ -42,6 +51,9 @@ export interface InitiativeSlice {
|
|
|
42
51
|
}
|
|
43
52
|
|
|
44
53
|
export interface InitiativeProjection {
|
|
54
|
+
short_name?: string;
|
|
55
|
+
title?: string;
|
|
56
|
+
source_issue?: string;
|
|
45
57
|
problem?: string;
|
|
46
58
|
result?: string;
|
|
47
59
|
design?: string;
|
|
@@ -128,6 +140,9 @@ export interface GithubInitiativeObservation {
|
|
|
128
140
|
}
|
|
129
141
|
|
|
130
142
|
export interface TaskProjection {
|
|
143
|
+
short_name?: string;
|
|
144
|
+
title?: string;
|
|
145
|
+
slice_ordinal?: number;
|
|
131
146
|
result?: string;
|
|
132
147
|
current_behavior?: string;
|
|
133
148
|
desired_behavior?: string;
|
|
@@ -188,6 +203,7 @@ interface GithubIssue {
|
|
|
188
203
|
body: string;
|
|
189
204
|
state: "open" | "closed";
|
|
190
205
|
state_reason: string | null;
|
|
206
|
+
labels: string[];
|
|
191
207
|
}
|
|
192
208
|
|
|
193
209
|
interface RepositorySnapshot {
|
|
@@ -276,12 +292,120 @@ function titleText(value: string): string {
|
|
|
276
292
|
return redactSecrets(value).replace(/\s+/g, " ");
|
|
277
293
|
}
|
|
278
294
|
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
295
|
+
/**
|
|
296
|
+
* A Planner-supplied display name for an Issue title. Display names are the
|
|
297
|
+
* only title source: the full goal/result prose stays in the body, and a
|
|
298
|
+
* missing or oversized display name fails closed instead of being truncated.
|
|
299
|
+
*/
|
|
300
|
+
function displayName(value: unknown, name: string, max: number): string {
|
|
301
|
+
if (typeof value !== "string" || !value.trim())
|
|
302
|
+
throw new Error(`${name} is required: publish a bounded display name instead of the full goal text`);
|
|
303
|
+
// Brackets are rejected on the raw value: redaction later introduces its own
|
|
304
|
+
// bracketed marker, which must never be mistaken for caller-supplied syntax.
|
|
305
|
+
if (/[[\]]/.test(value)) throw new Error(`${name} must not contain square brackets`);
|
|
306
|
+
return projectionText(value, name, max);
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
function issueTitle(value: string): string {
|
|
310
|
+
const title = titleText(value);
|
|
311
|
+
if (title.length > MAX_TITLE_LENGTH)
|
|
312
|
+
throw new Error(`GitHub Issue title must not exceed ${MAX_TITLE_LENGTH} characters: shorten the Planner display names`);
|
|
282
313
|
return title;
|
|
283
314
|
}
|
|
284
315
|
|
|
316
|
+
function initiativeDisplayNames(projection: InitiativeProjection | undefined): { shortName: string } {
|
|
317
|
+
const missing = (["short_name", "title"] as const).filter((field) => projection?.[field] === undefined);
|
|
318
|
+
if (missing.length)
|
|
319
|
+
throw new Error(`Initiative projection requires display names; missing ${missing.map((field) => `projection.${field}`).join(", ")}`);
|
|
320
|
+
return { shortName: displayName(projection?.short_name, "projection.short_name", MAX_DISPLAY_SHORT_NAME) };
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
function initiativeIssueTitle(initiativeId: string, projection: InitiativeProjection | undefined): string {
|
|
324
|
+
const { shortName } = initiativeDisplayNames(projection);
|
|
325
|
+
const title = displayName(projection?.title, "projection.title", MAX_DISPLAY_TITLE);
|
|
326
|
+
return issueTitle(`[${shortName}] ${title}`);
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
function taskDisplayNames(operation: Extract<TrackerOperation, { op: "upsert-task" }>): { shortName: string; title: string; ordinal: number } {
|
|
330
|
+
const projection = operation.projection;
|
|
331
|
+
if (projection?.short_name === undefined || projection.title === undefined || projection.slice_ordinal === undefined)
|
|
332
|
+
throw new Error(`Task ${operation.task_id} requires projection.short_name, projection.title, and projection.slice_ordinal display names`);
|
|
333
|
+
return {
|
|
334
|
+
shortName: displayName(projection.short_name, "projection.short_name", MAX_DISPLAY_SHORT_NAME),
|
|
335
|
+
title: displayName(projection.title, "projection.title", MAX_DISPLAY_TITLE),
|
|
336
|
+
ordinal: projection.slice_ordinal,
|
|
337
|
+
};
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
function taskIssueTitle(operation: Extract<TrackerOperation, { op: "upsert-task" }>, fallbackOrdinal?: number): string {
|
|
341
|
+
const { shortName, title, ordinal } = taskDisplayNames(operation);
|
|
342
|
+
return issueTitle(`[${shortName}] S${fallbackOrdinal ?? ordinal} ${title}`);
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/**
|
|
346
|
+
* The Slice position in the Initiative: the Parent's Slices checklist is the
|
|
347
|
+
* declared order, so a Child keeps the same `S<n>` across amendment batches
|
|
348
|
+
* instead of renumbering to its position inside the current batch.
|
|
349
|
+
*/
|
|
350
|
+
function sliceOrdinalFromChecklist(parentBody: string, sliceId: string, fallback: number): number {
|
|
351
|
+
const declared = [...parentBody.matchAll(/^- \[[ xX]\] <!-- immune-brain:slice-id=([A-Za-z0-9._:-]+) -->/gm)].map((match) => match[1]);
|
|
352
|
+
const index = declared.indexOf(sliceId);
|
|
353
|
+
return index === -1 ? fallback : index + 1;
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
/** Labels a published Task Issue must carry; the Parent carries none of them. */
|
|
357
|
+
function desiredTaskLabels(operation: Extract<TrackerOperation, { op: "upsert-task" }>): string[] {
|
|
358
|
+
return (operation.projection?.blocked_by ?? []).length
|
|
359
|
+
? [READY_FOR_AGENT_LABEL, BLOCKED_LABEL]
|
|
360
|
+
: [READY_FOR_AGENT_LABEL];
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
/** Converge managed labels from the observed set: add what is desired, remove only managed labels that are not. */
|
|
364
|
+
function labelMutationArgs(observed: string[], desired: string[]): string[] {
|
|
365
|
+
const args: string[] = [];
|
|
366
|
+
for (const label of desired) if (!observed.includes(label)) args.push("--add-label", label);
|
|
367
|
+
for (const label of MANAGED_ISSUE_LABELS)
|
|
368
|
+
if (observed.includes(label) && !desired.includes(label)) args.push("--remove-label", label);
|
|
369
|
+
return args;
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
async function repositoryLabels(root: string, gh: GhTransport, repository: RepositoryInfo): Promise<string[] | GithubTrackerResult> {
|
|
373
|
+
const execution = await gh.run(
|
|
374
|
+
["label", "list", "--repo", repository.name_with_owner, "--json", "name", "--limit", "1000"],
|
|
375
|
+
{ cwd: root },
|
|
376
|
+
);
|
|
377
|
+
if (execution.exit_code !== 0 || execution.output_exceeded)
|
|
378
|
+
return ghFailure("upsert-task", execution, "cannot query repository labels");
|
|
379
|
+
try {
|
|
380
|
+
const parsed = JSON.parse(execution.stdout) as unknown;
|
|
381
|
+
if (!Array.isArray(parsed)) throw new Error("gh returned malformed label list");
|
|
382
|
+
return parsed
|
|
383
|
+
.map((item) => (item as { name?: unknown })?.name)
|
|
384
|
+
.filter((name): name is string => typeof name === "string");
|
|
385
|
+
} catch (error) {
|
|
386
|
+
return result("upsert-task", "permanent_failure", error instanceof Error ? error.message : String(error));
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
/**
|
|
391
|
+
* Labels are never created by the tracker: a repository missing a required
|
|
392
|
+
* label fails closed before any Issue mutation, naming the exact label.
|
|
393
|
+
*/
|
|
394
|
+
async function labelAvailabilityFailure(
|
|
395
|
+
root: string,
|
|
396
|
+
gh: GhTransport,
|
|
397
|
+
repository: RepositoryInfo,
|
|
398
|
+
required: string[],
|
|
399
|
+
): Promise<GithubTrackerResult | null> {
|
|
400
|
+
if (!required.length) return null;
|
|
401
|
+
const labels = await repositoryLabels(root, gh, repository);
|
|
402
|
+
if (!Array.isArray(labels)) return labels;
|
|
403
|
+
const missing = required.filter((label) => !labels.includes(label));
|
|
404
|
+
return missing.length
|
|
405
|
+
? result("upsert-task", "permanent_failure", `repository labels missing: ${missing.join(", ")}; create them before publishing so Task Issues carry publication state`)
|
|
406
|
+
: null;
|
|
407
|
+
}
|
|
408
|
+
|
|
285
409
|
export function redactGithubDiagnostic(value: string): string {
|
|
286
410
|
return redactSecrets(value)
|
|
287
411
|
.replace(/\s+/g, " ")
|
|
@@ -414,6 +538,11 @@ function parseIssues(raw: string): GithubIssue[] {
|
|
|
414
538
|
body: typeof item.body === "string" ? item.body : "",
|
|
415
539
|
state: item.state,
|
|
416
540
|
state_reason: typeof item.state_reason === "string" ? item.state_reason.toLowerCase() : null,
|
|
541
|
+
labels: Array.isArray(item.labels)
|
|
542
|
+
? item.labels
|
|
543
|
+
.map((label) => (typeof label === "string" ? label : (label as { name?: unknown })?.name))
|
|
544
|
+
.filter((name): name is string => typeof name === "string")
|
|
545
|
+
: [],
|
|
417
546
|
};
|
|
418
547
|
});
|
|
419
548
|
}
|
|
@@ -721,12 +850,15 @@ function createInitiativeBody(
|
|
|
721
850
|
historicalSlices: string[] = [],
|
|
722
851
|
): string {
|
|
723
852
|
const projection = operation.projection ?? {};
|
|
853
|
+
const provenance = projection.source_issue
|
|
854
|
+
? `## Provenance\n\n- Derived from #${projection.source_issue}: the originating feature Issue for this Initiative.\n\n`
|
|
855
|
+
: "";
|
|
724
856
|
return `${[
|
|
725
857
|
PROTOCOL_MARKER,
|
|
726
858
|
KIND_INITIATIVE_MARKER,
|
|
727
859
|
marker("repo-id", repository.id),
|
|
728
860
|
marker("initiative-id", operation.initiative_id),
|
|
729
|
-
].join("\n")}\n\n
|
|
861
|
+
].join("\n")}\n\n${provenance}## How to use this Issue\n\n- Edit planning prose and Slice ordering directly after creation.\n- Keep each Slice marker attached to exactly one stable Slice entry.\n- The tracker never rewrites or closes this Parent after creation; the tracker never changes or closes it automatically.\n\n## Problem\n\n${publicText(projection.problem ?? "The Initiative addresses the bounded delivery described below.", "projection.problem")}\n\n## Result\n\n${publicText(projection.result ?? operation.goal, "projection.result")}\n\n## Initiative design\n\n${publicText(projection.design ?? "Each Child preserves the shared Initiative decisions and boundaries recorded here.", "projection.design")}\n\n## Decisions\n\n${listText(projection.decisions, "- No additional Initiative decisions recorded.")}\n\n## Testing strategy\n\n${publicText(projection.testing_strategy ?? "Each Child closes from its focused acceptance verification.", "projection.testing_strategy")}\n\n## Out of scope\n\n${listText(projection.out_of_scope, "- Unrelated work outside this Initiative.")}\n\n## Slices\n\n${operation.slices.length + historicalSlices.length === 0 ? "No Slices recorded yet." : [...historicalSlices, ...operation.slices.map((slice) => `- [ ] ${marker("slice-id", slice.id)} **${slice.id}**: ${slice.result ?? slice.goal}${slice.blocked_by?.length ? ` (blocked by: ${slice.blocked_by.join(", ")})` : ""}`)].join("\n")}\n\n${ISSUE_FOOTER}\n`;
|
|
730
862
|
}
|
|
731
863
|
|
|
732
864
|
async function createInitiative(
|
|
@@ -741,7 +873,7 @@ async function createInitiative(
|
|
|
741
873
|
const body = createInitiativeBody(source.repository, operation);
|
|
742
874
|
const oversized = bodyLimitFailure(operation.op, body);
|
|
743
875
|
if (oversized) return oversized;
|
|
744
|
-
const title =
|
|
876
|
+
const title = initiativeIssueTitle(operation.initiative_id, operation.projection);
|
|
745
877
|
if (found.kind === "missing") {
|
|
746
878
|
const mutation = await gh.run([
|
|
747
879
|
"issue", "create", "--repo", source.repository.name_with_owner,
|
|
@@ -761,8 +893,28 @@ async function createInitiative(
|
|
|
761
893
|
? result(operation.op, "created", "Initiative Issue created as the single GitHub source", confirmed.issue)
|
|
762
894
|
: result(operation.op, "retryable_failure", "Initiative Issue did not converge to the requested initial title and body", confirmed.issue);
|
|
763
895
|
}
|
|
764
|
-
if (found.issue.body === body && found.issue.title === title)
|
|
896
|
+
if (found.issue.body === body && found.issue.title === title) {
|
|
897
|
+
// A repeated complete batch is a convergence pass: managed labels the Parent
|
|
898
|
+
// must not carry are repaired here, without rewriting its content.
|
|
899
|
+
const labelArgs = labelMutationArgs(found.issue.labels, []);
|
|
900
|
+
if (labelArgs.length) {
|
|
901
|
+
const edited = await gh.run([
|
|
902
|
+
"issue", "edit", String(found.issue.number), "--repo", source.repository.name_with_owner,
|
|
903
|
+
...labelArgs,
|
|
904
|
+
], { cwd: root });
|
|
905
|
+
if (edited.exit_code !== 0 || edited.output_exceeded)
|
|
906
|
+
return ghFailure(operation.op, edited, "Initiative Parent label convergence failed");
|
|
907
|
+
const refreshed = await snapshot(root, gh, operation.op);
|
|
908
|
+
if ("contract" in refreshed) return refreshed;
|
|
909
|
+
const confirmed = lookup(refreshed.issues);
|
|
910
|
+
if (confirmed.kind !== "found")
|
|
911
|
+
return result(operation.op, "ambiguous_remote_state", "Initiative Parent became ambiguous after label convergence", found.issue);
|
|
912
|
+
return confirmed.issue.title === title && confirmed.issue.body === body && !labelMutationArgs(confirmed.issue.labels, []).length
|
|
913
|
+
? result(operation.op, "updated", "Initiative Issue managed labels converged", confirmed.issue)
|
|
914
|
+
: result(operation.op, "retryable_failure", "Initiative Issue did not converge to an unlabeled Parent", confirmed.issue);
|
|
915
|
+
}
|
|
765
916
|
return result(operation.op, "already_current", "Initiative Issue already carries the requested initial source", found.issue);
|
|
917
|
+
}
|
|
766
918
|
return result(
|
|
767
919
|
operation.op,
|
|
768
920
|
"permanent_failure",
|
|
@@ -785,7 +937,7 @@ function childBody(
|
|
|
785
937
|
marker("initiative-id", operation.initiative_id),
|
|
786
938
|
marker("slice-id", operation.slice_id),
|
|
787
939
|
marker("task-id", operation.task_id),
|
|
788
|
-
].join("\n")}\n\n
|
|
940
|
+
].join("\n")}\n\n## Parent\n\n| Initiative | \`${operation.initiative_id}\` |\n| Parent Issue | [#${parent.number}](${parent.url}) |\n| Slice | \`${operation.slice_id}\` |\n| Risk | \`${operation.risk}\` |\n\n## Current behavior\n\n${publicText(projection.current_behavior ?? "The current behavior is defined by the repository's existing contract.", "projection.current_behavior")}\n\n## Desired behavior\n\n${publicText(projection.desired_behavior ?? projection.result ?? operation.goal, "projection.desired_behavior")}\n\n## Key interfaces\n\n${listText(projection.key_interfaces, "- Canonical TaskIntent acceptance and Kernel lifecycle remain authoritative.")}\n\n## Acceptance criteria\n\n${acceptance}\n\n## Verification\n\n${publicText(projection.verification ?? "Run the focused acceptance verification declared by the TaskIntent.", "projection.verification")}\n\n## Blocked by\n\n${projection.blocked_by?.length ? projection.blocked_by.map((id) => `- \`${identifier(id, "blocked_by task_id")}\``).join("\n") : "None"}\n\n## Out of scope\n\n${listText(projection.out_of_scope, "- Scope not declared by the validated TaskIntent.")}\n\n## Agent handoff\n\n${publicText(projection.agent_handoff ?? "Implement only the bounded TaskIntent result and run the focused checks. Do not widen scope or treat GitHub as authorization.", "projection.agent_handoff")}\n\n${ISSUE_FOOTER}\n`;
|
|
789
941
|
}
|
|
790
942
|
|
|
791
943
|
/** Derived approved-final content and baseline-derived historical evidence for an amendment. */
|
|
@@ -846,28 +998,30 @@ function approvedAmendmentContent(
|
|
|
846
998
|
);
|
|
847
999
|
const historicalSlices = baselineHistoricalSlices(amendment.parent.body, amendedSliceIds, boundSliceIds);
|
|
848
1000
|
if (typeof historicalSlices === "string") return historicalSlices;
|
|
849
|
-
const pendingContent = new Map<string, { title: string; body: string }>();
|
|
850
1001
|
const sourceStub = {
|
|
851
1002
|
repository,
|
|
852
1003
|
issues: [parentIssue],
|
|
853
1004
|
};
|
|
854
1005
|
void sourceStub;
|
|
1006
|
+
// The approved final Parent body fixes the Slices checklist, so every Child
|
|
1007
|
+
// title is numbered from the Initiative order rather than from this batch.
|
|
1008
|
+
const parentBody = createInitiativeBody(sourceStub.repository, parent, historicalSlices);
|
|
1009
|
+
const oversizedParent = bodyLimitFailure("create-initiative", parentBody);
|
|
1010
|
+
if (oversizedParent) return `${oversizedParent.status}: ${oversizedParent.message}`;
|
|
1011
|
+
const pendingContent = new Map<string, { title: string; body: string }>();
|
|
855
1012
|
for (const operation of prepared.order) {
|
|
856
1013
|
const body = childBody(repository, operation, parentIssue);
|
|
857
1014
|
const oversized = bodyLimitFailure("upsert-task", body, MAX_TERMINAL_SUFFIX_BYTES);
|
|
858
1015
|
if (oversized) return `${oversized.status}: ${oversized.message}`;
|
|
859
1016
|
pendingContent.set(operation.task_id, {
|
|
860
|
-
title:
|
|
1017
|
+
title: taskIssueTitle(operation, sliceOrdinalFromChecklist(parentBody, operation.slice_id, operation.projection?.slice_ordinal ?? 1)),
|
|
861
1018
|
body,
|
|
862
1019
|
});
|
|
863
1020
|
}
|
|
864
|
-
const parentBody = createInitiativeBody(sourceStub.repository, parent, historicalSlices);
|
|
865
|
-
const oversizedParent = bodyLimitFailure("create-initiative", parentBody);
|
|
866
|
-
if (oversizedParent) return `${oversizedParent.status}: ${oversizedParent.message}`;
|
|
867
1021
|
return {
|
|
868
1022
|
pendingContent,
|
|
869
1023
|
parent: {
|
|
870
|
-
title:
|
|
1024
|
+
title: initiativeIssueTitle(parent.initiative_id, parent.projection),
|
|
871
1025
|
body: parentBody,
|
|
872
1026
|
},
|
|
873
1027
|
parentIssueNumber: parentIssue.number,
|
|
@@ -939,20 +1093,29 @@ async function amendInitiativeParent(
|
|
|
939
1093
|
return result(operation.op, "ambiguous_remote_state", "an amendment requires the Initiative Parent to remain open", found.issue);
|
|
940
1094
|
if (binding.issue_number !== found.issue.number)
|
|
941
1095
|
return result(operation.op, "ambiguous_remote_state", `amendment Parent is bound to Issue #${binding.issue_number} but observed Issue #${found.issue.number}`, found.issue);
|
|
942
|
-
if (found.issue.body === body && found.issue.title === title)
|
|
1096
|
+
if (found.issue.body === body && found.issue.title === title && !labelMutationArgs(found.issue.labels, []).length)
|
|
943
1097
|
return result(operation.op, "already_current", "Initiative Issue already carries the requested amended content", found.issue);
|
|
944
|
-
|
|
1098
|
+
// Content that already matches the approved final bytes still converges labels;
|
|
1099
|
+
// anything else must still be the approved amendment baseline.
|
|
1100
|
+
if ((found.issue.body !== body || found.issue.title !== title)
|
|
1101
|
+
&& (found.issue.body !== binding.body || found.issue.title !== binding.title))
|
|
945
1102
|
return result(operation.op, "ambiguous_remote_state", "Initiative Parent changed since the approved amendment baseline", found.issue);
|
|
1103
|
+
// The Parent carries no Task state labels: any managed label observed on it is
|
|
1104
|
+
// drift and is removed by the same edit that writes the amended content.
|
|
1105
|
+
const labelArgs = labelMutationArgs(found.issue.labels, []);
|
|
946
1106
|
const edited = await gh.run([
|
|
947
1107
|
"issue", "edit", String(found.issue.number), "--repo", source.repository.name_with_owner,
|
|
948
1108
|
"--title", title,
|
|
949
1109
|
"--body-file", "-",
|
|
1110
|
+
...labelArgs,
|
|
950
1111
|
], { cwd: root, stdin: body });
|
|
951
1112
|
if (edited.exit_code !== 0 || edited.output_exceeded) return ghFailure(operation.op, edited, "Initiative amendment edit failed");
|
|
952
1113
|
const refreshed = await snapshot(root, gh, operation.op);
|
|
953
1114
|
if ("contract" in refreshed) return refreshed;
|
|
954
1115
|
const confirmed = lookup(refreshed.issues);
|
|
955
1116
|
if (confirmed.kind !== "found") return result(operation.op, "ambiguous_remote_state", "Initiative Parent became ambiguous after amendment", found.issue);
|
|
1117
|
+
if (MANAGED_ISSUE_LABELS.some((label) => confirmed.issue.labels.includes(label)))
|
|
1118
|
+
return result(operation.op, "retryable_failure", "Initiative amendment did not converge to an unlabeled Parent", confirmed.issue);
|
|
956
1119
|
return confirmed.issue.body === body && confirmed.issue.title === title
|
|
957
1120
|
? result(operation.op, "updated", "Initiative Issue updated with approved amendment content", confirmed.issue)
|
|
958
1121
|
: result(operation.op, "retryable_failure", "Initiative amendment did not converge to the requested title and body", confirmed.issue);
|
|
@@ -966,6 +1129,8 @@ async function upsertTask(
|
|
|
966
1129
|
pendingBinding: InitiativeAmendmentBinding | undefined | null = null,
|
|
967
1130
|
amendmentContext: AmendmentExecutionContext | undefined = undefined,
|
|
968
1131
|
): Promise<GithubTrackerResult> {
|
|
1132
|
+
const labelFailure = await labelAvailabilityFailure(root, gh, source.repository, desiredTaskLabels(operation));
|
|
1133
|
+
if (labelFailure) return labelFailure;
|
|
969
1134
|
const parent = initiativeLookup(source.issues, source.repository.id, operation.initiative_id);
|
|
970
1135
|
if (parent.kind === "ambiguous") return result(operation.op, "ambiguous_remote_state", parent.message);
|
|
971
1136
|
if (parent.kind === "missing")
|
|
@@ -997,9 +1162,10 @@ async function upsertTask(
|
|
|
997
1162
|
const body = childBody(source.repository, operation, parent.issue);
|
|
998
1163
|
const oversized = bodyLimitFailure(operation.op, body, MAX_TERMINAL_SUFFIX_BYTES);
|
|
999
1164
|
if (oversized) return oversized;
|
|
1000
|
-
const title =
|
|
1165
|
+
const title = taskIssueTitle(operation, sliceOrdinalFromChecklist(parent.issue.body, operation.slice_id, operation.projection?.slice_ordinal ?? 1));
|
|
1001
1166
|
let child: GithubIssue;
|
|
1002
1167
|
let createdChild = false;
|
|
1168
|
+
let labelsConverged = false;
|
|
1003
1169
|
if (found.kind === "missing") {
|
|
1004
1170
|
// Pre-create re-read (amendment path): a concurrent writer may have already
|
|
1005
1171
|
// created this unbound Task between the initial snapshot and our create — the
|
|
@@ -1039,6 +1205,7 @@ async function upsertTask(
|
|
|
1039
1205
|
"issue", "create", "--repo", source.repository.name_with_owner,
|
|
1040
1206
|
"--title", title,
|
|
1041
1207
|
"--body-file", "-",
|
|
1208
|
+
...desiredTaskLabels(operation).flatMap((label) => ["--label", label]),
|
|
1042
1209
|
], { cwd: root, stdin: body });
|
|
1043
1210
|
const refreshed = await snapshot(root, gh, operation.op);
|
|
1044
1211
|
if ("contract" in refreshed) return refreshed;
|
|
@@ -1106,6 +1273,18 @@ async function upsertTask(
|
|
|
1106
1273
|
}
|
|
1107
1274
|
if (found.issue.body !== body || found.issue.title !== title)
|
|
1108
1275
|
return result(operation.op, "permanent_failure", "Task Issue already exists with a different title or Agent Brief; edit the GitHub source or retry the original projection before changing native relations", found.issue);
|
|
1276
|
+
// A repeated complete batch is a convergence pass: managed label drift is
|
|
1277
|
+
// repaired here without rewriting content the tracker never owns.
|
|
1278
|
+
const observedLabelArgs = labelMutationArgs(child.labels, desiredTaskLabels(operation));
|
|
1279
|
+
if (observedLabelArgs.length) {
|
|
1280
|
+
const edited = await gh.run([
|
|
1281
|
+
"issue", "edit", String(child.number), "--repo", source.repository.name_with_owner,
|
|
1282
|
+
...observedLabelArgs,
|
|
1283
|
+
], { cwd: root });
|
|
1284
|
+
if (edited.exit_code !== 0 || edited.output_exceeded)
|
|
1285
|
+
return ghFailure(operation.op, edited, `Task Issue #${child.number} label convergence failed`);
|
|
1286
|
+
labelsConverged = true;
|
|
1287
|
+
}
|
|
1109
1288
|
}
|
|
1110
1289
|
const attachment = await confirmAttachment(root, gh, operation.op, source.repository, parent.issue.number, child.number);
|
|
1111
1290
|
if (!("attached" in attachment)) return attachment;
|
|
@@ -1141,7 +1320,7 @@ async function upsertTask(
|
|
|
1141
1320
|
if (attachment.attached && dependencies.complete)
|
|
1142
1321
|
return createdChild
|
|
1143
1322
|
? result(operation.op, "created", "Task Issue created and attached with native blocking relations", child)
|
|
1144
|
-
: result(operation.op, "already_current", "Task Issue, native Sub-issue relation, and blocking relations are current", child);
|
|
1323
|
+
: result(operation.op, labelsConverged ? "updated" : "already_current", "Task Issue, native Sub-issue relation, and blocking relations are current", child);
|
|
1145
1324
|
return createdChild
|
|
1146
1325
|
? result(operation.op, "created", "Task Issue created and attached as a native Sub-issue", child)
|
|
1147
1326
|
: result(operation.op, "updated", "existing Task Issue attached as a native Sub-issue", child);
|
|
@@ -1272,7 +1451,13 @@ function normalizeProjection(value: TaskProjection | undefined): TaskProjection
|
|
|
1272
1451
|
if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("projection must be an object");
|
|
1273
1452
|
const blockedBy = normalizedProjectionList(value.blocked_by, "projection.blocked_by", 128)?.map((id) => identifier(id, "projection.blocked_by task_id"));
|
|
1274
1453
|
if (blockedBy && new Set(blockedBy).size !== blockedBy.length) throw new Error("projection.blocked_by must not contain duplicate Task IDs");
|
|
1454
|
+
if (value.slice_ordinal !== undefined
|
|
1455
|
+
&& (typeof value.slice_ordinal !== "number" || !Number.isSafeInteger(value.slice_ordinal) || value.slice_ordinal < 1 || value.slice_ordinal > 999))
|
|
1456
|
+
throw new Error("projection.slice_ordinal must be an integer between 1 and 999");
|
|
1275
1457
|
return {
|
|
1458
|
+
short_name: value.short_name,
|
|
1459
|
+
title: value.title,
|
|
1460
|
+
slice_ordinal: value.slice_ordinal,
|
|
1276
1461
|
result: value.result === undefined ? undefined : projectionText(value.result, "projection.result"),
|
|
1277
1462
|
current_behavior: value.current_behavior === undefined ? undefined : projectionText(value.current_behavior, "projection.current_behavior"),
|
|
1278
1463
|
desired_behavior: value.desired_behavior === undefined ? undefined : projectionText(value.desired_behavior, "projection.desired_behavior"),
|
|
@@ -1287,7 +1472,12 @@ function normalizeProjection(value: TaskProjection | undefined): TaskProjection
|
|
|
1287
1472
|
function normalizeInitiativeProjection(value: InitiativeProjection | undefined): InitiativeProjection | undefined {
|
|
1288
1473
|
if (value === undefined) return undefined;
|
|
1289
1474
|
if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("projection must be an object");
|
|
1475
|
+
if (value.source_issue !== undefined && (typeof value.source_issue !== "string" || !/^[1-9][0-9]{0,9}$/.test(value.source_issue)))
|
|
1476
|
+
throw new Error("projection.source_issue must be a GitHub Issue number");
|
|
1290
1477
|
return {
|
|
1478
|
+
short_name: value.short_name,
|
|
1479
|
+
title: value.title,
|
|
1480
|
+
source_issue: value.source_issue,
|
|
1291
1481
|
problem: value.problem === undefined ? undefined : projectionText(value.problem, "projection.problem"),
|
|
1292
1482
|
result: value.result === undefined ? undefined : projectionText(value.result, "projection.result"),
|
|
1293
1483
|
design: value.design === undefined ? undefined : projectionText(value.design, "projection.design"),
|
|
@@ -1318,7 +1508,7 @@ function validateOperation(operation: TrackerOperation): TrackerOperation {
|
|
|
1318
1508
|
};
|
|
1319
1509
|
}),
|
|
1320
1510
|
};
|
|
1321
|
-
|
|
1511
|
+
initiativeIssueTitle(normalized.initiative_id, normalized.projection);
|
|
1322
1512
|
return normalized;
|
|
1323
1513
|
}
|
|
1324
1514
|
if (operation.op === "upsert-task") {
|
|
@@ -1335,7 +1525,7 @@ function validateOperation(operation: TrackerOperation): TrackerOperation {
|
|
|
1335
1525
|
summary: projectionText(item.summary, `acceptance[${index}].summary`, 500),
|
|
1336
1526
|
})),
|
|
1337
1527
|
};
|
|
1338
|
-
|
|
1528
|
+
taskIssueTitle(normalized);
|
|
1339
1529
|
return normalized;
|
|
1340
1530
|
}
|
|
1341
1531
|
return {
|
|
@@ -1446,10 +1636,14 @@ function preflightPublication(root: string, input: InitiativePublicationInput):
|
|
|
1446
1636
|
if (input.projection[field] === undefined)
|
|
1447
1637
|
throw new Error(`a complete Initiative publication requires projection.${field}`);
|
|
1448
1638
|
}
|
|
1639
|
+
// The Initiative display names are validated once here, before any Task
|
|
1640
|
+
// projection is stamped with them, so a missing or oversized display name
|
|
1641
|
+
// fails the batch closed with zero remote writes.
|
|
1642
|
+
const initiativeShortName = initiativeDisplayNames(input.projection).shortName;
|
|
1449
1643
|
const publications = input.tasks.map((task, index) => {
|
|
1450
1644
|
if (!task || typeof task !== "object" || Array.isArray(task)) throw new Error(`tasks[${index}] must be an object`);
|
|
1451
1645
|
if (typeof task.intent !== "string") throw new Error(`tasks[${index}].intent must be a string`);
|
|
1452
|
-
return taskPublication(root, input.initiative_id, task.slice_id, task.intent, task.acceptance, task.projection);
|
|
1646
|
+
return taskPublication(root, input.initiative_id, task.slice_id, task.intent, task.acceptance, task.projection, index + 1, initiativeShortName);
|
|
1453
1647
|
});
|
|
1454
1648
|
const historicalIds = new Set<string>();
|
|
1455
1649
|
if (input.amendment) {
|
|
@@ -1820,7 +2014,7 @@ async function updatePendingChild(
|
|
|
1820
2014
|
const body = childBody(source.repository, op, parent.kind === "found" ? parent.issue : child);
|
|
1821
2015
|
const oversized = bodyLimitFailure(op.op, body, MAX_TERMINAL_SUFFIX_BYTES);
|
|
1822
2016
|
if (oversized) return oversized;
|
|
1823
|
-
const title =
|
|
2017
|
+
const title = taskIssueTitle(op, sliceOrdinalFromChecklist(parent.kind === "found" ? parent.issue.body : child.body, op.slice_id, op.projection?.slice_ordinal ?? 1));
|
|
1824
2018
|
// Parent content expectation for pre-write revalidation: on the amendment
|
|
1825
2019
|
// path the expectation is always the fixed approved final Parent bytes —
|
|
1826
2020
|
// never re-adopt freshly observed content as the baseline, so an edit that
|
|
@@ -1874,7 +2068,8 @@ async function updatePendingChild(
|
|
|
1874
2068
|
// that landed after the snapshot must be preserved exactly, never
|
|
1875
2069
|
// overwritten by bytes computed from stale data.
|
|
1876
2070
|
let observed = child;
|
|
1877
|
-
|
|
2071
|
+
const desiredLabels = desiredTaskLabels(op);
|
|
2072
|
+
if (child.title !== title || child.body !== finalBody || labelMutationArgs(child.labels, desiredLabels).length) {
|
|
1878
2073
|
// Re-read the Child immediately before writing: earlier blocker ownership
|
|
1879
2074
|
// reads may have raced a concurrent user edit. The bound issue_number must
|
|
1880
2075
|
// still hold and the remote content must still be baseline-or-approved-final.
|
|
@@ -1896,11 +2091,13 @@ async function updatePendingChild(
|
|
|
1896
2091
|
const writeBody = typeof observedEvent === "string"
|
|
1897
2092
|
? `${body.trimEnd()}${terminalSuffix(observedEvent)}`
|
|
1898
2093
|
: finalBody;
|
|
1899
|
-
|
|
2094
|
+
const labelArgs = labelMutationArgs(observed.labels, desiredLabels);
|
|
2095
|
+
if (observed.title !== title || observed.body !== writeBody || labelArgs.length) {
|
|
1900
2096
|
const edited = await gh.run([
|
|
1901
2097
|
"issue", "edit", String(child.number), "--repo", source.repository.name_with_owner,
|
|
1902
2098
|
"--title", title,
|
|
1903
2099
|
"--body-file", "-",
|
|
2100
|
+
...labelArgs,
|
|
1904
2101
|
], { cwd: root, stdin: writeBody });
|
|
1905
2102
|
if (edited.exit_code !== 0 || edited.output_exceeded) return ghFailure(op.op, edited, `pending Task Issue #${child.number} update failed`);
|
|
1906
2103
|
finalBody = writeBody;
|
|
@@ -1915,11 +2112,15 @@ async function updatePendingChild(
|
|
|
1915
2112
|
return result(op.op, "ambiguous_remote_state", `pending Task ${op.task_id} changed identity during amendment`, child);
|
|
1916
2113
|
if (reread.issue.title !== title || reread.issue.body !== finalBody)
|
|
1917
2114
|
return result(op.op, "retryable_failure", `pending Task ${op.task_id} update did not converge`, reread.issue);
|
|
2115
|
+
const labelsCurrent = !desiredLabels.some((label) => !reread.issue.labels.includes(label));
|
|
2116
|
+
if (!labelsCurrent)
|
|
2117
|
+
return result(op.op, "retryable_failure", `pending Task ${op.task_id} labels did not converge`, reread.issue);
|
|
1918
2118
|
const currentDependencies = await confirmBlockedBy(root, gh, op.op, refreshed.repository, reread.issue.number, desiredBlockers);
|
|
1919
2119
|
if (!("complete" in currentDependencies)) return currentDependencies;
|
|
1920
2120
|
if (!currentDependencies.complete)
|
|
1921
2121
|
return result(op.op, "retryable_failure", `pending Task ${op.task_id} dependencies did not converge`, reread.issue);
|
|
1922
|
-
const contentCurrent = child.title === title && child.body === finalBody
|
|
2122
|
+
const contentCurrent = child.title === title && child.body === finalBody
|
|
2123
|
+
&& labelMutationArgs(child.labels, desiredLabels).length === 0;
|
|
1923
2124
|
return contentCurrent
|
|
1924
2125
|
? result(op.op, "already_current", `pending Task ${op.task_id} already carries the approved amendment content`, reread.issue)
|
|
1925
2126
|
: result(op.op, "updated", `pending Task ${op.task_id} Agent Brief updated with approved amendment content`, reread.issue);
|
|
@@ -2089,6 +2290,12 @@ export async function runGithubInitiativePublication(
|
|
|
2089
2290
|
if (initialParent.kind === "ambiguous") return publicationResult("ambiguous_remote_state", initialParent.message);
|
|
2090
2291
|
if (amendment && initialParent.kind === "missing")
|
|
2091
2292
|
return publicationResult("permanent_failure", "an amendment requires the Initiative Parent to already exist");
|
|
2293
|
+
// Label availability is validated before any remote mutation: the tracker
|
|
2294
|
+
// never creates labels, so a missing label fails the whole batch closed
|
|
2295
|
+
// instead of half-publishing an Initiative.
|
|
2296
|
+
const requiredLabels = [...new Set(prepared.order.flatMap((operation) => desiredTaskLabels(operation)))];
|
|
2297
|
+
const labelFailure = await labelAvailabilityFailure(absoluteRoot, gh, initial.repository, requiredLabels);
|
|
2298
|
+
if (labelFailure) return publicationResult(labelFailure.status, labelFailure.message, labelFailure);
|
|
2092
2299
|
const parentForPreflight = initialParent.kind === "found" ? initialParent.issue : {
|
|
2093
2300
|
id: Number.MAX_SAFE_INTEGER,
|
|
2094
2301
|
number: Number.MAX_SAFE_INTEGER,
|
|
@@ -2097,6 +2304,7 @@ export async function runGithubInitiativePublication(
|
|
|
2097
2304
|
body: "",
|
|
2098
2305
|
state: "open" as const,
|
|
2099
2306
|
state_reason: null,
|
|
2307
|
+
labels: [],
|
|
2100
2308
|
};
|
|
2101
2309
|
const parentBodyFailure = bodyLimitFailure("create-initiative", createInitiativeBody(initial.repository, prepared.initiative));
|
|
2102
2310
|
if (parentBodyFailure) return publicationResult(parentBodyFailure.status, parentBodyFailure.message, parentBodyFailure);
|
|
@@ -2214,7 +2422,7 @@ export async function runGithubInitiativePublication(
|
|
|
2214
2422
|
const parentDrift = publicationIssueDrift(
|
|
2215
2423
|
parentResult,
|
|
2216
2424
|
parent.issue,
|
|
2217
|
-
amendmentContext ? amendmentContext.parent.title :
|
|
2425
|
+
amendmentContext ? amendmentContext.parent.title : initiativeIssueTitle(prepared.initiative.initiative_id, prepared.initiative.projection),
|
|
2218
2426
|
amendmentContext
|
|
2219
2427
|
? amendmentContext.parent.body
|
|
2220
2428
|
: createInitiativeBody(finalSource.repository, prepared.initiative),
|
|
@@ -2241,12 +2449,15 @@ export async function runGithubInitiativePublication(
|
|
|
2241
2449
|
const childDrift = publicationIssueDrift(
|
|
2242
2450
|
childResult,
|
|
2243
2451
|
child.issue,
|
|
2244
|
-
|
|
2452
|
+
taskIssueTitle(operation, sliceOrdinalFromChecklist(parent.issue.body, operation.slice_id, operation.projection?.slice_ordinal ?? 1)),
|
|
2245
2453
|
childBody(finalSource.repository, operation, parent.issue),
|
|
2246
2454
|
`Task ${operation.task_id}`,
|
|
2247
2455
|
{ allowTerminalSuffix: amendmentContext !== undefined },
|
|
2248
2456
|
);
|
|
2249
2457
|
if (childDrift) return publicationResult("ambiguous_remote_state", childDrift, parentResult, taskResults);
|
|
2458
|
+
const expectedLabels = desiredTaskLabels(operation);
|
|
2459
|
+
if (childResult.status !== "already_current" && expectedLabels.some((label) => !child.issue.labels.includes(label)))
|
|
2460
|
+
return publicationResult("ambiguous_remote_state", `Task ${operation.task_id} is missing publication labels after publication`, parentResult, taskResults);
|
|
2250
2461
|
expectedNumbers.push(child.issue.number);
|
|
2251
2462
|
const ownership = await confirmTerminalOwnership(absoluteRoot, gh, "upsert-task", finalSource, child.issue);
|
|
2252
2463
|
if (!("owned" in ownership)) return publicationResult(ownership.status, ownership.message, parentResult, taskResults);
|
|
@@ -2341,7 +2552,9 @@ function taskPublication(
|
|
|
2341
2552
|
sliceId: string,
|
|
2342
2553
|
intentPath: string,
|
|
2343
2554
|
acceptance: unknown,
|
|
2344
|
-
projection
|
|
2555
|
+
projection: TaskProjection | undefined,
|
|
2556
|
+
ordinal: number,
|
|
2557
|
+
initiativeShortName: string,
|
|
2345
2558
|
): PreparedPublicationTask {
|
|
2346
2559
|
const absoluteRoot = resolve(root);
|
|
2347
2560
|
const absolutePath = resolve(absoluteRoot, intentPath);
|
|
@@ -2367,6 +2580,18 @@ function taskPublication(
|
|
|
2367
2580
|
});
|
|
2368
2581
|
const missingIds = [...expectedIds].filter((id) => !publicById.has(id));
|
|
2369
2582
|
if (missingIds.length) throw new Error(`Task ${taskId} is missing public acceptance ids: ${missingIds.join(", ")}`);
|
|
2583
|
+
// Display names are stamped from the Initiative so every Child title carries
|
|
2584
|
+
// the same short name and its declared Slice position; an explicitly
|
|
2585
|
+
// supplied value must agree, never silently override.
|
|
2586
|
+
if (projection?.short_name !== undefined && projection.short_name !== initiativeShortName)
|
|
2587
|
+
throw new Error(`Task ${taskId} projection.short_name must match the Initiative short name`);
|
|
2588
|
+
if (projection?.slice_ordinal !== undefined && projection.slice_ordinal !== ordinal)
|
|
2589
|
+
throw new Error(`Task ${taskId} projection.slice_ordinal must match its declared Slice position ${ordinal}`);
|
|
2590
|
+
const stampedProjection: TaskProjection = {
|
|
2591
|
+
...(projection ?? {}),
|
|
2592
|
+
short_name: initiativeShortName,
|
|
2593
|
+
slice_ordinal: ordinal,
|
|
2594
|
+
};
|
|
2370
2595
|
return {
|
|
2371
2596
|
operation: validateOperation({
|
|
2372
2597
|
op: "upsert-task",
|
|
@@ -2376,7 +2601,7 @@ function taskPublication(
|
|
|
2376
2601
|
goal: intent.goal,
|
|
2377
2602
|
risk: intent.risk,
|
|
2378
2603
|
acceptance: intent.acceptance.map((item) => publicById.get(item.id)!),
|
|
2379
|
-
projection,
|
|
2604
|
+
projection: stampedProjection,
|
|
2380
2605
|
}) as Extract<TrackerOperation, { op: "upsert-task" }>,
|
|
2381
2606
|
intent_path: read.intent_ref.path,
|
|
2382
2607
|
intent_content_hash: read.content_hash,
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
// Generated by scripts/plugin_versioning.ts from the root package.json.
|
|
2
|
-
export const PLUGIN_VERSION = "3.6.
|
|
2
|
+
export const PLUGIN_VERSION = "3.6.9" as const;
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: imm-review-retro
|
|
3
|
+
description: Use when the user explicitly requests Immune-Brain ranking of models by cross-model review load or a project usage retro.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Immune-Brain: Review Retro
|
|
7
|
+
|
|
8
|
+
Use [`../../dist/imm-review-retro.md`](../../dist/imm-review-retro.md) as the
|
|
9
|
+
canonical contract index, not a whole-document read. This is a
|
|
10
|
+
standalone host-native analysis entry, not a Managed Path continuation
|
|
11
|
+
and not an `imm-loop` internal-role dispatch.
|
|
12
|
+
|
|
13
|
+
Mandatory constraints: read-only on pi session logs. Do not edit code, tests,
|
|
14
|
+
Specs, or workflow state. Do not write session logs or `.imm/` files.
|
|
15
|
+
|
|
16
|
+
Section routes - load a section's instructions only when its branch applies.
|
|
17
|
+
Read each linked heading body up to the next heading; nested sections and
|
|
18
|
+
references load only under their own condition. Never read the whole contract
|
|
19
|
+
or all references as an entry prerequisite.
|
|
20
|
+
|
|
21
|
+
- common: [Boundary](../../dist/imm-review-retro.md#boundary), [Invocation](../../dist/imm-review-retro.md#invocation)
|
|
22
|
+
- running the analyzer: [Counting rules](../../dist/imm-review-retro.md#counting-rules), [CLI](../../dist/imm-review-retro.md#cli)
|
|
23
|
+
- interpreting the report: [Report](../../dist/imm-review-retro.md#report), [Caveats](../../dist/imm-review-retro.md#caveats)
|
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
#!/usr/bin/env bun
|
|
2
|
+
/**
|
|
3
|
+
* Retro: how many code reviews did each model's own code trigger, across recent pi sessions.
|
|
4
|
+
*
|
|
5
|
+
* Usage: bun review_retro.ts <days> [--root <sessions-dir>] [--project <substr>] [--top N]
|
|
6
|
+
*
|
|
7
|
+
* Counting rules (the 口径 that keeps the numbers honest):
|
|
8
|
+
* review executed = Agent(subagent_type="Review") tool call
|
|
9
|
+
* avgSc / pass% = average score (0-10) and PASS rate from [SCORE: ...] tags in Review toolResult
|
|
10
|
+
* kernel:submit_review = the *registration* of that same review, reported separately (never added)
|
|
11
|
+
* attribution = the model behind the most recent edit/write before the review (the code's author)
|
|
12
|
+
* findings = imm_kernel_canary record_finding, deduped per session, harness bookkeeping split out
|
|
13
|
+
*/
|
|
14
|
+
import { createReadStream, existsSync, readdirSync } from "node:fs";
|
|
15
|
+
import { homedir } from "node:os";
|
|
16
|
+
import { join } from "node:path";
|
|
17
|
+
import { createInterface } from "node:readline";
|
|
18
|
+
import { pathToFileURL } from "node:url";
|
|
19
|
+
|
|
20
|
+
const BOOKKEEPING = /recorded cleanly|receipt recorded|round recorded|no findings?\b/i;
|
|
21
|
+
const EDIT_TOOLS = new Set(["edit", "write", "multiedit"]);
|
|
22
|
+
const SCORE_RE = /\[SCORE:\s*([\d.]+)\s*(?:\/\s*10)?\]/i;
|
|
23
|
+
const VERDICT_RE = /\[VERDICT:\s*(\w+)\]/i;
|
|
24
|
+
const RISK_RE = /\[RISK:\s*(\w+)\]/i;
|
|
25
|
+
const BLOCK_RE = /\[BLOCKING:\s*(\d+)\]/i;
|
|
26
|
+
const ADVIS_RE = /\[ADVISORY:\s*(\d+)\]/i;
|
|
27
|
+
|
|
28
|
+
type ReviewTag = {
|
|
29
|
+
score: number;
|
|
30
|
+
verdict: string;
|
|
31
|
+
risk: string;
|
|
32
|
+
blocking: number;
|
|
33
|
+
advisory: number;
|
|
34
|
+
};
|
|
35
|
+
|
|
36
|
+
function parseReviewTag(text: string): ReviewTag | null {
|
|
37
|
+
if (!text.includes("[SCORE:")) return null;
|
|
38
|
+
const sc = SCORE_RE.exec(text);
|
|
39
|
+
if (!sc) return null;
|
|
40
|
+
const score = Number(sc[1]);
|
|
41
|
+
if (!Number.isFinite(score)) return null;
|
|
42
|
+
return {
|
|
43
|
+
score,
|
|
44
|
+
verdict: (VERDICT_RE.exec(text)?.[1] ?? "UNKNOWN").toUpperCase(),
|
|
45
|
+
risk: (RISK_RE.exec(text)?.[1] ?? "UNKNOWN").toUpperCase(),
|
|
46
|
+
blocking: Number(BLOCK_RE.exec(text)?.[1] ?? 0),
|
|
47
|
+
advisory: Number(ADVIS_RE.exec(text)?.[1] ?? 0),
|
|
48
|
+
};
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function extractText(content: unknown): string {
|
|
52
|
+
if (typeof content === "string") return content;
|
|
53
|
+
if (Array.isArray(content)) {
|
|
54
|
+
return content
|
|
55
|
+
.map((item) => {
|
|
56
|
+
if (typeof item === "string") return item;
|
|
57
|
+
if (item && typeof item === "object") {
|
|
58
|
+
const rec = item as Record<string, unknown>;
|
|
59
|
+
if ("text" in rec) return String(rec.text);
|
|
60
|
+
if ("result" in rec) return String(rec.result);
|
|
61
|
+
}
|
|
62
|
+
return "";
|
|
63
|
+
})
|
|
64
|
+
.join("\n");
|
|
65
|
+
}
|
|
66
|
+
return "";
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function norm(a: unknown): Record<string, unknown> {
|
|
70
|
+
if (typeof a === "string") {
|
|
71
|
+
try {
|
|
72
|
+
return JSON.parse(a) as Record<string, unknown>;
|
|
73
|
+
} catch {
|
|
74
|
+
return {};
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
return a && typeof a === "object" ? (a as Record<string, unknown>) : {};
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function bump(map: Map<string, number>, key: string, n = 1): void {
|
|
81
|
+
map.set(key, (map.get(key) ?? 0) + n);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
type Args = { days: number; root: string; project: string; top: number };
|
|
85
|
+
|
|
86
|
+
function parseArgs(argv: string[]): Args {
|
|
87
|
+
const out: Args = {
|
|
88
|
+
days: NaN,
|
|
89
|
+
root: join(homedir(), ".pi/agent/sessions"),
|
|
90
|
+
project: "",
|
|
91
|
+
top: 15,
|
|
92
|
+
};
|
|
93
|
+
const rest: string[] = [];
|
|
94
|
+
for (let i = 0; i < argv.length; i++) {
|
|
95
|
+
const a = argv[i]!;
|
|
96
|
+
if (a === "--root") out.root = argv[++i] ?? "";
|
|
97
|
+
else if (a === "--project") out.project = argv[++i] ?? "";
|
|
98
|
+
else if (a === "--top") out.top = Number(argv[++i]);
|
|
99
|
+
else if (a.startsWith("-")) throw new Error(`unknown flag: ${a}`);
|
|
100
|
+
else rest.push(a);
|
|
101
|
+
}
|
|
102
|
+
out.days = Number(rest[0]);
|
|
103
|
+
if (!(out.days > 0) || rest[1] !== undefined) {
|
|
104
|
+
throw new Error("usage: review_retro.ts <days> [--root <dir>] [--project <substr>] [--top N]");
|
|
105
|
+
}
|
|
106
|
+
if (!Number.isFinite(out.top) || out.top <= 0) throw new Error("--top must be > 0");
|
|
107
|
+
return out;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
function listJsonl(root: string): string[] {
|
|
111
|
+
if (!existsSync(root)) return [];
|
|
112
|
+
const paths: string[] = [];
|
|
113
|
+
for (const dir of readdirSync(root, { withFileTypes: true })) {
|
|
114
|
+
if (!dir.isDirectory()) continue;
|
|
115
|
+
const folder = join(root, dir.name);
|
|
116
|
+
for (const name of readdirSync(folder)) {
|
|
117
|
+
if (name.endsWith(".jsonl")) paths.push(join(folder, name));
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
return paths;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
export async function run(argv: string[]): Promise<string> {
|
|
124
|
+
const args = parseArgs(argv);
|
|
125
|
+
const cut = new Date(Date.now() - args.days * 86400000).toISOString().slice(0, 19);
|
|
126
|
+
const home = homedir();
|
|
127
|
+
const dev = new Map<string, number>();
|
|
128
|
+
const turns = new Map<string, number>();
|
|
129
|
+
const rev = new Map<string, number>();
|
|
130
|
+
const sub = new Map<string, number>();
|
|
131
|
+
const tools = new Map<string, number>();
|
|
132
|
+
const proj = new Map<string, number>();
|
|
133
|
+
const episode = new Map<string, Set<string>>();
|
|
134
|
+
const tasks = new Map<string, Set<string>>();
|
|
135
|
+
const findUniq = new Map<string, Set<string>>();
|
|
136
|
+
const findingsRaw = new Map<string, number>();
|
|
137
|
+
const scores = new Map<string, number[]>();
|
|
138
|
+
const verdicts = new Map<string, number>();
|
|
139
|
+
const risks = new Map<string, number>();
|
|
140
|
+
let files = 0;
|
|
141
|
+
|
|
142
|
+
for (const path of listJsonl(args.root)) {
|
|
143
|
+
let editor: string | null = null;
|
|
144
|
+
let cwd = "";
|
|
145
|
+
let used = false;
|
|
146
|
+
const pending = new Map<string, string>();
|
|
147
|
+
const rl = createInterface({ input: createReadStream(path, { encoding: "utf8" }) });
|
|
148
|
+
for await (const raw of rl) {
|
|
149
|
+
const line = raw.trim();
|
|
150
|
+
if (!line) continue;
|
|
151
|
+
let o: Record<string, unknown>;
|
|
152
|
+
try {
|
|
153
|
+
o = JSON.parse(line) as Record<string, unknown>;
|
|
154
|
+
} catch {
|
|
155
|
+
continue;
|
|
156
|
+
}
|
|
157
|
+
if (o.type === "session") {
|
|
158
|
+
cwd = String(o.cwd ?? "");
|
|
159
|
+
continue;
|
|
160
|
+
}
|
|
161
|
+
const m = (o.message && typeof o.message === "object" ? o.message : {}) as Record<string, unknown>;
|
|
162
|
+
const role = m.role;
|
|
163
|
+
if (String(o.timestamp ?? "") < cut) continue;
|
|
164
|
+
if (args.project && !cwd.includes(args.project)) continue;
|
|
165
|
+
|
|
166
|
+
if (role === "assistant") {
|
|
167
|
+
used = true;
|
|
168
|
+
const mo = `${m.provider ?? "?"}/${m.model ?? "?"}`;
|
|
169
|
+
bump(turns, mo);
|
|
170
|
+
const content = Array.isArray(m.content) ? m.content : [];
|
|
171
|
+
for (const c of content) {
|
|
172
|
+
if (!c || typeof c !== "object") continue;
|
|
173
|
+
const call = c as Record<string, unknown>;
|
|
174
|
+
if (call.type !== "toolCall") continue;
|
|
175
|
+
const name = String(call.name ?? "");
|
|
176
|
+
const a = norm(call.arguments);
|
|
177
|
+
const cid = typeof call.id === "string" ? call.id : "";
|
|
178
|
+
bump(tools, name || "?");
|
|
179
|
+
if (EDIT_TOOLS.has(name)) {
|
|
180
|
+
bump(dev, mo);
|
|
181
|
+
editor = mo;
|
|
182
|
+
} else if (name === "Agent" && a.subagent_type === "Review") {
|
|
183
|
+
const owner = editor ?? "no-edit (review-only)";
|
|
184
|
+
bump(rev, owner);
|
|
185
|
+
const ep = episode.get(owner) ?? new Set();
|
|
186
|
+
ep.add(`${path}\0${String(a.description ?? "")}${String(a.prompt ?? "").slice(0, 240)}`);
|
|
187
|
+
episode.set(owner, ep);
|
|
188
|
+
const projKey = `${cwd.replace(home, "~")}\0${owner}`;
|
|
189
|
+
bump(proj, projKey);
|
|
190
|
+
if (cid) pending.set(cid, owner);
|
|
191
|
+
} else if (name === "imm_kernel_canary") {
|
|
192
|
+
const act = norm(a.action);
|
|
193
|
+
const op = act.op;
|
|
194
|
+
const owner = editor ?? "no-edit (review-only)";
|
|
195
|
+
if (op === "submit_review") {
|
|
196
|
+
bump(sub, owner);
|
|
197
|
+
const t = tasks.get(owner) ?? new Set();
|
|
198
|
+
t.add(`${cwd}\0${String(a.task_id ?? "")}`);
|
|
199
|
+
tasks.set(owner, t);
|
|
200
|
+
} else if (op === "record_finding") {
|
|
201
|
+
const f = norm(act.finding);
|
|
202
|
+
const summ = String(f.summary ?? "");
|
|
203
|
+
const kind = BOOKKEEPING.test(summ) ? "bookkeeping" : String(f.kind ?? "");
|
|
204
|
+
const rawKey = `${owner}\0${kind}`;
|
|
205
|
+
bump(findingsRaw, rawKey);
|
|
206
|
+
const uniq = findUniq.get(rawKey) ?? new Set();
|
|
207
|
+
uniq.add(`${path}\0${summ.slice(0, 160)}`);
|
|
208
|
+
findUniq.set(rawKey, uniq);
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
} else if (role === "toolResult") {
|
|
213
|
+
const tcid = String(m.toolCallId ?? "");
|
|
214
|
+
const owner = pending.get(tcid);
|
|
215
|
+
if (!owner) continue;
|
|
216
|
+
pending.delete(tcid);
|
|
217
|
+
const tag = parseReviewTag(extractText(m.content));
|
|
218
|
+
if (!tag) continue;
|
|
219
|
+
const sl = scores.get(owner) ?? [];
|
|
220
|
+
sl.push(tag.score);
|
|
221
|
+
scores.set(owner, sl);
|
|
222
|
+
bump(verdicts, `${owner}\0${tag.verdict}`);
|
|
223
|
+
bump(risks, `${owner}\0${tag.risk}`);
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
if (used) files += 1;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
const lines: string[] = [];
|
|
230
|
+
const emit = (s = "") => lines.push(s);
|
|
231
|
+
emit(`window: last ${args.days}d (UTC >= ${cut}Z) | sessions with activity: ${files} | root: ${args.root}`);
|
|
232
|
+
emit("review = Agent(Review) executed; attributed to the model that last edited the code under review");
|
|
233
|
+
emit("avgSc/pass% = parsed from Review [SCORE: .../10] [VERDICT: ...] tags (shows '-' if untagged)");
|
|
234
|
+
emit("");
|
|
235
|
+
const hdr =
|
|
236
|
+
`${"model".padEnd(42)}${"devEdits".padStart(9)}${"turns".padStart(6)}${"reviews".padStart(8)}${"uniq".padStart(5)}${"rev/100ed".padStart(10)}${"avgSc".padStart(6)}${"pass%".padStart(6)}${"registr".padStart(8)}${"block".padStart(6)}${"advis".padStart(6)}${"noisy".padStart(6)}`;
|
|
237
|
+
emit(hdr);
|
|
238
|
+
emit("-".repeat(hdr.length));
|
|
239
|
+
|
|
240
|
+
const models = [...new Set([...rev.keys(), ...dev.keys()])].sort(
|
|
241
|
+
(a, b) => (rev.get(b) ?? 0) - (rev.get(a) ?? 0),
|
|
242
|
+
);
|
|
243
|
+
const shown: string[] = [];
|
|
244
|
+
for (const mo of models) {
|
|
245
|
+
if (!rev.get(mo) && (dev.get(mo) ?? 0) < 30) continue;
|
|
246
|
+
shown.push(mo);
|
|
247
|
+
const uq = episode.get(mo)?.size ?? 0;
|
|
248
|
+
const d = dev.get(mo) ?? 0;
|
|
249
|
+
const r = rev.get(mo) ?? 0;
|
|
250
|
+
const rate = d ? ((100 * r) / d).toFixed(1) : "-";
|
|
251
|
+
const sl = scores.get(mo) ?? [];
|
|
252
|
+
const avgSc = sl.length ? (sl.reduce((x, y) => x + y, 0) / sl.length).toFixed(1) : "-";
|
|
253
|
+
const passCnt = verdicts.get(`${mo}\0PASS`) ?? 0;
|
|
254
|
+
const passPct = sl.length ? `${Math.round((100 * passCnt) / sl.length)}%` : "-";
|
|
255
|
+
const block = findUniq.get(`${mo}\0blocking`)?.size ?? 0;
|
|
256
|
+
const advis = findUniq.get(`${mo}\0advisory`)?.size ?? 0;
|
|
257
|
+
const noisy = findUniq.get(`${mo}\0bookkeeping`)?.size ?? 0;
|
|
258
|
+
emit(
|
|
259
|
+
`${mo.padEnd(42)}${String(d).padStart(9)}${String(turns.get(mo) ?? 0).padStart(6)}${String(r).padStart(8)}${String(uq).padStart(5)}${rate.padStart(10)}${avgSc.padStart(6)}${passPct.padStart(6)}${String(sub.get(mo) ?? 0).padStart(8)}${String(block).padStart(6)}${String(advis).padStart(6)}${String(noisy).padStart(6)}`,
|
|
260
|
+
);
|
|
261
|
+
}
|
|
262
|
+
emit("-".repeat(hdr.length));
|
|
263
|
+
const totalScores = [...scores.values()].flat();
|
|
264
|
+
const totAvg = totalScores.length
|
|
265
|
+
? (totalScores.reduce((x, y) => x + y, 0) / totalScores.length).toFixed(1)
|
|
266
|
+
: "-";
|
|
267
|
+
const totPass = shown.reduce((n, mo) => n + (verdicts.get(`${mo}\0PASS`) ?? 0), 0);
|
|
268
|
+
const totPassPct = totalScores.length ? `${Math.round((100 * totPass) / totalScores.length)}%` : "-";
|
|
269
|
+
const totBlock = [...findUniq.entries()].reduce((n, [k, v]) => n + (k.endsWith("\0blocking") ? v.size : 0), 0);
|
|
270
|
+
const totAdvis = [...findUniq.entries()].reduce((n, [k, v]) => n + (k.endsWith("\0advisory") ? v.size : 0), 0);
|
|
271
|
+
const totNoisy = [...findUniq.entries()].reduce((n, [k, v]) => n + (k.endsWith("\0bookkeeping") ? v.size : 0), 0);
|
|
272
|
+
const totDev = [...dev.values()].reduce((a, b) => a + b, 0);
|
|
273
|
+
const totTurns = [...turns.values()].reduce((a, b) => a + b, 0);
|
|
274
|
+
const totRev = [...rev.values()].reduce((a, b) => a + b, 0);
|
|
275
|
+
const totUniq = [...episode.values()].reduce((n, s) => n + s.size, 0);
|
|
276
|
+
const totSub = [...sub.values()].reduce((a, b) => a + b, 0);
|
|
277
|
+
emit(
|
|
278
|
+
`${"TOTAL".padEnd(42)}${String(totDev).padStart(9)}${String(totTurns).padStart(6)}${String(totRev).padStart(8)}${String(totUniq).padStart(5)}${"".padStart(10)}${totAvg.padStart(6)}${totPassPct.padStart(6)}${String(totSub).padStart(8)}${String(totBlock).padStart(6)}${String(totAdvis).padStart(6)}${String(totNoisy).padStart(6)}`,
|
|
279
|
+
);
|
|
280
|
+
|
|
281
|
+
emit("");
|
|
282
|
+
emit("=== usage (sessions / turns / edits / tools) ===");
|
|
283
|
+
emit(`sessions: ${files} | turns: ${totTurns} | edits: ${totDev}`);
|
|
284
|
+
const toolRows = [...tools.entries()].sort((a, b) => b[1] - a[1]);
|
|
285
|
+
if (toolRows.length === 0) emit("(no tool calls in this window)");
|
|
286
|
+
else for (const [name, n] of toolRows) emit(`${String(n).padStart(5)} ${name}`);
|
|
287
|
+
|
|
288
|
+
emit("");
|
|
289
|
+
emit("=== review quality & scores (new rubric) ===");
|
|
290
|
+
const scoredModels = [...scores.keys()]
|
|
291
|
+
.filter((m) => (scores.get(m) ?? []).length > 0)
|
|
292
|
+
.sort((a, b) => (scores.get(b)?.length ?? 0) - (scores.get(a)?.length ?? 0));
|
|
293
|
+
if (scoredModels.length) {
|
|
294
|
+
const qHdr = `${"model".padEnd(42)}${"scored".padStart(7)}${"avgScore".padStart(9)}${"PASS".padStart(6)}${"REVISE".padStart(7)}${"REJECT".padStart(7)}${"highRisk".padStart(9)}`;
|
|
295
|
+
emit(qHdr);
|
|
296
|
+
emit("-".repeat(qHdr.length));
|
|
297
|
+
for (const mo of scoredModels) {
|
|
298
|
+
const sl = scores.get(mo) ?? [];
|
|
299
|
+
const avgS = (sl.reduce((x, y) => x + y, 0) / sl.length).toFixed(2);
|
|
300
|
+
const pC = verdicts.get(`${mo}\0PASS`) ?? 0;
|
|
301
|
+
const revC = verdicts.get(`${mo}\0REVISE`) ?? 0;
|
|
302
|
+
const rejC = verdicts.get(`${mo}\0REJECT`) ?? 0;
|
|
303
|
+
const highR = (risks.get(`${mo}\0HIGH`) ?? 0) + (risks.get(`${mo}\0CRITICAL`) ?? 0);
|
|
304
|
+
emit(
|
|
305
|
+
`${mo.padEnd(42)}${String(sl.length).padStart(7)}${avgS.padStart(9)}${String(pC).padStart(6)}${String(revC).padStart(7)}${String(rejC).padStart(7)}${String(highR).padStart(9)}`,
|
|
306
|
+
);
|
|
307
|
+
}
|
|
308
|
+
} else {
|
|
309
|
+
emit("(no scored reviews found in this window yet; reviews with [SCORE: .../10] will appear here)");
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
emit("");
|
|
313
|
+
emit("=== where the reviews landed (project x author) ===");
|
|
314
|
+
const projRows = [...proj.entries()].sort((a, b) => b[1] - a[1]).slice(0, args.top);
|
|
315
|
+
for (const [key, v] of projRows) {
|
|
316
|
+
const [p, mo] = key.split("\0");
|
|
317
|
+
emit(`${String(v).padStart(5)} ${String(p).padEnd(58)} ${mo}`);
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
emit("");
|
|
321
|
+
emit("=== review rounds per kernel task (rework signal) ===");
|
|
322
|
+
const taskModels = [...tasks.keys()].sort((a, b) => (sub.get(b) ?? 0) - (sub.get(a) ?? 0));
|
|
323
|
+
for (const mo of taskModels) {
|
|
324
|
+
const nt = tasks.get(mo)?.size ?? 0;
|
|
325
|
+
if (!nt) continue;
|
|
326
|
+
const s = sub.get(mo) ?? 0;
|
|
327
|
+
emit(
|
|
328
|
+
`${(s / nt).toFixed(1).padStart(6)} rounds/task ${String(s).padStart(4)} registrations / ${String(nt).padStart(3)} tasks ${mo}`,
|
|
329
|
+
);
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
const rawSum = [...findingsRaw.values()].reduce((a, b) => a + b, 0);
|
|
333
|
+
const uniqSum = [...findUniq.values()].reduce((n, s) => n + s.size, 0);
|
|
334
|
+
if (rawSum && rawSum !== uniqSum) {
|
|
335
|
+
emit("");
|
|
336
|
+
emit(
|
|
337
|
+
`(findings: ${rawSum} raw calls deduped to ${uniqSum} distinct — re-submits of the same finding are counted once)`,
|
|
338
|
+
);
|
|
339
|
+
}
|
|
340
|
+
emit("caveats: 'bookkeep' = kernel self-entries mislabelled as findings, excluded from blocking/advisory;");
|
|
341
|
+
emit(" high rounds/task can be canary/QA harness re-registration, not human-visible rework.");
|
|
342
|
+
return `${lines.join("\n")}\n`;
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
const isMain = Boolean(process.argv[1]) && pathToFileURL(process.argv[1]!).href === import.meta.url;
|
|
346
|
+
if (isMain) {
|
|
347
|
+
run(process.argv.slice(2))
|
|
348
|
+
.then((text) => {
|
|
349
|
+
process.stdout.write(text);
|
|
350
|
+
})
|
|
351
|
+
.catch((err: unknown) => {
|
|
352
|
+
console.error(err);
|
|
353
|
+
process.exit(1);
|
|
354
|
+
});
|
|
355
|
+
}
|
|
@@ -56,3 +56,12 @@ skills:
|
|
|
56
56
|
output_artifacts: [maintain_report]
|
|
57
57
|
next_actions: []
|
|
58
58
|
boundary: Minimize tracked agent-instruction context after explicit manifest approval; no Managed authority mutation, contract installation, or reference-document creation.
|
|
59
|
+
- name: imm-review-retro
|
|
60
|
+
path: skills/imm-review-retro/SKILL.md
|
|
61
|
+
role: execute
|
|
62
|
+
title: Review Retro
|
|
63
|
+
role_class: discovery
|
|
64
|
+
canonical: true
|
|
65
|
+
output_artifacts: [retro_report]
|
|
66
|
+
next_actions: []
|
|
67
|
+
boundary: Rank models by review load and report project usage from pi session logs. Read-only. No Managed Path mutation.
|