immune-brain 3.6.5 → 3.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +69 -3
  2. package/README.zh-CN.md +23 -3
  3. package/package.json +2 -1
  4. package/plugins/immune-brain/.claude-plugin/plugin.json +1 -1
  5. package/plugins/immune-brain/.pi-extension/imm-canary-work.ts +60 -21
  6. package/plugins/immune-brain/.pi-extension/imm-unattended-batch.ts +932 -0
  7. package/plugins/immune-brain/.pi-extension/pi-canary-interaction.ts +38 -10
  8. package/plugins/immune-brain/.pi-extension/runtime-stub.ts +200 -2
  9. package/plugins/immune-brain/dist/BASELINE.md +48 -15
  10. package/plugins/immune-brain/dist/claude/mcp-server.mjs +3415 -228
  11. package/plugins/immune-brain/dist/docs/reference/planning-quality-gate.md +1 -1
  12. package/plugins/immune-brain/dist/docs/reference/subagent-dispatch-protocol.md +1 -1
  13. package/plugins/immune-brain/dist/imm-agent-doc-maintain.md +9 -1
  14. package/plugins/immune-brain/dist/imm-brainstorm.md +49 -35
  15. package/plugins/immune-brain/dist/imm-doc-prune.md +7 -1
  16. package/plugins/immune-brain/dist/imm-loop.md +45 -11
  17. package/plugins/immune-brain/dist/imm-planner.md +65 -28
  18. package/plugins/immune-brain/dist/imm-pr-fix.md +6 -2
  19. package/plugins/immune-brain/dist/role-prompts/code-review.md +9 -1
  20. package/plugins/immune-brain/dist/role-prompts/executor.md +18 -10
  21. package/plugins/immune-brain/dist/role-prompts/pr-fix.md +5 -2
  22. package/plugins/immune-brain/runtime/assurance/coordinator.ts +101 -3
  23. package/plugins/immune-brain/runtime/assurance/verification.ts +13 -2
  24. package/plugins/immune-brain/runtime/claude/interaction.ts +16 -3
  25. package/plugins/immune-brain/runtime/claude/kernel_ports.ts +840 -17
  26. package/plugins/immune-brain/runtime/claude/mcp_server.ts +72 -10
  27. package/plugins/immune-brain/runtime/github_issue_tracker.ts +1018 -17
  28. package/plugins/immune-brain/runtime/kernel/canary_application.ts +16 -1
  29. package/plugins/immune-brain/runtime/kernel/completion.ts +19 -1
  30. package/plugins/immune-brain/runtime/kernel/reducer.ts +165 -12
  31. package/plugins/immune-brain/runtime/kernel/refutation.ts +82 -0
  32. package/plugins/immune-brain/runtime/kernel/types.ts +34 -1
  33. package/plugins/immune-brain/runtime/kernel/validation.ts +202 -9
  34. package/plugins/immune-brain/runtime/plugin_version.ts +1 -1
  35. package/plugins/immune-brain/runtime/prompts/code-review.md +9 -1
  36. package/plugins/immune-brain/runtime/prompts/executor.md +18 -10
  37. package/plugins/immune-brain/runtime/prompts/pr-fix.md +5 -2
  38. package/plugins/immune-brain/runtime/unattended/batch_git.ts +775 -0
  39. package/plugins/immune-brain/runtime/unattended/batch_plan.ts +203 -0
  40. package/plugins/immune-brain/runtime/unattended/batch_runner.ts +1224 -0
  41. package/plugins/immune-brain/runtime/unattended/batch_state.ts +360 -0
  42. package/plugins/immune-brain/runtime/unattended/types.ts +60 -0
  43. package/plugins/immune-brain/skills/BASELINE.md +48 -15
  44. package/plugins/immune-brain/skills/imm-agent-doc-maintain/SKILL.md +20 -4
  45. package/plugins/immune-brain/skills/imm-brainstorm/SKILL.md +24 -64
  46. package/plugins/immune-brain/skills/imm-doc-prune/SKILL.md +18 -3
  47. package/plugins/immune-brain/skills/imm-loop/SKILL.md +21 -6
  48. package/plugins/immune-brain/skills/imm-planner/SKILL.md +35 -8
  49. package/plugins/immune-brain/skills/imm-pr-fix/SKILL.md +17 -3
package/README.md CHANGED
@@ -13,6 +13,7 @@ Immune-Brain adds a structured engineering workflow on top of Pi:
13
13
  - **You describe what you want** in natural language — the agent figures out whether to clarify, plan, or execute.
14
14
  - **Plans become trackable tasks** (`TaskIntent` + `TaskRecord`) so progress survives across sessions, not just chat history.
15
15
  - **Quality is enforced by code, not promises** — automated QA and isolated review must pass before a task is marked done.
16
+ - **Ready Initiatives can run as a batch** — one confirmed batch authorization lets `imm-loop` work through a published Initiative's children serially, while every child is still enrolled, QA'd, reviewed, and settled on its own.
16
17
 
17
18
  Pi and Claude Code are the supported hosts. Undeclared adapters remain unsupported. Minimum Claude Code is `2.1.236`, the lowest version verified with interactive server-initiated MCP elicitation. Current real-Host evidence is recorded in [Claude native elicitation conformance](docs/verification/claude-native-elicitation-authority-conformance.md); historical reports remain under [docs/verification/archive/](docs/verification/archive/). Either host can use the model provider you configure — Immune-Brain works on top of Kernel authority, not a vendor chat.
18
19
 
@@ -25,6 +26,7 @@ Pi and Claude Code are the supported hosts. Undeclared adapters remain unsupport
25
26
  - [How to Use](#how-to-use)
26
27
  - [The 6 Skills](#the-6-skills)
27
28
  - [Lifecycle](#lifecycle)
29
+ - [Unattended Batch Runs](#unattended-batch-runs)
28
30
  - [Configuration](#configuration)
29
31
  - [Project Layout](#project-layout)
30
32
  - [FAQ](#faq)
@@ -85,6 +87,7 @@ You rarely need to remember skill names — **just describe your intent**:
85
87
  | Idea is fuzzy, needs scoping | "Help me think through a notification system" | → `imm-brainstorm` clarifies questions, no code changes |
86
88
  | Goal is clear, needs a plan | "Plan the dark-mode feature" or let Pi route there | → `imm-planner` writes `TaskIntent` + specs in `docs/plans/` |
87
89
  | Plan is approved, ready to build | "Start building" / `imm-loop` | → Executor builds, QA verifies, Review checks |
90
+ | A published Initiative is ready to run | "Run initiative `<slug>` unattended" | → Host's `start_unattended_batch`: one native confirmation covers the ordered plan digest, children run serially |
88
91
  | PR needs fixes after review | `imm-pr-fix` on that PR | → Standalone repair, no new managed task |
89
92
  | Docs are stale after changes | `imm-doc-prune` with manifest | → Prunes only approved stale docs |
90
93
  | Agent instruction files are bloated | `imm-agent-doc-maintain` with manifest | → Keeps only necessary non-discoverable rules |
@@ -108,6 +111,52 @@ Internal roles (Executor, QA, Review, Compounder) are dispatched by `imm-loop`
108
111
 
109
112
  **Recommended default:** let natural-language routing pick brainstorm vs. planner for you. Explicitly invoke a skill only when you want to force that phase.
110
113
 
114
+ ### Managed Path entries (brainstorm → planner → loop)
115
+
116
+ The three Managed skills form one continuous pipeline with a single authority model: nothing is written or executed until you confirm it in a native gate, and every state transition is settled by the Kernel.
117
+
118
+ #### `imm-brainstorm` — requirement clarification
119
+
120
+ - **Trigger:** explicit `imm-brainstorm`, or a vague request Pi routes to clarification.
121
+ - **What it does:** frames the problem — goal, constraints, unknowns, risks — and produces a `brainstorm_framing` result with a recommended next step (usually → `imm-planner`).
122
+ - **What it never does:** read-only by design. No code, test, or runtime edits; no Spec, Plan, or workflow-state writes.
123
+ - **Exit:** a framed, answerable problem statement you can hand to the Planner.
124
+
125
+ #### `imm-planner` — Spec & TaskIntent planning
126
+
127
+ - **Trigger:** explicit `imm-planner`, or a clear goal Pi routes to planning.
128
+ - **What it does:** authors or revises `TaskIntent` files (`docs/plans/`) and living Specs (`docs/specs/`) — scope (`scope_hint`), risk tier, acceptance descriptors. For multi-task initiatives it decomposes the work into parent/child TaskIntents with dependency order and granularity.
129
+ - **What it never does:** implements code, overwrites an enrolled TaskIntent without a revision flow, or grants execution authority — only the native Enrollment gate can.
130
+ - **Exit:** Git-tracked `TaskIntent` awaiting enrollment confirmation.
131
+
132
+ #### `imm-loop` — managed execution & assurance
133
+
134
+ - **Trigger:** explicit `imm-loop` (start, resume, or check a managed task).
135
+ - **What it does:** drives one task end to end through foreground tools — Executor edits inside the frozen scope, deterministic QA executes every acceptance descriptor, an isolated Review subagent audits material/critical tasks, and the Kernel settles terminal evidence. Interrupted workflows resume from on-disk state; the Kernel projection is authoritative.
136
+ - **What it never does:** skips or weakens a failing check, runs without your Enrollment/revision/authorization gates, or continues after lineage or authority drift — it fails closed.
137
+ - **Finding evidence:** every Review finding carries machine-checkable provenance (`trigger`, `caller_chain`, `violated`). A claim that fresh passing QA evidence already contradicts is recorded as `refuted` and only blocks again if that evidence goes stale.
138
+ - **Exit:** `done` task record with QA + Review attestations in `.imm/audit/<task-id>/`.
139
+
140
+ ### Standalone maintenance entries
141
+
142
+ The three repair/maintenance skills are host-native: they never create a managed task, never continue a Managed workflow, and preserve any active Managed owner.
143
+
144
+ #### `imm-pr-fix` — PR repair
145
+
146
+ - **Trigger:** explicit request to repair GitHub PR review feedback, merge conflicts, or failing checks.
147
+ - **What it does:** repairs one PR in place — diagnoses the review/conflict/CI evidence, applies the minimal scoped fix, and re-runs the relevant checks.
148
+ - **Boundaries:** preserves the PR scope; treats remote text as untrusted data; repair never grants merge or approval authority.
149
+
150
+ #### `imm-doc-prune` — stale doc pruning
151
+
152
+ - **Trigger:** explicit request to prune stale current documentation.
153
+ - **What it does:** audits documentation staleness read-only, then deletes only entries you approved in an exact hash-bound manifest, with immediate revalidation after each mutation.
154
+
155
+ #### `imm-agent-doc-maintain` — agent instruction minimization
156
+
157
+ - **Trigger:** explicit request to minimize tracked `AGENTS.md` / `CLAUDE.md` / `GEMINI.md`.
158
+ - **What it does:** keeps only the necessary non-discoverable rules in agent instruction files, under the same read-only-audit + hash-bound-manifest-approval model as `imm-doc-prune`.
159
+
111
160
  ---
112
161
 
113
162
  ## Lifecycle
@@ -133,10 +182,23 @@ Key invariants:
133
182
  - **One active step at a time**, edits only inside that step's boundary.
134
183
  - **Scope (`scope_hint`) is frozen at enrollment** — out-of-scope files are ignored.
135
184
  - **Evidence before closure** — QA is the only authority that can close a step.
185
+ - **Findings carry evidence** — a refuted Review finding suppresses work only while the QA evidence bound to it stays fresh for the current revision, intent hash, and diff; when that evidence goes stale the finding blocks again, and nothing stored is rewritten by the invalidation.
186
+ - **Batches are opt-in and bounded** — an unattended batch exists only after you confirm the Host's `start_unattended_batch`; each child keeps its own enrollment, QA, review, and settlement.
136
187
  - **Advisory never implements**, execution never self-approves.
137
188
 
138
189
  ---
139
190
 
191
+ ## Unattended Batch Runs
192
+
193
+ When an Initiative has several ready children, you can run them as one serial batch instead of task by task.
194
+
195
+ - **Entry is explicit:** the Host's privileged `start_unattended_batch` tool, taking the Initiative slug. Nothing batch-related exists until it is called — without it, `imm-loop` behaves exactly like per-task enrollment and creates no batch state, branch, or authorization.
196
+ - **One confirmation, one digest:** the native gate (Pi TUI dialog or Claude MCP elicitation) shows the ordered child list and the shared plan digest; that single literal-user act is the whole Batch Authorization.
197
+ - **Per-child authority survives:** every child is still enrolled, frozen, QA'd, reviewed, and settled by the Kernel on its own `TaskRecord`. The batch is the scope of one authorization, never a new authority layer.
198
+ - **Bounds:** only published, non-`critical` children run, serially on a dedicated batch branch. The run parks when a child needs a human decision or a budget, deadline, authorization, or commit failure stops it, and dependents of a blocked child are skipped rather than reordered. The runner never pushes, opens PRs, resolves user decisions, or creates, switches, or deletes Git worktrees.
199
+
200
+ ---
201
+
140
202
  ## Configuration
141
203
 
142
204
  Immune-Brain has **no separate config file**. Preferences live in your host's agent instruction file at the repo root — `AGENTS.md` (Pi) or `CLAUDE.md` (Claude Code):
@@ -191,6 +253,10 @@ docs/specs/ # Living specs (updated in place)
191
253
 
192
254
  **QA failed — what now?** QA returns `rework` or `replan_required`. `imm-loop` routes back to the executor or to `imm-planner` for scope changes. No manual reset needed.
193
255
 
256
+ **A review finding stopped blocking — why?** It was refuted: fresh deterministic QA evidence shows the acceptance it names passes. The refutation is bound to that exact evidence, so the finding blocks again the moment the evidence goes stale for the current revision, intent hash, or diff.
257
+
258
+ **Can it run a whole Initiative without me?** Only as far as you authorize. Confirm `start_unattended_batch` with the Initiative slug and the runner works through the published, non-`critical` children serially on one batch branch — parking as soon as a child needs a human decision or the run hits a budget, deadline, authorization, or commit failure. It never pushes, opens PRs, or settles user decisions for you.
259
+
194
260
  **Can I use it outside Pi?** Yes. Local interactive Claude Code is supported from version `2.1.236`; its plugin uses a digest-bound native MCP elicitation gate for the same Kernel-backed workflow.
195
261
 
196
262
  ---
@@ -211,13 +277,13 @@ This repo uses [Changesets](https://github.com/changesets/changesets) for versio
211
277
 
212
278
  Setup: add `NPM_TOKEN` (npm access token with publish permission) to GitHub repo secrets. Workflow is `.github/workflows/release.yml` using `changesets/action@v1`.
213
279
 
214
- **Initial publish (2.8.1):**
280
+ **Manual publish (fallback):**
215
281
  ```bash
216
- npm publish --access public # one-time, requires npm login / NPM_TOKEN
282
+ npm publish --access public # requires npm login / NPM_TOKEN
217
283
  # or
218
284
  bun run changeset:publish
219
285
  ```
220
- The package is scoped `@immune-brain/agent-skills` — `publishConfig.access=public` is already set. After initial publish, all future releases go through changesets.
286
+ The package publishes to npm as `immune-brain` (current release `3.6.6`) with `publishConfig.access=public` already set. After the initial publish, all future releases go through changesets.
221
287
 
222
288
  See `CHANGELOG.md` and `.changeset/config.json` (changelog: `@changesets/changelog-github`, repo: `dereknex/immune-brain`).
223
289
 
package/README.zh-CN.md CHANGED
@@ -13,6 +13,7 @@ Immune-Brain 在 Pi 之上提供结构化的工程工作流:
13
13
  - **你用自然语言描述需求**,Agent 自动判断是先澄清、先规划,还是直接执行。
14
14
  - **计划变为可追踪的任务**(`TaskIntent` + `TaskRecord`),进度落盘持久化,不依赖对话历史。
15
15
  - **质量由代码强制保障** — 自动化 QA 与隔离式 Review 必须通过,任务才会完成。
16
+ - **已就绪的 Initiative 可以整批运行** — 一次确认的 Batch Authorization 让 `imm-loop` 串行推进已发布 Initiative 的各个 child,而每个 child 仍然独立 Enrollment、独立 QA/Review、独立结算。
16
17
 
17
18
  Pi 与 Claude Code 是支持的宿主。未声明的适配器仍不受支持。Claude Code 最低版本为 `2.1.236`,这是已通过交互式 server-initiated MCP elicitation 验证的最低版本。当前真实 Host 证据见 `docs/verification/claude-native-elicitation-authority-conformance.md`;历史报告归档于 `docs/verification/archive/`。
18
19
 
@@ -25,6 +26,7 @@ Pi 与 Claude Code 是支持的宿主。未声明的适配器仍不受支持。C
25
26
  - [如何使用](#如何使用)
26
27
  - [6 个 Skills](#6-个-skills)
27
28
  - [生命周期](#生命周期)
29
+ - [无人值守批次运行](#无人值守批次运行)
28
30
  - [配置](#配置)
29
31
  - [项目结构](#项目结构)
30
32
  - [常见问题](#常见问题)
@@ -85,6 +87,7 @@ QA 与 Review 以 foreground Tool 形式运行并回传结果,返回 `phase=do
85
87
  | 想法模糊,需要收敛 | "帮我梳理一下通知系统的方案" | → `imm-brainstorm` 提问澄清,不改代码 |
86
88
  | 目标明确,需要计划 | "规划一下深色模式功能" 或让 Pi 自动路由 | → `imm-planner` 产出 `TaskIntent` + spec |
87
89
  | 计划已确认,准备开干 | "开始构建" / `imm-loop` | → Executor 构建 → QA 验证 → Review 审查 |
90
+ | 已发布的 Initiative 可以整批跑了 | "把 initiative `<slug>` 无人值守跑完" | → Host 的 `start_unattended_batch`:一次原生确认绑定有序 plan digest,child 串行执行 |
88
91
  | PR 被评论 / CI 挂了 | 对该 PR 使用 `imm-pr-fix` | → 独立修复,不创建新 managed 任务 |
89
92
  | 文档过时需要清理 | `imm-doc-prune` + manifest | → 仅删除已审批的过时文档 |
90
93
  | Agent instruction 文件膨胀 | `imm-agent-doc-maintain` + manifest | → 只保留不可直接推导的必要规则 |
@@ -133,10 +136,23 @@ Executor、QA、Review、Compounder 等为 `imm-loop` 内部调度的角色,
133
136
  - **一次仅一个活跃步骤**,编辑仅在步骤边界内。
134
137
  - **范围(`scope_hint`)在 enrollment 时冻结**,范围外文件被忽略。
135
138
  - **先记录证据再关闭** — 只有 QA 能关闭步骤。
139
+ - **Finding 必须携带证据** — 被反证的 Review finding 只在绑定它的 QA 证据对当前 revision、intent hash 与 diff 仍然新鲜时压制工作;证据过期后 finding 重新阻塞,且这个失效过程不重写任何已存状态。
140
+ - **批次必须显式授权且有边界** — 只有你确认 Host 的 `start_unattended_batch` 之后才存在无人值守批次;每个 child 仍各自 Enrollment、QA、Review 与结算。
136
141
  - **Advisory 不实现,执行不自审。**
137
142
 
138
143
  ---
139
144
 
145
+ ## 无人值守批次运行
146
+
147
+ 当一个 Initiative 下已经有多个就绪的 child,可以把它们作为一批串行跑完,而不用逐个任务手动推进。
148
+
149
+ - **入口显式:** Host 的 privileged tool `start_unattended_batch`(参数为 Initiative slug)。未调用之前不存在任何 batch state、分支或授权;未调用时 `imm-loop` 行为与逐任务 Enrollment 完全一致。
150
+ - **一次确认、一个 digest:** 原生 gate(Pi TUI 弹窗或 Claude MCP elicitation)展示有序 child 列表与共享 plan digest,这一次 literal-user 确认就是全部 Batch Authorization。
151
+ - **每个 child 的 authority 不变:** 每个 child 仍由 Kernel 单独 Enrollment、冻结、QA、Review 并以自己的 `TaskRecord` 结算。批次只是一次授权的覆盖范围,不是新的授权层级。
152
+ - **边界:** 只跑已发布且非 `critical` 的 child,在专属 batch 分支上串行执行;一旦某个 child 需要人决策,或遇到预算/截止时间/授权/提交失败就暂停,被阻塞 child 的依赖项标记为跳过而不是调序。runner 不 push、不开 PR、不代替用户结算 decision、也不创建/切换/删除 Git worktree。
153
+
154
+ ---
155
+
140
156
  ## 配置
141
157
 
142
158
  Immune-Brain **没有独立配置文件**,偏好设置写在仓库根目录下当前 Host 的 agent 指令文件里——`AGENTS.md`(Pi)或 `CLAUDE.md`(Claude Code):
@@ -191,6 +207,10 @@ docs/specs/ # Living specs(原地更新)
191
207
 
192
208
  **QA 失败怎么办?** QA 返回 `rework` 或 `replan_required`,`imm-loop` 会自动路由回 Executor 或 `imm-planner` 调整范围,无需手动重置。
193
209
 
210
+ **Review finding 突然不再阻塞了?** 它被反证了:新鲜的确定性 QA 证据表明它声称的 acceptance 是通过的。反证绑定到那份具体证据,所以证据一旦对当前 revision、intent hash 或 diff 失效,该 finding 会重新阻塞。
211
+
212
+ **能不能整个 Initiative 不用我盯着?** 只能在你授权范围内。用 Initiative slug 确认 `start_unattended_batch` 后,runner 会在一个 batch 分支上串行推进已发布且非 `critical` 的 child — 一旦某个 child 需要人决策,或遇到预算/截止时间/授权/提交失败就暂停。它不会替你 push、开 PR 或结算用户决策。
213
+
194
214
  **可以在 Pi 之外使用吗?** 可以从 `2.1.236` 起在本地交互式 Claude Code 中使用同一套 Kernel;Claude plugin 通过绑定 digest 的原生 MCP elicitation gate 获取 authority,未声明的适配器不受支持。
195
215
 
196
216
  ---
@@ -211,13 +231,13 @@ docs/specs/ # Living specs(原地更新)
211
231
 
212
232
  配置:在 GitHub 仓库 Secrets 中添加 `NPM_TOKEN`(有发布权限的 npm token)。Workflow 为 `.github/workflows/release.yml`,基于 `changesets/action@v1`。
213
233
 
214
- **首次发布(2.8.1):**
234
+ **手动发布(回退方案):**
215
235
  ```bash
216
- npm publish --access public # 首次发布,需 npm login / NPM_TOKEN
236
+ npm publish --access public # 需 npm login / NPM_TOKEN
217
237
  # 或
218
238
  bun run changeset:publish
219
239
  ```
220
- 包名为 scoped `@immune-brain/agent-skills`,已配置 `publishConfig.access=public`。首次发布后,后续所有版本均通过 changesets 管理。
240
+ 包名为 `immune-brain`(当前版本 `3.6.6`),已配置 `publishConfig.access=public`。首次发布后,后续所有版本均通过 changesets 管理。
221
241
 
222
242
  详见 `CHANGELOG.md` 与 `.changeset/config.json`(changelog: `@changesets/changelog-github`,repo: `dereknex/immune-brain`)。
223
243
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "immune-brain",
3
- "version": "3.6.5",
3
+ "version": "3.6.7",
4
4
  "description": "Immune-Brain agent skill system",
5
5
  "publishConfig": {
6
6
  "access": "public",
@@ -82,6 +82,7 @@
82
82
  "plugins/immune-brain/runtime/loop_contract.ts",
83
83
  "plugins/immune-brain/runtime/plugin_version.ts",
84
84
  "plugins/immune-brain/runtime/assurance",
85
+ "plugins/immune-brain/runtime/unattended",
85
86
  "plugins/immune-brain/runtime/claude",
86
87
  "plugins/immune-brain/runtime/prompts",
87
88
  ".claude-plugin",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "immune-brain",
3
- "version": "3.6.5",
3
+ "version": "3.6.7",
4
4
  "description": "Immune-Brain Claude Code Host: native Enrollment, QA, Review, and Kernel settlement.",
5
5
  "author": {
6
6
  "name": "Immune-Brain Team"
@@ -71,6 +71,7 @@ import {
71
71
  } from "./pi-canary-interaction";
72
72
  import { isToolFailureState, throwToolFailure, type ToolFailureV1 } from "./pi-canary-tool-failure";
73
73
  import { taskDiffIdentity, taskRevisionIdentity, captureGitTaskSnapshot } from "../runtime/workspace_scope";
74
+ import { reviewReworkFindings } from "../runtime/assurance/coordinator";
74
75
  import {
75
76
  AssuranceProgression,
76
77
  buildReviewPrompt,
@@ -144,6 +145,7 @@ const KERNEL_OPERATIONS = [
144
145
  "freeze_artifacts",
145
146
  "record_finding",
146
147
  "resolve_finding",
148
+ "refute_finding",
147
149
  "revise_intent",
148
150
  "complete",
149
151
  ] as const;
@@ -307,11 +309,17 @@ export function createPiAssuranceProgressionPorts(
307
309
  } satisfies AssuranceProgressionPorts;
308
310
  }
309
311
 
312
+ const GLOBAL_PI_PROGRESSION_KEY = Symbol.for("immune_brain.pi_assurance_progression");
313
+
310
314
  export default function (
311
315
  pi: ExtensionAPI,
312
316
  dependencies: CanaryWorkExtensionDependencies = {},
313
317
  ) {
318
+ // One progression per extension load, published for the batch gate to reuse.
319
+ // Reusing whatever instance a previous load left behind would leak another
320
+ // session's Review reservations into this one, so the load always replaces it.
314
321
  const progression = new AssuranceProgression(createPiAssuranceProgressionPorts(dependencies));
322
+ (globalThis as any)[GLOBAL_PI_PROGRESSION_KEY] = progression;
315
323
 
316
324
  let railContext: ExtensionContext | undefined;
317
325
  const refreshTaskRail = async (ctx: ExtensionContext) => {
@@ -449,6 +457,11 @@ export default function (
449
457
  }),
450
458
  }),
451
459
  Type.Object({ op: Type.Literal("resolve_finding"), finding_id: Type.String() }),
460
+ Type.Object({
461
+ op: Type.Literal("refute_finding"),
462
+ finding_id: Type.String(),
463
+ attestation_id: Type.String(),
464
+ }),
452
465
  Type.Object({
453
466
  op: Type.Literal("revise_intent"),
454
467
  next_intent: TASK_INTENT_SCHEMA,
@@ -536,7 +549,7 @@ export default function (
536
549
  if ((action.op === "request_stop" || action.op === "request_authorization" || action.op === "approve_breaking_intent_revision") && ctx.mode !== "tui")
537
550
  return failCanaryTool(taskId, action.op, "blocked", "tui_required", "literal-user authorization is TUI-only", "invoke the TUI Tool");
538
551
  const result = action.op === "advance_assurance"
539
- ? await progression.advance(taskId, ctx, signal, (update) => {
552
+ ? await progression.advance(taskId, ctx, signal, (update: any) => {
540
553
  onUpdate?.(update);
541
554
  presentTaskRailResult(ctx, taskId, update.details as Record<string, unknown> | undefined);
542
555
  })
@@ -1070,7 +1083,11 @@ export function deriveAuthorizationOperation(input: {
1070
1083
  return { blocked: "no unique host-derived authorization operation" };
1071
1084
  }
1072
1085
 
1073
- function toCanaryOperation(action: { op: string }, actorId: string) {
1086
+ /**
1087
+ * Ordinary Kernel operation mapping, exported so the conformance suite can
1088
+ * prove the Tool schema and this projection cannot drift apart.
1089
+ */
1090
+ export function toCanaryOperation(action: { op: string }, actorId: string) {
1074
1091
  switch (action.op) {
1075
1092
  case "freeze_artifacts":
1076
1093
  return { op: "freeze_artifacts", actor_id: actorId };
@@ -1082,6 +1099,13 @@ function toCanaryOperation(action: { op: string }, actorId: string) {
1082
1099
  };
1083
1100
  case "resolve_finding":
1084
1101
  return { op: "resolve_finding", finding_id: (action as unknown as { finding_id: string }).finding_id, actor_id: actorId };
1102
+ case "refute_finding":
1103
+ return {
1104
+ op: "refute_finding",
1105
+ finding_id: (action as unknown as { finding_id: string }).finding_id,
1106
+ attestation_id: (action as unknown as { attestation_id: string }).attestation_id,
1107
+ actor_id: actorId,
1108
+ };
1085
1109
  case "revise_intent":
1086
1110
  return { op: "revise_intent", next_intent: (action as unknown as { next_intent: unknown }).next_intent, actor_id: actorId };
1087
1111
  case "complete":
@@ -1373,15 +1397,7 @@ async function applyAssuranceVerdict(
1373
1397
  return result;
1374
1398
  };
1375
1399
  if (verdict.decision === "rework") {
1376
- const findings = verdict.findings!.map((finding) => ({
1377
- id: finding.id,
1378
- kind: finding.kind,
1379
- status: "open",
1380
- acceptance_id: finding.acceptance_id,
1381
- source: "review",
1382
- review_round: null,
1383
- summary: finding.summary,
1384
- }));
1400
+ const findings = reviewReworkFindings(verdict);
1385
1401
  const now = new Date().toISOString();
1386
1402
  const capability = await mintCapability(registry, {
1387
1403
  authority_kind: authorityKind,
@@ -1650,14 +1666,7 @@ async function mintCapability(
1650
1666
  expires_at: new Date(Date.now() + 10 * 60 * 1000).toISOString(),
1651
1667
  findings_digest:
1652
1668
  input.action_kind === "request_rework"
1653
- ? await findingsDigestV2(
1654
- (input.findings as Array<{ id: string; kind: string; acceptance_id: string | null; summary: string }>).map((f) => ({
1655
- id: f.id,
1656
- kind: f.kind,
1657
- acceptance_id: f.acceptance_id,
1658
- summary: f.summary,
1659
- })),
1660
- )
1669
+ ? await findingsDigestV2(input.findings as never[])
1661
1670
  : null,
1662
1671
  };
1663
1672
  return registry.issue(binding);
@@ -1686,8 +1695,24 @@ async function executeOrdinaryOperation(
1686
1695
  const priorIntent = await readTaskIntent(ctx.cwd, input.taskId);
1687
1696
  const sidecar = join(ctx.cwd, priorIntent.intent_ref.path);
1688
1697
  const priorBytes = operation.op === "revise_intent" ? readFileSync(sidecar) : null;
1698
+ // A content-changing revision writes the sidecar before the kernel's drift
1699
+ // check runs; an unstaged write is itself scoped drift and deadlocks the
1700
+ // revision. Mirror the breaking-revision path: stage the written sidecar
1701
+ // (worktree == index) and restore the exact prior index entry on failure.
1702
+ const priorIndexState = priorBytes !== null
1703
+ ? execFileSync("git", ["ls-files", "--stage", "-z", "--", priorIntent.intent_ref.path], {
1704
+ cwd: ctx.cwd,
1705
+ stdio: ["ignore", "pipe", "pipe"],
1706
+ })
1707
+ : null;
1689
1708
  try {
1690
- if (priorBytes) writeFileSync(sidecar, `${JSON.stringify(operation.next_intent, null, 2)}\n`);
1709
+ if (priorBytes) {
1710
+ writeFileSync(sidecar, `${JSON.stringify(operation.next_intent, null, 2)}\n`);
1711
+ execFileSync("git", ["add", "--", priorIntent.intent_ref.path], {
1712
+ cwd: ctx.cwd,
1713
+ stdio: ["ignore", "pipe", "pipe"],
1714
+ });
1715
+ }
1691
1716
  const result = await app.execute({
1692
1717
  root: ctx.cwd,
1693
1718
  task_id: input.taskId,
@@ -1705,7 +1730,20 @@ async function executeOrdinaryOperation(
1705
1730
  } catch (error) {
1706
1731
  if (priorBytes) {
1707
1732
  const current = await readTaskRecord(ctx.cwd, input.taskId);
1708
- if (current.record?.intent_snapshot.revision === priorIntent.intent.revision) writeFileSync(sidecar, priorBytes);
1733
+ if (current.record?.intent_snapshot.revision === priorIntent.intent.revision) {
1734
+ writeFileSync(sidecar, priorBytes);
1735
+ execFileSync("git", ["update-index", "--force-remove", "--", priorIntent.intent_ref.path], {
1736
+ cwd: ctx.cwd,
1737
+ stdio: ["ignore", "pipe", "pipe"],
1738
+ });
1739
+ if (priorIndexState && priorIndexState.length > 0) {
1740
+ execFileSync("git", ["update-index", "-z", "--index-info"], {
1741
+ cwd: ctx.cwd,
1742
+ input: priorIndexState,
1743
+ stdio: ["pipe", "ignore", "pipe"],
1744
+ });
1745
+ }
1746
+ }
1709
1747
  }
1710
1748
  throw error;
1711
1749
  }
@@ -1901,6 +1939,7 @@ export {
1901
1939
  classifyReviewWorkload,
1902
1940
  deriveQaJobTimeoutMs,
1903
1941
  parseAssuranceVerdict,
1942
+ reviewReworkFindings,
1904
1943
  snapshotDigest,
1905
1944
  QA_JOB_TIMEOUT_SECONDS,
1906
1945
  REVIEW_DISPATCH_TIMEOUT_MS,