immune-brain 3.6.8 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -4
- package/README.zh-CN.md +10 -3
- package/package.json +3 -2
- package/plugins/immune-brain/.claude-plugin/plugin.json +1 -1
- package/plugins/immune-brain/.pi-extension/imm-canary-enroll.ts +18 -2
- package/plugins/immune-brain/.pi-extension/imm-canary-work.ts +76 -121
- package/plugins/immune-brain/.pi-extension/imm-unattended-batch.ts +106 -600
- package/plugins/immune-brain/.pi-extension/pi-canary-assurance-progression.ts +1 -0
- package/plugins/immune-brain/.pi-extension/pi-canary-verification.ts +3 -3
- package/plugins/immune-brain/.pi-extension/runtime-stub.ts +17 -43
- package/plugins/immune-brain/dist/claude/mcp-server.mjs +7581 -5042
- package/plugins/immune-brain/dist/docs/reference/planning-artifact-retention.md +11 -12
- package/plugins/immune-brain/dist/docs/reference/subagent-dispatch-protocol.md +1 -1
- package/plugins/immune-brain/dist/imm-loop.md +27 -25
- package/plugins/immune-brain/dist/imm-planner.md +53 -32
- package/plugins/immune-brain/dist/imm-review-retro.md +123 -0
- package/plugins/immune-brain/dist/registry.yaml +9 -0
- package/plugins/immune-brain/dist/role-prompts/code-review.md +3 -1
- package/plugins/immune-brain/dist/role-prompts/executor.md +4 -4
- package/plugins/immune-brain/runtime/assurance/coordinator.ts +183 -40
- package/plugins/immune-brain/runtime/assurance/delivery_workspace.ts +240 -0
- package/plugins/immune-brain/runtime/assurance/qa.ts +132 -58
- package/plugins/immune-brain/runtime/assurance/review_evidence.ts +15 -7
- package/plugins/immune-brain/runtime/assurance/verification.ts +246 -206
- package/plugins/immune-brain/runtime/authorization_operation.ts +20 -0
- package/plugins/immune-brain/runtime/claude/kernel_ports.ts +288 -721
- package/plugins/immune-brain/runtime/commands/kernel.ts +158 -67
- package/plugins/immune-brain/runtime/github_issue_tracker.ts +254 -29
- package/plugins/immune-brain/runtime/kernel/actor_identity.ts +33 -0
- package/plugins/immune-brain/runtime/kernel/application.ts +22 -6
- package/plugins/immune-brain/runtime/kernel/assurance_projection.ts +94 -5
- package/plugins/immune-brain/runtime/kernel/authority_port.ts +27 -6
- package/plugins/immune-brain/runtime/kernel/backend_claim.ts +43 -16
- package/plugins/immune-brain/runtime/kernel/batch_authority.ts +10 -6
- package/plugins/immune-brain/runtime/kernel/canary_application.ts +50 -63
- package/plugins/immune-brain/runtime/kernel/canary_eligibility.ts +13 -4
- package/plugins/immune-brain/runtime/kernel/completion.ts +5 -14
- package/plugins/immune-brain/runtime/kernel/enrollment.ts +124 -34
- package/plugins/immune-brain/runtime/kernel/enrollment_authority.ts +13 -5
- package/plugins/immune-brain/runtime/kernel/index.ts +3 -1
- package/plugins/immune-brain/runtime/kernel/intent.ts +7 -11
- package/plugins/immune-brain/runtime/kernel/legacy_audit.ts +4 -1
- package/plugins/immune-brain/runtime/kernel/legacy_task_record.ts +323 -0
- package/plugins/immune-brain/runtime/kernel/pi_canary_prepare.ts +10 -1
- package/plugins/immune-brain/runtime/kernel/reducer.ts +32 -31
- package/plugins/immune-brain/runtime/kernel/run_identity.ts +121 -0
- package/plugins/immune-brain/runtime/kernel/spec_binding.ts +100 -0
- package/plugins/immune-brain/runtime/kernel/sqlite_migration.ts +950 -0
- package/plugins/immune-brain/runtime/kernel/sqlite_store.ts +1193 -0
- package/plugins/immune-brain/runtime/kernel/storage.ts +1254 -1206
- package/plugins/immune-brain/runtime/kernel/storage_layout_migration.ts +129 -755
- package/plugins/immune-brain/runtime/kernel/storage_paths.ts +419 -46
- package/plugins/immune-brain/runtime/kernel/types.ts +12 -43
- package/plugins/immune-brain/runtime/kernel/validation.ts +60 -274
- package/plugins/immune-brain/runtime/managed_task_routing_policy.ts +0 -1
- package/plugins/immune-brain/runtime/plan_core.ts +27 -65
- package/plugins/immune-brain/runtime/plugin_version.ts +1 -1
- package/plugins/immune-brain/runtime/prompts/code-review.md +3 -1
- package/plugins/immune-brain/runtime/prompts/executor.md +4 -4
- package/plugins/immune-brain/runtime/staged_intent.ts +58 -0
- package/plugins/immune-brain/runtime/unattended/batch_git.ts +37 -7
- package/plugins/immune-brain/runtime/unattended/batch_plan.ts +42 -2
- package/plugins/immune-brain/runtime/unattended/batch_preflight.ts +771 -0
- package/plugins/immune-brain/runtime/unattended/batch_reasons.ts +189 -0
- package/plugins/immune-brain/runtime/unattended/batch_runner.ts +35 -0
- package/plugins/immune-brain/runtime/unattended/confirmation_deadline.ts +33 -0
- package/plugins/immune-brain/runtime/unattended/types.ts +14 -1
- package/plugins/immune-brain/runtime/v4_runtime.ts +19 -23
- package/plugins/immune-brain/runtime/verification_descriptor.ts +92 -136
- package/plugins/immune-brain/runtime/workspace_scope.ts +98 -13
- package/plugins/immune-brain/skills/imm-planner/SKILL.md +3 -3
- package/plugins/immune-brain/skills/imm-review-retro/SKILL.md +23 -0
- package/plugins/immune-brain/skills/imm-review-retro/scripts/review_retro.ts +355 -0
- package/plugins/immune-brain/skills/registry.yaml +9 -0
- package/plugins/immune-brain/bin/imm-retire-stale-wrapper +0 -4
- package/plugins/immune-brain/bin/imm-retired +0 -4
- package/plugins/immune-brain/runtime/authority_commit_receipts.ts +0 -716
- package/plugins/immune-brain/runtime/kernel/automatic_observations.ts +0 -451
- package/plugins/immune-brain/runtime/kernel/legacy.ts +0 -299
- package/plugins/immune-brain/runtime/kernel/observation.ts +0 -397
- package/plugins/immune-brain/runtime/kernel/readiness.ts +0 -282
- package/plugins/immune-brain/runtime/kernel/readiness_evidence.ts +0 -132
|
@@ -73,18 +73,17 @@ Non-terminal artifacts remain durable at their existing paths by default.
|
|
|
73
73
|
|
|
74
74
|
## Authority-Owned Lifecycle
|
|
75
75
|
|
|
76
|
-
An enrolled TaskIntent
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
or repair lifecycle state.
|
|
76
|
+
An enrolled TaskIntent remains under `docs/plans/` for the whole run. A complex
|
|
77
|
+
task may bind one active Spec under `docs/specs/` by content identity. Before QA,
|
|
78
|
+
the Kernel freezes those artifacts in place: Git objects are bound, source paths
|
|
79
|
+
do not move, and `TaskRecord.intent_ref.path` stays on the active sidecar.
|
|
80
|
+
Historical files already in `archive/` remain readable. A later edit of a bound
|
|
81
|
+
Spec changes the delivery identity, so prior assurance cannot be reused.
|
|
82
|
+
|
|
83
|
+
Authorized Review rework returns `artifact_state` to `active` without relocating
|
|
84
|
+
files. Completion and stop freeze in place; they do not archive planning
|
|
85
|
+
artifacts. A crash fails closed and converges only through the store transaction;
|
|
86
|
+
no background scanner or second status writer may infer or repair lifecycle state.
|
|
88
87
|
|
|
89
88
|
Terminal cleanup outside that lifecycle remains a bounded TaskIntent with an
|
|
90
89
|
explicit candidate list and a copy-paste verification command. It may accompany
|
|
@@ -80,7 +80,7 @@ Agent:
|
|
|
80
80
|
schedule: ""
|
|
81
81
|
```
|
|
82
82
|
|
|
83
|
-
Pi `Agent` has no `readonly` parameter; the empty tool policy, child type, and prompt contract enforce the read-only boundary. The Parent starts one foreground Agent at a time and consumes its direct result before deciding whether another child is needed. Advisory and discovery work does not
|
|
83
|
+
Pi `Agent` has no `readonly` parameter; the empty tool policy, child type, and prompt contract enforce the read-only boundary. The Parent starts one foreground Agent at a time and consumes its direct result before deciding whether another child is needed. Advisory and discovery work does not poll for deferred results or depend on completion notifications or host `followUp`. Kernel Review uses `subagent_type: "Review"`; its structured verdict is submitted by the Parent and validated against the current immutable snapshot. Loop `arch-explorer` uses `subagent_type: "Explore"`; other advisory/discovery and Loop internal roles use `general-purpose`.
|
|
84
84
|
|
|
85
85
|
## Scheduling And Visibility
|
|
86
86
|
|
|
@@ -11,9 +11,9 @@ This skill adheres to the **[BASELINE.md](BASELINE.md)**.
|
|
|
11
11
|
|
|
12
12
|
Only explicit `imm-loop` entry starts or resumes this loop. Ordinary host input
|
|
13
13
|
stays host-native; it never resumes a Managed owner implicitly. Read the current
|
|
14
|
-
Host's `
|
|
15
|
-
|
|
16
|
-
|
|
14
|
+
Host's `status` projection first and verify the exact active backend claim,
|
|
15
|
+
TaskIntent, and TaskRecord. Invalid or contradictory projections fail closed. A
|
|
16
|
+
candidate TaskIntent is not Enrollment authority.
|
|
17
17
|
|
|
18
18
|
Before the first Enrollment of a candidate TaskIntent, confirm the Planner
|
|
19
19
|
returned `tracker_associated` — but only when the candidate belongs to an
|
|
@@ -36,11 +36,12 @@ Historical prose Plans and State Ledgers are read-only history, not execution
|
|
|
36
36
|
instructions. Do not create Steps, workflow profiles, follow-up ledgers, or
|
|
37
37
|
successor Plans to drive a Kernel task.
|
|
38
38
|
|
|
39
|
-
At every internal role boundary call the read-only
|
|
40
|
-
`route` for current-context Executor work, bounded repair,
|
|
41
|
-
exploration, advisory review, Compounder, Kernel ownership, or
|
|
42
|
-
Use Kernel ownership for an enrolled task. This
|
|
43
|
-
not record execution evidence, mutate task state, or replace
|
|
39
|
+
At every internal role boundary call the invoking Host's read-only role-boundary
|
|
40
|
+
route. Use `route` for current-context Executor work, bounded repair,
|
|
41
|
+
architecture exploration, advisory review, Compounder, Kernel ownership, or
|
|
42
|
+
scope expansion. Use Kernel ownership for an enrolled task. This route projects
|
|
43
|
+
authority; it does not record execution evidence, mutate task state, or replace
|
|
44
|
+
Kernel operations.
|
|
44
45
|
Before a child dispatch, read the [Subagent Dispatch Protocol](docs/reference/subagent-dispatch-protocol.md#authorization-authority).
|
|
45
46
|
Never load an internal role as a public Skill or spawn another loop process.
|
|
46
47
|
The standalone `imm-pr-fix`, `imm-doc-prune`, and `imm-agent-doc-maintain` are host-native
|
|
@@ -51,26 +52,27 @@ and `pr-fix` repairs remain bounded by the enrolled TaskIntent.
|
|
|
51
52
|
|
|
52
53
|
Continue while the current projection has a valid action:
|
|
53
54
|
|
|
54
|
-
1. For active artifacts, implement only the enrolled acceptance within
|
|
55
|
-
`scope_hint` in the current conversation.
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
The Kernel
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
55
|
+
1. For active artifacts, implement only the enrolled acceptance within the
|
|
56
|
+
`scope_hint` envelope in the current conversation. New helpers or tests
|
|
57
|
+
inside an approved directory or glob do not require a revision. Run focused
|
|
58
|
+
checks. Executor checks are diagnostic evidence, not a QA or Review approval.
|
|
59
|
+
2. Call `advance_assurance` in the foreground and consume its direct terminal
|
|
60
|
+
result. The Kernel freezes the artifacts itself before QA: it binds Git
|
|
61
|
+
content identity in place without relocating source paths. A simple task has
|
|
62
|
+
TaskIntent only; a complex task may bind one active Spec. Deterministic QA
|
|
63
|
+
runs all descriptors in a disposable delivery materialization, never against
|
|
64
|
+
unchecked live worktree source. Do not dispatch a separate per-Step QA Agent.
|
|
65
|
+
3. On `review_ready`, invoke the returned `agent_params` as one exact foreground
|
|
64
66
|
Agent call, then pass its structured verdict to `submit_review`. Do not
|
|
65
67
|
replace this snapshot-bound reviewer with a generic role dispatch. The Parent
|
|
66
68
|
cannot issue its own QA or Review pass.
|
|
67
|
-
|
|
69
|
+
4. Follow the returned Kernel obligation. Fresh QA suffices for routine work;
|
|
68
70
|
material and critical work additionally require fresh independent Review.
|
|
69
71
|
Normal completion does not require a second user confirmation.
|
|
70
|
-
|
|
72
|
+
5. For rework, follow the projected artifact state before editing. Resolve
|
|
71
73
|
findings only after fixing and verifying their cause. Changed snapshots
|
|
72
|
-
invalidate old evidence;
|
|
73
|
-
|
|
74
|
+
invalidate old evidence; run the newly required obligations.
|
|
75
|
+
6. An unresolved decision pauses only dependent execution. On `awaiting_user`,
|
|
74
76
|
invoke `request_authorization` directly before ending the turn; use the
|
|
75
77
|
Decisions and Recovery route for its native-gate handling. End the turn if
|
|
76
78
|
the decision remains unresolved, is cancelled, or the gate fails. Otherwise
|
|
@@ -122,9 +124,9 @@ per completed child.
|
|
|
122
124
|
enrolled intent sidecars or ask for chat pre-confirmation.
|
|
123
125
|
- On `awaiting_user`, invoke `request_authorization` directly for a concrete
|
|
124
126
|
unresolved decision or rework authorization, not risk tier alone.
|
|
125
|
-
- When the user explicitly asks to stop
|
|
126
|
-
|
|
127
|
-
|
|
127
|
+
- When the user explicitly asks to stop an active task, invoke the Kernel stop
|
|
128
|
+
operation through the invoking Host directly. Its single native confirmation
|
|
129
|
+
authorizes existing Kernel stop settlement.
|
|
128
130
|
Cancellation is not task termination. A busy invocation must finish or be
|
|
129
131
|
cancelled through existing Host controls before requesting stop; never clear
|
|
130
132
|
claims manually or use this operation to force-kill QA.
|
|
@@ -115,8 +115,9 @@ and clarification. It does not apply to Enrolled Intent Revision.
|
|
|
115
115
|
Before authoring a TaskIntent, trace each expected behavior from its public or
|
|
116
116
|
runtime entry point through existing imports and callers to the highest focused
|
|
117
117
|
behavioral tests. Include generated or packaged mirrors and every owner of the
|
|
118
|
-
same state machine. Record the concrete paths in the Spec's discovery evidence
|
|
119
|
-
|
|
118
|
+
same state machine. Record the concrete paths in the Spec's discovery evidence when the work is
|
|
119
|
+
complex; simple TaskIntent-only work records them in `scope_hint`.
|
|
120
|
+
Do not author while a referenced sibling is unresolved. Use the smallest
|
|
120
121
|
coherent module directory for ordinary implementation scope. Keep Kernel,
|
|
121
122
|
authority, migration, secret, and security-sensitive scope exact to the files
|
|
122
123
|
proved necessary by the trace. Scope is closed by reference evidence, not by an
|
|
@@ -172,8 +173,13 @@ with `valid: true` and `enrollment_ready: true`. Resolve `../bin/imm-tracker` fr
|
|
|
172
173
|
Initiative slug and goal, Parent projection, and every Child's `slice_id`,
|
|
173
174
|
canonical TaskIntent path, bounded public `acceptance` summaries, and public
|
|
174
175
|
projection. The Parent projection requires
|
|
175
|
-
`problem`, `result`, and `design`, and may include
|
|
176
|
-
`
|
|
176
|
+
`short_name`, `title`, `problem`, `result`, and `design`, and may include
|
|
177
|
+
`source_issue`, `decisions`,
|
|
178
|
+
`testing_strategy`, and `out_of_scope`. `short_name` (1-32 characters) is the
|
|
179
|
+
stable short Initiative name used in every Issue title; `title` (1-60
|
|
180
|
+
characters) is the short Initiative display title; `source_issue` is the
|
|
181
|
+
originating feature Issue number, rendered as a Provenance link. `design`
|
|
182
|
+
records Initiative-level
|
|
177
183
|
invariants, Slice boundaries and ordering, shared interfaces or state flow, and
|
|
178
184
|
material compatibility decisions. Every Parent Slice must correspond to one
|
|
179
185
|
published Child; future checklist-only Slices are not allowed in the batch.
|
|
@@ -181,15 +187,24 @@ published Child; future checklist-only Slices are not allowed in the batch.
|
|
|
181
187
|
Each Child must provide public `acceptance` entries with `id` and a 1-500
|
|
182
188
|
character `summary`. Their IDs must match every canonical TaskIntent acceptance
|
|
183
189
|
ID exactly once. Canonical assertion prose is authority evidence and must never
|
|
184
|
-
be copied into public GitHub projection. Each Child projection
|
|
190
|
+
be copied into public GitHub projection. Each Child projection requires
|
|
191
|
+
`title` (1-60 characters), the short Slice display title, and may contain
|
|
185
192
|
`result`, `current_behavior`,
|
|
186
193
|
`desired_behavior`, `key_interfaces`, `verification`, `blocked_by` Task IDs,
|
|
187
|
-
`out_of_scope`, and `agent_handoff`. The tracker
|
|
194
|
+
`out_of_scope`, and `agent_handoff`. The tracker composes Issue titles from
|
|
195
|
+
these display names only — the Parent as `[<short_name>] <title>` and each
|
|
196
|
+
Child as `[<short_name>] S<n> <title>` with `n` the declared Slice position —
|
|
197
|
+
and fails the whole batch closed before any remote write when a display name
|
|
198
|
+
is missing or the composed title exceeds 80 characters; it never falls back to
|
|
199
|
+
goal prose and never truncates a title. The tracker rereads every canonical
|
|
188
200
|
TaskIntent for identity, risk, and acceptance IDs; projection fields and public
|
|
189
201
|
summaries never widen TaskIntent scope or authority. It validates the complete dependency graph before
|
|
190
202
|
remote writes, creates the Parent once, creates all Children, attaches every
|
|
191
203
|
Child as a native Sub-issue, creates native `blocked_by` relations, and rereads
|
|
192
|
-
the complete topology.
|
|
204
|
+
the complete topology. Every Child carries `ready-for-agent`, blocked Children
|
|
205
|
+
additionally carry `blocked`, and the Parent carries neither; the tracker never
|
|
206
|
+
creates labels, so a repository missing a required label fails the batch closed
|
|
207
|
+
before any remote write. The Child Agent Brief includes a direct Parent Issue link.
|
|
193
208
|
Internal role prompts, tool policies, review gates, model reservations, and
|
|
194
209
|
prompt digests never belong in this external handoff. If
|
|
195
210
|
`docs/initiatives/<slug>.md` exists, publication fails with a carrier conflict;
|
|
@@ -235,19 +250,25 @@ and an exact retry action; the strict no-amendment default is unchanged.
|
|
|
235
250
|
### Verification Descriptor Discipline
|
|
236
251
|
|
|
237
252
|
Every acceptance verification descriptor must be a focused, deterministic,
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
253
|
+
project-owned check that exercises only its acceptance assertion. Use
|
|
254
|
+
`assurance_kernel/verification_descriptor/v2` and reuse an existing project
|
|
255
|
+
script or host tool through its literal `command`; do not infer a language,
|
|
256
|
+
package manager, or runner. Add `environment.prepare` only when the check needs
|
|
257
|
+
explicit setup, and declare only the generated directories it needs in
|
|
258
|
+
`environment.writable_paths`. Never hide package installation inside an
|
|
259
|
+
acceptance command. Prefer the highest existing observable behavioral test seam
|
|
260
|
+
and the fewest sufficient seams; never use the full test suite, a build, network
|
|
261
|
+
access, or redundant heavyweight checks as acceptance. Cite relevant test prior
|
|
262
|
+
art and explain how the selected seam catches the intended regression. This is
|
|
263
|
+
a planning heuristic: it must not weaken acceptance-specific focused
|
|
264
|
+
verification descriptors or add a mandatory user confirmation. Use the smallest
|
|
265
|
+
`timeout_ms` and `max_output_bytes` that cover deterministic post-implementation
|
|
266
|
+
QA. A v1 descriptor is historical-only and requires explicit Intent revision
|
|
267
|
+
before execution.
|
|
247
268
|
|
|
248
269
|
## Core Responsibilities
|
|
249
270
|
|
|
250
|
-
- **Decomposition**: Convert requirements into a
|
|
271
|
+
- **Decomposition**: Convert requirements into one or more TaskIntents. Add a Spec under `docs/specs/` only for complex work. Treat Technical Design as one TaskIntent decomposition dimension alongside outcome, Verification, dependency, risk, rollback, compatibility, and authority.
|
|
251
272
|
- **Outcome Focus**: Each TaskIntent owns one independently verifiable outcome. Implementation batches are Executor work, not separately authorized read/edit/run Steps.
|
|
252
273
|
- **Planning granularity**: Keep a coherent outcome together when acceptance, risk, rollback, and authority can settle together. Use the TaskIntent decomposition rules below for independent outcomes. File count, tokens, compactions, elapsed time, and review rounds are evidence for judgment, not universal gates.
|
|
253
274
|
- **Historical artifacts**: v3 prose Plan mutation is retired. `imm-plan` is a read-only validator for archived Plans; create no new Roadmap, Phase, successor Plan, or State Ledger.
|
|
@@ -259,9 +280,9 @@ descriptors or add a mandatory user confirmation. Use the smallest `timeout_ms`
|
|
|
259
280
|
- **TaskIntent decomposition**: Use the selected design boundaries as one retain/split criterion for TaskIntent slices. Keep work in one TaskIntent when the selected views describe one coherent executable slice with shared acceptance, risk treatment, rollback, and authority. Split a successor TaskIntent when a service boundary, state-machine owner, migration/compatibility boundary, independently promotable layer, or sequence dependency needs independent verification, rollback, authorization, or settlement. Do not split merely because the design names several layers, files, or services. Treat trust-boundary changes as the same kind of decomposition evidence: a TaskIntent should normally change one primary trust-boundary invariant, while merely traversing several boundaries or updating both sides of one end-to-end authority chain does not require a split. Split separate trust invariants when they can be independently verified, rolled back, authorized, migrated, or settled. Keep multiple trust-boundary changes together only when they form one atomic security outcome and splitting would create an unsafe or unusable intermediate state; record that reason in the Spec. This is Planner judgment, not a TaskIntent schema field or an Enrollment counting rule. This does not revive prose Plan, Roadmap, or Phase authority.
|
|
260
281
|
- **Mermaid Use**: Mermaid is required only when a medium/high-risk design contains structure, sequence, data flow, or state transition relationships that a diagram materially clarifies. Mermaid is not a universal gate; a diagram supplements adjacent prose and never becomes a second design authority. Medium/High risk Specs record `**Diagram decision**: required|not_required` and a non-empty `**Diagram reason**:`. A `required` decision must have a Mermaid block; `not_required` explains why prose is sufficient. Low-risk Specs omit the empty ceremony and record neither field.
|
|
261
282
|
- **Verification**: Every acceptance assertion has a concrete focused descriptor that can fail on the intended regression. Hypothetical evidence is not execution-ready.
|
|
262
|
-
- **Executable Scope**: `scope_hint` is the mutation envelope, not discovery context. Close references across callers, tests, generated mirrors, and state-machine owners before authoring.
|
|
263
|
-
- **Devil's Advocate Preplan Audit**: Medium/High risk work records a `Devil's Advocate Audit` in
|
|
264
|
-
- **Execution posture**: Record `test-first` or `characterization-first`
|
|
283
|
+
- **Executable Scope**: `scope_hint` is the mutation envelope, not discovery context. Close references across callers, tests, generated mirrors, and state-machine owners before authoring. Simple tasks are TaskIntent-only. A complex task binds at most one active Spec by content identity; do not add archive paths for freeze, and do not relocate artifacts. Collect all known scope gaps in one revision request; ask again only when new evidence changes the boundary.
|
|
284
|
+
- **Devil's Advocate Preplan Audit**: Medium/High risk work records a `Devil's Advocate Audit` in its Spec covering rollback resilience, verification vanity, and spec dilution detection. Explain recovery from partial implementation, why verification detects the regression, and how accepted requirements remain covered. Simple TaskIntent-only work records outcome, boundary, and concrete verification on the Intent. Low-risk work omits the empty template.
|
|
285
|
+
- **Execution posture**: Record `test-first` or `characterization-first` on the Spec when the work is complex, otherwise on the TaskIntent, when explicitly requested or justified by fragile untested behavior. The Executor owns the local choreography; do not create prototype or RED/GREEN/REFACTOR authority Steps. Throwaway probes must have a cleanup condition and a durable decision output.
|
|
265
286
|
|
|
266
287
|
## Settlement-Design Contract
|
|
267
288
|
|
|
@@ -350,21 +371,21 @@ preparation does not apply the revision or authorize expanded execution.
|
|
|
350
371
|
planning heuristic: it must not weaken acceptance-specific focused
|
|
351
372
|
verification descriptors or add a mandatory user confirmation.
|
|
352
373
|
- **Review Mapping**: In-scope rework stays with the enrolled TaskIntent and explicit `imm-loop` entry. Cross-scope findings become a Planner decision delta with concrete missing paths and verification evidence; do not create a successor prose Plan.
|
|
353
|
-
- **Brainstorm Manifest Mapping**: Record every upstream `BR-*` item in a Spec `Brainstorm Trace
|
|
374
|
+
- **Brainstorm Manifest Mapping**: Record every upstream `BR-*` item in a Spec `Brainstorm Trace` when the work is complex, otherwise on the TaskIntent, mapped to acceptance, a captured decision, or an explicit reason for deferral or exclusion. Resolve every `BR-Q-*` item before handoff. Do not silently narrow confirmed framing.
|
|
354
375
|
- **Session Lifecycle Ownership**: The user chooses the current or a new session. Tokens, compactions, tool counts, elapsed time, and review rounds never trigger automatic session creation or termination. Recovery uses TaskRecord and the fresh Kernel projection.
|
|
355
376
|
- **Subagents**: Only when optional research is needed, read Research Dispatch and its shared dispatch reference. Default to inline evidence gathering. Plan conditional reviewers such as `security-reviewer` only if their trigger surfaces are explicit; do not manufacture them.
|
|
356
377
|
- **Enrolled Intent**: Follow Enrolled Intent Revision for a Loop-requested scope or acceptance change; candidate preparation never changes the current owner or grants execution authority.
|
|
357
378
|
- **CONTEXT.md Vocabulary**: Consult the relevant `CONTEXT.md` terms when domain meaning is unclear or changes; known file-local tasks do not require a full root-document read. `CONTEXT.md` is vocabulary and architecture navigation, not execution state.
|
|
358
|
-
- **Discovery Protocol**: Read `CONTEXT.md` `## Architecture Map` before broad searching; consult relevant `docs/solutions/` evidence under Clarification supplement's history trigger. Record concrete file pointers and reasons in the Spec. Do not read or write a legacy Step discovery cache.
|
|
359
|
-
- **Planning Quality Gate**: For elevated-risk work, verify contract surfaces, compatibility, interruption recovery, rollback, verification strength, and Brainstorm traceability in the Spec. Do not invoke retired Plan mutation or State Ledger synchronization.
|
|
379
|
+
- **Discovery Protocol**: Read `CONTEXT.md` `## Architecture Map` before broad searching; consult relevant `docs/solutions/` evidence under Clarification supplement's history trigger. Record concrete file pointers and reasons in the Spec when one exists, otherwise on the TaskIntent. Do not read or write a legacy Step discovery cache.
|
|
380
|
+
- **Planning Quality Gate**: For elevated-risk complex work, verify contract surfaces, compatibility, interruption recovery, rollback, verification strength, and Brainstorm traceability in the Spec. Simple TaskIntent-only work verifies those properties on the Intent. Do not invoke retired Plan mutation or State Ledger synchronization.
|
|
360
381
|
- **Parallel Probes**: Optional read-only probes must have bounded non-overlapping scopes, expected evidence, and no file or authority writes. They are advisory discovery, not persisted Step annotations. Probe failure falls back to inline investigation with a recorded reason.
|
|
361
382
|
|
|
362
383
|
## Research Dispatch
|
|
363
384
|
|
|
364
385
|
Follow [`docs/reference/subagent-dispatch-protocol.md`](docs/reference/subagent-dispatch-protocol.md) for the full dispatch lifecycle. This section defines planner-specific optional research dispatch.
|
|
365
386
|
|
|
366
|
-
Use
|
|
367
|
-
`advisory-reviewer` routing. Invoke the returned foreground Agent envelope
|
|
387
|
+
Use the invoking Host's read-only role-boundary route for bounded
|
|
388
|
+
`arch-explorer` and explicit-lens `advisory-reviewer` routing. Invoke the returned foreground Agent envelope
|
|
368
389
|
exactly. The Parent owns Spec/TaskIntent synthesis, Brainstorm traceability,
|
|
369
390
|
acceptance, and scope; children return evidence only. On Pi the `arch-explorer` envelope
|
|
370
391
|
uses `subagent_type: "Explore"`; invoke the returned envelope rather than
|
|
@@ -382,19 +403,19 @@ optional advisory dispatch fails, continue inline and record the reason.
|
|
|
382
403
|
|
|
383
404
|
## Boundary
|
|
384
405
|
|
|
385
|
-
- **Allowed**: Write
|
|
406
|
+
- **Allowed**: Write a TaskIntent. Add a Spec only for complex work. Initiative planning carriers and necessary domain vocabulary.
|
|
386
407
|
- **Blocked**: Implementation edits, direct Kernel-store writes, enrolled intent overwrites, and QA/Review decisions.
|
|
387
408
|
- **Workflow guard**: Execution continues through native Enrollment and explicit `imm-loop`. Planner owns design and decomposition, not execution authority.
|
|
388
409
|
|
|
389
410
|
## Output artifact
|
|
390
411
|
|
|
391
|
-
|
|
392
|
-
|
|
412
|
+
Canonical candidate `docs/plans/<task-id>.intent.json`. Simple future work does
|
|
413
|
+
not require a Spec. A complex task adds one Spec under `docs/specs/` bound by
|
|
414
|
+
immutable content identity, not by archive relocation. The Spec records outcome, discovery evidence,
|
|
393
415
|
decisions, assumptions, Technical Design when required, output language,
|
|
394
416
|
the Medium/High risk Devil's Advocate Audit, and acceptance/test mapping. Low
|
|
395
417
|
risk records outcome, boundary, and concrete verification without the empty
|
|
396
|
-
ceremony. Include a complete
|
|
397
|
-
`Brainstorm Trace` when consuming a Brainstorm manifest. TaskIntent is authored
|
|
418
|
+
ceremony. Include a complete Spec `Brainstorm Trace` for complex work that consumes a Brainstorm manifest; simple TaskIntent-only work records the same coverage on the Intent. TaskIntent is authored
|
|
398
419
|
and validated through `imm-kernel`; do not write a prose iteration Plan or sync
|
|
399
420
|
a State Ledger. Keep historical Plan validation strictly read-only.
|
|
400
421
|
|
|
@@ -417,7 +438,7 @@ a State Ledger. Keep historical Plan validation strictly read-only.
|
|
|
417
438
|
|
|
418
439
|
- Acceptance verification names only hypothetical evidence with no runnable descriptor.
|
|
419
440
|
- New work depends on a prose Plan validator, Step activation, or State Ledger.
|
|
420
|
-
- A Brainstorm manifest lacks
|
|
441
|
+
- A Brainstorm manifest lacks complete `BR-*` coverage on the Spec (complex) or TaskIntent (simple).
|
|
421
442
|
- A Medium/High risk Spec lacks a `Devil's Advocate Audit` covering rollback resilience, verification vanity, and spec dilution detection.
|
|
422
443
|
- New Spec prose ignores the document-language policy.
|
|
423
444
|
- Candidate artifacts escape the approved planning scope.
|
|
@@ -425,7 +446,7 @@ a State Ledger. Keep historical Plan validation strictly read-only.
|
|
|
425
446
|
## Verification
|
|
426
447
|
|
|
427
448
|
- Validate every candidate through `imm-kernel intent validate <path> --json` after authoring and staging. Require `valid: true` and `enrollment_ready: true` before Enrollment.
|
|
428
|
-
-
|
|
449
|
+
- For complex work, verify Spec design metadata, document language, reference closure, concrete descriptor paths, and complete Brainstorm traceability before handoff. Simple TaskIntent-only work verifies those properties on the Intent.
|
|
429
450
|
- Enrollment validates descriptor structure only. Deterministic QA owns descriptor execution after implementation; planning does not run the acceptance suite.
|
|
430
451
|
- Managed execution handoff is Git-tracked TaskIntent author/validate plus current-Host native Enrollment. Do not sync a v3 State Ledger or invoke a missing dispatcher.
|
|
431
452
|
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: imm-review-retro
|
|
3
|
+
description: Use when the user explicitly requests Immune-Brain ranking of models by cross-model review load or a project usage retro.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Immune-Brain: Review Retro
|
|
7
|
+
|
|
8
|
+
Rank models by how much code review their own edits triggered, and report
|
|
9
|
+
basic project usage over a look-back window the user supplies in days. This
|
|
10
|
+
is a standalone host-native analysis entry, not a Managed Path continuation
|
|
11
|
+
and not an `imm-loop` internal-role dispatch. It reviews no diff — a diff
|
|
12
|
+
review is `code-review`.
|
|
13
|
+
|
|
14
|
+
## Boundary
|
|
15
|
+
|
|
16
|
+
Allowed: read pi session JSONL under `~/.pi/agent/sessions` (or `--root`),
|
|
17
|
+
run the bundled analyzer, and write a stdout report.
|
|
18
|
+
|
|
19
|
+
Blocked: code, test, Spec, Plan, or `.imm/` edits; session-log writes;
|
|
20
|
+
Kernel, TaskIntent, or TaskRecord mutation; Compounder or scheduled runs;
|
|
21
|
+
`.imm/audit/` lifecycle statistics.
|
|
22
|
+
|
|
23
|
+
An already active Managed task remains owned by `imm-loop`. This Skill does
|
|
24
|
+
not create or resume Managed authority.
|
|
25
|
+
|
|
26
|
+
## Invocation
|
|
27
|
+
|
|
28
|
+
Requires explicit invocation: `imm-review-retro` or `/imm-review-retro`.
|
|
29
|
+
Ordinary questions such as "which model is worse" stay host-native and do
|
|
30
|
+
not enter this Skill.
|
|
31
|
+
|
|
32
|
+
The look-back window in days is required input. If the user named one, use
|
|
33
|
+
it. If not, ask before running, because the ranking moves with the window.
|
|
34
|
+
|
|
35
|
+
Default scan is the user's full session-log tree. Pass `--project <substr>`
|
|
36
|
+
when the user wants one repo or worktree. Do not invent a project filter.
|
|
37
|
+
|
|
38
|
+
No daemon, no cron, no CI, no automatic commit.
|
|
39
|
+
|
|
40
|
+
## Counting rules
|
|
41
|
+
|
|
42
|
+
These rules keep numbers comparable across runs. Read the analyzer header
|
|
43
|
+
aloud in the report so the 口径 stays visible.
|
|
44
|
+
|
|
45
|
+
- `review` = an `Agent` tool call with `subagent_type` equal to `Review`.
|
|
46
|
+
- Attribution = the model behind the most recent `edit`, `write`, or
|
|
47
|
+
`multiedit` in that session. If none, the row is `no-edit (review-only)`.
|
|
48
|
+
- `uniq` counts distinct (session, description+prompt prefix) pairs. A wide
|
|
49
|
+
gap versus `reviews` is the same review re-run on the same code.
|
|
50
|
+
- `rev/100ed` is `100 * reviews / devEdits`. Rank on both absolute `reviews`
|
|
51
|
+
and this intensity. A model can lead one axis and sit mid-pack on the
|
|
52
|
+
other.
|
|
53
|
+
- `avgSc` / `pass%` parse `[SCORE: …]` and `[VERDICT: …]` tags from the
|
|
54
|
+
matching Review `toolResult`. Untagged reviews show `-`.
|
|
55
|
+
- `registr` counts the Kernel `submit_review` registration in the session log.
|
|
56
|
+
It is the registration of the same review and is never added into `reviews`.
|
|
57
|
+
- `rounds/task` is registrations per distinct `(cwd, task_id)`. High values
|
|
58
|
+
can be canary/QA harness re-registration, not human-visible rework.
|
|
59
|
+
- Findings are `record_finding` calls, deduped per session. Summaries that
|
|
60
|
+
match `recorded cleanly`, `receipt recorded`, `round recorded`, or
|
|
61
|
+
`no finding(s)` are `bookkeep` / `noisy`, excluded from `block`/`advis`.
|
|
62
|
+
|
|
63
|
+
Usage counters on the same pass: sessions with activity, assistant turns,
|
|
64
|
+
edit counts, a tool-call name histogram, and the project × author table.
|
|
65
|
+
|
|
66
|
+
## CLI
|
|
67
|
+
|
|
68
|
+
Run the bundled analyzer. Prefer `bun`; `node` (≥23.6, type stripping) is
|
|
69
|
+
an allowed equivalent. The script is erasable TypeScript with `node:` APIs
|
|
70
|
+
only.
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
bun "<path-to-skill>/scripts/review_retro.ts" <days> [--root <sessions-dir>] [--project <substr>] [--top N]
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
- `<days>` must be `> 0`.
|
|
77
|
+
- `--root` defaults to `~/.pi/agent/sessions`.
|
|
78
|
+
- `--project` keeps sessions whose `cwd` contains the substring.
|
|
79
|
+
- `--top` is the project-table row cap (default 15).
|
|
80
|
+
- Malformed JSONL lines are skipped. `days <= 0` is a hard error.
|
|
81
|
+
|
|
82
|
+
Do not scan live `.imm/` directories. Tests use committed fixtures under
|
|
83
|
+
`tests/fixtures/review-retro/`.
|
|
84
|
+
|
|
85
|
+
## Report
|
|
86
|
+
|
|
87
|
+
Write-up order:
|
|
88
|
+
|
|
89
|
+
1. Window and 口径 in one line (copy the analyzer header).
|
|
90
|
+
2. Ranked model table, including scores.
|
|
91
|
+
3. Usage section: sessions, turns, edits, tool mix.
|
|
92
|
+
4. Quality and score findings.
|
|
93
|
+
5. Three to five bullets of what the table means (volume versus intensity,
|
|
94
|
+
quality versus rework, where it concentrated).
|
|
95
|
+
6. Caveats last.
|
|
96
|
+
|
|
97
|
+
Rank on both axes, never one. Name the axis you are ranking by, and call
|
|
98
|
+
out models that flip order between `reviews` and `rev/100ed`.
|
|
99
|
+
|
|
100
|
+
Separate one-pass from rework: compare `uniq` to `reviews`, and read
|
|
101
|
+
`rounds/task` on the kernel path.
|
|
102
|
+
|
|
103
|
+
Evaluate quality: high intensity plus high score is frequent review of
|
|
104
|
+
mostly minor issues; low intensity plus low score is rare review of severe
|
|
105
|
+
defects. Call out REJECT or highRisk ratings.
|
|
106
|
+
|
|
107
|
+
Ground each model in its projects. Cite the two or three worktrees where
|
|
108
|
+
that model's reviews concentrated.
|
|
109
|
+
|
|
110
|
+
## Caveats
|
|
111
|
+
|
|
112
|
+
Anything the script splits out as `bookkeep` stays visible next to the
|
|
113
|
+
column it contaminates. Flag any finding count you cannot trace to a real
|
|
114
|
+
defect.
|
|
115
|
+
|
|
116
|
+
High `rounds/task` can be canary/QA harness re-registration, not
|
|
117
|
+
human-visible rework.
|
|
118
|
+
|
|
119
|
+
This Skill does not persist snapshots or compute week-over-week diffs.
|
|
120
|
+
Re-run with a new window when the user wants a later period.
|
|
121
|
+
|
|
122
|
+
The personal python prototype under `~/.pi/agent/skills/review-retro/` is
|
|
123
|
+
not this Skill and is not modified by it.
|
|
@@ -56,3 +56,12 @@ skills:
|
|
|
56
56
|
output_artifacts: [maintain_report]
|
|
57
57
|
next_actions: []
|
|
58
58
|
boundary: Minimize tracked agent-instruction context after explicit manifest approval; no Managed authority mutation, contract installation, or reference-document creation.
|
|
59
|
+
- name: imm-review-retro
|
|
60
|
+
path: skills/imm-review-retro/SKILL.md
|
|
61
|
+
role: execute
|
|
62
|
+
title: Review Retro
|
|
63
|
+
role_class: discovery
|
|
64
|
+
canonical: true
|
|
65
|
+
output_artifacts: [retro_report]
|
|
66
|
+
next_actions: []
|
|
67
|
+
boundary: Rank models by review load and report project usage from pi session logs. Read-only. No Managed Path mutation.
|
|
@@ -38,6 +38,8 @@ boundary name). Do not invent fields, and never send an anchor yourself: the
|
|
|
38
38
|
Kernel derives it as the sha256 of the canonical `{violated.kind,
|
|
39
39
|
violated.ref, caller_chain}`, so an identical claim keeps one stable identity
|
|
40
40
|
across review rounds while a different call chain is a different claim. A
|
|
41
|
-
passing review
|
|
41
|
+
passing review carries no blocking findings; non-blocking notes may ride along as
|
|
42
|
+
`kind: "advisory"` findings, and every finding, advisory included, needs the
|
|
43
|
+
same machine-checkable evidence. If the
|
|
42
44
|
checkpoint is `awaiting_user_successor_decision`, stop without dispatch; only
|
|
43
45
|
a literal user may invoke `--approve-successor`.
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
# Internal role: executor
|
|
2
2
|
|
|
3
3
|
You are the Immune-Brain Executor role inside Loop. Implement exactly the
|
|
4
|
-
enrolled TaskIntent acceptance and `scope_hint` (or one accepted
|
|
5
|
-
same-boundary follow-up) in the current Parent conversation.
|
|
6
|
-
|
|
7
|
-
|
|
4
|
+
enrolled TaskIntent acceptance and `scope_hint` envelope (or one accepted
|
|
5
|
+
same-boundary follow-up) in the current Parent conversation. New helpers or
|
|
6
|
+
tests inside an approved directory or glob do not require a revision. Keep
|
|
7
|
+
every edit inside the authorized envelope; do not stage unrelated user files. Do not discover or load a Pi Skill.
|
|
8
8
|
|
|
9
9
|
Before handoff, run the permitted diagnostic checks and return commands and
|
|
10
10
|
outcomes to the Parent as structured diagnostic evidence. The read-only Loop
|