specpi 0.11.2 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -1
- package/NPM_RELEASE.md +3 -1
- package/README.md +38 -28
- package/SECURITY_MODEL.md +30 -0
- package/THIRD_PARTY.md +14 -0
- package/docs/delegation/README.md +264 -0
- package/docs/delegation/design-protocol.md +382 -0
- package/docs/delegation/design.md +525 -0
- package/docs/delegation/evaluation.md +307 -0
- package/docs/delegation/protocol.md +271 -0
- package/docs/delegation/research.md +216 -0
- package/extensions/command-guard/index.ts +118 -33
- package/extensions/delegation/core.mjs +772 -0
- package/extensions/delegation/errors.mjs +8 -0
- package/extensions/delegation/extension.mjs +475 -0
- package/extensions/delegation/index.ts +9 -0
- package/extensions/delegation/managed-files.mjs +13 -0
- package/extensions/delegation/native.mjs +155 -0
- package/extensions/delegation/presentation.mjs +315 -0
- package/extensions/delegation/protocol.mjs +296 -0
- package/extensions/delegation/provider.mjs +689 -0
- package/extensions/delegation/snapshot.mjs +532 -0
- package/extensions/delegation/worker.mjs +218 -0
- package/extensions/workflow-controls/index.ts +5 -1
- package/package.json +4 -2
- package/scripts/check-package.mjs +21 -3
- package/scripts/check-pi-package.mjs +4 -0
- package/scripts/check-syntax.mjs +61 -0
- package/scripts/specpi.mjs +6 -0
- package/site/logo.svg +1 -9
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
# Evidence behind the delegation design
|
|
2
|
+
|
|
3
|
+
Reviewed through 5 September 2026. This ledger distinguishes published studies,
|
|
4
|
+
preprints, production accounts and API contracts. Design implications are our
|
|
5
|
+
inferences. None of these sources benchmarks SpecPi's experimental implementation.
|
|
6
|
+
The [implemented guide](README.md) and [calls/time protocol](protocol.md) describe its
|
|
7
|
+
current behavior. The [archived design](design.md) preserves the broader proposal;
|
|
8
|
+
publication of supporting research does not establish that either design improves
|
|
9
|
+
SpecPi outcomes. Deterministic fixture tests and empirical evaluation are separate.
|
|
10
|
+
|
|
11
|
+
## Existing seven-study foundation
|
|
12
|
+
|
|
13
|
+
The [architecture article](../../site/single-agent/index.html) and its
|
|
14
|
+
[reviewed chart data](../../site/charts/research-data.json) contain the detailed
|
|
15
|
+
comparisons. The relevant implications for this design are:
|
|
16
|
+
|
|
17
|
+
| Primary source | Evidence relevant to the decision | Consequence for the protocol |
|
|
18
|
+
| ----------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------- |
|
|
19
|
+
| [Kim et al., Nature Machine Intelligence, 24 July 2026](https://www.nature.com/articles/s42256-026-01268-y) | Coordination helps some task classes and degrades others under matched system ceilings; software samples are small | Keep a parent-only route and evaluate by task class; no universal capability cutoff |
|
|
20
|
+
| [Tran and Kiela, preprint v2, 11 April 2026](https://arxiv.org/html/2604.02460v2) | Requested reasoning budget changes the single-versus-multi-agent ranking; actual accounting is imperfect | Give the single agent the same additional resources before attributing gains to delegation |
|
|
21
|
+
| [Wunderlich et al., ACL SRW, July 2026](https://aclanthology.org/2026.acl-srw.1/) | Configured ensembles can improve static-question accuracy at comparable modeled compute | Allow measured exceptions; distinguish ensembles from tool-using repository work |
|
|
22
|
+
| [SwarmBench, preprint v1, 31 August 2026](https://arxiv.org/html/2608.30661v1) | Swarm gains on several context-intensive tasks; only four task types have cost-matched comparisons | Prioritize bounded independent research; preserve metrics, model composition and cost denominators |
|
|
23
|
+
| [Anthropic research system, 13 June 2025](https://www.anthropic.com/engineering/multi-agent-research-system) | Parallel research can improve coverage, with substantially greater token use | Budget source collection and synthesis together; no automatic assumption of cheaper outcomes |
|
|
24
|
+
| [OneFlow, preprint v1, 18 January 2026](https://arxiv.org/html/2601.12307v1) | Useful workflow structure can survive single-conversation execution; latency changes depend on workflow | Test serial structured execution as an alternative; track caching and latency independently |
|
|
25
|
+
| [MAST, NeurIPS 2025](https://papers.nips.cc/paper_files/paper/2025/hash/b1041e52d3be19f0a9bc491657488e4a-Abstract-Datasets_and_Benchmarks_Track.html) | Taxonomy covers design, coordination and verification failures; no matched single-agent control | Require explicit coverage, provenance, incomplete states and parent acceptance |
|
|
26
|
+
|
|
27
|
+
## Additional empirical evidence
|
|
28
|
+
|
|
29
|
+
### CooperBench: communication does not solve integration
|
|
30
|
+
|
|
31
|
+
[CooperBench, v2, 26 January 2026](https://arxiv.org/html/2601.13295v2), Sections 2,
|
|
32
|
+
4–5 and Appendix C/Table 5, evaluates 652 paired-feature tasks across 12 repositories
|
|
33
|
+
and four languages. Tasks deliberately involve overlapping or interdependent code;
|
|
34
|
+
agents work in separate containers. Communication reduced merge conflicts for four
|
|
35
|
+
models without a significant cooperation-success gain for any model. GPT-5's
|
|
36
|
+
successful-task counts were 315 solo versus 183 cooperative. Communication could consume
|
|
37
|
+
up to 20% of execution steps.
|
|
38
|
+
|
|
39
|
+
**Inference:** file separation, successful messaging and a clean merge are weak
|
|
40
|
+
acceptance criteria. Keep one integration owner and specify behavior and interfaces,
|
|
41
|
+
not only paths. A writing-worker proposal needs its own end-to-end evidence.
|
|
42
|
+
|
|
43
|
+
**Limits:** the 100-action ceiling is per agent, not a demonstrated match of total
|
|
44
|
+
spending. Conflict-heavy coding tasks do not characterize independent research scouts.
|
|
45
|
+
Observed associations between early planning and fewer conflicts are not a causal
|
|
46
|
+
evaluation of a planning protocol.
|
|
47
|
+
|
|
48
|
+
### MTRouter: routing is a capability with a training cost
|
|
49
|
+
|
|
50
|
+
[MTRouter, ACL final proceedings, July 2026](https://aclanthology.org/2026.acl-long.2045.pdf),
|
|
51
|
+
Sections 3–4, Table 3 and Appendix A.2, studies sequential selection among six models.
|
|
52
|
+
On 359 in-distribution HLE questions, GPT-5 achieved 25.1±1.6% accuracy at $61.8, versus
|
|
53
|
+
26.0±2.3% at $35.0 for MTRouter. Values are mean±SD over three runs, and costs aggregate
|
|
54
|
+
evaluated episodes. Both use a $2 episode ceiling and a 30-turn limit. Training-data
|
|
55
|
+
collection used 29,693 trajectories at approximately $1,620.
|
|
56
|
+
|
|
57
|
+
**Inference:** evaluate model routing separately from delegation. Use observed
|
|
58
|
+
task-state evidence and account for training and switching costs; an untrained
|
|
59
|
+
confidence heuristic is not the studied router.
|
|
60
|
+
|
|
61
|
+
**Limits:** HLE and ScienceWorld are not repository coding; the small accuracy
|
|
62
|
+
difference does not establish superiority. Exact switch probabilities differ between
|
|
63
|
+
Figure 4 and its prose, so they are not used here. Cache-related explanations are not
|
|
64
|
+
isolated causal measurements.
|
|
65
|
+
|
|
66
|
+
### Paritok-4B: a smaller packet is not automatically a better packet
|
|
67
|
+
|
|
68
|
+
[Paritok-4B, v1, 25 August 2026](https://arxiv.org/html/2608.24188v1), Section 6.2,
|
|
69
|
+
Table 9 and Section 7, tests all 300 SWE-bench Lite instances. Its line-numbered
|
|
70
|
+
compression configuration retained 27.8% of context tokens by per-instance
|
|
71
|
+
macro-average. Uncompressed context solved 122/300 tasks (40.7%); compressed context
|
|
72
|
+
solved 109/300 (36.3%). Patch-application failures were five versus sixteen. The paired
|
|
73
|
+
solve difference had exact McNemar p=0.079.
|
|
74
|
+
|
|
75
|
+
**Inference:** preserve retrieval of original source inside the grant. Count omissions
|
|
76
|
+
and downstream accepted results, not just reduction in packet size. Select relevant
|
|
77
|
+
material before adding another model to compress it.
|
|
78
|
+
|
|
79
|
+
**Limits:** this is one tool-free Sonnet 4.5 request with oracle file context and
|
|
80
|
+
diff re-anchoring, not a running coding agent. Non-significance does not demonstrate
|
|
81
|
+
equivalence. Token reduction does not include compression, caching, latency or recovery
|
|
82
|
+
costs and must not be presented as end-to-end savings.
|
|
83
|
+
|
|
84
|
+
### OrchestraBench: recovery requires a useful change in state
|
|
85
|
+
|
|
86
|
+
[OrchestraBench, v1, 5 August 2026](https://arxiv.org/html/2608.05263v1), Sections
|
|
87
|
+
5.2–5.5, Table 8 and the trusted-state ablation, separates transient tool faults from
|
|
88
|
+
latent semantic corruption in controlled arithmetic chains. Blind retry reproduced
|
|
89
|
+
latent faults. In the N=180 trusted-state ablation, latent recovery fell from 0.67 to
|
|
90
|
+
0.08 when the trusted upstream value was removed. The paired comparison used 24 pairs.
|
|
91
|
+
|
|
92
|
+
**Inference:** a recovery request must identify the failed assumption and new evidence
|
|
93
|
+
or validated state. Preserve incomplete and uncertain outcomes. A replay of the same
|
|
94
|
+
instructions is not a recovery strategy.
|
|
95
|
+
|
|
96
|
+
**Limits:** these are small synthetic mechanism probes; some outcomes are deterministic
|
|
97
|
+
by construction. A simple TF-IDF comparator solved all ten adversarial routing cases.
|
|
98
|
+
The study does not justify a dedicated LLM router or production reliability claims.
|
|
99
|
+
|
|
100
|
+
## Additional production guidance
|
|
101
|
+
|
|
102
|
+
### Cognition: useful collaborators around a single writer
|
|
103
|
+
|
|
104
|
+
[Multi-Agents: What's Actually Working, 22 April 2026](https://cognition.com/blog/multi-agents-working)
|
|
105
|
+
updates Cognition's earlier skepticism. It describes useful fresh-context review and
|
|
106
|
+
capable-model consultation while retaining one writer. Its weaker-primary consultation
|
|
107
|
+
experiment improved cost and speed but hit a quality ceiling because the primary
|
|
108
|
+
struggled to recognize when and how to ask for help.
|
|
109
|
+
|
|
110
|
+
**Inference:** preserve a strong parent; keep review context free of implementation
|
|
111
|
+
rationalization; return concrete context requests when evidence is missing. Evaluate
|
|
112
|
+
different-model consultation against a same-model fresh context. Do not infer that
|
|
113
|
+
different context makes errors statistically independent.
|
|
114
|
+
|
|
115
|
+
**Limits:** this is a production account, not a controlled budget-matched comparison.
|
|
116
|
+
Full-history forking is one reported consultation technique, but SpecPi's privacy and
|
|
117
|
+
explicit-packet boundary means it is not adopted here. No reported product bug count
|
|
118
|
+
is used as a target for SpecPi.
|
|
119
|
+
|
|
120
|
+
### Anthropic: keep the standing harness small
|
|
121
|
+
|
|
122
|
+
[The new rules of context engineering, 24 July 2026](https://claude.com/blog/the-new-rules-of-context-engineering-for-claude-5-generation-models)
|
|
123
|
+
reports reducing Claude Code's system prompt by over 80% for newer models without a
|
|
124
|
+
measured loss on its coding evaluations. It emphasizes interface design, progressive
|
|
125
|
+
disclosure, and avoiding repeated instructions.
|
|
126
|
+
|
|
127
|
+
**Inference:** put constraints in the controller and broker; load short mode instructions
|
|
128
|
+
only when needed. Do not add a second planner, repeated role prompts, or several pages
|
|
129
|
+
of standing routing heuristics to every SpecPi session.
|
|
130
|
+
|
|
131
|
+
**Limits:** this is model- and harness-specific engineering guidance. It is not evidence
|
|
132
|
+
that deleting arbitrary safety checks, context, or instructions helps every model.
|
|
133
|
+
|
|
134
|
+
## Narrow admission policy
|
|
135
|
+
|
|
136
|
+
The implemented purposes are `review` of a frozen artifact and `scout` analysis of a
|
|
137
|
+
bounded evidence question. Both may use selected-source tools. A review gets original
|
|
138
|
+
requirements, relevant constraints and actual validation facts without the parent's
|
|
139
|
+
reasoning or verdict. A scout needs selected sources and an independently checkable
|
|
140
|
+
answer. A claimed parallel benefit also needs useful concurrent parent work; a fresh
|
|
141
|
+
review or context-isolated analysis may run while the parent waits.
|
|
142
|
+
|
|
143
|
+
These are **promising experimental scenarios, not measured SpecPi improvements**.
|
|
144
|
+
SwarmBench's cost-matched gains on three of four task types do not establish an
|
|
145
|
+
advantage for every investigation or this same-model snapshot-only implementation.
|
|
146
|
+
Cognition's cross-frontier consultation does not validate a separate generic same-model
|
|
147
|
+
consultation mode. Investigation and supplied-source research therefore share `scout`;
|
|
148
|
+
there is no live-web or model-routing route.
|
|
149
|
+
|
|
150
|
+
Small edits, routine lookups, coupled mutable work, repeated role answers, writing
|
|
151
|
+
teams and blind retries remain outside this policy. Parallel parent tool calls remain
|
|
152
|
+
an alternative to creating a worker. Mode/benefit checks and rejection of duplicate
|
|
153
|
+
normalized questions constrain structure; they cannot certify semantic independence.
|
|
154
|
+
The two-worker ceiling and all numeric quotas are engineering choices awaiting local
|
|
155
|
+
evaluation, not empirical optima extracted from the papers.
|
|
156
|
+
|
|
157
|
+
## Pi compatibility evidence
|
|
158
|
+
|
|
159
|
+
The experimental SDK integration checks required public capabilities rather than an
|
|
160
|
+
exact version list. Compatible Pi updates can activate without a SpecPi patch. Missing
|
|
161
|
+
APIs and incompatible session/provider behavior detected by the runtime checks fail closed.
|
|
162
|
+
The installer still bootstraps 0.84.4; its minimum-version contract is separate. The investigation used
|
|
163
|
+
public tagged source and official documentation without live provider calls or private
|
|
164
|
+
Pi state inspection. The current implementation uses public SDK `createAgentSession`,
|
|
165
|
+
in-memory sessions and a fresh Pi `ModelRuntime` with standard authentication,
|
|
166
|
+
environment and `models.json` resolution. Child transport/thinking budgets use configured
|
|
167
|
+
global settings without project settings. Parent model/thinking are explicit, with Pi
|
|
168
|
+
clamping; unsupported runtime-only authentication, selected extension-provider overrides,
|
|
169
|
+
model-specific headers, startup proxy configuration and safe descriptor mismatches fail
|
|
170
|
+
preflight. These integration limits leave the parent's setup unchanged.
|
|
171
|
+
|
|
172
|
+
Pi runs the child loop. SpecPi admits each SDK invocation, disables retries/compaction
|
|
173
|
+
and checks SDK-visible streaming. No ambient child resources or parent transcript are
|
|
174
|
+
loaded. Parent request hooks, ephemeral runtime settings and session affinity are not
|
|
175
|
+
automatically transferred. Full parent parity and hard raw-transport, hidden-attempt,
|
|
176
|
+
memory or invoice bounds are not claimed. SDK contract tests and comparative outcomes
|
|
177
|
+
remain distinct evidence requirements.
|
|
178
|
+
|
|
179
|
+
| Contract | Primary source | Design consequence |
|
|
180
|
+
| --------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- |
|
|
181
|
+
| Credential-blind completion facade | [ModelRegistry at v0.84.4](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/model-registry.ts#L59) | Useful building block; does not expose the configured streaming pipeline |
|
|
182
|
+
| Runtime auth and request preparation | [ModelRuntime](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/model-runtime.ts#L541) | Keep credential handling inside Pi; preserve composed provider behavior |
|
|
183
|
+
| Agent sessions and thinking translation | [SDK](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/sdk.ts#L283) | Use the Pi agent loop with explicit model/thinking; full parent policy is not automatic |
|
|
184
|
+
| Tool interception | [AgentSession](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/agent-session.ts#L451) | Built-in factories and `pi.exec()` do not automatically inherit Command Guard |
|
|
185
|
+
| Resource discovery | [ResourceLoader](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/resource-loader.ts#L36) | In-memory session storage alone does not create a sterile child |
|
|
186
|
+
| Tool scheduling | [Agent defaults](https://github.com/earendil-works/pi/blob/v0.84.4/packages/agent/src/agent.ts#L205) | Explicitly select execution policy and enforce each limit before work |
|
|
187
|
+
| Error and usage semantics | [Message types](https://github.com/earendil-works/pi/blob/v0.84.4/packages/ai/src/types.ts#L332) | Inspect terminal status; do not double-count reasoning output or assume errors are free |
|
|
188
|
+
| Session identity and navigation | [Extension types](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/extensions/types.ts#L522) | Invalidate actual navigation and task revisions, not every advancing leaf |
|
|
189
|
+
|
|
190
|
+
[Latest SDK](https://pi.dev/docs/latest/sdk), [latest extensions](https://pi.dev/docs/latest/extensions)
|
|
191
|
+
and [latest provider documentation](https://pi.dev/docs/latest/custom-provider) were
|
|
192
|
+
cross-checked. These are moving references, not proof of compatibility with a specific
|
|
193
|
+
version. The [Pi 0.85.0 release notes](https://github.com/earendil-works/pi/releases/tag/v0.85.0)
|
|
194
|
+
confirm its 4 September 2026 release. An isolated installation also reported CLI version
|
|
195
|
+
0.85.0, prompting review of that released SDK alongside the tagged 0.84.4 baseline cited
|
|
196
|
+
above. Isolated native and provider integration checks have passed on 0.85.0. These
|
|
197
|
+
prove the exercised fixtures, not every provider/setup or comparative task benefit.
|
|
198
|
+
The local activation failure on Pi 0.85.1 exposed the fragility of exact-version gating.
|
|
199
|
+
The [0.85.1 tag](https://github.com/earendil-works/pi/releases/tag/v0.85.1) and an isolated
|
|
200
|
+
installation were checked; its SDK session, agent-session and model-runtime modules
|
|
201
|
+
match 0.85.0. The provider and ordinary-startup integration suites also pass on 0.85.1,
|
|
202
|
+
including actual SDK tool replay, cancellation settlement, resource isolation and
|
|
203
|
+
reload behavior. Unit fixtures accept newer and absent version identifiers while
|
|
204
|
+
rejecting missing required APIs and unsupported provider routes. These synthetic
|
|
205
|
+
identifiers do not claim that future SDK releases have been tested. Version labels
|
|
206
|
+
now record test coverage rather than grant permission.
|
|
207
|
+
Full repository and package validation remain separate release checks; API presence
|
|
208
|
+
and a successful CLI version check cannot replace them.
|
|
209
|
+
|
|
210
|
+
## What remains unproven
|
|
211
|
+
|
|
212
|
+
The research does not establish an optimal default packet size, worker count, retry
|
|
213
|
+
count, or routing threshold for SpecPi. It does not demonstrate production reliability
|
|
214
|
+
for this implementation, a financial return, or a general advantage for parallel code
|
|
215
|
+
writers. The target design and experimental protocol turn these uncertainties into
|
|
216
|
+
testable choices. Runtime compatibility and measured user value remain distinct gates.
|
|
@@ -18,6 +18,7 @@ type State = {
|
|
|
18
18
|
categories: Record<string, number>;
|
|
19
19
|
rules: Record<string, number>;
|
|
20
20
|
criticalRule?: string;
|
|
21
|
+
onModeChanged?: () => void;
|
|
21
22
|
};
|
|
22
23
|
|
|
23
24
|
function validRecord(value: unknown): value is Record<string, unknown> {
|
|
@@ -138,12 +139,17 @@ function decisionPrompt(decision: any, cwd: string, affected: string): string {
|
|
|
138
139
|
return boundedReason(fields, 1200);
|
|
139
140
|
}
|
|
140
141
|
|
|
142
|
+
function lockSession(state: State): void {
|
|
143
|
+
state.mode = "locked";
|
|
144
|
+
state.generation += 1;
|
|
145
|
+
state.sessionApprovals.clear();
|
|
146
|
+
state.onModeChanged?.();
|
|
147
|
+
}
|
|
148
|
+
|
|
141
149
|
function deny(state: State, reason: string, critical = false): { block: true; reason: string } {
|
|
142
150
|
state.blocks += 1;
|
|
143
151
|
if (critical) {
|
|
144
|
-
state
|
|
145
|
-
state.generation += 1;
|
|
146
|
-
state.sessionApprovals.clear();
|
|
152
|
+
lockSession(state);
|
|
147
153
|
}
|
|
148
154
|
|
|
149
155
|
return { block: true, reason: boundedReason(reason) };
|
|
@@ -175,6 +181,34 @@ export default function registerCommandGuard(
|
|
|
175
181
|
categories: {},
|
|
176
182
|
rules: {},
|
|
177
183
|
};
|
|
184
|
+
state.onModeChanged = () => pi.events?.emit("specpi:guard-policy-changed", { reason: "guard policy changed" });
|
|
185
|
+
let guardStateSubscription: (() => void) | undefined;
|
|
186
|
+
const subscribeGuardState = () => {
|
|
187
|
+
if (!guardStateSubscription) {
|
|
188
|
+
guardStateSubscription = pi.events?.on?.("specpi:guard-state", (request: any) => {
|
|
189
|
+
request.reply({ mode: state.ready && !state.startupFailed ? state.mode : undefined });
|
|
190
|
+
});
|
|
191
|
+
}
|
|
192
|
+
};
|
|
193
|
+
|
|
194
|
+
subscribeGuardState();
|
|
195
|
+
const delegationPolicy = (input: unknown): { fingerprint: string; summary: string } | undefined => {
|
|
196
|
+
let replies = 0;
|
|
197
|
+
let policy: any;
|
|
198
|
+
pi.events?.emit("specpi:delegation-policy", {
|
|
199
|
+
input,
|
|
200
|
+
reply(value: any) {
|
|
201
|
+
replies += 1;
|
|
202
|
+
policy = value;
|
|
203
|
+
},
|
|
204
|
+
});
|
|
205
|
+
if (replies !== 1 || !/^[a-f0-9]{64}$/u.test(policy?.fingerprint) || typeof policy?.summary !== "string") {
|
|
206
|
+
return undefined;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
return { fingerprint: policy.fingerprint, summary: boundedReason(policy.summary, 1600) };
|
|
210
|
+
};
|
|
211
|
+
|
|
178
212
|
const reset = () => {
|
|
179
213
|
clearAnalysisCache();
|
|
180
214
|
state.mode = "guard";
|
|
@@ -192,10 +226,17 @@ export default function registerCommandGuard(
|
|
|
192
226
|
|
|
193
227
|
pi.on("session_start", async (_event, ctx) => {
|
|
194
228
|
reset();
|
|
229
|
+
const startupGeneration = state.generation;
|
|
195
230
|
try {
|
|
231
|
+
subscribeGuardState();
|
|
196
232
|
const choice = ctx.hasUI ? await startupChoice(ctx, startupTimeoutMs) : undefined;
|
|
233
|
+
if (state.generation !== startupGeneration) {
|
|
234
|
+
return;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
let mode: "guard" | "strict" | "off" = "guard";
|
|
197
238
|
if (choice === "Strict") {
|
|
198
|
-
|
|
239
|
+
mode = "strict";
|
|
199
240
|
} else if (choice === "Off for this session") {
|
|
200
241
|
const confirmed = ctx.hasUI
|
|
201
242
|
? await withTimeout(
|
|
@@ -207,10 +248,15 @@ export default function registerCommandGuard(
|
|
|
207
248
|
startupTimeoutMs,
|
|
208
249
|
)
|
|
209
250
|
: false;
|
|
210
|
-
state.
|
|
251
|
+
if (state.generation !== startupGeneration) {
|
|
252
|
+
return;
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
mode = confirmed ? "off" : "guard";
|
|
211
256
|
}
|
|
212
257
|
|
|
213
|
-
state.
|
|
258
|
+
state.mode = mode;
|
|
259
|
+
state.baseMode = mode;
|
|
214
260
|
state.ready = true;
|
|
215
261
|
state.startupFailed = false;
|
|
216
262
|
updateStatus(ctx, state);
|
|
@@ -221,6 +267,10 @@ export default function registerCommandGuard(
|
|
|
221
267
|
state.mode === "off" ? "warning" : "info",
|
|
222
268
|
);
|
|
223
269
|
} catch {
|
|
270
|
+
if (state.generation !== startupGeneration) {
|
|
271
|
+
return;
|
|
272
|
+
}
|
|
273
|
+
|
|
224
274
|
state.startupFailed = true;
|
|
225
275
|
state.ready = false;
|
|
226
276
|
try {
|
|
@@ -232,6 +282,11 @@ export default function registerCommandGuard(
|
|
|
232
282
|
});
|
|
233
283
|
pi.on("session_shutdown", (_event, ctx) => {
|
|
234
284
|
reset();
|
|
285
|
+
if (typeof guardStateSubscription === "function") {
|
|
286
|
+
guardStateSubscription();
|
|
287
|
+
guardStateSubscription = undefined;
|
|
288
|
+
}
|
|
289
|
+
|
|
235
290
|
try {
|
|
236
291
|
ctx.ui.setStatus("specpi-command-guard", undefined);
|
|
237
292
|
} catch {
|
|
@@ -256,6 +311,12 @@ export default function registerCommandGuard(
|
|
|
256
311
|
return;
|
|
257
312
|
}
|
|
258
313
|
|
|
314
|
+
if (!["guard", "strict", "off", "unlock", "clear-approvals"].includes(action)) {
|
|
315
|
+
ctx.ui.notify("Usage: /guard [status|guard|strict|off|unlock|clear-approvals]", "error");
|
|
316
|
+
|
|
317
|
+
return;
|
|
318
|
+
}
|
|
319
|
+
|
|
259
320
|
if (state.mode === "locked" && action !== "unlock") {
|
|
260
321
|
ctx.ui.notify(
|
|
261
322
|
"The command guard is locked. Use /guard unlock after reviewing the critical rule.",
|
|
@@ -268,6 +329,7 @@ export default function registerCommandGuard(
|
|
|
268
329
|
if (action === "clear-approvals") {
|
|
269
330
|
state.sessionApprovals.clear();
|
|
270
331
|
state.generation += 1;
|
|
332
|
+
state.onModeChanged?.();
|
|
271
333
|
ctx.ui.notify("Session approvals cleared.", "info");
|
|
272
334
|
|
|
273
335
|
return;
|
|
@@ -280,6 +342,7 @@ export default function registerCommandGuard(
|
|
|
280
342
|
return;
|
|
281
343
|
}
|
|
282
344
|
|
|
345
|
+
const approvalGeneration = state.generation;
|
|
283
346
|
const ok =
|
|
284
347
|
ctx.hasUI &&
|
|
285
348
|
(await withTimeout(
|
|
@@ -290,11 +353,12 @@ export default function registerCommandGuard(
|
|
|
290
353
|
false,
|
|
291
354
|
approvalTimeoutMs,
|
|
292
355
|
));
|
|
293
|
-
if (ok) {
|
|
356
|
+
if (ok && state.generation === approvalGeneration) {
|
|
294
357
|
state.mode = state.baseMode;
|
|
295
358
|
state.generation += 1;
|
|
296
359
|
state.sessionApprovals.clear();
|
|
297
360
|
state.criticalRule = undefined;
|
|
361
|
+
state.onModeChanged?.();
|
|
298
362
|
updateStatus(ctx, state);
|
|
299
363
|
ctx.ui.notify(`Command guard unlocked in ${state.baseMode} mode.`, "warning");
|
|
300
364
|
}
|
|
@@ -303,6 +367,11 @@ export default function registerCommandGuard(
|
|
|
303
367
|
}
|
|
304
368
|
|
|
305
369
|
if (action === "off") {
|
|
370
|
+
if (state.mode === "off") {
|
|
371
|
+
return;
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
const approvalGeneration = state.generation;
|
|
306
375
|
if (
|
|
307
376
|
!ctx.hasUI ||
|
|
308
377
|
!(await withTimeout(
|
|
@@ -312,7 +381,8 @@ export default function registerCommandGuard(
|
|
|
312
381
|
),
|
|
313
382
|
false,
|
|
314
383
|
approvalTimeoutMs,
|
|
315
|
-
))
|
|
384
|
+
)) ||
|
|
385
|
+
state.generation !== approvalGeneration
|
|
316
386
|
) {
|
|
317
387
|
return;
|
|
318
388
|
}
|
|
@@ -321,24 +391,31 @@ export default function registerCommandGuard(
|
|
|
321
391
|
state.baseMode = "off";
|
|
322
392
|
state.generation += 1;
|
|
323
393
|
state.sessionApprovals.clear();
|
|
394
|
+
state.onModeChanged?.();
|
|
324
395
|
updateStatus(ctx, state);
|
|
325
396
|
|
|
326
397
|
return;
|
|
327
398
|
}
|
|
328
399
|
|
|
329
400
|
if (action === "strict" || action === "guard") {
|
|
401
|
+
if (state.mode === action) {
|
|
402
|
+
return;
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
const approvalGeneration = state.generation;
|
|
330
406
|
if (
|
|
331
|
-
action === "guard" &&
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
407
|
+
(action === "guard" &&
|
|
408
|
+
state.mode === "strict" &&
|
|
409
|
+
(!ctx.hasUI ||
|
|
410
|
+
!(await withTimeout(
|
|
411
|
+
ctx.ui.confirm(
|
|
412
|
+
"Switch to Guard mode?",
|
|
413
|
+
"This weakens protection for the rest of this session.",
|
|
414
|
+
),
|
|
415
|
+
false,
|
|
416
|
+
approvalTimeoutMs,
|
|
417
|
+
)))) ||
|
|
418
|
+
state.generation !== approvalGeneration
|
|
342
419
|
) {
|
|
343
420
|
return;
|
|
344
421
|
}
|
|
@@ -347,12 +424,11 @@ export default function registerCommandGuard(
|
|
|
347
424
|
state.baseMode = action;
|
|
348
425
|
state.generation += 1;
|
|
349
426
|
state.sessionApprovals.clear();
|
|
427
|
+
state.onModeChanged?.();
|
|
350
428
|
updateStatus(ctx, state);
|
|
351
429
|
|
|
352
430
|
return;
|
|
353
431
|
}
|
|
354
|
-
|
|
355
|
-
ctx.ui.notify("Usage: /guard [status|guard|strict|off|unlock|clear-approvals]", "error");
|
|
356
432
|
},
|
|
357
433
|
});
|
|
358
434
|
|
|
@@ -462,9 +538,7 @@ export default function registerCommandGuard(
|
|
|
462
538
|
}
|
|
463
539
|
|
|
464
540
|
if (answer === "Lock session") {
|
|
465
|
-
state
|
|
466
|
-
state.generation += 1;
|
|
467
|
-
state.sessionApprovals.clear();
|
|
541
|
+
lockSession(state);
|
|
468
542
|
updateStatus(ctx, state);
|
|
469
543
|
|
|
470
544
|
return deny(state, "The session was locked by command-guard approval.");
|
|
@@ -557,9 +631,7 @@ export default function registerCommandGuard(
|
|
|
557
631
|
}
|
|
558
632
|
|
|
559
633
|
if (answer === "Lock session") {
|
|
560
|
-
state
|
|
561
|
-
state.generation += 1;
|
|
562
|
-
state.sessionApprovals.clear();
|
|
634
|
+
lockSession(state);
|
|
563
635
|
updateStatus(ctx, state);
|
|
564
636
|
|
|
565
637
|
return deny(state, "The session was locked by command-guard approval.");
|
|
@@ -573,7 +645,15 @@ export default function registerCommandGuard(
|
|
|
573
645
|
|
|
574
646
|
if (state.mode === "strict") {
|
|
575
647
|
recordDecision(state, { category: "unknown", ruleIds: ["tool.unknown-capability"] });
|
|
576
|
-
const
|
|
648
|
+
const capability = name === "delegate" ? delegationPolicy(input) : undefined;
|
|
649
|
+
if (name === "delegate" && !capability) {
|
|
650
|
+
return deny(state, "Delegation policy is unavailable or ambiguous; execution is denied.");
|
|
651
|
+
}
|
|
652
|
+
|
|
653
|
+
const effectiveInput = capability
|
|
654
|
+
? { input, delegationPolicyFingerprint: capability.fingerprint }
|
|
655
|
+
: input;
|
|
656
|
+
const approvalFingerprint = toolFingerprint(name, effectiveInput, ctx.cwd, state.mode);
|
|
577
657
|
if (!approvalFingerprint) {
|
|
578
658
|
return deny(state, "Unknown-tool approval input is malformed or exceeds the safety bound.");
|
|
579
659
|
}
|
|
@@ -589,7 +669,7 @@ export default function registerCommandGuard(
|
|
|
589
669
|
const approvalGeneration = state.generation;
|
|
590
670
|
const answer = await withTimeout(
|
|
591
671
|
ctx.ui.select(
|
|
592
|
-
`Unknown tool approval — name: ${boundedReason(name, 96)}; mode: ${state.mode}; capability is not in the reviewed command-guard catalog
|
|
672
|
+
`Unknown tool approval — name: ${boundedReason(name, 96)}; mode: ${state.mode}; ${capability?.summary ?? "capability is not in the reviewed command-guard catalog."}`,
|
|
593
673
|
["Deny (Recommended)", "Allow once", "Allow exact call for session", "Lock session"],
|
|
594
674
|
),
|
|
595
675
|
undefined,
|
|
@@ -598,7 +678,14 @@ export default function registerCommandGuard(
|
|
|
598
678
|
if (
|
|
599
679
|
state.generation !== approvalGeneration ||
|
|
600
680
|
state.mode === "locked" ||
|
|
601
|
-
toolFingerprint(
|
|
681
|
+
toolFingerprint(
|
|
682
|
+
name,
|
|
683
|
+
capability
|
|
684
|
+
? { input, delegationPolicyFingerprint: delegationPolicy(input)?.fingerprint }
|
|
685
|
+
: input,
|
|
686
|
+
ctx.cwd,
|
|
687
|
+
state.mode,
|
|
688
|
+
) !== approvalFingerprint
|
|
602
689
|
) {
|
|
603
690
|
return deny(state, "Command-guard state or input changed during approval; execution is denied.");
|
|
604
691
|
}
|
|
@@ -619,9 +706,7 @@ export default function registerCommandGuard(
|
|
|
619
706
|
}
|
|
620
707
|
|
|
621
708
|
if (answer === "Lock session") {
|
|
622
|
-
state
|
|
623
|
-
state.generation += 1;
|
|
624
|
-
state.sessionApprovals.clear();
|
|
709
|
+
lockSession(state);
|
|
625
710
|
updateStatus(ctx, state);
|
|
626
711
|
}
|
|
627
712
|
|