specpi 0.11.2 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -1
- package/NPM_RELEASE.md +3 -1
- package/README.md +41 -29
- package/SECURITY_MODEL.md +36 -0
- package/THIRD_PARTY.md +18 -2
- package/docs/browser-testing.md +76 -0
- package/docs/delegation/README.md +264 -0
- package/docs/delegation/design-protocol.md +382 -0
- package/docs/delegation/design.md +525 -0
- package/docs/delegation/evaluation.md +307 -0
- package/docs/delegation/protocol.md +271 -0
- package/docs/delegation/research.md +216 -0
- package/extensions/browser/core.d.mts +64 -0
- package/extensions/browser/diagnostics.ts +275 -0
- package/extensions/browser/index.ts +349 -57
- package/extensions/browser/interactions.ts +118 -0
- package/extensions/browser/lifecycle.ts +28 -0
- package/extensions/command-guard/index.ts +118 -33
- package/extensions/delegation/core.mjs +772 -0
- package/extensions/delegation/errors.mjs +8 -0
- package/extensions/delegation/extension.mjs +475 -0
- package/extensions/delegation/index.ts +9 -0
- package/extensions/delegation/managed-files.mjs +13 -0
- package/extensions/delegation/native.mjs +155 -0
- package/extensions/delegation/presentation.mjs +315 -0
- package/extensions/delegation/protocol.mjs +296 -0
- package/extensions/delegation/provider.mjs +689 -0
- package/extensions/delegation/snapshot.mjs +532 -0
- package/extensions/delegation/worker.mjs +218 -0
- package/extensions/tool-wishlist/verification.mjs +1 -0
- package/extensions/workflow-controls/index.ts +5 -1
- package/package.json +17 -4
- package/scripts/check-package.mjs +29 -3
- package/scripts/check-pi-package.mjs +5 -0
- package/scripts/check-syntax.mjs +61 -0
- package/scripts/run-browser-tests.mjs +61 -0
- package/scripts/setup-browser-tests.mjs +38 -0
- package/scripts/site-browser.mjs +272 -0
- package/scripts/specpi.mjs +24 -0
- package/site/logo.svg +1 -9
- package/templates/AGENTS.md +1 -0
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
# Evidence behind the delegation design
|
|
2
|
+
|
|
3
|
+
Reviewed through 5 September 2026. This ledger distinguishes published studies,
|
|
4
|
+
preprints, production accounts and API contracts. Design implications are our
|
|
5
|
+
inferences. None of these sources benchmarks SpecPi's experimental implementation.
|
|
6
|
+
The [implemented guide](README.md) and [calls/time protocol](protocol.md) describe its
|
|
7
|
+
current behavior. The [archived design](design.md) preserves the broader proposal;
|
|
8
|
+
publication of supporting research does not establish that either design improves
|
|
9
|
+
SpecPi outcomes. Deterministic fixture tests and empirical evaluation are separate.
|
|
10
|
+
|
|
11
|
+
## Existing seven-study foundation
|
|
12
|
+
|
|
13
|
+
The [architecture article](../../site/single-agent/index.html) and its
|
|
14
|
+
[reviewed chart data](../../site/charts/research-data.json) contain the detailed
|
|
15
|
+
comparisons. The relevant implications for this design are:
|
|
16
|
+
|
|
17
|
+
| Primary source | Evidence relevant to the decision | Consequence for the protocol |
|
|
18
|
+
| ----------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------- |
|
|
19
|
+
| [Kim et al., Nature Machine Intelligence, 24 July 2026](https://www.nature.com/articles/s42256-026-01268-y) | Coordination helps some task classes and degrades others under matched system ceilings; software samples are small | Keep a parent-only route and evaluate by task class; no universal capability cutoff |
|
|
20
|
+
| [Tran and Kiela, preprint v2, 11 April 2026](https://arxiv.org/html/2604.02460v2) | Requested reasoning budget changes the single-versus-multi-agent ranking; actual accounting is imperfect | Give the single agent the same additional resources before attributing gains to delegation |
|
|
21
|
+
| [Wunderlich et al., ACL SRW, July 2026](https://aclanthology.org/2026.acl-srw.1/) | Configured ensembles can improve static-question accuracy at comparable modeled compute | Allow measured exceptions; distinguish ensembles from tool-using repository work |
|
|
22
|
+
| [SwarmBench, preprint v1, 31 August 2026](https://arxiv.org/html/2608.30661v1) | Swarm gains on several context-intensive tasks; only four task types have cost-matched comparisons | Prioritize bounded independent research; preserve metrics, model composition and cost denominators |
|
|
23
|
+
| [Anthropic research system, 13 June 2025](https://www.anthropic.com/engineering/multi-agent-research-system) | Parallel research can improve coverage, with substantially greater token use | Budget source collection and synthesis together; no automatic assumption of cheaper outcomes |
|
|
24
|
+
| [OneFlow, preprint v1, 18 January 2026](https://arxiv.org/html/2601.12307v1) | Useful workflow structure can survive single-conversation execution; latency changes depend on workflow | Test serial structured execution as an alternative; track caching and latency independently |
|
|
25
|
+
| [MAST, NeurIPS 2025](https://papers.nips.cc/paper_files/paper/2025/hash/b1041e52d3be19f0a9bc491657488e4a-Abstract-Datasets_and_Benchmarks_Track.html) | Taxonomy covers design, coordination and verification failures; no matched single-agent control | Require explicit coverage, provenance, incomplete states and parent acceptance |
|
|
26
|
+
|
|
27
|
+
## Additional empirical evidence
|
|
28
|
+
|
|
29
|
+
### CooperBench: communication does not solve integration
|
|
30
|
+
|
|
31
|
+
[CooperBench, v2, 26 January 2026](https://arxiv.org/html/2601.13295v2), Sections 2,
|
|
32
|
+
4–5 and Appendix C/Table 5, evaluates 652 paired-feature tasks across 12 repositories
|
|
33
|
+
and four languages. Tasks deliberately involve overlapping or interdependent code;
|
|
34
|
+
agents work in separate containers. Communication reduced merge conflicts for four
|
|
35
|
+
models without a significant cooperation-success gain for any model. GPT-5's
|
|
36
|
+
successful-task counts were 315 solo versus 183 cooperative. Communication could consume
|
|
37
|
+
up to 20% of execution steps.
|
|
38
|
+
|
|
39
|
+
**Inference:** file separation, successful messaging and a clean merge are weak
|
|
40
|
+
acceptance criteria. Keep one integration owner and specify behavior and interfaces,
|
|
41
|
+
not only paths. A writing-worker proposal needs its own end-to-end evidence.
|
|
42
|
+
|
|
43
|
+
**Limits:** the 100-action ceiling is per agent, not a demonstrated match of total
|
|
44
|
+
spending. Conflict-heavy coding tasks do not characterize independent research scouts.
|
|
45
|
+
Observed associations between early planning and fewer conflicts are not a causal
|
|
46
|
+
evaluation of a planning protocol.
|
|
47
|
+
|
|
48
|
+
### MTRouter: routing is a capability with a training cost
|
|
49
|
+
|
|
50
|
+
[MTRouter, ACL final proceedings, July 2026](https://aclanthology.org/2026.acl-long.2045.pdf),
|
|
51
|
+
Sections 3–4, Table 3 and Appendix A.2, studies sequential selection among six models.
|
|
52
|
+
On 359 in-distribution HLE questions, GPT-5 achieved 25.1±1.6% accuracy at $61.8, versus
|
|
53
|
+
26.0±2.3% at $35.0 for MTRouter. Values are mean±SD over three runs, and costs aggregate
|
|
54
|
+
evaluated episodes. Both use a $2 episode ceiling and a 30-turn limit. Training-data
|
|
55
|
+
collection used 29,693 trajectories at approximately $1,620.
|
|
56
|
+
|
|
57
|
+
**Inference:** evaluate model routing separately from delegation. Use observed
|
|
58
|
+
task-state evidence and account for training and switching costs; an untrained
|
|
59
|
+
confidence heuristic is not the studied router.
|
|
60
|
+
|
|
61
|
+
**Limits:** HLE and ScienceWorld are not repository coding; the small accuracy
|
|
62
|
+
difference does not establish superiority. Exact switch probabilities differ between
|
|
63
|
+
Figure 4 and its prose, so they are not used here. Cache-related explanations are not
|
|
64
|
+
isolated causal measurements.
|
|
65
|
+
|
|
66
|
+
### Paritok-4B: a smaller packet is not automatically a better packet
|
|
67
|
+
|
|
68
|
+
[Paritok-4B, v1, 25 August 2026](https://arxiv.org/html/2608.24188v1), Section 6.2,
|
|
69
|
+
Table 9 and Section 7, tests all 300 SWE-bench Lite instances. Its line-numbered
|
|
70
|
+
compression configuration retained 27.8% of context tokens by per-instance
|
|
71
|
+
macro-average. Uncompressed context solved 122/300 tasks (40.7%); compressed context
|
|
72
|
+
solved 109/300 (36.3%). Patch-application failures were five versus sixteen. The paired
|
|
73
|
+
solve difference had exact McNemar p=0.079.
|
|
74
|
+
|
|
75
|
+
**Inference:** preserve retrieval of original source inside the grant. Count omissions
|
|
76
|
+
and downstream accepted results, not just reduction in packet size. Select relevant
|
|
77
|
+
material before adding another model to compress it.
|
|
78
|
+
|
|
79
|
+
**Limits:** this is one tool-free Sonnet 4.5 request with oracle file context and
|
|
80
|
+
diff re-anchoring, not a running coding agent. Non-significance does not demonstrate
|
|
81
|
+
equivalence. Token reduction does not include compression, caching, latency or recovery
|
|
82
|
+
costs and must not be presented as end-to-end savings.
|
|
83
|
+
|
|
84
|
+
### OrchestraBench: recovery requires a useful change in state
|
|
85
|
+
|
|
86
|
+
[OrchestraBench, v1, 5 August 2026](https://arxiv.org/html/2608.05263v1), Sections
|
|
87
|
+
5.2–5.5, Table 8 and the trusted-state ablation, separates transient tool faults from
|
|
88
|
+
latent semantic corruption in controlled arithmetic chains. Blind retry reproduced
|
|
89
|
+
latent faults. In the N=180 trusted-state ablation, latent recovery fell from 0.67 to
|
|
90
|
+
0.08 when the trusted upstream value was removed. The paired comparison used 24 pairs.
|
|
91
|
+
|
|
92
|
+
**Inference:** a recovery request must identify the failed assumption and new evidence
|
|
93
|
+
or validated state. Preserve incomplete and uncertain outcomes. A replay of the same
|
|
94
|
+
instructions is not a recovery strategy.
|
|
95
|
+
|
|
96
|
+
**Limits:** these are small synthetic mechanism probes; some outcomes are deterministic
|
|
97
|
+
by construction. A simple TF-IDF comparator solved all ten adversarial routing cases.
|
|
98
|
+
The study does not justify a dedicated LLM router or production reliability claims.
|
|
99
|
+
|
|
100
|
+
## Additional production guidance
|
|
101
|
+
|
|
102
|
+
### Cognition: useful collaborators around a single writer
|
|
103
|
+
|
|
104
|
+
[Multi-Agents: What's Actually Working, 22 April 2026](https://cognition.com/blog/multi-agents-working)
|
|
105
|
+
updates Cognition's earlier skepticism. It describes useful fresh-context review and
|
|
106
|
+
capable-model consultation while retaining one writer. Its weaker-primary consultation
|
|
107
|
+
experiment improved cost and speed but hit a quality ceiling because the primary
|
|
108
|
+
struggled to recognize when and how to ask for help.
|
|
109
|
+
|
|
110
|
+
**Inference:** preserve a strong parent; keep review context free of implementation
|
|
111
|
+
rationalization; return concrete context requests when evidence is missing. Evaluate
|
|
112
|
+
different-model consultation against a same-model fresh context. Do not infer that
|
|
113
|
+
different context makes errors statistically independent.
|
|
114
|
+
|
|
115
|
+
**Limits:** this is a production account, not a controlled budget-matched comparison.
|
|
116
|
+
Full-history forking is one reported consultation technique, but SpecPi's privacy and
|
|
117
|
+
explicit-packet boundary means it is not adopted here. No reported product bug count
|
|
118
|
+
is used as a target for SpecPi.
|
|
119
|
+
|
|
120
|
+
### Anthropic: keep the standing harness small
|
|
121
|
+
|
|
122
|
+
[The new rules of context engineering, 24 July 2026](https://claude.com/blog/the-new-rules-of-context-engineering-for-claude-5-generation-models)
|
|
123
|
+
reports reducing Claude Code's system prompt by over 80% for newer models without a
|
|
124
|
+
measured loss on its coding evaluations. It emphasizes interface design, progressive
|
|
125
|
+
disclosure, and avoiding repeated instructions.
|
|
126
|
+
|
|
127
|
+
**Inference:** put constraints in the controller and broker; load short mode instructions
|
|
128
|
+
only when needed. Do not add a second planner, repeated role prompts, or several pages
|
|
129
|
+
of standing routing heuristics to every SpecPi session.
|
|
130
|
+
|
|
131
|
+
**Limits:** this is model- and harness-specific engineering guidance. It is not evidence
|
|
132
|
+
that deleting arbitrary safety checks, context, or instructions helps every model.
|
|
133
|
+
|
|
134
|
+
## Narrow admission policy
|
|
135
|
+
|
|
136
|
+
The implemented purposes are `review` of a frozen artifact and `scout` analysis of a
|
|
137
|
+
bounded evidence question. Both may use selected-source tools. A review gets original
|
|
138
|
+
requirements, relevant constraints and actual validation facts without the parent's
|
|
139
|
+
reasoning or verdict. A scout needs selected sources and an independently checkable
|
|
140
|
+
answer. A claimed parallel benefit also needs useful concurrent parent work; a fresh
|
|
141
|
+
review or context-isolated analysis may run while the parent waits.
|
|
142
|
+
|
|
143
|
+
These are **promising experimental scenarios, not measured SpecPi improvements**.
|
|
144
|
+
SwarmBench's cost-matched gains on three of four task types do not establish an
|
|
145
|
+
advantage for every investigation or this same-model snapshot-only implementation.
|
|
146
|
+
Cognition's cross-frontier consultation does not validate a separate generic same-model
|
|
147
|
+
consultation mode. Investigation and supplied-source research therefore share `scout`;
|
|
148
|
+
there is no live-web or model-routing route.
|
|
149
|
+
|
|
150
|
+
Small edits, routine lookups, coupled mutable work, repeated role answers, writing
|
|
151
|
+
teams and blind retries remain outside this policy. Parallel parent tool calls remain
|
|
152
|
+
an alternative to creating a worker. Mode/benefit checks and rejection of duplicate
|
|
153
|
+
normalized questions constrain structure; they cannot certify semantic independence.
|
|
154
|
+
The two-worker ceiling and all numeric quotas are engineering choices awaiting local
|
|
155
|
+
evaluation, not empirical optima extracted from the papers.
|
|
156
|
+
|
|
157
|
+
## Pi compatibility evidence
|
|
158
|
+
|
|
159
|
+
The experimental SDK integration checks required public capabilities rather than an
|
|
160
|
+
exact version list. Compatible Pi updates can activate without a SpecPi patch. Missing
|
|
161
|
+
APIs and incompatible session/provider behavior detected by the runtime checks fail closed.
|
|
162
|
+
The installer still bootstraps 0.84.4; its minimum-version contract is separate. The investigation used
|
|
163
|
+
public tagged source and official documentation without live provider calls or private
|
|
164
|
+
Pi state inspection. The current implementation uses public SDK `createAgentSession`,
|
|
165
|
+
in-memory sessions and a fresh Pi `ModelRuntime` with standard authentication,
|
|
166
|
+
environment and `models.json` resolution. Child transport/thinking budgets use configured
|
|
167
|
+
global settings without project settings. Parent model/thinking are explicit, with Pi
|
|
168
|
+
clamping; unsupported runtime-only authentication, selected extension-provider overrides,
|
|
169
|
+
model-specific headers, startup proxy configuration and safe descriptor mismatches fail
|
|
170
|
+
preflight. These integration limits leave the parent's setup unchanged.
|
|
171
|
+
|
|
172
|
+
Pi runs the child loop. SpecPi admits each SDK invocation, disables retries/compaction
|
|
173
|
+
and checks SDK-visible streaming. No ambient child resources or parent transcript are
|
|
174
|
+
loaded. Parent request hooks, ephemeral runtime settings and session affinity are not
|
|
175
|
+
automatically transferred. Full parent parity and hard raw-transport, hidden-attempt,
|
|
176
|
+
memory or invoice bounds are not claimed. SDK contract tests and comparative outcomes
|
|
177
|
+
remain distinct evidence requirements.
|
|
178
|
+
|
|
179
|
+
| Contract | Primary source | Design consequence |
|
|
180
|
+
| --------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- |
|
|
181
|
+
| Credential-blind completion facade | [ModelRegistry at v0.84.4](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/model-registry.ts#L59) | Useful building block; does not expose the configured streaming pipeline |
|
|
182
|
+
| Runtime auth and request preparation | [ModelRuntime](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/model-runtime.ts#L541) | Keep credential handling inside Pi; preserve composed provider behavior |
|
|
183
|
+
| Agent sessions and thinking translation | [SDK](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/sdk.ts#L283) | Use the Pi agent loop with explicit model/thinking; full parent policy is not automatic |
|
|
184
|
+
| Tool interception | [AgentSession](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/agent-session.ts#L451) | Built-in factories and `pi.exec()` do not automatically inherit Command Guard |
|
|
185
|
+
| Resource discovery | [ResourceLoader](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/resource-loader.ts#L36) | In-memory session storage alone does not create a sterile child |
|
|
186
|
+
| Tool scheduling | [Agent defaults](https://github.com/earendil-works/pi/blob/v0.84.4/packages/agent/src/agent.ts#L205) | Explicitly select execution policy and enforce each limit before work |
|
|
187
|
+
| Error and usage semantics | [Message types](https://github.com/earendil-works/pi/blob/v0.84.4/packages/ai/src/types.ts#L332) | Inspect terminal status; do not double-count reasoning output or assume errors are free |
|
|
188
|
+
| Session identity and navigation | [Extension types](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/extensions/types.ts#L522) | Invalidate actual navigation and task revisions, not every advancing leaf |
|
|
189
|
+
|
|
190
|
+
[Latest SDK](https://pi.dev/docs/latest/sdk), [latest extensions](https://pi.dev/docs/latest/extensions)
|
|
191
|
+
and [latest provider documentation](https://pi.dev/docs/latest/custom-provider) were
|
|
192
|
+
cross-checked. These are moving references, not proof of compatibility with a specific
|
|
193
|
+
version. The [Pi 0.85.0 release notes](https://github.com/earendil-works/pi/releases/tag/v0.85.0)
|
|
194
|
+
confirm its 4 September 2026 release. An isolated installation also reported CLI version
|
|
195
|
+
0.85.0, prompting review of that released SDK alongside the tagged 0.84.4 baseline cited
|
|
196
|
+
above. Isolated native and provider integration checks have passed on 0.85.0. These
|
|
197
|
+
prove the exercised fixtures, not every provider/setup or comparative task benefit.
|
|
198
|
+
The local activation failure on Pi 0.85.1 exposed the fragility of exact-version gating.
|
|
199
|
+
The [0.85.1 tag](https://github.com/earendil-works/pi/releases/tag/v0.85.1) and an isolated
|
|
200
|
+
installation were checked; its SDK session, agent-session and model-runtime modules
|
|
201
|
+
match 0.85.0. The provider and ordinary-startup integration suites also pass on 0.85.1,
|
|
202
|
+
including actual SDK tool replay, cancellation settlement, resource isolation and
|
|
203
|
+
reload behavior. Unit fixtures accept newer and absent version identifiers while
|
|
204
|
+
rejecting missing required APIs and unsupported provider routes. These synthetic
|
|
205
|
+
identifiers do not claim that future SDK releases have been tested. Version labels
|
|
206
|
+
now record test coverage rather than grant permission.
|
|
207
|
+
Full repository and package validation remain separate release checks; API presence
|
|
208
|
+
and a successful CLI version check cannot replace them.
|
|
209
|
+
|
|
210
|
+
## What remains unproven
|
|
211
|
+
|
|
212
|
+
The research does not establish an optimal default packet size, worker count, retry
|
|
213
|
+
count, or routing threshold for SpecPi. It does not demonstrate production reliability
|
|
214
|
+
for this implementation, a financial return, or a general advantage for parallel code
|
|
215
|
+
writers. The target design and experimental protocol turn these uncertainties into
|
|
216
|
+
testable choices. Runtime compatibility and measured user value remain distinct gates.
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type * as Playwright from "playwright";
|
|
2
|
+
|
|
3
|
+
export type Viewport = { width: number; height: number };
|
|
4
|
+
export type ViewportInput = { preset?: "desktop" | "tablet" | "mobile"; width?: number; height?: number };
|
|
5
|
+
export type PngImage = Viewport & { data: Buffer };
|
|
6
|
+
export type BrowserRuntime = {
|
|
7
|
+
playwright: typeof Playwright;
|
|
8
|
+
PNG: {
|
|
9
|
+
new (size: Viewport): PngImage;
|
|
10
|
+
sync: { read(data: Buffer): PngImage; write(image: PngImage): Buffer };
|
|
11
|
+
};
|
|
12
|
+
pixelmatch: (
|
|
13
|
+
a: Buffer,
|
|
14
|
+
b: Buffer,
|
|
15
|
+
output: Buffer,
|
|
16
|
+
width: number,
|
|
17
|
+
height: number,
|
|
18
|
+
options: { threshold: number },
|
|
19
|
+
) => number;
|
|
20
|
+
};
|
|
21
|
+
export declare const MAX_CAPTURE_DIMENSION: number;
|
|
22
|
+
export declare const MAX_CAPTURE_PIXELS: number;
|
|
23
|
+
export declare const MAX_INLINE_IMAGE_BYTES: number;
|
|
24
|
+
export declare const MAX_PNG_BYTES: number;
|
|
25
|
+
export declare const VIEWPORT_PRESETS: Readonly<Record<"desktop" | "tablet" | "mobile", Viewport>>;
|
|
26
|
+
export declare const DEFAULT_DIFF_THRESHOLD: number;
|
|
27
|
+
export declare const DEFAULT_MAX_DIFF_PIXEL_RATIO: number;
|
|
28
|
+
export declare const MAX_VIEWPORT_PIXELS: number;
|
|
29
|
+
export declare function assertDistinctPaths(entries: Array<[string, string]>): void;
|
|
30
|
+
export declare function assertPngResourceBounds(data: Buffer, label?: string): Viewport;
|
|
31
|
+
export declare function readPngDimensions(data: Buffer, label?: string): Viewport;
|
|
32
|
+
export declare function comparePngBuffers(
|
|
33
|
+
a: Buffer,
|
|
34
|
+
b: Buffer,
|
|
35
|
+
runtime: Pick<BrowserRuntime, "PNG" | "pixelmatch">,
|
|
36
|
+
options?: { threshold?: number; maxDiffPixelRatio?: number },
|
|
37
|
+
): {
|
|
38
|
+
pass: boolean;
|
|
39
|
+
dimensionsMatch: boolean;
|
|
40
|
+
baseline: Viewport;
|
|
41
|
+
current: Viewport;
|
|
42
|
+
diffPixels: number;
|
|
43
|
+
diffPixelRatio: number;
|
|
44
|
+
diffBuffer: Buffer;
|
|
45
|
+
threshold: number;
|
|
46
|
+
maxDiffPixelRatio: number;
|
|
47
|
+
};
|
|
48
|
+
export declare function getAgentDir(extensionUrl: string): string;
|
|
49
|
+
export declare function loadBrowserRuntime(runtimeDir: string): Promise<BrowserRuntime>;
|
|
50
|
+
export declare function makeArtifactPath(
|
|
51
|
+
agentDir: string,
|
|
52
|
+
sessionId: string | undefined,
|
|
53
|
+
kind: string,
|
|
54
|
+
extension?: string,
|
|
55
|
+
): string;
|
|
56
|
+
export declare function sanitizeArtifactSegment(value: string): string;
|
|
57
|
+
export declare function normalizeBrowserUrl(value: string): string;
|
|
58
|
+
export declare function publishBuffer(
|
|
59
|
+
file: string,
|
|
60
|
+
data: Buffer,
|
|
61
|
+
options?: { overwrite?: boolean; signal?: AbortSignal },
|
|
62
|
+
): Promise<void>;
|
|
63
|
+
export declare function resolveUserPath(cwd: string, value: string, label?: string): string;
|
|
64
|
+
export declare function resolveViewport(input?: ViewportInput): Viewport;
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
import crypto from "node:crypto";
|
|
2
|
+
import type { ConsoleMessage, Page, Request, Response } from "playwright";
|
|
3
|
+
|
|
4
|
+
export const DIAGNOSTIC_CATEGORIES = ["pageerror", "console", "requestfailed", "http"] as const;
|
|
5
|
+
export type DiagnosticCategory = (typeof DIAGNOSTIC_CATEGORIES)[number];
|
|
6
|
+
export const MAX_DIAGNOSTIC_RECORDS = 200;
|
|
7
|
+
export const MAX_DIAGNOSTIC_BYTES = 256 * 1024;
|
|
8
|
+
export const MAX_RECORD_BYTES = 2048;
|
|
9
|
+
|
|
10
|
+
export function boundedInteger(value: number | undefined, fallback: number, maximum: number, minimum = 1): number {
|
|
11
|
+
const result = value ?? fallback;
|
|
12
|
+
if (!Number.isInteger(result) || result < minimum || result > maximum) {
|
|
13
|
+
throw new Error(`Expected an integer from ${minimum} to ${maximum}.`);
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
return result;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
function prefix(value: string, limit: number): string {
|
|
20
|
+
return value.slice(0, limit).toWellFormed();
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function diagnosticUrl(value: string) {
|
|
24
|
+
try {
|
|
25
|
+
const url = new URL(value.slice(0, 16384));
|
|
26
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") {
|
|
27
|
+
return { text: "[non-http location]", truncated: false };
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
url.username = "";
|
|
31
|
+
url.password = "";
|
|
32
|
+
url.search = "";
|
|
33
|
+
url.hash = "";
|
|
34
|
+
|
|
35
|
+
return { text: prefix(url.href, 300), truncated: value.length > 16384 || url.href.length > 300 };
|
|
36
|
+
} catch {
|
|
37
|
+
return { text: "[location omitted]", truncated: value.length > 16384 };
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export function sanitizeUrl(value: string): string {
|
|
42
|
+
return diagnosticUrl(value).text;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** Best effort only. Application messages and URL paths may contain arbitrary secrets. */
|
|
46
|
+
function diagnosticText(value: string, limit = 700) {
|
|
47
|
+
// Bound processing as well as retention. Redact before taking the final display prefix.
|
|
48
|
+
let locationTruncated = false;
|
|
49
|
+
const sanitized = value
|
|
50
|
+
.slice(0, 16384)
|
|
51
|
+
.replace(/\u001b\][^\u0007\u001b]*(?:\u0007|\u001b\\|$)/gu, " ")
|
|
52
|
+
.replace(/\u001b\[[0-?]*[ -/]*[@-~]/gu, " ")
|
|
53
|
+
.replace(/[\u0000-\u001f\u007f-\u009f\u202a-\u202e\u2066-\u2069]/gu, " ")
|
|
54
|
+
.replace(/https?:\/\/[^\s<>"']+/giu, (url) => {
|
|
55
|
+
const location = diagnosticUrl(url);
|
|
56
|
+
locationTruncated ||= location.truncated;
|
|
57
|
+
|
|
58
|
+
return location.text;
|
|
59
|
+
})
|
|
60
|
+
.replace(/\b(?:bearer|basic)\s+[^\s,;]+/giu, "[authorization redacted]")
|
|
61
|
+
.replace(
|
|
62
|
+
/(["']?(?:password|passwd|secret|token|api[_-]?key|authorization|cookie|credential|access[_-]?key)["']?\s*[:=]\s*)(?:"[^"]*(?:"|$)|'[^']*(?:'|$)|[^\s,;]+)/giu,
|
|
63
|
+
"$1[redacted]",
|
|
64
|
+
);
|
|
65
|
+
|
|
66
|
+
return {
|
|
67
|
+
text: prefix(sanitized, limit),
|
|
68
|
+
truncated: value.length > 16384 || sanitized.length > limit || locationTruncated,
|
|
69
|
+
};
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
export function sanitizeDiagnostic(value: string, limit = 700): string {
|
|
73
|
+
return diagnosticText(value, limit).text;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
type DiagnosticRecord = {
|
|
77
|
+
sequence: number;
|
|
78
|
+
navigation: number;
|
|
79
|
+
category: DiagnosticCategory;
|
|
80
|
+
message: string;
|
|
81
|
+
url: string;
|
|
82
|
+
method?: string;
|
|
83
|
+
status?: number;
|
|
84
|
+
resourceType?: string;
|
|
85
|
+
truncated: boolean;
|
|
86
|
+
};
|
|
87
|
+
export type DiagnosticQuery = {
|
|
88
|
+
maxEntries?: number;
|
|
89
|
+
maxChars?: number;
|
|
90
|
+
category?: DiagnosticCategory;
|
|
91
|
+
cursor?: string;
|
|
92
|
+
clear?: boolean;
|
|
93
|
+
};
|
|
94
|
+
|
|
95
|
+
export class BrowserDiagnostics {
|
|
96
|
+
private context = crypto.randomUUID();
|
|
97
|
+
private sequence = 0;
|
|
98
|
+
private navigation = 0;
|
|
99
|
+
private floor = 0;
|
|
100
|
+
private dropped = 0;
|
|
101
|
+
private bytes = 0;
|
|
102
|
+
private records: DiagnosticRecord[] = [];
|
|
103
|
+
|
|
104
|
+
reset(): void {
|
|
105
|
+
this.context = crypto.randomUUID();
|
|
106
|
+
this.sequence = 0;
|
|
107
|
+
this.navigation = 0;
|
|
108
|
+
this.floor = 0;
|
|
109
|
+
this.dropped = 0;
|
|
110
|
+
this.bytes = 0;
|
|
111
|
+
this.records = [];
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
navigated(): void {
|
|
115
|
+
this.navigation += 1;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
record(
|
|
119
|
+
category: DiagnosticCategory,
|
|
120
|
+
message: string,
|
|
121
|
+
url = "",
|
|
122
|
+
extra: { method?: string; status?: number; resourceType?: string } = {},
|
|
123
|
+
): void {
|
|
124
|
+
const clean = diagnosticText(message);
|
|
125
|
+
const location = url ? diagnosticUrl(url) : undefined;
|
|
126
|
+
const method = extra.method ? diagnosticText(extra.method, 20) : undefined;
|
|
127
|
+
const resourceType = extra.resourceType ? diagnosticText(extra.resourceType, 30) : undefined;
|
|
128
|
+
const entry: DiagnosticRecord = {
|
|
129
|
+
sequence: ++this.sequence,
|
|
130
|
+
navigation: this.navigation,
|
|
131
|
+
category,
|
|
132
|
+
message: clean.text,
|
|
133
|
+
url: location?.text ?? "",
|
|
134
|
+
method: method?.text,
|
|
135
|
+
status: extra.status,
|
|
136
|
+
resourceType: resourceType?.text,
|
|
137
|
+
truncated:
|
|
138
|
+
clean.truncated ||
|
|
139
|
+
location?.truncated === true ||
|
|
140
|
+
method?.truncated === true ||
|
|
141
|
+
resourceType?.truncated === true,
|
|
142
|
+
};
|
|
143
|
+
while (Buffer.byteLength(JSON.stringify(entry)) > MAX_RECORD_BYTES) {
|
|
144
|
+
entry.truncated = true;
|
|
145
|
+
if (entry.message.length) {
|
|
146
|
+
entry.message = prefix(entry.message, Math.floor(entry.message.length / 2));
|
|
147
|
+
} else {
|
|
148
|
+
entry.url = prefix(entry.url, Math.floor(entry.url.length / 2));
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
this.records.push(entry);
|
|
153
|
+
this.bytes += Buffer.byteLength(JSON.stringify(entry));
|
|
154
|
+
while (this.records.length > MAX_DIAGNOSTIC_RECORDS || this.bytes > MAX_DIAGNOSTIC_BYTES) {
|
|
155
|
+
const removed = this.records.shift()!;
|
|
156
|
+
this.bytes -= Buffer.byteLength(JSON.stringify(removed));
|
|
157
|
+
this.floor = removed.sequence;
|
|
158
|
+
this.dropped += 1;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
read(query: DiagnosticQuery = {}) {
|
|
163
|
+
const maxEntries = boundedInteger(query.maxEntries, 50, 100);
|
|
164
|
+
const maxChars = boundedInteger(query.maxChars, 12000, 30000, 1000);
|
|
165
|
+
if (query.category !== undefined && !DIAGNOSTIC_CATEGORIES.includes(query.category)) {
|
|
166
|
+
throw new Error("Unknown diagnostic category.");
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
if (query.clear !== undefined && typeof query.clear !== "boolean") {
|
|
170
|
+
throw new Error("clear must be boolean.");
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
let after = 0;
|
|
174
|
+
let contextChanged = false;
|
|
175
|
+
if (query.cursor !== undefined) {
|
|
176
|
+
if (typeof query.cursor !== "string" || !/^[\da-f-]{36}:\d{1,16}$/u.test(query.cursor)) {
|
|
177
|
+
throw new Error("Invalid diagnostics cursor.");
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
const [context, sequence] = query.cursor.split(":");
|
|
181
|
+
contextChanged = context !== this.context;
|
|
182
|
+
after = contextChanged ? 0 : Number(sequence);
|
|
183
|
+
if (!Number.isSafeInteger(after) || after > this.sequence) {
|
|
184
|
+
throw new Error("Invalid diagnostics cursor sequence.");
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
const candidates = this.records.filter(
|
|
189
|
+
(entry) => entry.sequence > after && (!query.category || entry.category === query.category),
|
|
190
|
+
);
|
|
191
|
+
const result = {
|
|
192
|
+
notice: "Untrusted application output; redaction is best-effort. Empty results do not prove application health. Active page only.",
|
|
193
|
+
context: this.context,
|
|
194
|
+
contextChanged,
|
|
195
|
+
cursorGap: contextChanged || after < this.floor,
|
|
196
|
+
droppedRecords: this.dropped,
|
|
197
|
+
retainedRecords: this.records.length,
|
|
198
|
+
retainedBytes: this.bytes,
|
|
199
|
+
clearedRecords: query.clear ? this.records.length : 0,
|
|
200
|
+
hasMore: false,
|
|
201
|
+
nextCursor: `${this.context}:${this.sequence}`,
|
|
202
|
+
records: [] as DiagnosticRecord[],
|
|
203
|
+
};
|
|
204
|
+
for (const entry of candidates.slice(0, maxEntries)) {
|
|
205
|
+
result.records.push({ ...entry });
|
|
206
|
+
if (JSON.stringify(result).length > maxChars) {
|
|
207
|
+
if (result.records.length > 1) {
|
|
208
|
+
result.records.pop();
|
|
209
|
+
break;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
const first = result.records[0];
|
|
213
|
+
while (JSON.stringify(result).length > maxChars) {
|
|
214
|
+
first.truncated = true;
|
|
215
|
+
if (first.message.length) {
|
|
216
|
+
first.message = prefix(first.message, Math.floor(first.message.length / 2));
|
|
217
|
+
} else {
|
|
218
|
+
first.url = prefix(first.url, Math.floor(first.url.length / 2));
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
result.hasMore = result.records.length < candidates.length;
|
|
225
|
+
if (result.hasMore) {
|
|
226
|
+
result.nextCursor = `${this.context}:${result.records.at(-1)?.sequence ?? after}`;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
if (query.clear) {
|
|
230
|
+
// Atomic synchronous read-and-clear of the entire buffer, including filtered/unreturned records.
|
|
231
|
+
this.records = [];
|
|
232
|
+
this.bytes = 0;
|
|
233
|
+
this.floor = this.sequence;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
return result;
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
attach(page: Page): () => void {
|
|
240
|
+
const onError = (error: Error) => this.record("pageerror", error.message);
|
|
241
|
+
const onConsole = (message: ConsoleMessage) => {
|
|
242
|
+
if (message.type() === "error") {
|
|
243
|
+
this.record("console", message.text(), message.location().url);
|
|
244
|
+
}
|
|
245
|
+
};
|
|
246
|
+
|
|
247
|
+
const onRequest = (request: Request) =>
|
|
248
|
+
this.record("requestfailed", request.failure()?.errorText ?? "Request failed", request.url(), {
|
|
249
|
+
method: request.method(),
|
|
250
|
+
resourceType: request.resourceType(),
|
|
251
|
+
});
|
|
252
|
+
const onResponse = (response: Response) => {
|
|
253
|
+
if (response.status() >= 400) {
|
|
254
|
+
const request = response.request();
|
|
255
|
+
this.record("http", `HTTP ${response.status()}`, response.url(), {
|
|
256
|
+
status: response.status(),
|
|
257
|
+
method: request.method(),
|
|
258
|
+
resourceType: request.resourceType(),
|
|
259
|
+
});
|
|
260
|
+
}
|
|
261
|
+
};
|
|
262
|
+
|
|
263
|
+
page.on("pageerror", onError);
|
|
264
|
+
page.on("console", onConsole);
|
|
265
|
+
page.on("requestfailed", onRequest);
|
|
266
|
+
page.on("response", onResponse);
|
|
267
|
+
|
|
268
|
+
return () => {
|
|
269
|
+
page.off("pageerror", onError);
|
|
270
|
+
page.off("console", onConsole);
|
|
271
|
+
page.off("requestfailed", onRequest);
|
|
272
|
+
page.off("response", onResponse);
|
|
273
|
+
};
|
|
274
|
+
}
|
|
275
|
+
}
|