specpi 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/CHANGELOG.md +21 -1
  2. package/NPM_RELEASE.md +3 -1
  3. package/README.md +41 -29
  4. package/SECURITY_MODEL.md +36 -0
  5. package/THIRD_PARTY.md +18 -2
  6. package/docs/browser-testing.md +76 -0
  7. package/docs/delegation/README.md +264 -0
  8. package/docs/delegation/design-protocol.md +382 -0
  9. package/docs/delegation/design.md +525 -0
  10. package/docs/delegation/evaluation.md +307 -0
  11. package/docs/delegation/protocol.md +271 -0
  12. package/docs/delegation/research.md +216 -0
  13. package/extensions/browser/core.d.mts +64 -0
  14. package/extensions/browser/diagnostics.ts +275 -0
  15. package/extensions/browser/index.ts +349 -57
  16. package/extensions/browser/interactions.ts +118 -0
  17. package/extensions/browser/lifecycle.ts +28 -0
  18. package/extensions/command-guard/index.ts +118 -33
  19. package/extensions/delegation/core.mjs +772 -0
  20. package/extensions/delegation/errors.mjs +8 -0
  21. package/extensions/delegation/extension.mjs +475 -0
  22. package/extensions/delegation/index.ts +9 -0
  23. package/extensions/delegation/managed-files.mjs +13 -0
  24. package/extensions/delegation/native.mjs +155 -0
  25. package/extensions/delegation/presentation.mjs +315 -0
  26. package/extensions/delegation/protocol.mjs +296 -0
  27. package/extensions/delegation/provider.mjs +689 -0
  28. package/extensions/delegation/snapshot.mjs +532 -0
  29. package/extensions/delegation/worker.mjs +218 -0
  30. package/extensions/tool-wishlist/verification.mjs +1 -0
  31. package/extensions/workflow-controls/index.ts +5 -1
  32. package/package.json +17 -4
  33. package/scripts/check-package.mjs +29 -3
  34. package/scripts/check-pi-package.mjs +5 -0
  35. package/scripts/check-syntax.mjs +61 -0
  36. package/scripts/run-browser-tests.mjs +61 -0
  37. package/scripts/setup-browser-tests.mjs +38 -0
  38. package/scripts/site-browser.mjs +272 -0
  39. package/scripts/specpi.mjs +24 -0
  40. package/site/logo.svg +1 -9
  41. package/templates/AGENTS.md +1 -0
@@ -0,0 +1,216 @@
1
+ # Evidence behind the delegation design
2
+
3
+ Reviewed through 5 September 2026. This ledger distinguishes published studies,
4
+ preprints, production accounts and API contracts. Design implications are our
5
+ inferences. None of these sources benchmarks SpecPi's experimental implementation.
6
+ The [implemented guide](README.md) and [calls/time protocol](protocol.md) describe its
7
+ current behavior. The [archived design](design.md) preserves the broader proposal;
8
+ publication of supporting research does not establish that either design improves
9
+ SpecPi outcomes. Deterministic fixture tests and empirical evaluation are separate.
10
+
11
+ ## Existing seven-study foundation
12
+
13
+ The [architecture article](../../site/single-agent/index.html) and its
14
+ [reviewed chart data](../../site/charts/research-data.json) contain the detailed
15
+ comparisons. The relevant implications for this design are:
16
+
17
+ | Primary source | Evidence relevant to the decision | Consequence for the protocol |
18
+ | ----------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------- |
19
+ | [Kim et al., Nature Machine Intelligence, 24 July 2026](https://www.nature.com/articles/s42256-026-01268-y) | Coordination helps some task classes and degrades others under matched system ceilings; software samples are small | Keep a parent-only route and evaluate by task class; no universal capability cutoff |
20
+ | [Tran and Kiela, preprint v2, 11 April 2026](https://arxiv.org/html/2604.02460v2) | Requested reasoning budget changes the single-versus-multi-agent ranking; actual accounting is imperfect | Give the single agent the same additional resources before attributing gains to delegation |
21
+ | [Wunderlich et al., ACL SRW, July 2026](https://aclanthology.org/2026.acl-srw.1/) | Configured ensembles can improve static-question accuracy at comparable modeled compute | Allow measured exceptions; distinguish ensembles from tool-using repository work |
22
+ | [SwarmBench, preprint v1, 31 August 2026](https://arxiv.org/html/2608.30661v1) | Swarm gains on several context-intensive tasks; only four task types have cost-matched comparisons | Prioritize bounded independent research; preserve metrics, model composition and cost denominators |
23
+ | [Anthropic research system, 13 June 2025](https://www.anthropic.com/engineering/multi-agent-research-system) | Parallel research can improve coverage, with substantially greater token use | Budget source collection and synthesis together; no automatic assumption of cheaper outcomes |
24
+ | [OneFlow, preprint v1, 18 January 2026](https://arxiv.org/html/2601.12307v1) | Useful workflow structure can survive single-conversation execution; latency changes depend on workflow | Test serial structured execution as an alternative; track caching and latency independently |
25
+ | [MAST, NeurIPS 2025](https://papers.nips.cc/paper_files/paper/2025/hash/b1041e52d3be19f0a9bc491657488e4a-Abstract-Datasets_and_Benchmarks_Track.html) | Taxonomy covers design, coordination and verification failures; no matched single-agent control | Require explicit coverage, provenance, incomplete states and parent acceptance |
26
+
27
+ ## Additional empirical evidence
28
+
29
+ ### CooperBench: communication does not solve integration
30
+
31
+ [CooperBench, v2, 26 January 2026](https://arxiv.org/html/2601.13295v2), Sections 2,
32
+ 4–5 and Appendix C/Table 5, evaluates 652 paired-feature tasks across 12 repositories
33
+ and four languages. Tasks deliberately involve overlapping or interdependent code;
34
+ agents work in separate containers. Communication reduced merge conflicts for four
35
+ models without a significant cooperation-success gain for any model. GPT-5's
36
+ successful-task counts were 315 solo versus 183 cooperative. Communication could consume
37
+ up to 20% of execution steps.
38
+
39
+ **Inference:** file separation, successful messaging and a clean merge are weak
40
+ acceptance criteria. Keep one integration owner and specify behavior and interfaces,
41
+ not only paths. A writing-worker proposal needs its own end-to-end evidence.
42
+
43
+ **Limits:** the 100-action ceiling is per agent, not a demonstrated match of total
44
+ spending. Conflict-heavy coding tasks do not characterize independent research scouts.
45
+ Observed associations between early planning and fewer conflicts are not a causal
46
+ evaluation of a planning protocol.
47
+
48
+ ### MTRouter: routing is a capability with a training cost
49
+
50
+ [MTRouter, ACL final proceedings, July 2026](https://aclanthology.org/2026.acl-long.2045.pdf),
51
+ Sections 3–4, Table 3 and Appendix A.2, studies sequential selection among six models.
52
+ On 359 in-distribution HLE questions, GPT-5 achieved 25.1±1.6% accuracy at $61.8, versus
53
+ 26.0±2.3% at $35.0 for MTRouter. Values are mean±SD over three runs, and costs aggregate
54
+ evaluated episodes. Both use a $2 episode ceiling and a 30-turn limit. Training-data
55
+ collection used 29,693 trajectories at approximately $1,620.
56
+
57
+ **Inference:** evaluate model routing separately from delegation. Use observed
58
+ task-state evidence and account for training and switching costs; an untrained
59
+ confidence heuristic is not the studied router.
60
+
61
+ **Limits:** HLE and ScienceWorld are not repository coding; the small accuracy
62
+ difference does not establish superiority. Exact switch probabilities differ between
63
+ Figure 4 and its prose, so they are not used here. Cache-related explanations are not
64
+ isolated causal measurements.
65
+
66
+ ### Paritok-4B: a smaller packet is not automatically a better packet
67
+
68
+ [Paritok-4B, v1, 25 August 2026](https://arxiv.org/html/2608.24188v1), Section 6.2,
69
+ Table 9 and Section 7, tests all 300 SWE-bench Lite instances. Its line-numbered
70
+ compression configuration retained 27.8% of context tokens by per-instance
71
+ macro-average. Uncompressed context solved 122/300 tasks (40.7%); compressed context
72
+ solved 109/300 (36.3%). Patch-application failures were five versus sixteen. The paired
73
+ solve difference had exact McNemar p=0.079.
74
+
75
+ **Inference:** preserve retrieval of original source inside the grant. Count omissions
76
+ and downstream accepted results, not just reduction in packet size. Select relevant
77
+ material before adding another model to compress it.
78
+
79
+ **Limits:** this is one tool-free Sonnet 4.5 request with oracle file context and
80
+ diff re-anchoring, not a running coding agent. Non-significance does not demonstrate
81
+ equivalence. Token reduction does not include compression, caching, latency or recovery
82
+ costs and must not be presented as end-to-end savings.
83
+
84
+ ### OrchestraBench: recovery requires a useful change in state
85
+
86
+ [OrchestraBench, v1, 5 August 2026](https://arxiv.org/html/2608.05263v1), Sections
87
+ 5.2–5.5, Table 8 and the trusted-state ablation, separates transient tool faults from
88
+ latent semantic corruption in controlled arithmetic chains. Blind retry reproduced
89
+ latent faults. In the N=180 trusted-state ablation, latent recovery fell from 0.67 to
90
+ 0.08 when the trusted upstream value was removed. The paired comparison used 24 pairs.
91
+
92
+ **Inference:** a recovery request must identify the failed assumption and new evidence
93
+ or validated state. Preserve incomplete and uncertain outcomes. A replay of the same
94
+ instructions is not a recovery strategy.
95
+
96
+ **Limits:** these are small synthetic mechanism probes; some outcomes are deterministic
97
+ by construction. A simple TF-IDF comparator solved all ten adversarial routing cases.
98
+ The study does not justify a dedicated LLM router or production reliability claims.
99
+
100
+ ## Additional production guidance
101
+
102
+ ### Cognition: useful collaborators around a single writer
103
+
104
+ [Multi-Agents: What's Actually Working, 22 April 2026](https://cognition.com/blog/multi-agents-working)
105
+ updates Cognition's earlier skepticism. It describes useful fresh-context review and
106
+ capable-model consultation while retaining one writer. Its weaker-primary consultation
107
+ experiment improved cost and speed but hit a quality ceiling because the primary
108
+ struggled to recognize when and how to ask for help.
109
+
110
+ **Inference:** preserve a strong parent; keep review context free of implementation
111
+ rationalization; return concrete context requests when evidence is missing. Evaluate
112
+ different-model consultation against a same-model fresh context. Do not infer that
113
+ different context makes errors statistically independent.
114
+
115
+ **Limits:** this is a production account, not a controlled budget-matched comparison.
116
+ Full-history forking is one reported consultation technique, but SpecPi's privacy and
117
+ explicit-packet boundary means it is not adopted here. No reported product bug count
118
+ is used as a target for SpecPi.
119
+
120
+ ### Anthropic: keep the standing harness small
121
+
122
+ [The new rules of context engineering, 24 July 2026](https://claude.com/blog/the-new-rules-of-context-engineering-for-claude-5-generation-models)
123
+ reports reducing Claude Code's system prompt by over 80% for newer models without a
124
+ measured loss on its coding evaluations. It emphasizes interface design, progressive
125
+ disclosure, and avoiding repeated instructions.
126
+
127
+ **Inference:** put constraints in the controller and broker; load short mode instructions
128
+ only when needed. Do not add a second planner, repeated role prompts, or several pages
129
+ of standing routing heuristics to every SpecPi session.
130
+
131
+ **Limits:** this is model- and harness-specific engineering guidance. It is not evidence
132
+ that deleting arbitrary safety checks, context, or instructions helps every model.
133
+
134
+ ## Narrow admission policy
135
+
136
+ The implemented purposes are `review` of a frozen artifact and `scout` analysis of a
137
+ bounded evidence question. Both may use selected-source tools. A review gets original
138
+ requirements, relevant constraints and actual validation facts without the parent's
139
+ reasoning or verdict. A scout needs selected sources and an independently checkable
140
+ answer. A claimed parallel benefit also needs useful concurrent parent work; a fresh
141
+ review or context-isolated analysis may run while the parent waits.
142
+
143
+ These are **promising experimental scenarios, not measured SpecPi improvements**.
144
+ SwarmBench's cost-matched gains on three of four task types do not establish an
145
+ advantage for every investigation or this same-model snapshot-only implementation.
146
+ Cognition's cross-frontier consultation does not validate a separate generic same-model
147
+ consultation mode. Investigation and supplied-source research therefore share `scout`;
148
+ there is no live-web or model-routing route.
149
+
150
+ Small edits, routine lookups, coupled mutable work, repeated role answers, writing
151
+ teams and blind retries remain outside this policy. Parallel parent tool calls remain
152
+ an alternative to creating a worker. Mode/benefit checks and rejection of duplicate
153
+ normalized questions constrain structure; they cannot certify semantic independence.
154
+ The two-worker ceiling and all numeric quotas are engineering choices awaiting local
155
+ evaluation, not empirical optima extracted from the papers.
156
+
157
+ ## Pi compatibility evidence
158
+
159
+ The experimental SDK integration checks required public capabilities rather than an
160
+ exact version list. Compatible Pi updates can activate without a SpecPi patch. Missing
161
+ APIs and incompatible session/provider behavior detected by the runtime checks fail closed.
162
+ The installer still bootstraps 0.84.4; its minimum-version contract is separate. The investigation used
163
+ public tagged source and official documentation without live provider calls or private
164
+ Pi state inspection. The current implementation uses public SDK `createAgentSession`,
165
+ in-memory sessions and a fresh Pi `ModelRuntime` with standard authentication,
166
+ environment and `models.json` resolution. Child transport/thinking budgets use configured
167
+ global settings without project settings. Parent model/thinking are explicit, with Pi
168
+ clamping; unsupported runtime-only authentication, selected extension-provider overrides,
169
+ model-specific headers, startup proxy configuration and safe descriptor mismatches fail
170
+ preflight. These integration limits leave the parent's setup unchanged.
171
+
172
+ Pi runs the child loop. SpecPi admits each SDK invocation, disables retries/compaction
173
+ and checks SDK-visible streaming. No ambient child resources or parent transcript are
174
+ loaded. Parent request hooks, ephemeral runtime settings and session affinity are not
175
+ automatically transferred. Full parent parity and hard raw-transport, hidden-attempt,
176
+ memory or invoice bounds are not claimed. SDK contract tests and comparative outcomes
177
+ remain distinct evidence requirements.
178
+
179
+ | Contract | Primary source | Design consequence |
180
+ | --------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- |
181
+ | Credential-blind completion facade | [ModelRegistry at v0.84.4](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/model-registry.ts#L59) | Useful building block; does not expose the configured streaming pipeline |
182
+ | Runtime auth and request preparation | [ModelRuntime](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/model-runtime.ts#L541) | Keep credential handling inside Pi; preserve composed provider behavior |
183
+ | Agent sessions and thinking translation | [SDK](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/sdk.ts#L283) | Use the Pi agent loop with explicit model/thinking; full parent policy is not automatic |
184
+ | Tool interception | [AgentSession](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/agent-session.ts#L451) | Built-in factories and `pi.exec()` do not automatically inherit Command Guard |
185
+ | Resource discovery | [ResourceLoader](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/resource-loader.ts#L36) | In-memory session storage alone does not create a sterile child |
186
+ | Tool scheduling | [Agent defaults](https://github.com/earendil-works/pi/blob/v0.84.4/packages/agent/src/agent.ts#L205) | Explicitly select execution policy and enforce each limit before work |
187
+ | Error and usage semantics | [Message types](https://github.com/earendil-works/pi/blob/v0.84.4/packages/ai/src/types.ts#L332) | Inspect terminal status; do not double-count reasoning output or assume errors are free |
188
+ | Session identity and navigation | [Extension types](https://github.com/earendil-works/pi/blob/v0.84.4/packages/coding-agent/src/core/extensions/types.ts#L522) | Invalidate actual navigation and task revisions, not every advancing leaf |
189
+
190
+ [Latest SDK](https://pi.dev/docs/latest/sdk), [latest extensions](https://pi.dev/docs/latest/extensions)
191
+ and [latest provider documentation](https://pi.dev/docs/latest/custom-provider) were
192
+ cross-checked. These are moving references, not proof of compatibility with a specific
193
+ version. The [Pi 0.85.0 release notes](https://github.com/earendil-works/pi/releases/tag/v0.85.0)
194
+ confirm its 4 September 2026 release. An isolated installation also reported CLI version
195
+ 0.85.0, prompting review of that released SDK alongside the tagged 0.84.4 baseline cited
196
+ above. Isolated native and provider integration checks have passed on 0.85.0. These
197
+ prove the exercised fixtures, not every provider/setup or comparative task benefit.
198
+ The local activation failure on Pi 0.85.1 exposed the fragility of exact-version gating.
199
+ The [0.85.1 tag](https://github.com/earendil-works/pi/releases/tag/v0.85.1) and an isolated
200
+ installation were checked; its SDK session, agent-session and model-runtime modules
201
+ match 0.85.0. The provider and ordinary-startup integration suites also pass on 0.85.1,
202
+ including actual SDK tool replay, cancellation settlement, resource isolation and
203
+ reload behavior. Unit fixtures accept newer and absent version identifiers while
204
+ rejecting missing required APIs and unsupported provider routes. These synthetic
205
+ identifiers do not claim that future SDK releases have been tested. Version labels
206
+ now record test coverage rather than grant permission.
207
+ Full repository and package validation remain separate release checks; API presence
208
+ and a successful CLI version check cannot replace them.
209
+
210
+ ## What remains unproven
211
+
212
+ The research does not establish an optimal default packet size, worker count, retry
213
+ count, or routing threshold for SpecPi. It does not demonstrate production reliability
214
+ for this implementation, a financial return, or a general advantage for parallel code
215
+ writers. The target design and experimental protocol turn these uncertainties into
216
+ testable choices. Runtime compatibility and measured user value remain distinct gates.
@@ -0,0 +1,64 @@
1
+ import type * as Playwright from "playwright";
2
+
3
+ export type Viewport = { width: number; height: number };
4
+ export type ViewportInput = { preset?: "desktop" | "tablet" | "mobile"; width?: number; height?: number };
5
+ export type PngImage = Viewport & { data: Buffer };
6
+ export type BrowserRuntime = {
7
+ playwright: typeof Playwright;
8
+ PNG: {
9
+ new (size: Viewport): PngImage;
10
+ sync: { read(data: Buffer): PngImage; write(image: PngImage): Buffer };
11
+ };
12
+ pixelmatch: (
13
+ a: Buffer,
14
+ b: Buffer,
15
+ output: Buffer,
16
+ width: number,
17
+ height: number,
18
+ options: { threshold: number },
19
+ ) => number;
20
+ };
21
+ export declare const MAX_CAPTURE_DIMENSION: number;
22
+ export declare const MAX_CAPTURE_PIXELS: number;
23
+ export declare const MAX_INLINE_IMAGE_BYTES: number;
24
+ export declare const MAX_PNG_BYTES: number;
25
+ export declare const VIEWPORT_PRESETS: Readonly<Record<"desktop" | "tablet" | "mobile", Viewport>>;
26
+ export declare const DEFAULT_DIFF_THRESHOLD: number;
27
+ export declare const DEFAULT_MAX_DIFF_PIXEL_RATIO: number;
28
+ export declare const MAX_VIEWPORT_PIXELS: number;
29
+ export declare function assertDistinctPaths(entries: Array<[string, string]>): void;
30
+ export declare function assertPngResourceBounds(data: Buffer, label?: string): Viewport;
31
+ export declare function readPngDimensions(data: Buffer, label?: string): Viewport;
32
+ export declare function comparePngBuffers(
33
+ a: Buffer,
34
+ b: Buffer,
35
+ runtime: Pick<BrowserRuntime, "PNG" | "pixelmatch">,
36
+ options?: { threshold?: number; maxDiffPixelRatio?: number },
37
+ ): {
38
+ pass: boolean;
39
+ dimensionsMatch: boolean;
40
+ baseline: Viewport;
41
+ current: Viewport;
42
+ diffPixels: number;
43
+ diffPixelRatio: number;
44
+ diffBuffer: Buffer;
45
+ threshold: number;
46
+ maxDiffPixelRatio: number;
47
+ };
48
+ export declare function getAgentDir(extensionUrl: string): string;
49
+ export declare function loadBrowserRuntime(runtimeDir: string): Promise<BrowserRuntime>;
50
+ export declare function makeArtifactPath(
51
+ agentDir: string,
52
+ sessionId: string | undefined,
53
+ kind: string,
54
+ extension?: string,
55
+ ): string;
56
+ export declare function sanitizeArtifactSegment(value: string): string;
57
+ export declare function normalizeBrowserUrl(value: string): string;
58
+ export declare function publishBuffer(
59
+ file: string,
60
+ data: Buffer,
61
+ options?: { overwrite?: boolean; signal?: AbortSignal },
62
+ ): Promise<void>;
63
+ export declare function resolveUserPath(cwd: string, value: string, label?: string): string;
64
+ export declare function resolveViewport(input?: ViewportInput): Viewport;
@@ -0,0 +1,275 @@
1
+ import crypto from "node:crypto";
2
+ import type { ConsoleMessage, Page, Request, Response } from "playwright";
3
+
4
+ export const DIAGNOSTIC_CATEGORIES = ["pageerror", "console", "requestfailed", "http"] as const;
5
+ export type DiagnosticCategory = (typeof DIAGNOSTIC_CATEGORIES)[number];
6
+ export const MAX_DIAGNOSTIC_RECORDS = 200;
7
+ export const MAX_DIAGNOSTIC_BYTES = 256 * 1024;
8
+ export const MAX_RECORD_BYTES = 2048;
9
+
10
+ export function boundedInteger(value: number | undefined, fallback: number, maximum: number, minimum = 1): number {
11
+ const result = value ?? fallback;
12
+ if (!Number.isInteger(result) || result < minimum || result > maximum) {
13
+ throw new Error(`Expected an integer from ${minimum} to ${maximum}.`);
14
+ }
15
+
16
+ return result;
17
+ }
18
+
19
+ function prefix(value: string, limit: number): string {
20
+ return value.slice(0, limit).toWellFormed();
21
+ }
22
+
23
+ function diagnosticUrl(value: string) {
24
+ try {
25
+ const url = new URL(value.slice(0, 16384));
26
+ if (url.protocol !== "http:" && url.protocol !== "https:") {
27
+ return { text: "[non-http location]", truncated: false };
28
+ }
29
+
30
+ url.username = "";
31
+ url.password = "";
32
+ url.search = "";
33
+ url.hash = "";
34
+
35
+ return { text: prefix(url.href, 300), truncated: value.length > 16384 || url.href.length > 300 };
36
+ } catch {
37
+ return { text: "[location omitted]", truncated: value.length > 16384 };
38
+ }
39
+ }
40
+
41
+ export function sanitizeUrl(value: string): string {
42
+ return diagnosticUrl(value).text;
43
+ }
44
+
45
+ /** Best effort only. Application messages and URL paths may contain arbitrary secrets. */
46
+ function diagnosticText(value: string, limit = 700) {
47
+ // Bound processing as well as retention. Redact before taking the final display prefix.
48
+ let locationTruncated = false;
49
+ const sanitized = value
50
+ .slice(0, 16384)
51
+ .replace(/\u001b\][^\u0007\u001b]*(?:\u0007|\u001b\\|$)/gu, " ")
52
+ .replace(/\u001b\[[0-?]*[ -/]*[@-~]/gu, " ")
53
+ .replace(/[\u0000-\u001f\u007f-\u009f\u202a-\u202e\u2066-\u2069]/gu, " ")
54
+ .replace(/https?:\/\/[^\s<>"']+/giu, (url) => {
55
+ const location = diagnosticUrl(url);
56
+ locationTruncated ||= location.truncated;
57
+
58
+ return location.text;
59
+ })
60
+ .replace(/\b(?:bearer|basic)\s+[^\s,;]+/giu, "[authorization redacted]")
61
+ .replace(
62
+ /(["']?(?:password|passwd|secret|token|api[_-]?key|authorization|cookie|credential|access[_-]?key)["']?\s*[:=]\s*)(?:"[^"]*(?:"|$)|'[^']*(?:'|$)|[^\s,;]+)/giu,
63
+ "$1[redacted]",
64
+ );
65
+
66
+ return {
67
+ text: prefix(sanitized, limit),
68
+ truncated: value.length > 16384 || sanitized.length > limit || locationTruncated,
69
+ };
70
+ }
71
+
72
+ export function sanitizeDiagnostic(value: string, limit = 700): string {
73
+ return diagnosticText(value, limit).text;
74
+ }
75
+
76
+ type DiagnosticRecord = {
77
+ sequence: number;
78
+ navigation: number;
79
+ category: DiagnosticCategory;
80
+ message: string;
81
+ url: string;
82
+ method?: string;
83
+ status?: number;
84
+ resourceType?: string;
85
+ truncated: boolean;
86
+ };
87
+ export type DiagnosticQuery = {
88
+ maxEntries?: number;
89
+ maxChars?: number;
90
+ category?: DiagnosticCategory;
91
+ cursor?: string;
92
+ clear?: boolean;
93
+ };
94
+
95
+ export class BrowserDiagnostics {
96
+ private context = crypto.randomUUID();
97
+ private sequence = 0;
98
+ private navigation = 0;
99
+ private floor = 0;
100
+ private dropped = 0;
101
+ private bytes = 0;
102
+ private records: DiagnosticRecord[] = [];
103
+
104
+ reset(): void {
105
+ this.context = crypto.randomUUID();
106
+ this.sequence = 0;
107
+ this.navigation = 0;
108
+ this.floor = 0;
109
+ this.dropped = 0;
110
+ this.bytes = 0;
111
+ this.records = [];
112
+ }
113
+
114
+ navigated(): void {
115
+ this.navigation += 1;
116
+ }
117
+
118
+ record(
119
+ category: DiagnosticCategory,
120
+ message: string,
121
+ url = "",
122
+ extra: { method?: string; status?: number; resourceType?: string } = {},
123
+ ): void {
124
+ const clean = diagnosticText(message);
125
+ const location = url ? diagnosticUrl(url) : undefined;
126
+ const method = extra.method ? diagnosticText(extra.method, 20) : undefined;
127
+ const resourceType = extra.resourceType ? diagnosticText(extra.resourceType, 30) : undefined;
128
+ const entry: DiagnosticRecord = {
129
+ sequence: ++this.sequence,
130
+ navigation: this.navigation,
131
+ category,
132
+ message: clean.text,
133
+ url: location?.text ?? "",
134
+ method: method?.text,
135
+ status: extra.status,
136
+ resourceType: resourceType?.text,
137
+ truncated:
138
+ clean.truncated ||
139
+ location?.truncated === true ||
140
+ method?.truncated === true ||
141
+ resourceType?.truncated === true,
142
+ };
143
+ while (Buffer.byteLength(JSON.stringify(entry)) > MAX_RECORD_BYTES) {
144
+ entry.truncated = true;
145
+ if (entry.message.length) {
146
+ entry.message = prefix(entry.message, Math.floor(entry.message.length / 2));
147
+ } else {
148
+ entry.url = prefix(entry.url, Math.floor(entry.url.length / 2));
149
+ }
150
+ }
151
+
152
+ this.records.push(entry);
153
+ this.bytes += Buffer.byteLength(JSON.stringify(entry));
154
+ while (this.records.length > MAX_DIAGNOSTIC_RECORDS || this.bytes > MAX_DIAGNOSTIC_BYTES) {
155
+ const removed = this.records.shift()!;
156
+ this.bytes -= Buffer.byteLength(JSON.stringify(removed));
157
+ this.floor = removed.sequence;
158
+ this.dropped += 1;
159
+ }
160
+ }
161
+
162
+ read(query: DiagnosticQuery = {}) {
163
+ const maxEntries = boundedInteger(query.maxEntries, 50, 100);
164
+ const maxChars = boundedInteger(query.maxChars, 12000, 30000, 1000);
165
+ if (query.category !== undefined && !DIAGNOSTIC_CATEGORIES.includes(query.category)) {
166
+ throw new Error("Unknown diagnostic category.");
167
+ }
168
+
169
+ if (query.clear !== undefined && typeof query.clear !== "boolean") {
170
+ throw new Error("clear must be boolean.");
171
+ }
172
+
173
+ let after = 0;
174
+ let contextChanged = false;
175
+ if (query.cursor !== undefined) {
176
+ if (typeof query.cursor !== "string" || !/^[\da-f-]{36}:\d{1,16}$/u.test(query.cursor)) {
177
+ throw new Error("Invalid diagnostics cursor.");
178
+ }
179
+
180
+ const [context, sequence] = query.cursor.split(":");
181
+ contextChanged = context !== this.context;
182
+ after = contextChanged ? 0 : Number(sequence);
183
+ if (!Number.isSafeInteger(after) || after > this.sequence) {
184
+ throw new Error("Invalid diagnostics cursor sequence.");
185
+ }
186
+ }
187
+
188
+ const candidates = this.records.filter(
189
+ (entry) => entry.sequence > after && (!query.category || entry.category === query.category),
190
+ );
191
+ const result = {
192
+ notice: "Untrusted application output; redaction is best-effort. Empty results do not prove application health. Active page only.",
193
+ context: this.context,
194
+ contextChanged,
195
+ cursorGap: contextChanged || after < this.floor,
196
+ droppedRecords: this.dropped,
197
+ retainedRecords: this.records.length,
198
+ retainedBytes: this.bytes,
199
+ clearedRecords: query.clear ? this.records.length : 0,
200
+ hasMore: false,
201
+ nextCursor: `${this.context}:${this.sequence}`,
202
+ records: [] as DiagnosticRecord[],
203
+ };
204
+ for (const entry of candidates.slice(0, maxEntries)) {
205
+ result.records.push({ ...entry });
206
+ if (JSON.stringify(result).length > maxChars) {
207
+ if (result.records.length > 1) {
208
+ result.records.pop();
209
+ break;
210
+ }
211
+
212
+ const first = result.records[0];
213
+ while (JSON.stringify(result).length > maxChars) {
214
+ first.truncated = true;
215
+ if (first.message.length) {
216
+ first.message = prefix(first.message, Math.floor(first.message.length / 2));
217
+ } else {
218
+ first.url = prefix(first.url, Math.floor(first.url.length / 2));
219
+ }
220
+ }
221
+ }
222
+ }
223
+
224
+ result.hasMore = result.records.length < candidates.length;
225
+ if (result.hasMore) {
226
+ result.nextCursor = `${this.context}:${result.records.at(-1)?.sequence ?? after}`;
227
+ }
228
+
229
+ if (query.clear) {
230
+ // Atomic synchronous read-and-clear of the entire buffer, including filtered/unreturned records.
231
+ this.records = [];
232
+ this.bytes = 0;
233
+ this.floor = this.sequence;
234
+ }
235
+
236
+ return result;
237
+ }
238
+
239
+ attach(page: Page): () => void {
240
+ const onError = (error: Error) => this.record("pageerror", error.message);
241
+ const onConsole = (message: ConsoleMessage) => {
242
+ if (message.type() === "error") {
243
+ this.record("console", message.text(), message.location().url);
244
+ }
245
+ };
246
+
247
+ const onRequest = (request: Request) =>
248
+ this.record("requestfailed", request.failure()?.errorText ?? "Request failed", request.url(), {
249
+ method: request.method(),
250
+ resourceType: request.resourceType(),
251
+ });
252
+ const onResponse = (response: Response) => {
253
+ if (response.status() >= 400) {
254
+ const request = response.request();
255
+ this.record("http", `HTTP ${response.status()}`, response.url(), {
256
+ status: response.status(),
257
+ method: request.method(),
258
+ resourceType: request.resourceType(),
259
+ });
260
+ }
261
+ };
262
+
263
+ page.on("pageerror", onError);
264
+ page.on("console", onConsole);
265
+ page.on("requestfailed", onRequest);
266
+ page.on("response", onResponse);
267
+
268
+ return () => {
269
+ page.off("pageerror", onError);
270
+ page.off("console", onConsole);
271
+ page.off("requestfailed", onRequest);
272
+ page.off("response", onResponse);
273
+ };
274
+ }
275
+ }