@trim21/personal-pi-extensions 0.1.556 → 0.1.558

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,16 +1,19 @@
1
1
  /**
2
- * gh-readonly 的代理配置与请求层。
2
+ * HTTP 代理层:gh-readonly 与 web_fetch 共用的出网配置。
3
3
  *
4
4
  * 配置来源(配置文件优先,未写的字段回退到环境变量):
5
- * - ~/.pi/agent/gh.json: { "proxy": "http://127.0.0.1:7890", "noProxy": "localhost,.corp" }
5
+ * - ~/.pi/agent/proxy.json: { "proxy": "http://127.0.0.1:7890", "noProxy": "localhost,.corp" }
6
6
  * - HTTPS_PROXY / HTTP_PROXY / ALL_PROXY(小写变体同样接受)、NO_PROXY
7
7
  *
8
- * 两条出口共用同一份配置,且在扩展加载时一次性读完:
8
+ * 这份配置是全局出网代理,不专属 GitHub:gh 子进程、octokit 请求、web_fetch 都走它。
9
+ *
10
+ * 三条出口共用同一份配置,且在扩展加载时一次性读完:
9
11
  * - gh CLI 子进程:env 给出要注入子进程的 HTTP(S)_PROXY / NO_PROXY 等变量
10
- * - octokit 请求:fetch 是挂了代理 dispatcher 的 fetch;未配置代理时就是全局 fetch
12
+ * - octokit 请求:fetch 是挂了代理 dispatcher 的 fetch(octokit 只认自定义 fetch)
13
+ * - web_fetch:同样用 fetch(Node 的全局 fetch 不认 HTTPS_PROXY 环境变量)
11
14
  *
12
- * 配置有错(JSON 语法错、字段类型不符、proxy 不是 http(s) URL)直接抛错——扩展
13
- * 加载即失败,而不是带着一份被忽略的配置静默直连。
15
+ * 未配置代理时 fetch 就是全局 fetch。配置有错(JSON 语法错、字段类型不符、proxy 不是
16
+ * http(s) URL)直接抛错——扩展加载即失败,而不是带着一份被忽略的配置静默直连。
14
17
  */
15
18
 
16
19
  import { readFileSync } from "node:fs";
@@ -22,25 +25,25 @@ import { EnvHttpProxyAgent } from "undici";
22
25
 
23
26
  import { parseWithSchema } from "./parse-with-schema.js";
24
27
 
25
- const ghConfigSchema = Type.Object({
28
+ const proxyConfigSchema = Type.Object({
26
29
  proxy: Type.Optional(Type.String()),
27
30
  noProxy: Type.Optional(Type.String()),
28
31
  });
29
32
 
30
- export interface GhProxySettings {
33
+ export interface HttpProxySettings {
31
34
  /** 代理 URL(http/https);undefined 表示不使用代理。 */
32
35
  proxy?: string;
33
36
  /** 不走代理的 host 列表(逗号分隔),语义同 NO_PROXY。 */
34
37
  noProxy?: string;
35
38
  }
36
39
 
37
- export function ghProxyConfigPath(): string {
38
- return join(homedir(), ".pi", "agent", "gh.json");
40
+ export function proxyConfigPath(): string {
41
+ return join(homedir(), ".pi", "agent", "proxy.json");
39
42
  }
40
43
 
41
- /** 解析 gh.json 的内容;字段类型不符时抛出带字段路径的错误。 */
42
- export function parseGhProxyConfig(value: unknown): GhProxySettings {
43
- const parsed = parseWithSchema(ghConfigSchema, value);
44
+ /** 解析 proxy.json 的内容;字段类型不符时抛出带字段路径的错误。 */
45
+ export function parseProxyConfig(value: unknown): HttpProxySettings {
46
+ const parsed = parseWithSchema(proxyConfigSchema, value);
44
47
  const proxy = parsed.proxy?.trim();
45
48
  const noProxy = parsed.noProxy?.trim();
46
49
  return { ...(proxy && { proxy }), ...(noProxy && { noProxy }) };
@@ -84,10 +87,10 @@ function normalizeProxy(value: string): string {
84
87
  * 这里是仓库里允许的同步例外)。文件不存在 = 未配置;文件读不了、JSON 非法或
85
88
  * 字段不符都直接抛。
86
89
  */
87
- export function readGhProxySettings(
88
- configPath: string = ghProxyConfigPath(),
90
+ export function readProxySettings(
91
+ configPath: string = proxyConfigPath(),
89
92
  env: NodeJS.ProcessEnv = process.env,
90
- ): GhProxySettings {
93
+ ): HttpProxySettings {
91
94
  let raw: string | undefined;
92
95
  try {
93
96
  raw = readFileSync(configPath, "utf8");
@@ -99,10 +102,10 @@ export function readGhProxySettings(
99
102
  }
100
103
  }
101
104
 
102
- let file: GhProxySettings = {};
105
+ let file: HttpProxySettings = {};
103
106
  if (raw !== undefined) {
104
107
  try {
105
- file = parseGhProxyConfig(JSON.parse(raw));
108
+ file = parseProxyConfig(JSON.parse(raw));
106
109
  } catch (error) {
107
110
  throw new Error(`${configPath}: ${error instanceof Error ? error.message : String(error)}`, {
108
111
  cause: error,
@@ -119,7 +122,7 @@ export function readGhProxySettings(
119
122
  }
120
123
 
121
124
  /** 要注入 gh 子进程的代理环境变量;未配置代理时为空对象(子进程继承父进程环境)。 */
122
- export function proxyEnvVars(settings: GhProxySettings): NodeJS.ProcessEnv {
125
+ export function proxyEnvVars(settings: HttpProxySettings): NodeJS.ProcessEnv {
123
126
  const { proxy, noProxy } = settings;
124
127
  if (!proxy) return {};
125
128
  return {
@@ -154,9 +157,9 @@ function createProxyDispatcher(
154
157
  }) as unknown as NonNullable<RequestInit["dispatcher"]>;
155
158
  }
156
159
 
157
- export interface GhProxy {
160
+ export interface HttpProxy {
158
161
  /** 生效的代理设置(配置文件与环境变量合并后的结果)。 */
159
- readonly settings: GhProxySettings;
162
+ readonly settings: HttpProxySettings;
160
163
  /** 要注入 gh 子进程的代理环境变量;未配置代理时为空对象。 */
161
164
  readonly env: NodeJS.ProcessEnv;
162
165
  /** 走代理的 fetch;未配置代理时就是全局 fetch。 */
@@ -168,11 +171,11 @@ export interface GhProxy {
168
171
  * `globalThis.fetch` 本身:每次调用都取当前的全局 fetch,否则首个请求之后替换
169
172
  * `globalThis.fetch`(插桩、测试替身)就不再生效。
170
173
  */
171
- export function createGhProxy(
172
- configPath: string = ghProxyConfigPath(),
174
+ export function createHttpProxy(
175
+ configPath: string = proxyConfigPath(),
173
176
  env: NodeJS.ProcessEnv = process.env,
174
- ): GhProxy {
175
- const settings = readGhProxySettings(configPath, env);
177
+ ): HttpProxy {
178
+ const settings = readProxySettings(configPath, env);
176
179
  const { proxy, noProxy } = settings;
177
180
  let dispatcher: NonNullable<RequestInit["dispatcher"]> | undefined;
178
181
 
package/src/web/fetch.ts CHANGED
@@ -1,13 +1,19 @@
1
1
  /**
2
- * `web_fetch` 工具:抓取 URL 并提取正文为 markdown。
2
+ * `web_fetch` 工具:抓取 URL 并提取正文为 markdown,或按 `output_path` 原样落盘。
3
3
  *
4
4
  * SSRF 防护:DNS 预解析 + 拒绝私有/保留地址 + 每跳重定向重新校验,
5
5
  * 防止把 agent 变成内网探测口。正文提取用 readability 主内容算法。
6
6
  *
7
+ * 出网走 `src/lib/proxy.ts` 的代理层(~/.pi/agent/proxy.json,回退 HTTPS_PROXY 等环境
8
+ * 变量):Node 的全局 fetch 不认代理环境变量,GitHub 的用户附件、release 资产这类只在
9
+ * 代理可达的 host 上,必须从这里挂出去,否则沙箱内一律 fetch failed。
10
+ *
7
11
  * 本文件是独立扩展入口(见 package.json 的 pi.extensions),可在配置里单独禁用。
8
12
  */
9
13
  import { lookup } from "node:dns/promises";
14
+ import { mkdir, open, rm } from "node:fs/promises";
10
15
  import { isIP } from "node:net";
16
+ import { dirname } from "node:path";
11
17
 
12
18
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
13
19
  import { Readability } from "@mozilla/readability";
@@ -15,9 +21,17 @@ import { parseHTML } from "linkedom";
15
21
  import TurndownService from "turndown";
16
22
  import { Type } from "typebox";
17
23
 
24
+ import { resolvePathArg } from "../lib/path.js";
25
+ import { createHttpProxy } from "../lib/proxy.js";
26
+ import { guardWriteAccess } from "../lib/write-guard.js";
27
+
28
+ const httpProxy = createHttpProxy();
29
+
18
30
  const MAX_REDIRECTS = 5;
19
31
  const TIMEOUT_MS = 30_000;
20
32
  const MAX_BYTES = 5 * 1024 * 1024;
33
+ /** 落盘模式的上限:附件、镜像、release 资产都比网页大得多,文本模式仍用 MAX_BYTES。 */
34
+ const MAX_FILE_BYTES = 200 * 1024 * 1024;
21
35
  const MIN_USEFUL_CONTENT = 200;
22
36
  const MAX_MARKDOWN_BYTES = 100 * 1024;
23
37
 
@@ -84,12 +98,20 @@ interface FetchedPage {
84
98
  markdown: string;
85
99
  }
86
100
 
87
- /** 手动跟随重定向,每跳重新做 SSRF 校验(防 DNS rebinding 简化处理) */
101
+ /** 落盘结果:最终 URL、本地路径、响应的 content-type 与写出的字节数。 */
102
+ interface SavedFile {
103
+ url: string;
104
+ filePath: string;
105
+ contentType: string;
106
+ bytes: number;
107
+ }
108
+
109
+ /** 手动跟随重定向,每跳重新做 SSRF 校验(防 DNS rebinding 简化处理)。 */
88
110
  async function fetchWithRedirects(
89
111
  url: URL,
90
112
  signal: AbortSignal | undefined,
91
- fetchFn: typeof fetch = fetch,
92
- ): Promise<Response> {
113
+ fetchFn: typeof fetch = httpProxy.fetch,
114
+ ): Promise<{ response: Response; url: URL }> {
93
115
  let current = url;
94
116
  for (let redirects = 0; ; redirects++) {
95
117
  await assertPublicHostname(current.hostname);
@@ -109,11 +131,19 @@ async function fetchWithRedirects(
109
131
  }
110
132
  continue;
111
133
  }
112
- return response;
134
+ return { response, url: current };
113
135
  }
114
136
  }
115
137
 
116
- export async function fetchPage(url: string, signal?: AbortSignal): Promise<FetchedPage> {
138
+ /** 已确认 2xx 的响应,连同手动重定向后跟踪到的最终地址。 */
139
+ interface OpenedResponse {
140
+ response: Response;
141
+ /** 手动重定向下 `response.url` 可能是空的,最终地址由重定向循环给出。 */
142
+ url: string;
143
+ }
144
+
145
+ /** 校验 URL、逐跳跟随重定向、要求 2xx;响应体怎么处理由调用方决定。 */
146
+ async function openResponse(url: string, signal?: AbortSignal): Promise<OpenedResponse> {
117
147
  let target: URL;
118
148
  try {
119
149
  target = new URL(url);
@@ -124,10 +154,15 @@ export async function fetchPage(url: string, signal?: AbortSignal): Promise<Fetc
124
154
  throw new Error(`只支持 http/https,收到: ${target.protocol}`);
125
155
  }
126
156
 
127
- const response = await fetchWithRedirects(target, signal);
157
+ const { response, url: finalUrl } = await fetchWithRedirects(target, signal);
128
158
  if (!response.ok) {
129
159
  throw new Error(`HTTP ${response.status} ${response.statusText}`);
130
160
  }
161
+ return { response, url: finalUrl.href };
162
+ }
163
+
164
+ export async function fetchPage(url: string, signal?: AbortSignal): Promise<FetchedPage> {
165
+ const { response, url: finalUrl } = await openResponse(url, signal);
131
166
  const contentType = response.headers.get("content-type") ?? "";
132
167
  const category = classifyContentType(contentType);
133
168
  if (category === null) {
@@ -155,10 +190,58 @@ export async function fetchPage(url: string, signal?: AbortSignal): Promise<Fetc
155
190
  }
156
191
 
157
192
  if (category === "html") {
158
- return extractMarkdown(body, response.url);
193
+ return extractMarkdown(body, finalUrl);
159
194
  }
160
195
  // JSON / XML / text/*:原样返回
161
- return { url: response.url, title: response.url, markdown: body.trim() };
196
+ return { url: finalUrl, title: finalUrl, markdown: body.trim() };
197
+ }
198
+
199
+ /**
200
+ * 把响应体原样写进 `filePath`(二进制安全:不做 content-type 白名单、不解码、不转换),
201
+ * 用于 GitHub 用户附件、release 资产这类不能当正文读的下载。
202
+ * 任何一步失败都会删掉半成品,不留下看着完整其实截断的文件。
203
+ */
204
+ export async function saveUrlToFile(
205
+ url: string,
206
+ filePath: string,
207
+ signal?: AbortSignal,
208
+ ): Promise<SavedFile> {
209
+ const { response, url: finalUrl } = await openResponse(url, signal);
210
+
211
+ const declaredLength = Number(response.headers.get("content-length") ?? "0");
212
+ if (declaredLength > MAX_FILE_BYTES) {
213
+ throw new Error(`文件过大 (${declaredLength} bytes),上限 ${MAX_FILE_BYTES}`);
214
+ }
215
+
216
+ await mkdir(dirname(filePath), { recursive: true });
217
+ const handle = await open(filePath, "w");
218
+ let bytes = 0;
219
+ try {
220
+ if (response.body) {
221
+ const reader = response.body.getReader();
222
+ for (;;) {
223
+ const chunk = (await reader.read()) as { done: boolean; value: Uint8Array };
224
+ if (chunk.done) break;
225
+ bytes += chunk.value.byteLength;
226
+ if (bytes > MAX_FILE_BYTES) {
227
+ throw new Error(`文件过大,上限 ${MAX_FILE_BYTES} bytes`);
228
+ }
229
+ await handle.write(chunk.value);
230
+ }
231
+ }
232
+ } catch (error) {
233
+ await handle.close();
234
+ await rm(filePath, { force: true });
235
+ throw error;
236
+ }
237
+ await handle.close();
238
+
239
+ return {
240
+ url: finalUrl,
241
+ filePath,
242
+ contentType: response.headers.get("content-type") ?? "",
243
+ bytes,
244
+ };
162
245
  }
163
246
 
164
247
  /** 按 mime 主体分类响应;html 走 readability,其余文本类原样返回 */
@@ -232,14 +315,48 @@ export default function webFetchTool(pi: ExtensionAPI): void {
232
315
  label: "Web Fetch",
233
316
  description:
234
317
  "Fetch a URL and return its content as markdown (HTML pages) or raw text " +
235
- "(JSON/XML/plain-text API responses). SSRF-protected: refuses private/internal " +
236
- "addresses.",
237
- promptSnippet: "Fetch a web page or API response",
318
+ "(JSON/XML/plain-text API responses). With output_path the body is saved to that " +
319
+ "file verbatim instead — any content type, no extraction, no truncation — and the " +
320
+ "result is the JSON summary {url, file_path, content_type, bytes} rather than the " +
321
+ "content; use it for images, logs and other attachments (GitHub user-attachments " +
322
+ "links from issue bodies, release assets, raw files). Give the file the extension " +
323
+ "matching the response's content_type: the Read tool decides image support by " +
324
+ "extension. Requests honour the proxy in ~/.pi/agent/proxy.json, so they reach hosts " +
325
+ "the shell sandbox blocks. SSRF-protected: refuses private/internal addresses.",
326
+ promptSnippet: "Fetch a web page, API response, or download a file",
238
327
  parameters: Type.Object({
239
328
  url: Type.String({ description: "The URL to fetch" }),
329
+ output_path: Type.Optional(
330
+ Type.String({
331
+ description:
332
+ "Save the response body to this path verbatim instead of returning it (absolute, or relative to the session cwd; ~ is expanded). Parent directories are created and an existing file is overwritten.",
333
+ }),
334
+ ),
240
335
  }),
241
- async execute(_id, params, signal) {
336
+ async execute(_id, params, signal, _onUpdate, ctx) {
337
+ const destination =
338
+ params.output_path === undefined ? undefined : resolvePathArg(ctx.cwd, params.output_path);
339
+ // 落盘位置的审批与写文件工具同一套:工作区与 /tmp 自动放行,其余问用户。
340
+ // 放在 try 外面,拒绝的原因(user deny)不该被改写成「抓取失败」。
341
+ if (destination !== undefined) {
342
+ await guardWriteAccess(ctx, { toolName: "web_fetch", absolutePath: destination });
343
+ }
344
+
242
345
  try {
346
+ if (destination !== undefined) {
347
+ const file = await saveUrlToFile(params.url, destination, signal);
348
+ const payload = {
349
+ url: file.url,
350
+ file_path: file.filePath,
351
+ content_type: file.contentType,
352
+ bytes: file.bytes,
353
+ };
354
+ return {
355
+ content: [{ type: "text", text: JSON.stringify(payload, null, 2) }],
356
+ details: payload,
357
+ };
358
+ }
359
+
243
360
  const page = await fetchPage(params.url, signal);
244
361
  const { text, truncated } = truncateMarkdown(page.markdown);
245
362
  const details: Record<string, unknown> = {