@trim21/personal-pi-extensions 0.0.436 → 0.1.437

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@trim21/personal-pi-extensions",
3
- "version": "0.0.436",
3
+ "version": "0.1.437",
4
4
  "type": "module",
5
5
  "description": "Custom pi coding-agent extensions: bwrap sandbox, workspace guard, opencode edit, and more",
6
6
  "keywords": [
@@ -70,7 +70,8 @@
70
70
  "src/spawn-agent.ts",
71
71
  "src/system-prompt/index.ts",
72
72
  "src/talk/index.ts",
73
- "src/web/index.ts"
73
+ "src/web/search.ts",
74
+ "src/web/fetch.ts"
74
75
  ],
75
76
  "skills": [
76
77
  "src/talk/skills"
package/src/web/fetch.ts CHANGED
@@ -1,20 +1,25 @@
1
1
  /**
2
- * web_fetch:抓取 URL 并提取正文为 markdown。
2
+ * `web_fetch` 工具:抓取 URL 并提取正文为 markdown。
3
3
  *
4
4
  * SSRF 防护:DNS 预解析 + 拒绝私有/保留地址 + 每跳重定向重新校验,
5
5
  * 防止把 agent 变成内网探测口。正文提取用 readability 主内容算法。
6
+ *
7
+ * 本文件是独立扩展入口(见 package.json 的 pi.extensions),可在配置里单独禁用。
6
8
  */
7
9
  import { lookup } from "node:dns/promises";
8
10
  import { isIP } from "node:net";
9
11
 
12
+ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
10
13
  import { Readability } from "@mozilla/readability";
11
14
  import { parseHTML } from "linkedom";
12
15
  import TurndownService from "turndown";
16
+ import { Type } from "typebox";
13
17
 
14
18
  const MAX_REDIRECTS = 5;
15
19
  const TIMEOUT_MS = 30_000;
16
20
  const MAX_BYTES = 5 * 1024 * 1024;
17
21
  const MIN_USEFUL_CONTENT = 200;
22
+ const MAX_MARKDOWN_BYTES = 100 * 1024;
18
23
 
19
24
  function isPrivateIpv4(ip: string): boolean {
20
25
  const parts = ip.split(".").map(Number);
@@ -212,3 +217,50 @@ export function extractMarkdown(html: string, sourceUrl: string): FetchedPage {
212
217
  .trim();
213
218
  return { url: sourceUrl, title, markdown };
214
219
  }
220
+
221
+ function truncateMarkdown(text: string): { text: string; truncated: boolean } {
222
+ if (Buffer.byteLength(text, "utf8") <= MAX_MARKDOWN_BYTES) return { text, truncated: false };
223
+ const bytes = Buffer.from(text, "utf8");
224
+ const sliced = bytes.subarray(0, MAX_MARKDOWN_BYTES).toString("utf8");
225
+ const cut = sliced.lastIndexOf("\n", sliced.length - 1);
226
+ return { text: (cut > 0 ? sliced.slice(0, cut) : sliced) + "\n…(已截断)", truncated: true };
227
+ }
228
+
229
+ export default function webFetchTool(pi: ExtensionAPI): void {
230
+ pi.registerTool({
231
+ name: "web_fetch",
232
+ label: "Web Fetch",
233
+ description:
234
+ "Fetch a URL and return its content as markdown (HTML pages) or raw text " +
235
+ "(JSON/XML/plain-text API responses). SSRF-protected: refuses private/internal " +
236
+ "addresses.",
237
+ promptSnippet: "Fetch a web page or API response",
238
+ parameters: Type.Object({
239
+ url: Type.String({ description: "The URL to fetch" }),
240
+ }),
241
+ async execute(_id, params, signal) {
242
+ try {
243
+ const page = await fetchPage(params.url, signal);
244
+ const { text, truncated } = truncateMarkdown(page.markdown);
245
+ const details: Record<string, unknown> = {
246
+ url: page.url,
247
+ title: page.title,
248
+ bytes: Buffer.byteLength(page.markdown, "utf8"),
249
+ truncated,
250
+ };
251
+ return {
252
+ content: [{ type: "text", text }],
253
+ details,
254
+ };
255
+ } catch (error) {
256
+ const message = error instanceof Error ? error.message : String(error);
257
+ const details: Record<string, unknown> = { error: message, url: params.url };
258
+ return {
259
+ isError: true,
260
+ content: [{ type: "text", text: `抓取失败: ${message}` }],
261
+ details,
262
+ };
263
+ }
264
+ },
265
+ });
266
+ }
package/src/web/search.ts CHANGED
@@ -1,12 +1,18 @@
1
1
  /**
2
- * Search1API 搜索:请求构造、响应解析与结果整理。
2
+ * `web_search` 工具:Search1API 搜索的请求构造、响应解析、结果整理与注册。
3
3
  *
4
+ * key 读 ~/.pi/web-search.json 的 search1apiApiKey 或 SEARCH1API_KEY 环境变量。
4
5
  * 搜索响应不做 AI 预消化,直接返回整理后的结构化结果(title/url/snippet,
5
6
  * crawl_results 开启时含内联正文)——模型自己读,零额外模型调用。
7
+ *
8
+ * 本文件是独立扩展入口(见 package.json 的 pi.extensions),可在配置里单独禁用。
6
9
  */
10
+ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
7
11
  import { Type } from "typebox";
8
12
  import { Value } from "typebox/value";
9
13
 
14
+ import { loadSearch1ApiKey } from "./config.js";
15
+
10
16
  const SEARCH_URL = "https://api.search1api.com/search";
11
17
  const TIMEOUT_MS = 60_000;
12
18
 
@@ -143,3 +149,118 @@ export async function searchWeb(
143
149
  const parsed = Value.Parse(searchResponseSchema, JSON.parse(raw));
144
150
  return { query, hits: mapHits(parsed.results) };
145
151
  }
152
+
153
+ export default function webSearchTool(pi: ExtensionAPI): void {
154
+ pi.registerTool({
155
+ name: "web_search",
156
+ label: "Web Search",
157
+ description:
158
+ "Search the web via Search1API and return structured results " +
159
+ "(title/url/snippet, optionally inline content).",
160
+ promptSnippet: "Search the web",
161
+ parameters: Type.Object({
162
+ query: Type.Optional(
163
+ Type.String({ description: "Search query (mutually exclusive with queries)" }),
164
+ ),
165
+ queries: Type.Optional(
166
+ Type.Array(Type.String(), { description: "Multiple queries searched in sequence" }),
167
+ ),
168
+ numResults: Type.Optional(
169
+ Type.Number({ description: "Results per query (default: 5, max: 50)" }),
170
+ ),
171
+ recencyFilter: Type.Optional(
172
+ Type.String({ description: "Filter by recency: day, week, month, year" }),
173
+ ),
174
+ domainFilter: Type.Optional(
175
+ Type.Array(Type.String(), { description: "Limit to domains; prefix with - to exclude" }),
176
+ ),
177
+ searchService: Type.Optional(
178
+ Type.String({
179
+ description:
180
+ "Search engine: google, bing, duckduckgo, github, arxiv, reddit, youtube, etc.",
181
+ }),
182
+ ),
183
+ includeContent: Type.Optional(
184
+ Type.Boolean({ description: "Inline-fetch content for the top results (up to 5)" }),
185
+ ),
186
+ }),
187
+ async execute(_id, params, signal, onUpdate) {
188
+ try {
189
+ const queries = [...(params.query ? [params.query] : []), ...(params.queries ?? [])]
190
+ .map((q) => q.trim())
191
+ .filter(Boolean)
192
+ .slice(0, 4);
193
+ if (queries.length === 0) {
194
+ return {
195
+ isError: true,
196
+ content: [{ type: "text", text: "需要提供 query 或 queries(至少一个搜索词)。" }],
197
+ details: { error: "no query" },
198
+ };
199
+ }
200
+
201
+ const apiKey = await loadSearch1ApiKey();
202
+ if (!apiKey) {
203
+ return {
204
+ isError: true,
205
+ content: [
206
+ {
207
+ type: "text",
208
+ text: "未找到 Search1API key:请在 ~/.pi/web-search.json 配置 search1apiApiKey,或设置 SEARCH1API_KEY 环境变量。",
209
+ },
210
+ ],
211
+ details: { error: "search1api key not configured" },
212
+ };
213
+ }
214
+
215
+ onUpdate?.({
216
+ content: [{ type: "text", text: `正在搜索: ${queries.join(" / ")}` }],
217
+ details: {},
218
+ });
219
+
220
+ const common = {
221
+ numResults: params.numResults,
222
+ recencyFilter: params.recencyFilter,
223
+ domainFilter: params.domainFilter,
224
+ searchService: params.searchService,
225
+ includeContent: params.includeContent,
226
+ signal,
227
+ };
228
+ const results = await Promise.all(queries.map((query) => searchWeb(query, apiKey, common)));
229
+
230
+ // 按 URL 去重合并
231
+ const seen = new Set<string>();
232
+ const hits: SearchHit[] = [];
233
+ for (const result of results) {
234
+ for (const hit of result.hits) {
235
+ if (seen.has(hit.url)) continue;
236
+ seen.add(hit.url);
237
+ hits.push(hit);
238
+ }
239
+ }
240
+ if (hits.length === 0) {
241
+ return {
242
+ content: [{ type: "text", text: "没有找到结果。" }],
243
+ details: { query: queries, count: 0 },
244
+ };
245
+ }
246
+
247
+ return {
248
+ content: [{ type: "text", text: JSON.stringify(hits, null, 2) }],
249
+ details: {
250
+ query: queries,
251
+ provider: "search1api",
252
+ count: hits.length,
253
+ results: hits,
254
+ },
255
+ };
256
+ } catch (error) {
257
+ const message = error instanceof Error ? error.message : String(error);
258
+ return {
259
+ isError: true,
260
+ content: [{ type: "text", text: `搜索失败: ${message}` }],
261
+ details: { error: message },
262
+ };
263
+ }
264
+ },
265
+ });
266
+ }
package/src/web/index.ts DELETED
@@ -1,175 +0,0 @@
1
- /**
2
- * web 扩展:自研 web_search + web_fetch,替代 pi-web-access。
3
- *
4
- * - web_search:Search1API 搜索(key 读 ~/.pi/web-search.json 的
5
- * search1apiApiKey 或 SEARCH1API_KEY),直接返回整理后的结构化结果。
6
- * - web_fetch:抓取 URL,SSRF 防护 + readability 提取正文为 markdown。
7
- */
8
- import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
9
- import { Type } from "typebox";
10
-
11
- import { loadSearch1ApiKey } from "./config.js";
12
- import { fetchPage } from "./fetch.js";
13
- import { type SearchHit, searchWeb } from "./search.js";
14
-
15
- const MAX_MARKDOWN_BYTES = 100 * 1024;
16
-
17
- function truncateMarkdown(text: string): { text: string; truncated: boolean } {
18
- if (Buffer.byteLength(text, "utf8") <= MAX_MARKDOWN_BYTES) return { text, truncated: false };
19
- const bytes = Buffer.from(text, "utf8");
20
- const sliced = bytes.subarray(0, MAX_MARKDOWN_BYTES).toString("utf8");
21
- const cut = sliced.lastIndexOf("\n", sliced.length - 1);
22
- return { text: (cut > 0 ? sliced.slice(0, cut) : sliced) + "\n…(已截断)", truncated: true };
23
- }
24
-
25
- export default function webTools(pi: ExtensionAPI) {
26
- pi.registerTool({
27
- name: "web_search",
28
- label: "Web Search",
29
- description:
30
- "Search the web via Search1API and return structured results " +
31
- "(title/url/snippet, optionally inline content).",
32
- promptSnippet: "Search the web",
33
- parameters: Type.Object({
34
- query: Type.Optional(
35
- Type.String({ description: "Search query (mutually exclusive with queries)" }),
36
- ),
37
- queries: Type.Optional(
38
- Type.Array(Type.String(), { description: "Multiple queries searched in sequence" }),
39
- ),
40
- numResults: Type.Optional(
41
- Type.Number({ description: "Results per query (default: 5, max: 50)" }),
42
- ),
43
- recencyFilter: Type.Optional(
44
- Type.String({ description: "Filter by recency: day, week, month, year" }),
45
- ),
46
- domainFilter: Type.Optional(
47
- Type.Array(Type.String(), { description: "Limit to domains; prefix with - to exclude" }),
48
- ),
49
- searchService: Type.Optional(
50
- Type.String({
51
- description:
52
- "Search engine: google, bing, duckduckgo, github, arxiv, reddit, youtube, etc.",
53
- }),
54
- ),
55
- includeContent: Type.Optional(
56
- Type.Boolean({ description: "Inline-fetch content for the top results (up to 5)" }),
57
- ),
58
- }),
59
- async execute(_id, params, signal, onUpdate) {
60
- try {
61
- const queries = [...(params.query ? [params.query] : []), ...(params.queries ?? [])]
62
- .map((q) => q.trim())
63
- .filter(Boolean)
64
- .slice(0, 4);
65
- if (queries.length === 0) {
66
- return {
67
- isError: true,
68
- content: [{ type: "text", text: "需要提供 query 或 queries(至少一个搜索词)。" }],
69
- details: { error: "no query" },
70
- };
71
- }
72
-
73
- const apiKey = await loadSearch1ApiKey();
74
- if (!apiKey) {
75
- return {
76
- isError: true,
77
- content: [
78
- {
79
- type: "text",
80
- text: "未找到 Search1API key:请在 ~/.pi/web-search.json 配置 search1apiApiKey,或设置 SEARCH1API_KEY 环境变量。",
81
- },
82
- ],
83
- details: { error: "search1api key not configured" },
84
- };
85
- }
86
-
87
- onUpdate?.({
88
- content: [{ type: "text", text: `正在搜索: ${queries.join(" / ")}` }],
89
- details: {},
90
- });
91
-
92
- const common = {
93
- numResults: params.numResults,
94
- recencyFilter: params.recencyFilter,
95
- domainFilter: params.domainFilter,
96
- searchService: params.searchService,
97
- includeContent: params.includeContent,
98
- signal,
99
- };
100
- const results = await Promise.all(queries.map((query) => searchWeb(query, apiKey, common)));
101
-
102
- // 按 URL 去重合并
103
- const seen = new Set<string>();
104
- const hits: SearchHit[] = [];
105
- for (const result of results) {
106
- for (const hit of result.hits) {
107
- if (seen.has(hit.url)) continue;
108
- seen.add(hit.url);
109
- hits.push(hit);
110
- }
111
- }
112
- if (hits.length === 0) {
113
- return {
114
- content: [{ type: "text", text: "没有找到结果。" }],
115
- details: { query: queries, count: 0 },
116
- };
117
- }
118
-
119
- return {
120
- content: [{ type: "text", text: JSON.stringify(hits, null, 2) }],
121
- details: {
122
- query: queries,
123
- provider: "search1api",
124
- count: hits.length,
125
- results: hits,
126
- },
127
- };
128
- } catch (error) {
129
- const message = error instanceof Error ? error.message : String(error);
130
- return {
131
- isError: true,
132
- content: [{ type: "text", text: `搜索失败: ${message}` }],
133
- details: { error: message },
134
- };
135
- }
136
- },
137
- });
138
-
139
- pi.registerTool({
140
- name: "web_fetch",
141
- label: "Web Fetch",
142
- description:
143
- "Fetch a URL and return its content as markdown (HTML pages) or raw text " +
144
- "(JSON/XML/plain-text API responses). SSRF-protected: refuses private/internal " +
145
- "addresses.",
146
- promptSnippet: "Fetch a web page or API response",
147
- parameters: Type.Object({
148
- url: Type.String({ description: "The URL to fetch" }),
149
- }),
150
- async execute(_id, params, signal) {
151
- try {
152
- const page = await fetchPage(params.url, signal);
153
- const { text, truncated } = truncateMarkdown(page.markdown);
154
- const details: Record<string, unknown> = {
155
- url: page.url,
156
- title: page.title,
157
- bytes: Buffer.byteLength(page.markdown, "utf8"),
158
- truncated,
159
- };
160
- return {
161
- content: [{ type: "text", text }],
162
- details,
163
- };
164
- } catch (error) {
165
- const message = error instanceof Error ? error.message : String(error);
166
- const details: Record<string, unknown> = { error: message, url: params.url };
167
- return {
168
- isError: true,
169
- content: [{ type: "text", text: `抓取失败: ${message}` }],
170
- details,
171
- };
172
- }
173
- },
174
- });
175
- }