@epoch-agent/plugin-web 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,219 @@
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright [yyyy] [name of copyright owner]
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
203
+
204
+ --------------------------------------------------------------------------------
205
+
206
+ epoch-agent
207
+ Copyright 2024-2026 bowen
208
+
209
+ Portions of this software are derived from the Gemini CLI
210
+ (https://github.com/google-gemini/gemini-cli), Copyright 2025 Google LLC,
211
+ licensed under the Apache License, Version 2.0. Those files have been modified;
212
+ each one retains its original copyright notice and SPDX-License-Identifier
213
+ header. See the NOTICE file in the project source repository for the complete
214
+ list of derived files.
215
+
216
+ This software also quotes, with attribution, two lines of prose from OpenAI
217
+ Codex (https://github.com/openai/codex), Copyright 2025 OpenAI, licensed under
218
+ the Apache License, Version 2.0. No source file is a derived work of Codex.
219
+ See the NOTICE file for details.
package/README.md ADDED
@@ -0,0 +1,106 @@
1
+ # @epoch-agent/plugin-web
2
+
3
+ 网页抓取与搜索插件。**两个工具**:`web_fetch`(总是有)和 `web_search`
4
+ (**配了搜索后端才注册**)。
5
+
6
+ - ✅ **做**:HTTP GET → HTML→Markdown/Text/HTML、SSRF 校验、逐跳重定向校验、大小限制;
7
+ web 搜索(tavily / brave / searxng 三个后端)
8
+ - ❌ **不做**:不执行 JS、不渲染页面、不带浏览器;**不抓搜索引擎的 HTML 结果页**
9
+ (违反 ToS、随时会碎、要处理验证码 —— 没 key 就诚实地不提供这个工具)
10
+ - **依赖**:只有 node 内置模块。[protocol](../../protocol) 是 peer,
11
+ HTML 转换是手写的(无第三方库)
12
+
13
+ ## 工具参数
14
+
15
+ ```json
16
+ { "url": "https://example.com/doc", "format": "markdown", "timeout": 30 }
17
+ { "query": "node 22 fetch keepalive", "maxResults": 5, "recencyDays": 30 }
18
+ ```
19
+
20
+ **两个工具的参数表在 [docs/TOOLS.md](../../../docs/TOOLS.md#网页)** —— 那份是工具清单
21
+ 的唯一真源,这里只写这个包自己的实现取舍。
22
+
23
+ ## `web_search`:搜什么,不抓什么
24
+
25
+ **只返回标题 / URL / 摘要,不返回正文。** 一次搜索 5 条结果全抓正文轻松几万 token,
26
+ 而其中多半是模型看一眼摘要就丢掉的。分工是:搜 → 模型挑 → `web_fetch` 抓那一条。
27
+
28
+ ### 三个后端,按凭据自动挑
29
+
30
+ 三家(`tavily` / `brave` / `searxng`)各要什么凭据见
31
+ [docs/TOOLS.md](../../../docs/TOOLS.md#web_search-要一个搜索后端)。
32
+
33
+ 不配 `search.provider` 时按那张表的顺序探测,第一个有凭据的赢。**显式配了哪家就只认哪家**
34
+ —— 缺 key 时报诊断而不是悄悄换一家:换一家意味着查询发去了另一家公司,
35
+ 那是用户没同意的事。
36
+
37
+ 一个后端都没有时 `web_search` **不注册**,启动诊断给 `skipped` 并列出三家各缺什么。
38
+ 注册一个一调就报错的工具比不注册更糟:模型看见工具表里有它就会反复试,每次烧一轮。
39
+
40
+ > SearXNG **默认没开** JSON 输出,实例的 `settings.yml` 里要有
41
+ > `search: { formats: [html, json] }`。没开时它返回 403,我们把这句话写进了错误里 ——
42
+ > 否则用户会以为是自己地址写错了。
43
+
44
+ ### 搜索结果是**风险最高**的一种不可信内容
45
+
46
+ 它和抓来的网页一样带 `openWorldHint: true`(见下一节),但风险更高一档:
47
+ 网页要等 agent 主动去抓,而搜索结果是攻击者可以针对某个查询词**主动投放**的
48
+ (SEO 投毒 + 提示注入)。所以 [core 的 system prompt](../../core/src/agent/prompt.ts)
49
+ 在不可信内容策略里单独点了它的名。
50
+
51
+ 结果里的 URL **不自动跟随**。模型要抓时走 `web_fetch`,SSRF 校验和 `network`
52
+ 权限判定一道都不会因为「这是搜出来的」而放松。
53
+
54
+ 失败(key 错 / 配额用尽 / 网络不通)返回一条**说明原因**的失败结果,不抛异常 ——
55
+ 搜索挂了不该让整轮对话断在这儿。同一个查询在**同一会话内**缓存,省配额。
56
+
57
+ ## SSRF 防护是三层,不是一层
58
+
59
+ 1. **协议白名单**——只放 `http` / `https`
60
+ 2. **主机名与 IP 字面量黑名单**——`localhost`、云元数据域名、私网段
61
+ (10/8、127/8、169.254/16、172.16/12、192.168/16、CGNAT、组播与保留段,IPv6 同理)
62
+ 3. **DNS 解析后再校验一次**——攻击者控制的域名可以解析到 `127.0.0.1`,
63
+ 光看主机名字面量挡不住。这是最常见的绕过手法
64
+
65
+ 外加**逐跳重定向校验**:不用 `fetch` 的 `redirect: 'follow'`,而是手动跟随、每一跳
66
+ 重新做完整校验(最多 5 跳)。默认跟随只校验第一个 URL,一个公网地址 302 到
67
+ `169.254.169.254` 就能直接读到云元数据。
68
+
69
+ 其余限制:响应上限 5 MB,Cloudflare 挑战会重试一次。
70
+
71
+ ## 三个 annotation 不只是元数据
72
+
73
+ ```ts
74
+ annotations: { readOnlyHint: true, idempotentHint: false, openWorldHint: true }
75
+ ```
76
+
77
+ `openWorldHint: true` 会让 [core](../../core) 的 tool-executor 把输出包进
78
+ `<tool_output untrusted="true">`。抓来的网页内容作者既不是用户也不是我们,必须标成
79
+ 「数据」而不是「指令」再回灌上下文。
80
+
81
+ `describeTarget: (args) => args.url` 让审批缓存按 URL 记——不给它就只能退回参数 JSON,
82
+ 「批准抓这个站」下次照样弹窗。
83
+
84
+ ## 文件
85
+
86
+ | 文件 | 内容 |
87
+ | ---------------------------------- | ----------------------------------------------- |
88
+ | `index.ts` | `webPlugin` + `createWebSearchTool`(条件注册) |
89
+ | `tools/web-fetch.ts` | 抓取 + 重定向 + 大小限制 + 重试 |
90
+ | `tools/web-search.ts` | 搜索工具:参数钳制 + 会话内缓存 + 失败不炸轮 |
91
+ | `search/provider.ts` | 后端接口 + 共用的 HTTP / 错误分类 |
92
+ | `search/registry.ts` | 按配置 + 凭据挑后端(挑不到就不注册) |
93
+ | `search/{tavily,brave,searxng}.ts` | 三个后端各自的请求 / 响应翻译 |
94
+ | `utils/url-safety.ts` | 三层 SSRF 校验 |
95
+ | `utils/html-to-markdown.ts` | 手写 HTML→Markdown / →Text |
96
+
97
+ ## 开发
98
+
99
+ ```bash
100
+ pnpm --filter @epoch-agent/plugin-web test
101
+ ```
102
+
103
+ 四个用例文件:`ssrf.test.ts`、`url-safety.test.ts`、`html-to-markdown.test.ts`、
104
+ `web-search.test.ts`。改 SSRF 判定必须先看前两个——里面记的是具体的绕过手法,
105
+ 不是凑数的。搜索那份**全部用假 `fetch`**:真打网络的用例要么依赖别人的 API key、
106
+ 要么在墙内直接红,两种都不该进 `pnpm check`。
@@ -0,0 +1,229 @@
1
+ import { SearchProviderType, EpochConfig, ToolContext, ToolResult, EpochTool, EpochPlugin } from '@epoch-agent/protocol';
2
+
3
+ /**
4
+ * 搜索后端的接口(方案 37)。
5
+ *
6
+ * ## 为什么 provider 化而不是绑死一家
7
+ *
8
+ * 搜索 API 全都要 key、都有配额、都可能被墙。绑死一家等于给用户一个大概率
9
+ * 不可用的功能。设计口径和 `ProviderRouter`(15 个 LLM provider 按凭据挑)一致,
10
+ * 但**不复用代码** —— 那边要管降级链、工厂缓存、运行期换模型,这里只要
11
+ * 「能不能用」和「搜一次」两件事。
12
+ *
13
+ * ## 每个 provider 只负责三件事
14
+ *
15
+ * 1. 说自己能不能用(`unavailableReason`)—— 装配层据此决定注册不注册工具
16
+ * 2. 把我们的查询翻成它的 HTTP 请求
17
+ * 3. 把它的响应翻回 `SearchResult[]`
18
+ *
19
+ * **不负责**:SSRF 校验(结果里的 URL 不自动跟随,跟随时由 `web_fetch` 校验)、
20
+ * 缓存(在工具那一层,跨 provider 共用)、错误分类(同上)。
21
+ */
22
+ /** 一条搜索结果。**没有网页正文** —— 正文要模型自己调 `web_fetch` */
23
+ interface SearchResult {
24
+ title: string;
25
+ url: string;
26
+ /** 摘要,通常是 provider 自己截的一两句话 */
27
+ snippet: string;
28
+ /** ISO 日期,provider 给了才有 */
29
+ publishedAt?: string;
30
+ }
31
+ interface SearchOptions {
32
+ /** 要几条。provider 自己钳到它的上限 */
33
+ maxResults: number;
34
+ /** 只要最近 N 天的。provider 不支持时**忽略并说明**,不是报错 */
35
+ recencyDays?: number;
36
+ /** 只在这些域名里搜 */
37
+ domains?: string[];
38
+ /** 排除这些域名 */
39
+ excludeDomains?: string[];
40
+ /** 中止信号,工具那一层给的超时 */
41
+ signal?: AbortSignal;
42
+ }
43
+ interface SearchOutcome {
44
+ results: SearchResult[];
45
+ /**
46
+ * 这次调用有哪些参数**没生效**(provider 不支持)。
47
+ *
48
+ * 走这条通道而不是静默忽略:模型写了 `recencyDays: 7` 却拿到三年前的结果,
49
+ * 它没法知道是「最近确实没有」还是「这个参数根本没传下去」,
50
+ * 于是会反复重试同一个查询。
51
+ */
52
+ ignored?: string[];
53
+ }
54
+ interface SearchProvider {
55
+ readonly name: string;
56
+ /**
57
+ * 现在能不能用;能用返回 `null`,不能用返回**一句给人看的原因**
58
+ * (含怎么配)。装配层拿它写启动诊断。
59
+ *
60
+ * 每次调用都重新判:key 是从环境变量读的,而凭据预取发生在装配早期,
61
+ * 缓存下来会得到一个「启动那一刻」的答案。
62
+ */
63
+ unavailableReason(): string | null;
64
+ search(query: string, opts: SearchOptions): Promise<SearchOutcome>;
65
+ }
66
+ /** provider 报出来的失败。工具那一层把它翻成给模型看的话,**不让整轮失败** */
67
+ declare class SearchError extends Error {
68
+ /** HTTP 状态码,网络层失败时没有 */
69
+ readonly status?: number | undefined;
70
+ constructor(message: string,
71
+ /** HTTP 状态码,网络层失败时没有 */
72
+ status?: number | undefined);
73
+ }
74
+
75
+ /**
76
+ * 按配置 + 凭据挑一个搜索后端(方案 37)。
77
+ *
78
+ * ```
79
+ * 显式配了 provider → 只认那一家。缺 key 就报出来,**不静默回落到别家**
80
+ * 没配 → 按顺序试 tavily → brave → searxng,第一个能用的赢
81
+ * 一个都不能用 → 返回 null,工具**不注册**
82
+ * ```
83
+ *
84
+ * ## 「配了却缺 key」为什么不回落
85
+ *
86
+ * 用户显式写 `provider: brave` 是一个决定(他可能有隐私要求、或者公司只报销这家)。
87
+ * 悄悄换成 tavily 意味着他的查询发去了另一家公司 —— 那是他没同意的事。
88
+ * 报出来让他自己选,是唯一诚实的做法。
89
+ *
90
+ * ## 「一个都没有」为什么不注册工具,而不是注册一个会报错的
91
+ *
92
+ * 注册一个一调就失败的工具,模型会反复试(它看到工具表里有 `web_search`,
93
+ * 就会认为搜索是可行的),每次都烧一轮。不注册的话它从一开始就知道没这个能力。
94
+ * 诊断走 `skipped` 而不是 `failed` —— 没配搜索不是故障,是没开这个功能。
95
+ */
96
+
97
+ /** 一次后端选择的结果 */
98
+ type SearchSelection = {
99
+ provider: SearchProvider;
100
+ detail: string;
101
+ }
102
+ /** 没有可用后端。`detail` 是给启动诊断的一句话,含怎么配 */
103
+ | {
104
+ provider: null;
105
+ detail: string;
106
+ };
107
+ declare function createSearchProvider(type: SearchProviderType, config: EpochConfig['search']): SearchProvider;
108
+ /**
109
+ * 挑一个能用的后端。
110
+ *
111
+ * @param config `EpochConfig.search`。整段缺席 = 用户没开这个功能,
112
+ * 但**仍然会按凭据自动探测** —— 环境里已经有 `TAVILY_API_KEY` 的人
113
+ * 不该被要求再写一行配置才能用上。
114
+ */
115
+ declare function selectSearchProvider(config: EpochConfig['search']): SearchSelection;
116
+
117
+ /**
118
+ * SearXNG —— **自托管、无 key** 的那一档(方案 37)。
119
+ *
120
+ * 它是这个方案里唯一一个不依赖第三方账号的后端:内网、隐私敏感、
121
+ * 或者单纯不想再办一个 API key 的场景,答案是它。
122
+ *
123
+ * 用 JSON 格式接口(`/search?format=json`)。注意 SearXNG **默认没开** JSON 输出,
124
+ * 实例的 `settings.yml` 里要有:
125
+ *
126
+ * ```yaml
127
+ * search:
128
+ * formats: [html, json]
129
+ * ```
130
+ *
131
+ * 没开时它返回 HTTP 403 —— 那句提示写进了错误里,否则用户会以为是自己地址写错了。
132
+ */
133
+
134
+ interface SearxngOptions {
135
+ /** 实例地址,例如 `https://searx.example.com` */
136
+ baseUrl?: string;
137
+ }
138
+ declare function createSearxngProvider(opts: SearxngOptions): SearchProvider;
139
+
140
+ /**
141
+ * Tavily —— 专为 LLM 做的搜索 API(方案 37)。
142
+ *
143
+ * 选它做第一顺位是因为它返回的 `content` 本来就是**为模型准备的摘要**,
144
+ * 而不是搜索引擎那种带省略号的片段。同样一次搜索,模型判断「这条值不值得抓正文」
145
+ * 的准确率更高,于是二次 `web_fetch` 更少。
146
+ *
147
+ * key:`TAVILY_API_KEY`。
148
+ */
149
+
150
+ interface TavilyOptions {
151
+ /** 不给就从 `TAVILY_API_KEY` 读 —— **每次读**,不缓存(见 provider.ts) */
152
+ apiKey?: string;
153
+ }
154
+ declare function createTavilyProvider(opts?: TavilyOptions): SearchProvider;
155
+
156
+ /**
157
+ * Brave Search API(方案 37)。
158
+ *
159
+ * 有免费额度、隐私口碑好,是「不想把查询交给大厂」但又不想自建 SearXNG 时的选项。
160
+ *
161
+ * key:`BRAVE_SEARCH_API_KEY`。
162
+ */
163
+
164
+ interface BraveOptions {
165
+ apiKey?: string;
166
+ }
167
+ declare function createBraveProvider(opts?: BraveOptions): SearchProvider;
168
+
169
+ /**
170
+ * web_search —— 搜一串结果回来,**不抓正文**(方案 37)。
171
+ *
172
+ * ⚠️ 方案 37 的 md 2026-08-10 按去留规则删了,**而它没有独立的验收记录** ——
173
+ * 判据当时是搬进本文件这段头注释的(见 `.agents/plans/README.md` 那一轮的记账)。
174
+ * 也就是说这里就是真源,别再往方案 md 找。
175
+ *
176
+ * ## 为什么不顺手把正文也抓了
177
+ *
178
+ * 一次搜索 5 条结果,全抓正文轻松几万 token,而其中多半是模型看一眼摘要就会
179
+ * 丢掉的。分工是:`web_search` 给标题 + URL + 摘要,模型自己判断哪条值得看,
180
+ * 再调 `web_fetch`。工具描述里必须把这句话写出来,否则模型会期望这里直接给正文,
181
+ * 然后对着摘要抱怨内容不全。
182
+ *
183
+ * ## 搜索结果是**风险最高的一种**不可信内容
184
+ *
185
+ * `openWorldHint: true` 让 core 的 tool-executor 把输出包进
186
+ * `<tool_output untrusted="true">`。这不是走个形式:搜索结果是攻击者可以**主动
187
+ * 投放**的内容(SEO 投毒 + 提示注入),比读到一个仓库里的 README 危险得多。
188
+ * 结果里的 URL **不自动跟随** —— 模型要抓时走 `web_fetch`,那条路有 SSRF 校验
189
+ * 和 `network` 权限判定,一道都不会因为「这是搜出来的」而放松。
190
+ *
191
+ * ## 失败不炸轮
192
+ *
193
+ * 配额用尽 / key 错了 / 网络不通 —— 全部返回一条**说明为什么**的失败结果,
194
+ * 而不是抛异常。模型收到「搜索不可用:配额用尽」之后可以接着干别的;
195
+ * 抛出去的话这一轮就断在这儿了。
196
+ *
197
+ * ## 刻意不记搜索次数(方案 37 的一个明确决定)
198
+ *
199
+ * 原方案有一条「搜索调用次数进 `epoch status`」,**没做,也不打算做**:
200
+ * 那要在 runtime 上再挂一个计数器、穿过装配层递到 CLI,而它的收益
201
+ * (用户知道自己这轮搜了几次)远小于那条接线的常驻成本。
202
+ * 真需要按次计费或配额告警时,`Telemetry` 那条通道本来就是干这个的
203
+ * —— 工具调用已经有 `TOOL_CALL_COUNT` 指标,按工具名一分就是搜索次数,
204
+ * 不用为它单开一条链路。
205
+ */
206
+
207
+ interface WebSearchDeps {
208
+ provider: SearchProvider;
209
+ /** 一次返回几条的缺省值,来自 `config.search.maxResults` */
210
+ defaultMaxResults?: number;
211
+ }
212
+ /** 给用例用:清掉缓存 */
213
+ declare function clearSearchCache(): void;
214
+ /** 造一个绑定了具体后端的 `web_search` 执行函数 */
215
+ declare function createWebSearchExec(deps: WebSearchDeps): (args: Record<string, unknown>, ctx: ToolContext) => Promise<ToolResult>;
216
+
217
+ declare const webPlugin: EpochPlugin;
218
+ /**
219
+ * `web_search` 工具(方案 37)。
220
+ *
221
+ * **不在 `webPlugin.tools` 里**,因为它是有条件的:没有可用的搜索后端时
222
+ * 压根不该注册(见 `search/registry.ts` 的文件头)。装配层挑到后端才调这个工厂,
223
+ * 形状与 `createTodoTool` / `createMemoryTool` 那几个条件工具一致。
224
+ */
225
+ declare function createWebSearchTool(provider: SearchProvider, opts?: {
226
+ defaultMaxResults?: number;
227
+ }): EpochTool;
228
+
229
+ export { SearchError, type SearchOptions, type SearchOutcome, type SearchProvider, type SearchResult, type SearchSelection, clearSearchCache, createBraveProvider, createSearchProvider, createSearxngProvider, createTavilyProvider, createWebSearchExec, createWebSearchTool, selectSearchProvider, webPlugin };
package/dist/index.js ADDED
@@ -0,0 +1,743 @@
1
+ // src/utils/html-to-markdown.ts
2
+ function htmlToMarkdown(html) {
3
+ let cleaned = html.replace(/<script[\s\S]*?<\/script>/gi, "").replace(/<style[\s\S]*?<\/style>/gi, "").replace(/<meta[^>]*>/gi, "").replace(/<!--[\s\S]*?-->/g, "");
4
+ cleaned = cleaned.replace(
5
+ /<h([1-6])[^>]*>([\s\S]*?)<\/h\1>/gi,
6
+ (_, n, text) => "\n\n" + "#".repeat(Number(n)) + " " + stripTags(text).trim() + "\n\n"
7
+ );
8
+ cleaned = cleaned.replace(
9
+ /<p[^>]*>([\s\S]*?)<\/p>/gi,
10
+ (_, text) => "\n\n" + stripTags(text).trim() + "\n\n"
11
+ );
12
+ cleaned = cleaned.replace(/<br\s*\/?>/gi, "\n");
13
+ cleaned = cleaned.replace(/<a[^>]*href="([^"]*)"[^>]*>([\s\S]*?)<\/a>/gi, (_, url, text) => {
14
+ const t = stripTags(text).trim();
15
+ return t ? `[${t}](${url})` : url;
16
+ });
17
+ cleaned = cleaned.replace(
18
+ /<img[^>]*src="([^"]*)"[^>]*alt="([^"]*)"[^>]*>/gi,
19
+ (_, url, alt) => `![${alt}](${url})`
20
+ );
21
+ cleaned = cleaned.replace(/<img[^>]*src="([^"]*)"[^>]*>/gi, (_, url) => `![](${url})`);
22
+ cleaned = cleaned.replace(/<(strong|b)[^>]*>([\s\S]*?)<\/(strong|b)>/gi, "**$2**");
23
+ cleaned = cleaned.replace(/<(em|i)[^>]*>([\s\S]*?)<\/(em|i)>/gi, "*$2*");
24
+ cleaned = cleaned.replace(/<code[^>]*>([\s\S]*?)<\/code>/gi, "`$1`");
25
+ cleaned = cleaned.replace(
26
+ /<pre[^>]*>[\s\S]*?<code[^>]*>([\s\S]*?)<\/code>[\s\S]*?<\/pre>/gi,
27
+ (_, code) => "\n```\n" + decodeEntities(code).trim() + "\n```\n"
28
+ );
29
+ cleaned = cleaned.replace(
30
+ /<li[^>]*>([\s\S]*?)<\/li>/gi,
31
+ (_, text) => "- " + stripTags(text).trim() + "\n"
32
+ );
33
+ cleaned = cleaned.replace(/<ol[^>]*>([\s\S]*?)<\/ol>/gi, (_, listContent) => {
34
+ let i = 1;
35
+ return listContent.replace(
36
+ /<li[^>]*>([\s\S]*?)<\/li>/gi,
37
+ (_2, text) => `${i++}. ` + stripTags(text).trim() + "\n"
38
+ );
39
+ });
40
+ cleaned = stripTags(cleaned);
41
+ cleaned = decodeEntities(cleaned);
42
+ cleaned = cleaned.replace(/\n{3,}/g, "\n\n").trim();
43
+ return cleaned;
44
+ }
45
+ function htmlToText(html) {
46
+ return stripTags(
47
+ html.replace(/<script[\s\S]*?<\/script>/gi, "").replace(/<style[\s\S]*?<\/style>/gi, "").replace(/<br\s*\/?>/gi, "\n")
48
+ ).replace(/&nbsp;/g, " ").replace(/\n{3,}/g, "\n\n").trim();
49
+ }
50
+ function stripTags(html) {
51
+ return html.replace(/<[^>]*>/g, "");
52
+ }
53
+ function decodeEntities(text) {
54
+ return text.replace(/&amp;/g, "&").replace(/&lt;/g, "<").replace(/&gt;/g, ">").replace(/&quot;/g, '"').replace(/&#39;/g, "'").replace(/&nbsp;/g, " ");
55
+ }
56
+
57
+ // src/utils/url-safety.ts
58
+ import { lookup } from "dns/promises";
59
+ import { isIP } from "net";
60
+ var ALLOWED_PROTOCOLS = ["http:", "https:"];
61
+ var BLOCKED_IPV4 = [
62
+ /^0\./,
63
+ // 当前网络 / 未指定
64
+ /^10\./,
65
+ // class A private
66
+ /^127\./,
67
+ // loopback
68
+ /^169\.254\./,
69
+ // link-local(含云元数据 169.254.169.254)
70
+ /^172\.(1[6-9]|2\d|3[01])\./,
71
+ // class B private
72
+ /^192\.0\.0\./,
73
+ // IETF protocol assignments
74
+ /^192\.0\.2\./,
75
+ // TEST-NET-1
76
+ /^192\.168\./,
77
+ // class C private
78
+ /^198\.1[89]\./,
79
+ // benchmark
80
+ /^198\.51\.100\./,
81
+ // TEST-NET-2
82
+ /^203\.0\.113\./,
83
+ // TEST-NET-3
84
+ /^2(2[4-9]|3\d)\./,
85
+ // 224-239 组播
86
+ /^2(4[0-9]|5[0-5])\./,
87
+ // 240-255 保留 / 广播
88
+ /^100\.(6[4-9]|[7-9]\d|1[01]\d|12[0-7])\./
89
+ // 100.64/10 CGNAT
90
+ ];
91
+ var BLOCKED_HOSTNAMES = /* @__PURE__ */ new Set([
92
+ "localhost",
93
+ "localhost.localdomain",
94
+ "ip6-localhost",
95
+ "ip6-loopback",
96
+ // 云元数据端点
97
+ "metadata.google.internal",
98
+ "metadata.goog",
99
+ "metadata",
100
+ "instance-data",
101
+ // 常见的「解析到回环」的公共域名,用来绕过主机名检查
102
+ "localtest.me",
103
+ "lvh.me",
104
+ "vcap.me"
105
+ ]);
106
+ var BLOCKED_SUFFIXES = [".localhost", ".local", ".internal", ".localtest.me", ".lvh.me"];
107
+ function isBlockedIPv6(addr) {
108
+ const a = addr.toLowerCase().replace(/^\[|\]$/g, "");
109
+ if (a === "::" || a === "::1") return true;
110
+ if (/^fe[89ab][0-9a-f]:/.test(a)) return true;
111
+ if (/^f[cd][0-9a-f]{2}:/.test(a)) return true;
112
+ const mapped = a.match(/^::ffff:(\d+\.\d+\.\d+\.\d+)$/);
113
+ if (mapped && mapped[1]) return isBlockedIPv4(mapped[1]);
114
+ return false;
115
+ }
116
+ function isBlockedIPv4(addr) {
117
+ return BLOCKED_IPV4.some((p) => p.test(addr));
118
+ }
119
+ function isBlockedAddress(addr) {
120
+ const v = isIP(addr);
121
+ if (v === 4) return isBlockedIPv4(addr);
122
+ if (v === 6) return isBlockedIPv6(addr);
123
+ return false;
124
+ }
125
+ function checkUrlSafety(url) {
126
+ let parsed;
127
+ try {
128
+ parsed = new URL(url);
129
+ } catch {
130
+ return { safe: false, reason: "\u65E0\u6548 URL" };
131
+ }
132
+ if (!ALLOWED_PROTOCOLS.includes(parsed.protocol)) {
133
+ return { safe: false, reason: `\u7981\u6B62\u534F\u8BAE: ${parsed.protocol}` };
134
+ }
135
+ if (parsed.username || parsed.password) {
136
+ return { safe: false, reason: "URL \u4E0D\u5141\u8BB8\u5305\u542B\u7528\u6237\u540D/\u5BC6\u7801" };
137
+ }
138
+ const host = parsed.hostname.toLowerCase().replace(/^\[|\]$/g, "").replace(/\.$/, "");
139
+ if (!host) return { safe: false, reason: "\u7F3A\u5C11\u4E3B\u673A\u540D" };
140
+ if (BLOCKED_HOSTNAMES.has(host)) {
141
+ return { safe: false, reason: `\u7981\u6B62\u8BBF\u95EE\u672C\u5730/\u5143\u6570\u636E\u4E3B\u673A: ${host}` };
142
+ }
143
+ if (BLOCKED_SUFFIXES.some((s) => host.endsWith(s))) {
144
+ return { safe: false, reason: `\u7981\u6B62\u8BBF\u95EE\u5185\u90E8\u57DF: ${host}` };
145
+ }
146
+ if (isIP(host) !== 0 && isBlockedAddress(host)) {
147
+ return { safe: false, reason: `\u7981\u6B62\u8BBF\u95EE\u79C1\u6709/\u4FDD\u7559\u5730\u5740: ${host}` };
148
+ }
149
+ return { safe: true };
150
+ }
151
+ async function checkUrlSafetyAsync(url) {
152
+ const sync = checkUrlSafety(url);
153
+ if (!sync.safe) return sync;
154
+ const host = new URL(url).hostname.toLowerCase().replace(/^\[|\]$/g, "");
155
+ if (isIP(host) !== 0) return sync;
156
+ try {
157
+ const records = await lookup(host, { all: true, verbatim: true });
158
+ for (const r of records) {
159
+ if (isBlockedAddress(r.address)) {
160
+ return {
161
+ safe: false,
162
+ reason: `${host} \u89E3\u6790\u5230\u79C1\u6709/\u4FDD\u7559\u5730\u5740 ${r.address}\uFF0C\u62D2\u7EDD\u8BBF\u95EE`
163
+ };
164
+ }
165
+ }
166
+ } catch {
167
+ return sync;
168
+ }
169
+ return sync;
170
+ }
171
+
172
+ // src/tools/web-fetch.ts
173
+ var MAX_SIZE = 5 * 1024 * 1024;
174
+ var MAX_REDIRECTS = 5;
175
+ var BROWSER_UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/143.0.0.0 Safari/537.36";
176
+ function acceptFor(format2) {
177
+ if (format2 === "markdown") return "text/markdown;q=1.0, text/html;q=0.9, text/plain;q=0.8";
178
+ if (format2 === "text") return "text/plain;q=1.0, text/html;q=0.9";
179
+ return "text/html;q=1.0";
180
+ }
181
+ async function fetchWithSafeRedirects(startUrl, init) {
182
+ let current = startUrl;
183
+ for (let hop = 0; hop <= MAX_REDIRECTS; hop++) {
184
+ const safety = await checkUrlSafetyAsync(current);
185
+ if (!safety.safe) {
186
+ const where = hop === 0 ? "URL \u4E0D\u5B89\u5168" : `\u91CD\u5B9A\u5411\u7B2C ${hop} \u8DF3\u88AB\u62D2`;
187
+ return { error: `${where}: ${safety.reason}` };
188
+ }
189
+ const response = await fetch(current, { ...init, redirect: "manual" });
190
+ if (response.status >= 300 && response.status < 400) {
191
+ const location = response.headers.get("location");
192
+ if (!location) return { response, finalUrl: current };
193
+ await response.arrayBuffer().catch(() => void 0);
194
+ let next;
195
+ try {
196
+ next = new URL(location, current).toString();
197
+ } catch {
198
+ return { error: `\u91CD\u5B9A\u5411\u5730\u5740\u65E0\u6CD5\u89E3\u6790: ${location}` };
199
+ }
200
+ current = next;
201
+ continue;
202
+ }
203
+ return { response, finalUrl: current };
204
+ }
205
+ return { error: `\u91CD\u5B9A\u5411\u8D85\u8FC7 ${MAX_REDIRECTS} \u8DF3\uFF0C\u5DF2\u653E\u5F03` };
206
+ }
207
+ async function execWebFetch(args, _ctx) {
208
+ const url = String(args["url"] || "").trim();
209
+ if (!url) return fail(5001, "URL \u4E0D\u80FD\u4E3A\u7A7A");
210
+ const format2 = args["format"] || "markdown";
211
+ const timeout = Math.min(args["timeout"] || 30, 120) * 1e3;
212
+ const controller = new AbortController();
213
+ const timer = setTimeout(() => controller.abort(), timeout);
214
+ try {
215
+ const headers = {
216
+ "User-Agent": BROWSER_UA,
217
+ Accept: acceptFor(format2),
218
+ "Accept-Language": "en-US,en;q=0.9"
219
+ };
220
+ const first = await fetchWithSafeRedirects(url, { signal: controller.signal, headers });
221
+ if ("error" in first) return fail(5002, first.error);
222
+ let { response } = first;
223
+ const { finalUrl } = first;
224
+ if (response.status === 403 && response.headers.get("cf-mitigated") === "challenge") {
225
+ const retry = await fetchWithSafeRedirects(finalUrl, {
226
+ signal: controller.signal,
227
+ headers: { "User-Agent": "epoch-agent/1.0", Accept: "text/html" }
228
+ });
229
+ if ("error" in retry) return fail(5002, retry.error);
230
+ response = retry.response;
231
+ }
232
+ if (!response.ok) return fail(5003, `HTTP ${response.status}: ${response.statusText}`);
233
+ const contentType = response.headers.get("content-type") || "";
234
+ const contentLen = response.headers.get("content-length");
235
+ if (contentLen && Number(contentLen) > MAX_SIZE)
236
+ return fail(5004, "\u54CD\u5E94\u8FC7\u5927 (exceeds 5MB limit)");
237
+ const body = await readCapped(response, MAX_SIZE);
238
+ if (body === null) return fail(5004, "\u54CD\u5E94\u8FC7\u5927 (exceeds 5MB limit)");
239
+ const raw = new TextDecoder().decode(body);
240
+ const isHtml = contentType.includes("text/html") || raw.includes("<html") || raw.includes("<HTML");
241
+ let output;
242
+ if (isHtml) {
243
+ output = format2 === "text" ? htmlToText(raw) : htmlToMarkdown(raw);
244
+ } else {
245
+ output = raw.slice(0, 5e4);
246
+ }
247
+ const title = extractTitle(raw, isHtml);
248
+ const prefix = finalUrl !== url ? `> \u6700\u7EC8\u5730\u5740: ${finalUrl}
249
+
250
+ ` : "";
251
+ return {
252
+ success: true,
253
+ output: prefix + (title ? `# ${title}
254
+
255
+ ${output}` : output),
256
+ duration: 0
257
+ };
258
+ } catch (err) {
259
+ if (err instanceof Error && err.name === "AbortError") {
260
+ return fail(5005, "\u8BF7\u6C42\u8D85\u65F6");
261
+ }
262
+ return fail(5006, `\u7F51\u7EDC\u9519\u8BEF: ${err instanceof Error ? err.message : String(err)}`);
263
+ } finally {
264
+ clearTimeout(timer);
265
+ }
266
+ }
267
+ async function readCapped(response, max) {
268
+ if (!response.body) {
269
+ const buf = new Uint8Array(await response.arrayBuffer());
270
+ return buf.byteLength > max ? null : buf;
271
+ }
272
+ const reader = response.body.getReader();
273
+ const chunks = [];
274
+ let total = 0;
275
+ for (; ; ) {
276
+ const { done, value } = await reader.read();
277
+ if (done) break;
278
+ if (!value) continue;
279
+ total += value.byteLength;
280
+ if (total > max) {
281
+ await reader.cancel().catch(() => void 0);
282
+ return null;
283
+ }
284
+ chunks.push(value);
285
+ }
286
+ const out = new Uint8Array(total);
287
+ let offset = 0;
288
+ for (const c of chunks) {
289
+ out.set(c, offset);
290
+ offset += c.byteLength;
291
+ }
292
+ return out;
293
+ }
294
+ function fail(code, message) {
295
+ return { success: false, output: "", error: { code, message }, duration: 0 };
296
+ }
297
+ function extractTitle(html, isHtml) {
298
+ if (!isHtml) return void 0;
299
+ const m = html.match(/<title[^>]*>([\s\S]*?)<\/title>/i);
300
+ return m && m[1] ? m[1].trim().slice(0, 200) : void 0;
301
+ }
302
+
303
+ // src/search/provider.ts
304
+ var SearchError = class extends Error {
305
+ constructor(message, status) {
306
+ super(message);
307
+ this.status = status;
308
+ this.name = "SearchError";
309
+ }
310
+ status;
311
+ };
312
+ async function searchFetch(url, init, who) {
313
+ let response;
314
+ try {
315
+ response = await fetch(url, init);
316
+ } catch (err) {
317
+ if (err instanceof Error && err.name === "AbortError") {
318
+ throw new SearchError(`${who} \u641C\u7D22\u8D85\u65F6`);
319
+ }
320
+ throw new SearchError(`${who} \u8FDE\u4E0D\u4E0A\uFF1A${err instanceof Error ? err.message : String(err)}`);
321
+ }
322
+ if (!response.ok) {
323
+ const hint = response.status === 401 || response.status === 403 ? "\uFF08API key \u65E0\u6548\u6216\u6CA1\u6709\u6743\u9650\uFF09" : response.status === 429 ? "\uFF08\u914D\u989D\u7528\u5C3D\u6216\u89E6\u53D1\u9650\u6D41\uFF09" : "";
324
+ throw new SearchError(`${who} \u8FD4\u56DE HTTP ${response.status}${hint}`, response.status);
325
+ }
326
+ try {
327
+ return await response.json();
328
+ } catch {
329
+ throw new SearchError(`${who} \u7684\u54CD\u5E94\u4E0D\u662F\u5408\u6CD5 JSON`);
330
+ }
331
+ }
332
+ function str(obj, key) {
333
+ if (typeof obj !== "object" || obj === null) return "";
334
+ const value = obj[key];
335
+ return typeof value === "string" ? value : "";
336
+ }
337
+ function toResult(raw, keys) {
338
+ const url = str(raw, keys.url);
339
+ if (!url) return null;
340
+ const publishedAt = keys.date ? str(raw, keys.date) : "";
341
+ return {
342
+ title: str(raw, keys.title) || url,
343
+ url,
344
+ snippet: str(raw, keys.snippet),
345
+ ...publishedAt ? { publishedAt } : {}
346
+ };
347
+ }
348
+
349
+ // src/tools/web-search.ts
350
+ var DEFAULT_MAX_RESULTS = 5;
351
+ var HARD_MAX_RESULTS = 20;
352
+ var TIMEOUT_MS = 2e4;
353
+ var MAX_SNIPPET = 500;
354
+ var cache = /* @__PURE__ */ new Map();
355
+ function clearSearchCache() {
356
+ cache.clear();
357
+ }
358
+ function cacheKey(sessionId, query, opts) {
359
+ return `${sessionId}\0${query.trim().toLowerCase()}\0${JSON.stringify(opts)}`;
360
+ }
361
+ function format(results, ignored) {
362
+ const lines = results.map((r, i) => {
363
+ const date = r.publishedAt ? ` \xB7 ${r.publishedAt}` : "";
364
+ const snippet = r.snippet.length > MAX_SNIPPET ? `${r.snippet.slice(0, MAX_SNIPPET)}\u2026` : r.snippet;
365
+ return `${i + 1}. ${r.title}${date}
366
+ ${r.url}
367
+ ${snippet}`;
368
+ });
369
+ const head = `${results.length} \u6761\u7ED3\u679C\uFF08\u53EA\u6709\u6458\u8981\uFF0C\u8981\u6B63\u6587\u8BF7\u5BF9\u5177\u4F53 URL \u8C03 web_fetch\uFF09`;
370
+ const tail = ignored.length > 0 ? [`
371
+ \u6CE8\u610F\uFF1A${ignored.join("\uFF1B")}`] : [];
372
+ return [head, "", ...lines, ...tail].join("\n");
373
+ }
374
+ function stringArray(value) {
375
+ if (!Array.isArray(value)) return void 0;
376
+ const out = value.filter((v) => typeof v === "string" && v.trim().length > 0);
377
+ return out.length > 0 ? out : void 0;
378
+ }
379
+ function createWebSearchExec(deps) {
380
+ return async function execWebSearch(args, ctx) {
381
+ const started = Date.now();
382
+ const query = String(args["query"] ?? "").trim();
383
+ if (!query) {
384
+ return {
385
+ success: false,
386
+ output: "",
387
+ error: { code: 5101, message: "\u67E5\u8BE2\u8BCD\u4E0D\u80FD\u4E3A\u7A7A" },
388
+ duration: 0
389
+ };
390
+ }
391
+ const requested = Number(args["maxResults"]);
392
+ const maxResults = Math.min(
393
+ Number.isFinite(requested) && requested > 0 ? Math.floor(requested) : deps.defaultMaxResults ?? DEFAULT_MAX_RESULTS,
394
+ HARD_MAX_RESULTS
395
+ );
396
+ const recency = Number(args["recencyDays"]);
397
+ const options = {
398
+ maxResults,
399
+ ...Number.isFinite(recency) && recency > 0 ? { recencyDays: Math.floor(recency) } : {},
400
+ ...stringArray(args["domains"]) ? { domains: stringArray(args["domains"]) } : {},
401
+ ...stringArray(args["excludeDomains"]) ? { excludeDomains: stringArray(args["excludeDomains"]) } : {}
402
+ };
403
+ const key = cacheKey(ctx.sessionId, query, options);
404
+ const cached = cache.get(key);
405
+ if (cached) {
406
+ return {
407
+ success: true,
408
+ output: format(cached, ["\u672C\u6B21\u7ED3\u679C\u6765\u81EA\u672C\u4F1A\u8BDD\u7F13\u5B58\uFF0C\u672A\u91CD\u65B0\u8054\u7F51"]),
409
+ duration: Date.now() - started
410
+ };
411
+ }
412
+ const controller = new AbortController();
413
+ const timer = setTimeout(() => controller.abort(), TIMEOUT_MS);
414
+ try {
415
+ const outcome = await deps.provider.search(query, { ...options, signal: controller.signal });
416
+ cache.set(key, outcome.results);
417
+ if (outcome.results.length === 0) {
418
+ return {
419
+ success: true,
420
+ output: `\u6CA1\u6709\u641C\u5230\u7ED3\u679C\uFF08${deps.provider.name}\uFF09\u3002\u6362\u4E2A\u8BF4\u6CD5\u6216\u653E\u5BBD\u6761\u4EF6\u518D\u8BD5\u3002`,
421
+ duration: Date.now() - started
422
+ };
423
+ }
424
+ return {
425
+ success: true,
426
+ output: format(outcome.results, outcome.ignored ?? []),
427
+ duration: Date.now() - started
428
+ };
429
+ } catch (err) {
430
+ const message = err instanceof SearchError ? err.message : `\u641C\u7D22\u5931\u8D25\uFF1A${err instanceof Error ? err.message : String(err)}`;
431
+ return {
432
+ success: false,
433
+ output: "",
434
+ error: {
435
+ code: 5102,
436
+ message,
437
+ suggestion: "\u8FD9\u4E0D\u662F\u81F4\u547D\u9519\u8BEF\uFF0C\u53EF\u4EE5\u7EE7\u7EED\u522B\u7684\u5DE5\u4F5C\uFF0C\u6216\u7A0D\u540E\u91CD\u8BD5"
438
+ },
439
+ duration: Date.now() - started
440
+ };
441
+ } finally {
442
+ clearTimeout(timer);
443
+ }
444
+ };
445
+ }
446
+
447
+ // src/search/brave.ts
448
+ var ENDPOINT = "https://api.search.brave.com/res/v1/web/search";
449
+ var MAX_RESULTS = 20;
450
+ function createBraveProvider(opts = {}) {
451
+ const key = () => (opts.apiKey ?? process.env["BRAVE_SEARCH_API_KEY"] ?? "").trim();
452
+ return {
453
+ name: "brave",
454
+ unavailableReason() {
455
+ return key() ? null : "\u7F3A\u5C11 BRAVE_SEARCH_API_KEY\uFF08\u5728 https://brave.com/search/api \u7533\u8BF7\u540E\u5199\u8FDB ~/.epoch/.env\uFF09";
456
+ },
457
+ async search(query, options) {
458
+ const url = new URL(ENDPOINT);
459
+ url.searchParams.set("q", buildQuery(query, options));
460
+ url.searchParams.set("count", String(Math.min(options.maxResults, MAX_RESULTS)));
461
+ const freshness = toFreshness(options.recencyDays);
462
+ if (freshness) url.searchParams.set("freshness", freshness);
463
+ const json = await searchFetch(
464
+ url.toString(),
465
+ {
466
+ headers: {
467
+ Accept: "application/json",
468
+ "Accept-Encoding": "gzip",
469
+ "X-Subscription-Token": key()
470
+ },
471
+ ...options.signal ? { signal: options.signal } : {}
472
+ },
473
+ "brave"
474
+ );
475
+ const web = json.web;
476
+ const raw = Array.isArray(web?.results) ? web.results : [];
477
+ const results = [];
478
+ for (const item of raw) {
479
+ const hit = toResult(item, {
480
+ title: "title",
481
+ url: "url",
482
+ snippet: "description",
483
+ date: "age"
484
+ });
485
+ if (hit) results.push(hit);
486
+ if (results.length >= options.maxResults) break;
487
+ }
488
+ const ignored = [];
489
+ if (options.recencyDays !== void 0 && !freshness) {
490
+ ignored.push("recencyDays\uFF08brave \u7684 freshness \u53EA\u5230\u300C\u4E00\u5E74\u5185\u300D\uFF0C\u66F4\u5927\u7684\u5929\u6570\u5DF2\u5FFD\u7565\uFF09");
491
+ }
492
+ return { results, ...ignored.length > 0 ? { ignored } : {} };
493
+ }
494
+ };
495
+ }
496
+ function buildQuery(query, opts) {
497
+ const parts = [query];
498
+ for (const d of opts.domains ?? []) parts.push(`site:${d}`);
499
+ for (const d of opts.excludeDomains ?? []) parts.push(`-site:${d}`);
500
+ return parts.join(" ");
501
+ }
502
+ function toFreshness(days) {
503
+ if (days === void 0 || days <= 0) return void 0;
504
+ if (days <= 1) return "pd";
505
+ if (days <= 7) return "pw";
506
+ if (days <= 31) return "pm";
507
+ if (days <= 366) return "py";
508
+ return void 0;
509
+ }
510
+
511
+ // src/search/searxng.ts
512
+ function createSearxngProvider(opts) {
513
+ return {
514
+ name: "searxng",
515
+ unavailableReason() {
516
+ if (!opts.baseUrl?.trim()) {
517
+ return "searxng \u9700\u8981\u81EA\u5EFA\u5B9E\u4F8B\u5730\u5740\uFF08epoch config set search.baseUrl https://\u4F60\u7684\u5B9E\u4F8B\uFF09";
518
+ }
519
+ try {
520
+ const url = new URL(opts.baseUrl);
521
+ if (url.protocol !== "http:" && url.protocol !== "https:") {
522
+ return `searxng \u7684 baseUrl \u5FC5\u987B\u662F http(s)\uFF0C\u73B0\u5728\u662F ${url.protocol}`;
523
+ }
524
+ } catch {
525
+ return `searxng \u7684 baseUrl \u4E0D\u662F\u5408\u6CD5 URL\uFF1A${opts.baseUrl}`;
526
+ }
527
+ return null;
528
+ },
529
+ async search(query, options) {
530
+ const base = opts.baseUrl.replace(/\/+$/, "");
531
+ const url = new URL(`${base}/search`);
532
+ url.searchParams.set("q", buildQuery2(query, options));
533
+ url.searchParams.set("format", "json");
534
+ const range = timeRange(options.recencyDays);
535
+ if (range) url.searchParams.set("time_range", range);
536
+ const body = await searchFetch(
537
+ url.toString(),
538
+ {
539
+ headers: { Accept: "application/json" },
540
+ ...options.signal ? { signal: options.signal } : {}
541
+ },
542
+ "searxng"
543
+ ).catch((err) => {
544
+ if (err instanceof SearchError && err.status === 403) {
545
+ throw new SearchError(
546
+ "searxng \u8FD4\u56DE 403 \u2014\u2014 \u591A\u534A\u662F\u5B9E\u4F8B\u6CA1\u5F00 JSON \u8F93\u51FA\uFF08settings.yml \u91CC search.formats \u8981\u542B json\uFF09",
547
+ 403
548
+ );
549
+ }
550
+ throw err;
551
+ });
552
+ const raw = Array.isArray(body.results) ? body.results : [];
553
+ const results = [];
554
+ for (const item of raw) {
555
+ const hit = toResult(item, {
556
+ title: "title",
557
+ url: "url",
558
+ snippet: "content",
559
+ date: "publishedDate"
560
+ });
561
+ if (hit) results.push(hit);
562
+ if (results.length >= options.maxResults) break;
563
+ }
564
+ const ignored = [];
565
+ if (options.recencyDays !== void 0 && !range) {
566
+ ignored.push("recencyDays\uFF08searxng \u53EA\u652F\u6301 day/week/month/year \u56DB\u6863\uFF0C\u5929\u6570\u592A\u5927\u5DF2\u5FFD\u7565\uFF09");
567
+ }
568
+ return { results, ...ignored.length > 0 ? { ignored } : {} };
569
+ }
570
+ };
571
+ }
572
+ function buildQuery2(query, opts) {
573
+ const parts = [query];
574
+ for (const d of opts.domains ?? []) parts.push(`site:${d}`);
575
+ for (const d of opts.excludeDomains ?? []) parts.push(`-site:${d}`);
576
+ return parts.join(" ");
577
+ }
578
+ function timeRange(days) {
579
+ if (days === void 0 || days <= 0) return void 0;
580
+ if (days <= 1) return "day";
581
+ if (days <= 7) return "week";
582
+ if (days <= 31) return "month";
583
+ if (days <= 366) return "year";
584
+ return void 0;
585
+ }
586
+
587
+ // src/search/tavily.ts
588
+ var ENDPOINT2 = "https://api.tavily.com/search";
589
+ var MAX_RESULTS2 = 20;
590
+ function createTavilyProvider(opts = {}) {
591
+ const key = () => (opts.apiKey ?? process.env["TAVILY_API_KEY"] ?? "").trim();
592
+ return {
593
+ name: "tavily",
594
+ unavailableReason() {
595
+ return key() ? null : "\u7F3A\u5C11 TAVILY_API_KEY\uFF08\u5728 https://tavily.com \u7533\u8BF7\u540E\u5199\u8FDB ~/.epoch/.env\uFF09";
596
+ },
597
+ async search(query, options) {
598
+ const body = {
599
+ query,
600
+ max_results: Math.min(options.maxResults, MAX_RESULTS2),
601
+ // `basic` 而不是 `advanced`:后者贵一倍,而我们本来就不要正文 ——
602
+ // 模型看完 snippet 自己去 web_fetch
603
+ search_depth: "basic"
604
+ };
605
+ if (options.recencyDays !== void 0 && options.recencyDays > 0) {
606
+ body["days"] = Math.ceil(options.recencyDays);
607
+ }
608
+ if (options.domains?.length) body["include_domains"] = options.domains;
609
+ if (options.excludeDomains?.length) body["exclude_domains"] = options.excludeDomains;
610
+ const json = await searchFetch(
611
+ ENDPOINT2,
612
+ {
613
+ method: "POST",
614
+ headers: {
615
+ "Content-Type": "application/json",
616
+ // Bearer 而不是 body 里的 api_key 字段:后者是旧写法,
617
+ // 且会让 key 出现在任何记录了请求体的地方
618
+ Authorization: `Bearer ${key()}`
619
+ },
620
+ body: JSON.stringify(body),
621
+ ...options.signal ? { signal: options.signal } : {}
622
+ },
623
+ "tavily"
624
+ );
625
+ const raw = Array.isArray(json.results) ? json.results : [];
626
+ const results = [];
627
+ for (const item of raw) {
628
+ const hit = toResult(item, {
629
+ title: "title",
630
+ url: "url",
631
+ snippet: "content",
632
+ date: "published_date"
633
+ });
634
+ if (hit) results.push(hit);
635
+ if (results.length >= options.maxResults) break;
636
+ }
637
+ return { results };
638
+ }
639
+ };
640
+ }
641
+
642
+ // src/search/registry.ts
643
+ var AUTO_ORDER = ["tavily", "brave", "searxng"];
644
+ function createSearchProvider(type, config) {
645
+ if (type === "searxng") {
646
+ return createSearxngProvider(config?.baseUrl ? { baseUrl: config.baseUrl } : {});
647
+ }
648
+ return type === "brave" ? createBraveProvider() : createTavilyProvider();
649
+ }
650
+ function selectSearchProvider(config) {
651
+ const explicit = config?.provider;
652
+ if (explicit) {
653
+ const provider = createSearchProvider(explicit, config);
654
+ const reason = provider.unavailableReason();
655
+ return reason ? { provider: null, detail: `\u914D\u7F6E\u6307\u5B9A\u4E86 ${explicit}\uFF0C\u4F46${reason}` } : { provider, detail: `\u4F7F\u7528 ${explicit}` };
656
+ }
657
+ const tried = [];
658
+ for (const type of AUTO_ORDER) {
659
+ const provider = createSearchProvider(type, config);
660
+ const reason = provider.unavailableReason();
661
+ if (!reason) return { provider, detail: `\u81EA\u52A8\u9009\u7528 ${type}\uFF08\u6309\u51ED\u636E\u63A2\u6D4B\uFF09` };
662
+ tried.push(reason);
663
+ }
664
+ return {
665
+ provider: null,
666
+ // 把三家各自缺什么都列出来:用户多半只想配一家,让他自己挑最方便的那个
667
+ detail: `\u672A\u914D\u7F6E\u641C\u7D22\u540E\u7AEF\uFF0Cweb_search \u672A\u6CE8\u518C\u3002\u53EF\u9009\uFF1A${tried.join("\uFF1B")}`
668
+ };
669
+ }
670
+
671
+ // src/index.ts
672
+ var webPlugin = {
673
+ name: "@epoch-agent/plugin-web",
674
+ version: "0.0.0",
675
+ description: "Web \u6293\u53D6\u63D2\u4EF6 \u2014 URL \u5B89\u5168\u6821\u9A8C + HTML\u2192Markdown + SSRF \u9632\u62A4",
676
+ tools: [
677
+ {
678
+ name: "web_fetch",
679
+ description: "\u6293\u53D6\u7F51\u9875\u5185\u5BB9\u3002\u53C2\u6570: url(\u5FC5\u586B), format(markdown|text|html, \u9ED8\u8BA4 markdown), timeout(\u79D2, \u9ED8\u8BA4 30, \u6700\u5927 120)\u3002\u81EA\u52A8\u5C06 HTML \u8F6C\u4E3A Markdown\uFF0C\u8FC7\u6EE4 script/style\uFF0CSSRF \u5B89\u5168\u6821\u9A8C\u3002",
680
+ parameters: {
681
+ type: "object",
682
+ properties: {
683
+ url: { type: "string", description: "\u7F51\u9875 URL\uFF08\u5FC5\u987B\u4EE5 http:// \u6216 https:// \u5F00\u5934\uFF09" },
684
+ format: { type: "string", enum: ["markdown", "text", "html"], description: "\u8FD4\u56DE\u683C\u5F0F" },
685
+ timeout: { type: "number", description: "\u8D85\u65F6\u79D2\u6570\uFF08\u9ED8\u8BA4 30\uFF0C\u6700\u5927 120\uFF09" }
686
+ },
687
+ required: ["url"]
688
+ },
689
+ // `openWorldHint: true` 不只是元数据:core 的 tool-executor 用它决定
690
+ // 要不要把输出包进 `<tool_output untrusted="true">`。抓来的网页内容
691
+ // 作者既不是用户也不是我们,必须标成「数据」而不是「指令」回灌上下文。
692
+ annotations: { readOnlyHint: true, idempotentHint: false, openWorldHint: true },
693
+ operation: "network",
694
+ // 审批目标是 URL —— 五候选里没有 `url`,不给 describeTarget 就只能退回
695
+ // 参数 JSON,「批准抓这个站」下次照样弹窗
696
+ describeTarget: (args) => String(args["url"] ?? ""),
697
+ execute: execWebFetch
698
+ }
699
+ ]
700
+ };
701
+ function createWebSearchTool(provider, opts = {}) {
702
+ return {
703
+ name: "web_search",
704
+ description: `\u7528 ${provider.name} \u641C\u7D22\u7F51\u9875\u3002\u53C2\u6570: query(\u5FC5\u586B), maxResults(\u9ED8\u8BA4 5, \u6700\u5927 20), recencyDays(\u53EA\u8981\u6700\u8FD1 N \u5929), domains(\u53EA\u641C\u8FD9\u4E9B\u57DF\u540D), excludeDomains(\u6392\u9664\u8FD9\u4E9B\u57DF\u540D)\u3002**\u53EA\u8FD4\u56DE\u6807\u9898 / URL / \u6458\u8981\uFF0C\u4E0D\u8FD4\u56DE\u7F51\u9875\u6B63\u6587** \u2014\u2014 \u770B\u5B8C\u6458\u8981\u5224\u65AD\u54EA\u6761\u503C\u5F97\u8BFB\uFF0C\u518D\u5BF9\u90A3\u4E2A URL \u8C03 web_fetch \u53D6\u6B63\u6587\u3002`,
705
+ parameters: {
706
+ type: "object",
707
+ properties: {
708
+ query: { type: "string", description: "\u641C\u7D22\u8BCD" },
709
+ maxResults: { type: "number", description: "\u8FD4\u56DE\u51E0\u6761\uFF08\u9ED8\u8BA4 5\uFF0C\u6700\u5927 20\uFF09" },
710
+ recencyDays: { type: "number", description: "\u53EA\u8981\u6700\u8FD1 N \u5929\u5185\u53D1\u5E03\u7684" },
711
+ domains: {
712
+ type: "array",
713
+ items: { type: "string" },
714
+ description: '\u53EA\u5728\u8FD9\u4E9B\u57DF\u540D\u91CC\u641C\uFF0C\u4F8B\u5982 ["nodejs.org"]'
715
+ },
716
+ excludeDomains: { type: "array", items: { type: "string" }, description: "\u6392\u9664\u8FD9\u4E9B\u57DF\u540D" }
717
+ },
718
+ required: ["query"]
719
+ },
720
+ // `openWorldHint: true` 是这里最重要的一行:core 的 tool-executor 用它决定
721
+ // 要不要把输出包进 `<tool_output untrusted="true">`。搜索结果是攻击者
722
+ // **可以主动投放**的内容(SEO 投毒 + 提示注入),是不可信内容里风险最高的一种。
723
+ // `idempotentHint: false` —— 同一个词今天明天搜出来的东西不一样
724
+ annotations: { readOnlyHint: true, idempotentHint: false, openWorldHint: true },
725
+ operation: "network",
726
+ // 审批目标是查询词而不是参数 JSON:`terminal` 那边批准的是命令,
727
+ // 这边批准的就该是「搜这个词」
728
+ describeTarget: (args) => String(args["query"] ?? ""),
729
+ execute: createWebSearchExec({ provider, ...opts })
730
+ };
731
+ }
732
+ export {
733
+ SearchError,
734
+ clearSearchCache,
735
+ createBraveProvider,
736
+ createSearchProvider,
737
+ createSearxngProvider,
738
+ createTavilyProvider,
739
+ createWebSearchExec,
740
+ createWebSearchTool,
741
+ selectSearchProvider,
742
+ webPlugin
743
+ };
package/package.json ADDED
@@ -0,0 +1,42 @@
1
+ {
2
+ "name": "@epoch-agent/plugin-web",
3
+ "version": "0.1.0",
4
+ "private": false,
5
+ "description": "Web 抓取插件 — URL 安全校验 + HTML→Markdown + SSRF 防护",
6
+ "repository": {
7
+ "type": "git",
8
+ "url": "https://github.com/Ddbor/epoch-agent.git"
9
+ },
10
+ "license": "Apache-2.0",
11
+ "author": "epoch-agent",
12
+ "type": "module",
13
+ "exports": {
14
+ ".": {
15
+ "types": "./dist/index.d.ts",
16
+ "import": "./dist/index.js"
17
+ }
18
+ },
19
+ "main": "./dist/index.js",
20
+ "types": "./dist/index.d.ts",
21
+ "files": [
22
+ "dist"
23
+ ],
24
+ "dependencies": {
25
+ "@epoch-agent/protocol": "0.1.0"
26
+ },
27
+ "devDependencies": {
28
+ "@types/node": "^22.0.0",
29
+ "tsup": "^8.0.0",
30
+ "typescript": "^5.8.3",
31
+ "vitest": "^3.0.0"
32
+ },
33
+ "engines": {
34
+ "node": ">=22.0.0"
35
+ },
36
+ "scripts": {
37
+ "build": "tsup",
38
+ "dev": "tsup --watch",
39
+ "test": "vitest run",
40
+ "typecheck": "tsc --noEmit"
41
+ }
42
+ }