@miphamai/cli 0.24.4 → 0.24.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
|
@@ -286,9 +286,9 @@ providers:
|
|
|
286
286
|
|
|
287
287
|
### 5.1 — Built-in Skills
|
|
288
288
|
|
|
289
|
-
Mipham Code ships with
|
|
289
|
+
Mipham Code ships with 17 built-in skills loaded automatically:
|
|
290
290
|
|
|
291
|
-
- **Standard (
|
|
291
|
+
- **Standard (14)**: code-review, compassionate-communication, doc-generator, github-ops, memory, mipham-code-setup, security-review, self-review, superpower, systematic-debugging, tdd, test-driven-development, web-access, web-search
|
|
292
292
|
- **Mipham (3)**: om-artifact, om-model-optimize, om-security
|
|
293
293
|
|
|
294
294
|
### 5.2 — Community Skills
|
|
@@ -1,54 +1,213 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: web-access
|
|
3
|
-
description:
|
|
4
|
-
version:
|
|
3
|
+
description: All network operations — web search, page fetching, authenticated browsing, social media scraping, dynamic page rendering. Routes to the correct tool (WebSearch/WebFetch/ComputerUse browser) based on the task.
|
|
4
|
+
version: 2.0.0
|
|
5
|
+
user-invocable: true
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- Bash
|
|
8
|
+
- WebFetch
|
|
9
|
+
- WebSearch
|
|
10
|
+
- ComputerUse
|
|
11
|
+
- Read
|
|
5
12
|
---
|
|
6
13
|
|
|
7
|
-
# Web Access
|
|
14
|
+
# Web Access — Executable Workflow
|
|
8
15
|
|
|
9
|
-
|
|
16
|
+
**Type**: Flexible — use the decision tree to route to the right tool, then adapt to the specific site.
|
|
10
17
|
|
|
11
|
-
|
|
18
|
+
**Purpose**: All network-bound operations go through this skill. It routes the request to the correct underlying tool and handles authentication, rendering, and extraction strategy.
|
|
12
19
|
|
|
13
|
-
|
|
14
|
-
- **Page scraping**: Extracting content from web pages
|
|
15
|
-
- **Authenticated access**: Sites requiring login (via browser automation)
|
|
16
|
-
- **Social media**: Content from Xiaohongshu, Weibo, Twitter, etc.
|
|
17
|
-
- **Dynamic content**: JavaScript-rendered pages requiring a real browser
|
|
18
|
-
- **API interaction**: REST/GraphQL endpoints with proper auth
|
|
20
|
+
**Triggers**: "search for", "look up", "find information about", "fetch this URL", "scrape", "browser", "login to", "check this website", "web", "online"
|
|
19
21
|
|
|
20
|
-
|
|
22
|
+
---
|
|
23
|
+
|
|
24
|
+
## Phase 0: Route to the Right Tool (ALWAYS RUN FIRST)
|
|
25
|
+
|
|
26
|
+
```
|
|
27
|
+
User request involves network?
|
|
28
|
+
├── Search engine query (find/discover/look up)?
|
|
29
|
+
│ └── → WebSearch tool
|
|
30
|
+
│ Query best practices: specific + versioned + technical terms
|
|
31
|
+
│
|
|
32
|
+
├── Read a known URL (docs/article/API)?
|
|
33
|
+
│ └── → WebFetch tool
|
|
34
|
+
│ HTTP auto-upgrades to HTTPS, HTML converts to markdown
|
|
35
|
+
│ Cached for 15 minutes — re-fetch only if stale
|
|
36
|
+
│
|
|
37
|
+
├── Login-required site? JavaScript SPA? Form submission?
|
|
38
|
+
│ └── → ComputerUse browser automation
|
|
39
|
+
│ browser_navigate → browser_snapshot → browser_click
|
|
40
|
+
│
|
|
41
|
+
├── Social media (Xiaohongshu, Weibo, Twitter, etc.)?
|
|
42
|
+
│ └── → ComputerUse browser (render JS, handle auth)
|
|
43
|
+
│ OR → WebFetch if public page
|
|
44
|
+
│
|
|
45
|
+
└── API endpoint (REST/GraphQL)?
|
|
46
|
+
└── → WebFetch with prompt for structured extraction
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
---
|
|
50
|
+
|
|
51
|
+
## Phase 1: Web Search
|
|
52
|
+
|
|
53
|
+
Use `WebSearch` for discovery queries — finding documentation, news, troubleshooting, comparisons.
|
|
54
|
+
|
|
55
|
+
### Query Construction
|
|
56
|
+
|
|
57
|
+
```
|
|
58
|
+
❌ "React" → too broad
|
|
59
|
+
❌ "React problems" → ambiguous
|
|
60
|
+
✅ "React 19 useEffect double mount fix 2026" → specific + versioned
|
|
61
|
+
✅ "Next.js 14 App Router caching behavior" → targeted
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### Domain Filtering
|
|
65
|
+
|
|
66
|
+
Use `allowed_domains` for authoritative sources:
|
|
67
|
+
|
|
68
|
+
- `docs.github.com` — GitHub docs
|
|
69
|
+
- `nextjs.org` — Next.js official
|
|
70
|
+
- `developer.mozilla.org` — MDN
|
|
71
|
+
- `nodejs.org` — Node.js official
|
|
72
|
+
|
|
73
|
+
Use `blocked_domains` to exclude noise (e.g., exclude `w3schools.com` when looking for MDN).
|
|
74
|
+
|
|
75
|
+
### Verification
|
|
76
|
+
|
|
77
|
+
- Cross-reference claims across 2+ independent sources
|
|
78
|
+
- Prefer results from current year
|
|
79
|
+
- Authority: official docs > well-known blogs > Stack Overflow > random forums
|
|
21
80
|
|
|
22
|
-
###
|
|
81
|
+
### Source Attribution
|
|
23
82
|
|
|
24
|
-
|
|
83
|
+
Always end responses with:
|
|
25
84
|
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
- Technical troubleshooting
|
|
29
|
-
- Technology comparisons
|
|
85
|
+
```markdown
|
|
86
|
+
Sources:
|
|
30
87
|
|
|
31
|
-
|
|
88
|
+
- [Title](URL) — brief note
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
---
|
|
92
|
+
|
|
93
|
+
## Phase 2: Web Fetch
|
|
94
|
+
|
|
95
|
+
Use `WebFetch` for reading a specific URL.
|
|
96
|
+
|
|
97
|
+
### What it does
|
|
98
|
+
|
|
99
|
+
- Auto-upgrades HTTP → HTTPS
|
|
100
|
+
- Converts HTML to Markdown (headings, links, images, lists, code blocks)
|
|
101
|
+
- Strips scripts, styles, nav, header, footer before conversion
|
|
102
|
+
- Caches results for 15 minutes per URL
|
|
103
|
+
- Detects cross-host redirects and reports them
|
|
104
|
+
- Truncates content at 100K characters
|
|
105
|
+
|
|
106
|
+
### Prompt Parameter
|
|
107
|
+
|
|
108
|
+
Use `prompt` to guide extraction focus:
|
|
109
|
+
|
|
110
|
+
```
|
|
111
|
+
WebFetch: url="https://docs.example.com", prompt="find the authentication API section"
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
### When NOT to use WebFetch
|
|
115
|
+
|
|
116
|
+
- Search queries → use WebSearch
|
|
117
|
+
- Login-required pages → use ComputerUse browser
|
|
118
|
+
- Large file downloads → use Bash + curl/wget
|
|
119
|
+
- API endpoints returning JSON → WebFetch works, returns raw JSON
|
|
120
|
+
|
|
121
|
+
---
|
|
32
122
|
|
|
33
|
-
|
|
123
|
+
## Phase 3: Browser Automation (ComputerUse)
|
|
34
124
|
|
|
35
|
-
|
|
36
|
-
- Checking API responses
|
|
37
|
-
- Extracting article content
|
|
38
|
-
- Verifying links
|
|
125
|
+
Use `ComputerUse` for interactive browsing — login, form submission, JavaScript rendering.
|
|
39
126
|
|
|
40
|
-
###
|
|
127
|
+
### Available Actions
|
|
41
128
|
|
|
42
|
-
|
|
129
|
+
| Action | Purpose |
|
|
130
|
+
| ------------------ | ------------------------------------------- |
|
|
131
|
+
| `browser_navigate` | Go to a URL |
|
|
132
|
+
| `browser_snapshot` | Capture accessibility tree (page structure) |
|
|
133
|
+
| `browser_click` | Click an element by UID |
|
|
134
|
+
| `screenshot` | Capture visible viewport |
|
|
135
|
+
| `launch` | Open a desktop application |
|
|
43
136
|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
137
|
+
### Workflow for Authenticated Sites
|
|
138
|
+
|
|
139
|
+
```
|
|
140
|
+
1. browser_navigate → login page
|
|
141
|
+
2. browser_snapshot → find form fields (UIDs)
|
|
142
|
+
3. Ask user for credentials (NEVER auto-fill)
|
|
143
|
+
4. browser_click → submit
|
|
144
|
+
5. browser_navigate → target page
|
|
145
|
+
6. browser_snapshot → extract content
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
### Workflow for SPAs (React/Vue/Angular)
|
|
149
|
+
|
|
150
|
+
```
|
|
151
|
+
1. browser_navigate → SPA URL
|
|
152
|
+
2. Wait 2-3 seconds (JavaScript render)
|
|
153
|
+
3. browser_snapshot → extract rendered content
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
### Prerequisites
|
|
157
|
+
|
|
158
|
+
- Playwright must be installed: `npm install playwright`
|
|
159
|
+
- First launch opens a visible browser window (headless: false)
|
|
160
|
+
|
|
161
|
+
---
|
|
162
|
+
|
|
163
|
+
## Phase 4: Extraction & Synthesis
|
|
164
|
+
|
|
165
|
+
After fetching content (via any method):
|
|
166
|
+
|
|
167
|
+
### Content Extraction
|
|
168
|
+
|
|
169
|
+
1. Identify relevant sections using the prompt/h3 headings
|
|
170
|
+
2. Extract key facts, code examples, API signatures
|
|
171
|
+
3. Note the source URL for attribution
|
|
172
|
+
|
|
173
|
+
### Cross-Referencing
|
|
174
|
+
|
|
175
|
+
1. Verify technical claims across 2+ sources
|
|
176
|
+
2. Flag contradictions between sources
|
|
177
|
+
3. Note version/deprecation warnings
|
|
178
|
+
|
|
179
|
+
### Output Format
|
|
180
|
+
|
|
181
|
+
```markdown
|
|
182
|
+
## [Topic]
|
|
183
|
+
|
|
184
|
+
[Key finding with source attribution]
|
|
185
|
+
|
|
186
|
+
### Details
|
|
187
|
+
|
|
188
|
+
[Structured content from page]
|
|
189
|
+
|
|
190
|
+
Sources:
|
|
191
|
+
|
|
192
|
+
- [Title](URL)
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
---
|
|
47
196
|
|
|
48
197
|
## Security Rules
|
|
49
198
|
|
|
50
|
-
- Never submit credentials without explicit user approval
|
|
51
|
-
- Respect robots.txt and rate limiting
|
|
199
|
+
- **Never submit credentials** without explicit user approval
|
|
200
|
+
- Respect `robots.txt` and rate limiting
|
|
52
201
|
- Do not scrape PII or sensitive data
|
|
53
|
-
-
|
|
54
|
-
- Only HTTPS for remote requests
|
|
202
|
+
- All URLs validated against SSRF before fetching
|
|
203
|
+
- Only HTTPS for remote requests (HTTP auto-upgraded)
|
|
204
|
+
- Cross-host redirects reported to caller (not silently followed)
|
|
205
|
+
|
|
206
|
+
---
|
|
207
|
+
|
|
208
|
+
## When NOT to Use This Skill
|
|
209
|
+
|
|
210
|
+
- Pure logic / algorithmic questions (reasoning, not research)
|
|
211
|
+
- Questions answerable from code already in context
|
|
212
|
+
- Opinions / subjective recommendations (search for data, not consensus)
|
|
213
|
+
- Downloading large binaries → use Bash + curl
|
|
@@ -1,75 +1,176 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: web-search
|
|
3
|
-
description: Search the web for current information — documentation, news, technical references, and
|
|
4
|
-
version:
|
|
3
|
+
description: Search the web for current information — documentation, news, technical references, troubleshooting, and research. Routes queries through Brave Search API with domain filtering and source verification.
|
|
4
|
+
version: 3.0.0
|
|
5
|
+
user-invocable: true
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- WebSearch
|
|
8
|
+
- WebFetch
|
|
5
9
|
---
|
|
6
10
|
|
|
7
|
-
# Web Search
|
|
11
|
+
# Web Search — Executable Workflow
|
|
8
12
|
|
|
9
|
-
|
|
13
|
+
**Type**: Flexible — follow the query construction rules strictly, then adapt verification depth to the task.
|
|
10
14
|
|
|
11
|
-
|
|
15
|
+
**Purpose**: Find accurate, current information from the web. This skill covers query formulation, domain filtering, result verification, and when to follow up with WebFetch for deep reading.
|
|
12
16
|
|
|
13
|
-
|
|
14
|
-
- **Current events**: News, releases, incidents (anything after training cutoff)
|
|
15
|
-
- **Troubleshooting**: Error messages, stack traces, known issues
|
|
16
|
-
- **Comparisons**: Technology trade-offs, benchmark data
|
|
17
|
-
- **Code examples**: Real-world usage patterns, configuration snippets
|
|
17
|
+
**Triggers**: "search for", "look up", "find", "what is", "how to", "latest", "current", "news about", "documentation for", "research"
|
|
18
18
|
|
|
19
|
-
|
|
19
|
+
---
|
|
20
20
|
|
|
21
|
-
|
|
21
|
+
## Phase 0: Decide Whether to Search (ALWAYS RUN FIRST)
|
|
22
22
|
|
|
23
23
|
```
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
24
|
+
Question involves...
|
|
25
|
+
├── Current events, news, recent releases?
|
|
26
|
+
│ └── YES → Search (model training cutoff limitation)
|
|
27
|
+
│
|
|
28
|
+
├── Library/framework documentation?
|
|
29
|
+
│ └── YES → Search (version-specific, up-to-date)
|
|
30
|
+
│
|
|
31
|
+
├── Error messages, stack traces?
|
|
32
|
+
│ └── YES → Search (known issues, fixes)
|
|
33
|
+
│
|
|
34
|
+
├── Technology comparisons, benchmarks?
|
|
35
|
+
│ └── YES → Search (current data)
|
|
36
|
+
│
|
|
37
|
+
├── Pure logic, algorithms, math?
|
|
38
|
+
│ └── NO → Reason directly (no external data needed)
|
|
39
|
+
│
|
|
40
|
+
├── Question answerable from code in context?
|
|
41
|
+
│ └── NO → Use existing context (faster, no network)
|
|
42
|
+
│
|
|
43
|
+
└── Opinion / subjective?
|
|
44
|
+
└── MAYBE → Search for data points, not consensus
|
|
28
45
|
```
|
|
29
46
|
|
|
30
|
-
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## Phase 1: Construct the Query
|
|
50
|
+
|
|
51
|
+
### Rules (apply in order)
|
|
52
|
+
|
|
53
|
+
1. **Be specific**: include version numbers, dates, proper nouns
|
|
54
|
+
2. **Use technical terms**: framework/language jargon over natural language
|
|
55
|
+
3. **Include context**: OS, environment, constraints if relevant
|
|
56
|
+
4. **English preferred**: technical content is richer in English
|
|
57
|
+
|
|
58
|
+
### Examples
|
|
31
59
|
|
|
32
60
|
```
|
|
33
|
-
❌ "
|
|
61
|
+
❌ "React" → too broad
|
|
62
|
+
❌ "React problems" → ambiguous
|
|
63
|
+
❌ "how to make website fast" → natural language
|
|
64
|
+
✅ "React 19 useEffect double mount fix" → specific + versioned
|
|
34
65
|
✅ "Core Web Vitals LCP optimization Next.js 14"
|
|
66
|
+
✅ "Prisma 5 findMany nested include filter TypeScript"
|
|
67
|
+
✅ "playwright click button not working 2026"
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
### For Chinese-Language Queries
|
|
71
|
+
|
|
72
|
+
Chinese queries work but yield fewer technical results:
|
|
73
|
+
|
|
74
|
+
```
|
|
75
|
+
✅ "React 19 useEffect 执行两次 修复" → mixed language for best results
|
|
76
|
+
✅ "Vue 3 Composition API 最佳实践 2026"
|
|
35
77
|
```
|
|
36
78
|
|
|
37
|
-
|
|
79
|
+
---
|
|
80
|
+
|
|
81
|
+
## Phase 2: Filter & Verify Results
|
|
82
|
+
|
|
83
|
+
### Domain Authority Tiers
|
|
38
84
|
|
|
85
|
+
| Tier | Domains | Weight |
|
|
86
|
+
| ----------------- | --------------------------------------------------------------------- | ------- |
|
|
87
|
+
| **Official** | docs.github.com, nextjs.org, nodejs.org, python.org, rust-lang.org | Highest |
|
|
88
|
+
| **Authoritative** | developer.mozilla.org, web.dev, kubernetes.io | High |
|
|
89
|
+
| **Trusted** | stackoverflow.com (high-score), dev.to, medium.com (verified authors) | Medium |
|
|
90
|
+
| **Low** | personal blogs, random forums, w3schools | Low |
|
|
91
|
+
|
|
92
|
+
### Use allowed_domains for targeted searches
|
|
93
|
+
|
|
94
|
+
```json
|
|
95
|
+
{ "query": "Next.js caching", "allowed_domains": ["nextjs.org", "github.com"] }
|
|
39
96
|
```
|
|
40
|
-
|
|
41
|
-
|
|
97
|
+
|
|
98
|
+
### Use blocked_domains to exclude noise
|
|
99
|
+
|
|
100
|
+
```json
|
|
101
|
+
{ "query": "JavaScript array methods", "blocked_domains": ["w3schools.com"] }
|
|
42
102
|
```
|
|
43
103
|
|
|
44
|
-
###
|
|
104
|
+
### Cross-Reference Rule
|
|
45
105
|
|
|
46
|
-
|
|
106
|
+
- **Critical claims** (API behavior, security): 2+ independent sources
|
|
107
|
+
- **Code examples**: test before recommending
|
|
108
|
+
- **Version info**: check publish date (prefer current year)
|
|
47
109
|
|
|
48
|
-
|
|
49
|
-
- `nextjs.org` — Next.js official docs
|
|
50
|
-
- `developer.mozilla.org` — MDN Web Docs
|
|
51
|
-
- `nodejs.org` — Node.js official
|
|
110
|
+
---
|
|
52
111
|
|
|
53
|
-
##
|
|
112
|
+
## Phase 3: Deep Read (When Needed)
|
|
54
113
|
|
|
55
|
-
|
|
56
|
-
- **Recency**: Prefer results from the current year; note article dates
|
|
57
|
-
- **Authority**: Official docs > well-known blogs > Stack Overflow > random forums
|
|
58
|
-
- **Cite sources**: Always include source URLs in responses
|
|
114
|
+
After search returns results, decide whether to deep-read:
|
|
59
115
|
|
|
60
|
-
|
|
116
|
+
```
|
|
117
|
+
Search result looks promising?
|
|
118
|
+
├── Snippet answers the question fully?
|
|
119
|
+
│ └── → Use snippet + cite source (done)
|
|
120
|
+
│
|
|
121
|
+
├── Need code examples / detailed API docs?
|
|
122
|
+
│ └── → WebFetch the page URL
|
|
123
|
+
│ Use prompt to focus extraction
|
|
124
|
+
│
|
|
125
|
+
├── Multiple sources needed for verification?
|
|
126
|
+
│ └── → WebFetch top 2-3 results
|
|
127
|
+
│ Cross-reference and flag contradictions
|
|
128
|
+
│
|
|
129
|
+
└── Page is JavaScript SPA / login-walled?
|
|
130
|
+
└── → Delegate to web-access skill (ComputerUse browser)
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
---
|
|
61
134
|
|
|
62
|
-
|
|
135
|
+
## Phase 4: Report Results
|
|
136
|
+
|
|
137
|
+
### Format
|
|
63
138
|
|
|
64
139
|
```markdown
|
|
140
|
+
## [Topic]
|
|
141
|
+
|
|
142
|
+
[Answer with inline citations]
|
|
143
|
+
|
|
144
|
+
### Details (if deep-read was done)
|
|
145
|
+
|
|
146
|
+
[Structured content from fetched pages]
|
|
147
|
+
|
|
65
148
|
Sources:
|
|
66
149
|
|
|
67
|
-
- [Title](URL) —
|
|
68
|
-
- [Title](URL) —
|
|
150
|
+
- [Title](URL) — [1-sentence note on what was found there]
|
|
151
|
+
- [Title](URL) — [1-sentence note]
|
|
69
152
|
```
|
|
70
153
|
|
|
71
|
-
|
|
154
|
+
### Attribution Rules
|
|
155
|
+
|
|
156
|
+
- Always include source URLs
|
|
157
|
+
- Note if a source is official docs vs community
|
|
158
|
+
- Flag outdated content (e.g., "article from 2024, may be stale")
|
|
159
|
+
- Distinguish between facts (need citation) and reasoning (your own)
|
|
160
|
+
|
|
161
|
+
---
|
|
162
|
+
|
|
163
|
+
## Search API Configuration
|
|
164
|
+
|
|
165
|
+
Web search uses **Brave Search API** (free tier: 2,000 queries/month).
|
|
166
|
+
|
|
167
|
+
If search returns "not configured":
|
|
168
|
+
|
|
169
|
+
1. Get a free API key at https://brave.com/search/api/
|
|
170
|
+
2. Set: `export BRAVE_API_KEY="BSA..."`
|
|
171
|
+
3. Restart Mipham Code
|
|
172
|
+
|
|
173
|
+
Alternatives (additional API keys supported):
|
|
72
174
|
|
|
73
|
-
-
|
|
74
|
-
-
|
|
75
|
-
- Opinions and subjective recommendations (search for data, not consensus)
|
|
175
|
+
- `TAVILY_API_KEY` — https://tavily.com
|
|
176
|
+
- `SERPAPI_API_KEY` — https://serpapi.com
|
|
@@ -1,21 +1,180 @@
|
|
|
1
1
|
import type { ToolDefinition } from '../../shared/index.ts'
|
|
2
2
|
import { validateUrl } from '../../security/url'
|
|
3
3
|
|
|
4
|
+
// ── In-memory cache (15-min TTL per URL) ──
|
|
5
|
+
|
|
6
|
+
interface CacheEntry {
|
|
7
|
+
content: string
|
|
8
|
+
timestamp: number
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
const CACHE_TTL = 15 * 60 * 1000 // 15 minutes
|
|
12
|
+
const cache = new Map<string, CacheEntry>()
|
|
13
|
+
|
|
14
|
+
function getCached(url: string): string | undefined {
|
|
15
|
+
const entry = cache.get(url)
|
|
16
|
+
if (!entry) return undefined
|
|
17
|
+
if (Date.now() - entry.timestamp > CACHE_TTL) {
|
|
18
|
+
cache.delete(url)
|
|
19
|
+
return undefined
|
|
20
|
+
}
|
|
21
|
+
return entry.content
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function setCache(url: string, content: string): void {
|
|
25
|
+
// Evict oldest entries if cache grows too large (max 200 URLs)
|
|
26
|
+
if (cache.size >= 200) {
|
|
27
|
+
const oldest = [...cache.entries()].sort((a, b) => a[1].timestamp - b[1].timestamp)
|
|
28
|
+
for (let i = 0; i < 20 && oldest[i]; i++) {
|
|
29
|
+
cache.delete(oldest[i]![0])
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
cache.set(url, { content, timestamp: Date.now() })
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// ── HTTP→HTTPS upgrade ──
|
|
36
|
+
|
|
37
|
+
function upgradeToHttps(url: string): string {
|
|
38
|
+
if (url.startsWith('http://')) {
|
|
39
|
+
return url.replace('http://', 'https://')
|
|
40
|
+
}
|
|
41
|
+
return url
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
// ── HTML → Markdown conversion ──
|
|
45
|
+
|
|
46
|
+
function htmlToMarkdown(html: string, baseUrl: string): string {
|
|
47
|
+
let text = html
|
|
48
|
+
|
|
49
|
+
// Remove scripts, styles, nav, header, footer
|
|
50
|
+
text = text.replace(/<script[^>]*>[\s\S]*?<\/script>/gi, '')
|
|
51
|
+
text = text.replace(/<style[^>]*>[\s\S]*?<\/style>/gi, '')
|
|
52
|
+
text = text.replace(/<nav[^>]*>[\s\S]*?<\/nav>/gi, '')
|
|
53
|
+
text = text.replace(/<header[^>]*>[\s\S]*?<\/header>/gi, '')
|
|
54
|
+
text = text.replace(/<footer[^>]*>[\s\S]*?<\/footer>/gi, '')
|
|
55
|
+
|
|
56
|
+
// Convert headings
|
|
57
|
+
text = text.replace(/<h1[^>]*>([\s\S]*?)<\/h1>/gi, (_, c) => `\n# ${stripTags(c).trim()}\n`)
|
|
58
|
+
text = text.replace(/<h2[^>]*>([\s\S]*?)<\/h2>/gi, (_, c) => `\n## ${stripTags(c).trim()}\n`)
|
|
59
|
+
text = text.replace(/<h3[^>]*>([\s\S]*?)<\/h3>/gi, (_, c) => `\n### ${stripTags(c).trim()}\n`)
|
|
60
|
+
text = text.replace(/<h4[^>]*>([\s\S]*?)<\/h4>/gi, (_, c) => `\n#### ${stripTags(c).trim()}\n`)
|
|
61
|
+
text = text.replace(/<h5[^>]*>([\s\S]*?)<\/h5>/gi, (_, c) => `\n##### ${stripTags(c).trim()}\n`)
|
|
62
|
+
text = text.replace(/<h6[^>]*>([\s\S]*?)<\/h6>/gi, (_, c) => `\n###### ${stripTags(c).trim()}\n`)
|
|
63
|
+
|
|
64
|
+
// Convert links: <a href="...">text</a> → [text](url)
|
|
65
|
+
text = text.replace(/<a[^>]*href=["']([^"']*)["'][^>]*>([\s\S]*?)<\/a>/gi, (_, href, content) => {
|
|
66
|
+
const resolved = resolveUrl(href, baseUrl)
|
|
67
|
+
return `[${stripTags(content).trim()}](${resolved})`
|
|
68
|
+
})
|
|
69
|
+
|
|
70
|
+
// Convert images: <img ... src="..." ...> → 
|
|
71
|
+
text = text.replace(
|
|
72
|
+
/<img[^>]*src=["']([^"']*)["'][^>]*alt=["']([^"']*)["'][^>]*\/?>/gi,
|
|
73
|
+
(_, src, alt) => {
|
|
74
|
+
const resolved = resolveUrl(src, baseUrl)
|
|
75
|
+
return ``
|
|
76
|
+
},
|
|
77
|
+
)
|
|
78
|
+
text = text.replace(/<img[^>]*src=["']([^"']*)["'][^>]*\/?>/gi, (_, src) => {
|
|
79
|
+
const resolved = resolveUrl(src, baseUrl)
|
|
80
|
+
return ``
|
|
81
|
+
})
|
|
82
|
+
|
|
83
|
+
// Convert lists
|
|
84
|
+
text = text.replace(/<li[^>]*>([\s\S]*?)<\/li>/gi, (_, c) => `- ${stripTags(c).trim()}\n`)
|
|
85
|
+
text = text.replace(/<\/ul>/gi, '\n')
|
|
86
|
+
text = text.replace(/<\/ol>/gi, '\n')
|
|
87
|
+
|
|
88
|
+
// Convert code blocks
|
|
89
|
+
text = text.replace(/<pre[^>]*><code[^>]*>([\s\S]*?)<\/code><\/pre>/gi, (_, c) => {
|
|
90
|
+
const decoded = c
|
|
91
|
+
.replace(/</g, '<')
|
|
92
|
+
.replace(/>/g, '>')
|
|
93
|
+
.replace(/&/g, '&')
|
|
94
|
+
.replace(/"/g, '"')
|
|
95
|
+
return `\n\`\`\`\n${decoded.trim()}\n\`\`\`\n`
|
|
96
|
+
})
|
|
97
|
+
text = text.replace(/<code[^>]*>([\s\S]*?)<\/code>/gi, (_, c) => `\`${c.trim()}\``)
|
|
98
|
+
|
|
99
|
+
// Convert inline formatting
|
|
100
|
+
text = text.replace(/<strong[^>]*>([\s\S]*?)<\/strong>/gi, '**$1**')
|
|
101
|
+
text = text.replace(/<b[^>]*>([\s\S]*?)<\/b>/gi, '**$1**')
|
|
102
|
+
text = text.replace(/<em[^>]*>([\s\S]*?)<\/em>/gi, '*$1*')
|
|
103
|
+
text = text.replace(/<i[^>]*>([\s\S]*?)<\/i>/gi, '*$1*')
|
|
104
|
+
|
|
105
|
+
// Convert paragraph and line break tags
|
|
106
|
+
text = text.replace(/<br\s*\/?>/gi, '\n')
|
|
107
|
+
text = text.replace(/<\/p>/gi, '\n\n')
|
|
108
|
+
text = text.replace(/<p[^>]*>/gi, '')
|
|
109
|
+
|
|
110
|
+
// Strip remaining HTML tags
|
|
111
|
+
text = text.replace(/<[^>]*>/g, '')
|
|
112
|
+
|
|
113
|
+
// Decode HTML entities
|
|
114
|
+
text = text.replace(/</g, '<')
|
|
115
|
+
text = text.replace(/>/g, '>')
|
|
116
|
+
text = text.replace(/&/g, '&')
|
|
117
|
+
text = text.replace(/"/g, '"')
|
|
118
|
+
text = text.replace(/'/g, "'")
|
|
119
|
+
text = text.replace(/'/g, "'")
|
|
120
|
+
text = text.replace(/ /g, ' ')
|
|
121
|
+
|
|
122
|
+
// Collapse whitespace (preserve intentional line breaks)
|
|
123
|
+
text = text
|
|
124
|
+
.split('\n')
|
|
125
|
+
.map((l) => l.replace(/\s+/g, ' ').trim())
|
|
126
|
+
.join('\n')
|
|
127
|
+
text = text.replace(/\n{3,}/g, '\n\n')
|
|
128
|
+
|
|
129
|
+
return text.trim()
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
function stripTags(html: string): string {
|
|
133
|
+
return html.replace(/<[^>]*>/g, '')
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
function resolveUrl(href: string, baseUrl: string): string {
|
|
137
|
+
try {
|
|
138
|
+
return new URL(href, baseUrl).toString()
|
|
139
|
+
} catch {
|
|
140
|
+
return href
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// ── Tool Definition ──
|
|
145
|
+
|
|
4
146
|
export const webFetchTool: ToolDefinition = {
|
|
5
147
|
name: 'WebFetch',
|
|
6
|
-
description:
|
|
148
|
+
description:
|
|
149
|
+
'Fetches a URL, converts the page to markdown. HTTP is upgraded to HTTPS. Cross-host redirects are returned to the caller. Responses are cached for 15 minutes per URL.',
|
|
7
150
|
category: 'network',
|
|
8
151
|
permission: 'auto',
|
|
9
152
|
parameters: {
|
|
10
153
|
type: 'object',
|
|
11
154
|
properties: {
|
|
12
155
|
url: { type: 'string', format: 'uri', description: 'URL to fetch' },
|
|
13
|
-
prompt: {
|
|
156
|
+
prompt: {
|
|
157
|
+
type: 'string',
|
|
158
|
+
description:
|
|
159
|
+
'What to extract from the page (e.g., "find the API docs for authentication"). The tool returns the full page; the prompt helps focus extraction.',
|
|
160
|
+
},
|
|
14
161
|
},
|
|
15
162
|
required: ['url'],
|
|
16
163
|
},
|
|
17
164
|
async execute(params, _ctx) {
|
|
18
|
-
const
|
|
165
|
+
const rawUrl = params.url as string
|
|
166
|
+
const prompt = (params.prompt as string) || ''
|
|
167
|
+
const url = upgradeToHttps(rawUrl)
|
|
168
|
+
|
|
169
|
+
// Check cache
|
|
170
|
+
const cached = getCached(url)
|
|
171
|
+
if (cached) {
|
|
172
|
+
return {
|
|
173
|
+
success: true,
|
|
174
|
+
content: cached,
|
|
175
|
+
metadata: { cached: true, url },
|
|
176
|
+
}
|
|
177
|
+
}
|
|
19
178
|
|
|
20
179
|
// SSRF protection: validate URL before fetching
|
|
21
180
|
const validationError = validateUrl(url)
|
|
@@ -28,25 +187,29 @@ export const webFetchTool: ToolDefinition = {
|
|
|
28
187
|
const timer = setTimeout(() => controller.abort(), 30_000) // 30s timeout
|
|
29
188
|
|
|
30
189
|
const response = await fetch(url, {
|
|
31
|
-
headers: {
|
|
190
|
+
headers: {
|
|
191
|
+
'User-Agent': 'Mipham-Code/0.24.0',
|
|
192
|
+
Accept: 'text/html,application/xhtml+xml,*/*',
|
|
193
|
+
},
|
|
32
194
|
redirect: 'follow',
|
|
33
195
|
signal: controller.signal,
|
|
34
196
|
})
|
|
35
197
|
|
|
36
198
|
clearTimeout(timer)
|
|
37
199
|
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
200
|
+
// Cross-host redirect reporting
|
|
201
|
+
if (response.url) {
|
|
202
|
+
const finalHost = new URL(response.url).hostname
|
|
203
|
+
const originalHost = new URL(url).hostname
|
|
204
|
+
if (finalHost !== originalHost) {
|
|
205
|
+
return {
|
|
206
|
+
success: true,
|
|
207
|
+
content: `Redirected to: ${response.url}\n\nFetch from this URL directly to retrieve content.`,
|
|
208
|
+
metadata: { redirected: true, originalUrl: url, finalUrl: response.url },
|
|
209
|
+
}
|
|
43
210
|
}
|
|
44
|
-
}
|
|
45
211
|
|
|
46
|
-
|
|
47
|
-
// `redirect: 'follow'` may have landed on an internal IP — check again.
|
|
48
|
-
// Guard: response.url may not exist in some environments (e.g. mocked fetch).
|
|
49
|
-
if (response.url) {
|
|
212
|
+
// SSRF defense: re-validate the resolved URL after redirects
|
|
50
213
|
const redirectError = validateUrl(response.url)
|
|
51
214
|
if (redirectError) {
|
|
52
215
|
return {
|
|
@@ -57,14 +220,43 @@ export const webFetchTool: ToolDefinition = {
|
|
|
57
220
|
}
|
|
58
221
|
}
|
|
59
222
|
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
const
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
223
|
+
// Determine content type; only convert HTML to markdown
|
|
224
|
+
const contentType = response.headers.get('content-type') || ''
|
|
225
|
+
const isHtml = contentType.includes('text/html') || contentType.includes('application/xhtml')
|
|
226
|
+
|
|
227
|
+
if (!response.ok) {
|
|
228
|
+
return {
|
|
229
|
+
success: false,
|
|
230
|
+
content: '',
|
|
231
|
+
error: `HTTP ${response.status}: ${response.statusText}`,
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
let content: string
|
|
236
|
+
|
|
237
|
+
if (isHtml) {
|
|
238
|
+
const html = await response.text()
|
|
239
|
+
const baseUrl = response.url || url
|
|
240
|
+
content = htmlToMarkdown(html, baseUrl)
|
|
241
|
+
} else {
|
|
242
|
+
// Plain text / JSON / etc. — return as-is
|
|
243
|
+
content = await response.text()
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
// Truncate to 100K characters
|
|
247
|
+
if (content.length > 100_000) {
|
|
248
|
+
content = content.slice(0, 100_000) + '\n\n... (truncated)'
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
// Cache the result
|
|
252
|
+
setCache(url, content)
|
|
253
|
+
|
|
254
|
+
// Include prompt context if provided
|
|
255
|
+
const header = prompt
|
|
256
|
+
? `── WebFetch: ${url} ──\nPrompt: ${prompt}\n\n`
|
|
257
|
+
: `── WebFetch: ${url} ──\n\n`
|
|
258
|
+
|
|
259
|
+
return { success: true, content: header + content, metadata: { url, size: content.length } }
|
|
68
260
|
} catch (err) {
|
|
69
261
|
const message =
|
|
70
262
|
err instanceof Error && err.name === 'AbortError'
|
|
@@ -1,8 +1,85 @@
|
|
|
1
1
|
import type { ToolDefinition } from '../../shared/index.ts'
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* Brave Search API integration.
|
|
5
|
+
*
|
|
6
|
+
* API: https://api.search.brave.com/res/v1/web/search
|
|
7
|
+
* Free tier: 2000 queries/month (no credit card required)
|
|
8
|
+
* Sign up: https://brave.com/search/api/
|
|
9
|
+
*
|
|
10
|
+
* Set BRAVE_API_KEY in your environment to enable web search.
|
|
11
|
+
* Falls back to a helpful message when the key is not configured.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
const BRAVE_API = 'https://api.search.brave.com/res/v1/web/search'
|
|
15
|
+
const MAX_RESULTS = 10
|
|
16
|
+
|
|
17
|
+
interface BraveWebResult {
|
|
18
|
+
title: string
|
|
19
|
+
url: string
|
|
20
|
+
description: string
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
interface BraveAPIResponse {
|
|
24
|
+
web?: {
|
|
25
|
+
results?: BraveWebResult[]
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
function getApiKey(): string | undefined {
|
|
30
|
+
return process.env.BRAVE_API_KEY || undefined
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function filterByDomains(
|
|
34
|
+
results: BraveWebResult[],
|
|
35
|
+
allowed?: string[],
|
|
36
|
+
blocked?: string[],
|
|
37
|
+
): BraveWebResult[] {
|
|
38
|
+
let filtered = results
|
|
39
|
+
|
|
40
|
+
if (allowed && allowed.length > 0) {
|
|
41
|
+
filtered = filtered.filter((r) => {
|
|
42
|
+
try {
|
|
43
|
+
const host = new URL(r.url).hostname
|
|
44
|
+
return allowed.some((d) => host === d || host.endsWith('.' + d))
|
|
45
|
+
} catch {
|
|
46
|
+
return false
|
|
47
|
+
}
|
|
48
|
+
})
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
if (blocked && blocked.length > 0) {
|
|
52
|
+
filtered = filtered.filter((r) => {
|
|
53
|
+
try {
|
|
54
|
+
const host = new URL(r.url).hostname
|
|
55
|
+
return !blocked.some((d) => host === d || host.endsWith('.' + d))
|
|
56
|
+
} catch {
|
|
57
|
+
return true
|
|
58
|
+
}
|
|
59
|
+
})
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
return filtered
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function formatResults(results: BraveWebResult[]): string {
|
|
66
|
+
if (results.length === 0) {
|
|
67
|
+
return 'No results found.'
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
return results
|
|
71
|
+
.map((r, i) => {
|
|
72
|
+
const title = r.title.replace(/\n/g, ' ').trim()
|
|
73
|
+
const desc = (r.description || '').replace(/\n/g, ' ').trim()
|
|
74
|
+
return `${i + 1}. **${title}**\n ${r.url}\n ${desc}`
|
|
75
|
+
})
|
|
76
|
+
.join('\n\n')
|
|
77
|
+
}
|
|
78
|
+
|
|
3
79
|
export const webSearchTool: ToolDefinition = {
|
|
4
80
|
name: 'WebSearch',
|
|
5
|
-
description:
|
|
81
|
+
description:
|
|
82
|
+
'Search the web via Brave Search API. Returns result blocks with titles, URLs, and descriptions. Set BRAVE_API_KEY to enable.',
|
|
6
83
|
category: 'network',
|
|
7
84
|
permission: 'auto',
|
|
8
85
|
parameters: {
|
|
@@ -24,9 +101,106 @@ export const webSearchTool: ToolDefinition = {
|
|
|
24
101
|
},
|
|
25
102
|
async execute(params, _ctx) {
|
|
26
103
|
const query = params.query as string
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
104
|
+
const allowedDomains = params.allowed_domains as string[] | undefined
|
|
105
|
+
const blockedDomains = params.blocked_domains as string[] | undefined
|
|
106
|
+
|
|
107
|
+
const apiKey = getApiKey()
|
|
108
|
+
if (!apiKey) {
|
|
109
|
+
return {
|
|
110
|
+
success: true,
|
|
111
|
+
content: [
|
|
112
|
+
'── Web Search (not configured) ──',
|
|
113
|
+
'',
|
|
114
|
+
`Query: "${query}"`,
|
|
115
|
+
'',
|
|
116
|
+
'Web search is not yet configured. To enable it:',
|
|
117
|
+
'',
|
|
118
|
+
'1. Get a free API key at https://brave.com/search/api/',
|
|
119
|
+
' (2,000 queries/month, no credit card required)',
|
|
120
|
+
'',
|
|
121
|
+
'2. Set the key in your environment:',
|
|
122
|
+
' export BRAVE_API_KEY="BSA..."',
|
|
123
|
+
'',
|
|
124
|
+
'3. Restart Mipham Code and try again.',
|
|
125
|
+
'',
|
|
126
|
+
'Alternatives (add API key for any):',
|
|
127
|
+
' TAVILY_API_KEY — https://tavily.com',
|
|
128
|
+
' SERPAPI_API_KEY — https://serpapi.com',
|
|
129
|
+
].join('\n'),
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
try {
|
|
134
|
+
const url = new URL(BRAVE_API)
|
|
135
|
+
url.searchParams.set('q', query)
|
|
136
|
+
url.searchParams.set('count', String(MAX_RESULTS))
|
|
137
|
+
|
|
138
|
+
const controller = new AbortController()
|
|
139
|
+
const timer = setTimeout(() => controller.abort(), 15_000) // 15s timeout
|
|
140
|
+
|
|
141
|
+
const response = await fetch(url.toString(), {
|
|
142
|
+
headers: {
|
|
143
|
+
Accept: 'application/json',
|
|
144
|
+
'Accept-Encoding': 'gzip',
|
|
145
|
+
'X-Subscription-Token': apiKey,
|
|
146
|
+
},
|
|
147
|
+
signal: controller.signal,
|
|
148
|
+
})
|
|
149
|
+
|
|
150
|
+
clearTimeout(timer)
|
|
151
|
+
|
|
152
|
+
if (!response.ok) {
|
|
153
|
+
const body = await response.text().catch(() => '')
|
|
154
|
+
// 429 = rate limit, 401 = bad key
|
|
155
|
+
if (response.status === 429) {
|
|
156
|
+
return {
|
|
157
|
+
success: false,
|
|
158
|
+
content: '',
|
|
159
|
+
error:
|
|
160
|
+
'Brave Search API rate limit reached (2,000/month on free tier). Try again later.',
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
if (response.status === 401) {
|
|
164
|
+
return {
|
|
165
|
+
success: false,
|
|
166
|
+
content: '',
|
|
167
|
+
error: 'Invalid BRAVE_API_KEY. Check your key at https://brave.com/search/api/.',
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
return {
|
|
171
|
+
success: false,
|
|
172
|
+
content: '',
|
|
173
|
+
error: `Brave Search API error (${response.status}): ${body.slice(0, 200)}`,
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
const data = (await response.json()) as BraveAPIResponse
|
|
178
|
+
const rawResults = data.web?.results || []
|
|
179
|
+
|
|
180
|
+
const filtered = filterByDomains(rawResults, allowedDomains, blockedDomains)
|
|
181
|
+
const formatted = formatResults(filtered)
|
|
182
|
+
|
|
183
|
+
return {
|
|
184
|
+
success: true,
|
|
185
|
+
content: [
|
|
186
|
+
`── Web Search: "${query}" ──`,
|
|
187
|
+
`${filtered.length} result(s)`,
|
|
188
|
+
'',
|
|
189
|
+
formatted,
|
|
190
|
+
'',
|
|
191
|
+
rawResults.length > filtered.length
|
|
192
|
+
? `(${rawResults.length - filtered.length} result(s) filtered by domain rules)`
|
|
193
|
+
: '',
|
|
194
|
+
]
|
|
195
|
+
.filter(Boolean)
|
|
196
|
+
.join('\n'),
|
|
197
|
+
}
|
|
198
|
+
} catch (err) {
|
|
199
|
+
const message =
|
|
200
|
+
err instanceof Error && err.name === 'AbortError'
|
|
201
|
+
? 'Search timed out (15s). Try a more specific query.'
|
|
202
|
+
: `Search failed: ${String(err)}`
|
|
203
|
+
return { success: false, content: '', error: message }
|
|
30
204
|
}
|
|
31
205
|
},
|
|
32
206
|
}
|