crawlforge-mcp-server 4.9.0 → 5.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +6 -5
- package/README.md +19 -3
- package/package.json +10 -12
- package/server.js +315 -214
- package/src/core/ActionExecutor.js +117 -33
- package/src/core/AgentOrchestrator.js +8 -2
- package/src/core/AuthManager.js +51 -17
- package/src/core/ChangeTracker.js +26 -10
- package/src/core/JobManager.js +9 -1
- package/src/core/LocalizationManager.js +19 -6
- package/src/core/ResearchOrchestrator.js +173 -35
- package/src/core/SnapshotManager.js +162 -165
- package/src/core/StealthBrowserManager.js +25 -3
- package/src/core/WebhookDispatcher.js +19 -14
- package/src/core/analysis/ContentAnalyzer.js +52 -7
- package/src/core/crawlers/BFSCrawler.js +27 -3
- package/src/core/processing/BrowserProcessor.js +19 -1
- package/src/core/processing/PDFProcessor.js +129 -65
- package/src/core/queue/QueueManager.js +3 -2
- package/src/schemas/toolOutputSchemas.js +269 -0
- package/src/server/auth/oauth.js +37 -7
- package/src/server/specHygiene.js +192 -0
- package/src/server/taskSupport.js +233 -0
- package/src/server/toolFilter.js +98 -0
- package/src/server/transports/streamableHttp.js +148 -11
- package/src/server/withAuth.js +11 -4
- package/src/skills/agent-skills/crawlforge-getting-started/SKILL.md +15 -0
- package/src/tools/advanced/ScrapeWithActionsTool.js +43 -52
- package/src/tools/advanced/batchScrape/index.js +128 -27
- package/src/tools/advanced/batchScrape/worker.js +55 -5
- package/src/tools/advanced/scrapeWithActions/recorder.js +3 -0
- package/src/tools/basic/_fetch.js +125 -70
- package/src/tools/basic/extractLinks.js +14 -12
- package/src/tools/basic/scrapeStructured.js +21 -4
- package/src/tools/crawl/crawlDeep.js +110 -48
- package/src/tools/crawl/mapSite.js +25 -6
- package/src/tools/extract/_fetchAndParse.js +98 -1
- package/src/tools/extract/extractContent.js +7 -4
- package/src/tools/extract/extractStructured.js +125 -84
- package/src/tools/extract/extractWithLlm.js +10 -2
- package/src/tools/extract/processDocument.js +54 -6
- package/src/tools/extract/summarizeContent.js +7 -1
- package/src/tools/llmstxt/generateLLMsTxt.js +8 -6
- package/src/tools/research/deepResearch.js +51 -31
- package/src/tools/scrape/_brandingExtractor.js +49 -11
- package/src/tools/scrape/unifiedScrape.js +27 -17
- package/src/tools/search/providers/searxng.js +5 -1
- package/src/tools/search/ranking/ResultDeduplicator.js +9 -1
- package/src/tools/search/ranking/ResultRanker.js +17 -2
- package/src/tools/search/searchWeb.js +31 -14
- package/src/tools/search/serpRank.js +23 -0
- package/src/tools/templates/TemplateRegistry.js +7 -1
- package/src/tools/tracking/trackChanges/index.js +87 -26
- package/src/tools/tracking/trackChanges/schema.js +2 -2
- package/src/utils/CircuitBreaker.js +11 -9
- package/src/utils/contentUtils.js +66 -53
- package/src/utils/secretMask.js +1 -1
- package/src/utils/sitemapParser.js +11 -9
- package/src/utils/ssrfGuard.js +212 -40
- package/src/utils/urlNormalizer.js +2 -2
|
@@ -12,10 +12,90 @@
|
|
|
12
12
|
|
|
13
13
|
import { load } from 'cheerio';
|
|
14
14
|
import { safeFetch } from '../../utils/ssrfGuard.js';
|
|
15
|
+
import { config } from '../../constants/config.js';
|
|
15
16
|
|
|
16
17
|
const DEFAULT_USER_AGENT = 'Mozilla/5.0 (compatible; CrawlForge-MCP/3.0)';
|
|
17
18
|
const DEFAULT_TIMEOUT_MS = 15000;
|
|
18
19
|
|
|
20
|
+
/**
|
|
21
|
+
* Read a response body as text while enforcing config.fetch.maxBodySize —
|
|
22
|
+
* the same cap _fetch.js applies to basic tools, so a large/hostile response
|
|
23
|
+
* (multi-hundred-MB file, endpoint streaming zeros) can't be buffered whole
|
|
24
|
+
* into a JS string and handed to cheerio/JSDOM/Turndown downstream.
|
|
25
|
+
* Content-Length is checked up front; actual bytes read are counted as a
|
|
26
|
+
* backstop for servers that omit or lie about it.
|
|
27
|
+
* @param {Response} response
|
|
28
|
+
* @returns {Promise<string>}
|
|
29
|
+
*/
|
|
30
|
+
async function readTextWithSizeCap(response) {
|
|
31
|
+
const maxBodySize = config.fetch.maxBodySize;
|
|
32
|
+
|
|
33
|
+
const contentLengthHeader = response.headers?.get?.('content-length') ?? null;
|
|
34
|
+
if (contentLengthHeader !== null) {
|
|
35
|
+
const declared = parseInt(contentLengthHeader, 10);
|
|
36
|
+
if (!isNaN(declared) && declared > maxBodySize) {
|
|
37
|
+
throw new Error(
|
|
38
|
+
`Response body too large: Content-Length ${declared} exceeds limit of ${maxBodySize} bytes`
|
|
39
|
+
);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// Responses without a ReadableStream body (test mocks, already-buffered
|
|
44
|
+
// responses) fall back to the native .text() with no additional guard.
|
|
45
|
+
if (!response.body || typeof response.body.getReader !== 'function') {
|
|
46
|
+
return response.text();
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const reader = response.body.getReader();
|
|
50
|
+
const chunks = [];
|
|
51
|
+
let totalBytes = 0;
|
|
52
|
+
|
|
53
|
+
while (true) {
|
|
54
|
+
const { done, value } = await reader.read();
|
|
55
|
+
if (done) break;
|
|
56
|
+
totalBytes += value.byteLength;
|
|
57
|
+
if (totalBytes > maxBodySize) {
|
|
58
|
+
reader.cancel();
|
|
59
|
+
throw new Error(
|
|
60
|
+
`Response body too large: exceeded limit of ${maxBodySize} bytes`
|
|
61
|
+
);
|
|
62
|
+
}
|
|
63
|
+
chunks.push(value);
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const mergedBytes = new Uint8Array(totalBytes);
|
|
67
|
+
let offset = 0;
|
|
68
|
+
for (const chunk of chunks) {
|
|
69
|
+
mergedBytes.set(chunk, offset);
|
|
70
|
+
offset += chunk.byteLength;
|
|
71
|
+
}
|
|
72
|
+
return new TextDecoder().decode(mergedBytes);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Classify a Content-Type header for the purposes of HTML parsing.
|
|
77
|
+
* Missing header is treated as 'html' (permissive default — many servers,
|
|
78
|
+
* and most test doubles, omit it for what is genuinely HTML).
|
|
79
|
+
* @param {string|null} contentType
|
|
80
|
+
* @returns {'html'|'text'|'binary'}
|
|
81
|
+
*/
|
|
82
|
+
function classifyContentType(contentType) {
|
|
83
|
+
if (!contentType) return 'html';
|
|
84
|
+
const type = contentType.split(';')[0].trim().toLowerCase();
|
|
85
|
+
if (
|
|
86
|
+
type === 'text/html' ||
|
|
87
|
+
type === 'application/xhtml+xml' ||
|
|
88
|
+
type === 'application/xml' ||
|
|
89
|
+
type === 'text/xml' ||
|
|
90
|
+
type.endsWith('+xml') ||
|
|
91
|
+
type.startsWith('text/')
|
|
92
|
+
) {
|
|
93
|
+
return type === 'text/plain' ? 'text' : 'html';
|
|
94
|
+
}
|
|
95
|
+
if (type === 'application/json') return 'text';
|
|
96
|
+
return 'binary';
|
|
97
|
+
}
|
|
98
|
+
|
|
19
99
|
/**
|
|
20
100
|
* Fetch a URL and return parsed HTML via Cheerio.
|
|
21
101
|
*
|
|
@@ -45,7 +125,24 @@ export async function fetchAndParse(url, options = {}) {
|
|
|
45
125
|
throw new Error(`HTTP ${response.status}: ${response.statusText}`);
|
|
46
126
|
}
|
|
47
127
|
|
|
48
|
-
const
|
|
128
|
+
const contentType = response.headers?.get?.('content-type') || null;
|
|
129
|
+
const classification = classifyContentType(contentType);
|
|
130
|
+
|
|
131
|
+
if (classification === 'binary') {
|
|
132
|
+
throw new Error(
|
|
133
|
+
`Unsupported content type "${contentType}" — this looks like binary content, not HTML/text. Use process_document for PDFs/documents/binary files.`
|
|
134
|
+
);
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
const html = await readTextWithSizeCap(response);
|
|
138
|
+
|
|
139
|
+
// text/plain and application/json aren't markup: running them through the
|
|
140
|
+
// HTML parser risks misinterpreting substrings (e.g. a "<script>" value
|
|
141
|
+
// inside a JSON string) as real tags and stripping/mangling content.
|
|
142
|
+
if (classification === 'text') {
|
|
143
|
+
return { html, $: load(''), textContent: html.trim(), finalUrl: response.url };
|
|
144
|
+
}
|
|
145
|
+
|
|
49
146
|
const $ = load(html);
|
|
50
147
|
|
|
51
148
|
if (stripTags.length > 0) {
|
|
@@ -298,14 +298,17 @@ export class ExtractContentTool {
|
|
|
298
298
|
* @returns {Promise<boolean>} - Whether JavaScript is needed
|
|
299
299
|
*/
|
|
300
300
|
async shouldUseJavaScript(url) {
|
|
301
|
-
// Simple heuristics for determining if JavaScript is needed
|
|
301
|
+
// Simple heuristics for determining if JavaScript is needed. Strip the
|
|
302
|
+
// fragment first: an ordinary document anchor (e.g. #install) isn't a
|
|
303
|
+
// signal for client-side routing, and anchoring the path pattern to full
|
|
304
|
+
// segments avoids false positives like "/apple" or "/spaces".
|
|
305
|
+
const urlWithoutFragment = url.split('#')[0];
|
|
302
306
|
const jsIndicators = [
|
|
303
|
-
/\/(app|spa|dashboard|admin)/,
|
|
304
|
-
/#/,
|
|
307
|
+
/\/(app|spa|dashboard|admin)(\/|$)/,
|
|
305
308
|
/\.(js|jsx|ts|tsx)$/
|
|
306
309
|
];
|
|
307
310
|
|
|
308
|
-
return jsIndicators.some(pattern => pattern.test(
|
|
311
|
+
return jsIndicators.some(pattern => pattern.test(urlWithoutFragment));
|
|
309
312
|
}
|
|
310
313
|
|
|
311
314
|
/**
|
|
@@ -109,10 +109,13 @@ export class ExtractStructuredTool {
|
|
|
109
109
|
// Step 3: Try LLM extraction first
|
|
110
110
|
let extractionResult = null;
|
|
111
111
|
let extractionMethod = 'llm';
|
|
112
|
+
let llmErrorMessage = null;
|
|
113
|
+
let llmAvailable = false;
|
|
112
114
|
|
|
113
115
|
try {
|
|
114
116
|
const llm = this._ensureLLMManager(llmConfig || {});
|
|
115
|
-
|
|
117
|
+
llmAvailable = llm.isAvailable();
|
|
118
|
+
if (llmAvailable) {
|
|
116
119
|
extractionResult = await llm.extractStructured(textContent, schema, {
|
|
117
120
|
prompt: prompt || '',
|
|
118
121
|
maxContentLength: 6000
|
|
@@ -120,12 +123,36 @@ export class ExtractStructuredTool {
|
|
|
120
123
|
extractionMethod = 'llm';
|
|
121
124
|
}
|
|
122
125
|
} catch (llmError) {
|
|
123
|
-
// LLM failed — will fall through to CSS fallback
|
|
126
|
+
// LLM failed — will fall through to CSS fallback. Keep the message so
|
|
127
|
+
// callers can tell "LLM broken" apart from "no LLM configured".
|
|
124
128
|
extractionResult = null;
|
|
129
|
+
llmErrorMessage = llmError.message;
|
|
125
130
|
}
|
|
126
131
|
|
|
127
132
|
// Step 4: CSS selector fallback if LLM unavailable or failed
|
|
128
133
|
if (!extractionResult && fallbackToSelectors !== false) {
|
|
134
|
+
// D1.4: no LLM configured and the schema demands more than 3 required
|
|
135
|
+
// fields — confirm before running the lower-fidelity CSS fallback.
|
|
136
|
+
const requiredCount = (schema.required || []).length;
|
|
137
|
+
if (!llmAvailable && requiredCount > 3) {
|
|
138
|
+
const proceed = await this._elicitation.confirm(
|
|
139
|
+
`No LLM provider is configured and the requested schema has ${requiredCount} required fields. ` +
|
|
140
|
+
`extract_structured will fall back to lower-fidelity CSS selector extraction, which may miss required fields.`,
|
|
141
|
+
{ url, required_fields: requiredCount }
|
|
142
|
+
);
|
|
143
|
+
if (!proceed) {
|
|
144
|
+
return {
|
|
145
|
+
url,
|
|
146
|
+
data: {},
|
|
147
|
+
extraction_method: 'none',
|
|
148
|
+
confidence: 0,
|
|
149
|
+
schema_used: schema,
|
|
150
|
+
processingTime: Date.now() - startTime,
|
|
151
|
+
error: 'Extraction cancelled by user (elicitation declined).',
|
|
152
|
+
validation: { valid: false, errors: ['Extraction cancelled by user (elicitation declined).'] }
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
}
|
|
129
156
|
extractionResult = this._cssExtraction($, schema, selectorHints || {});
|
|
130
157
|
extractionMethod = 'css_fallback';
|
|
131
158
|
}
|
|
@@ -140,6 +167,11 @@ export class ExtractStructuredTool {
|
|
|
140
167
|
// Step 6: Calculate confidence
|
|
141
168
|
const confidence = this._calculateConfidence(extractionResult, extractionMethod);
|
|
142
169
|
|
|
170
|
+
const extractionNotes = extractionResult.extractionNotes || [];
|
|
171
|
+
if (llmErrorMessage) {
|
|
172
|
+
extractionNotes.push(`LLM extraction failed: ${llmErrorMessage}`);
|
|
173
|
+
}
|
|
174
|
+
|
|
143
175
|
return {
|
|
144
176
|
url,
|
|
145
177
|
data: extractionResult.data || {},
|
|
@@ -151,7 +183,7 @@ export class ExtractStructuredTool {
|
|
|
151
183
|
valid: extractionResult.valid || false,
|
|
152
184
|
errors: extractionResult.validationErrors || []
|
|
153
185
|
},
|
|
154
|
-
extractionNotes
|
|
186
|
+
extractionNotes
|
|
155
187
|
};
|
|
156
188
|
|
|
157
189
|
} catch (error) {
|
|
@@ -178,98 +210,77 @@ export class ExtractStructuredTool {
|
|
|
178
210
|
let fieldsFound = 0;
|
|
179
211
|
|
|
180
212
|
for (const [key, fieldSchema] of Object.entries(properties)) {
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
//
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
const
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
213
|
+
// Schema keys become CSS selector fragments below (class/id/data-attr).
|
|
214
|
+
// A key with spaces/parens/quotes produces an invalid selector that
|
|
215
|
+
// throws ("Attribute selector didn't terminate") — catch that per-field
|
|
216
|
+
// so one bad key can't discard extraction results for every other field.
|
|
217
|
+
try {
|
|
218
|
+
const isArrayField = fieldSchema.type === 'array';
|
|
219
|
+
|
|
220
|
+
// Use explicit selector hint if provided
|
|
221
|
+
const selector = selectorHints[key];
|
|
222
|
+
if (selector) {
|
|
223
|
+
const els = $(selector);
|
|
224
|
+
if (els.length > 0) {
|
|
225
|
+
if (isArrayField || els.length > 1) {
|
|
226
|
+
const values = els.map((_, el) => $(el).text().trim()).get().filter(Boolean);
|
|
227
|
+
if (values.length > 0) {
|
|
228
|
+
extracted[key] = values;
|
|
229
|
+
fieldsFound++;
|
|
230
|
+
continue;
|
|
231
|
+
}
|
|
232
|
+
} else {
|
|
233
|
+
const rawValue = els.first().text().trim();
|
|
234
|
+
if (rawValue) {
|
|
235
|
+
extracted[key] = this._coerceValue(rawValue, fieldSchema);
|
|
236
|
+
fieldsFound++;
|
|
237
|
+
continue;
|
|
238
|
+
}
|
|
201
239
|
}
|
|
202
240
|
}
|
|
203
241
|
}
|
|
204
|
-
}
|
|
205
242
|
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
243
|
+
// For array fields: detect ul/ol > li patterns before meta/common selectors
|
|
244
|
+
if (isArrayField) {
|
|
245
|
+
const listSelectors = [
|
|
246
|
+
`ul.${key} > li`, `ol.${key} > li`,
|
|
247
|
+
`#${key} > li`, `[data-${key}] > li`,
|
|
248
|
+
`ul[class*="${key}"] > li`, `ol[class*="${key}"] > li`
|
|
249
|
+
];
|
|
250
|
+
let listValues = null;
|
|
251
|
+
for (const lsel of listSelectors) {
|
|
252
|
+
const items = $(lsel);
|
|
253
|
+
if (items.length > 0) {
|
|
254
|
+
listValues = items.map((_, el) => $(el).text().trim()).get().filter(Boolean);
|
|
255
|
+
break;
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
if (listValues && listValues.length > 0) {
|
|
259
|
+
extracted[key] = listValues;
|
|
260
|
+
fieldsFound++;
|
|
261
|
+
continue;
|
|
219
262
|
}
|
|
220
263
|
}
|
|
221
|
-
|
|
222
|
-
|
|
264
|
+
|
|
265
|
+
// Try common patterns: meta tags, headings, semantic elements
|
|
266
|
+
const metaContent = $(`meta[name="${key}"], meta[property="${key}"], meta[property="og:${key}"]`).attr('content');
|
|
267
|
+
if (metaContent) {
|
|
268
|
+
extracted[key] = this._coerceValue(metaContent, fieldSchema);
|
|
223
269
|
fieldsFound++;
|
|
224
270
|
continue;
|
|
225
271
|
}
|
|
226
|
-
}
|
|
227
|
-
|
|
228
|
-
// Try common patterns: meta tags, headings, semantic elements
|
|
229
|
-
const metaContent = $(`meta[name="${key}"], meta[property="${key}"], meta[property="og:${key}"]`).attr('content');
|
|
230
|
-
if (metaContent) {
|
|
231
|
-
extracted[key] = this._coerceValue(metaContent, fieldSchema);
|
|
232
|
-
fieldsFound++;
|
|
233
|
-
continue;
|
|
234
|
-
}
|
|
235
272
|
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
for (const sel of commonSelectors) {
|
|
245
|
-
const el = $(sel);
|
|
246
|
-
if (el.length > 0) {
|
|
247
|
-
if (isArrayField && el.length > 1) {
|
|
248
|
-
const values = el.map((_, item) => $(item).text().trim()).get().filter(Boolean);
|
|
249
|
-
if (values.length > 0) {
|
|
250
|
-
extracted[key] = values;
|
|
251
|
-
fieldsFound++;
|
|
252
|
-
break;
|
|
253
|
-
}
|
|
254
|
-
} else {
|
|
255
|
-
const rawValue = el.first().text().trim();
|
|
256
|
-
if (rawValue) {
|
|
257
|
-
extracted[key] = this._coerceValue(rawValue, fieldSchema);
|
|
258
|
-
fieldsFound++;
|
|
259
|
-
break;
|
|
260
|
-
}
|
|
261
|
-
}
|
|
262
|
-
}
|
|
263
|
-
}
|
|
273
|
+
// Try matching by common selectors based on field name
|
|
274
|
+
const commonSelectors = [
|
|
275
|
+
`[itemprop="${key}"]`,
|
|
276
|
+
`[data-${key}]`,
|
|
277
|
+
`.${key}`,
|
|
278
|
+
`#${key}`
|
|
279
|
+
];
|
|
264
280
|
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
const semanticSelectors = SEMANTIC_FIELD_SELECTORS[key.toLowerCase()];
|
|
269
|
-
if (semanticSelectors) {
|
|
270
|
-
for (const sel of semanticSelectors) {
|
|
271
|
-
const el = $(sel);
|
|
272
|
-
if (el.length === 0) continue;
|
|
281
|
+
for (const sel of commonSelectors) {
|
|
282
|
+
const el = $(sel);
|
|
283
|
+
if (el.length > 0) {
|
|
273
284
|
if (isArrayField && el.length > 1) {
|
|
274
285
|
const values = el.map((_, item) => $(item).text().trim()).get().filter(Boolean);
|
|
275
286
|
if (values.length > 0) {
|
|
@@ -287,6 +298,36 @@ export class ExtractStructuredTool {
|
|
|
287
298
|
}
|
|
288
299
|
}
|
|
289
300
|
}
|
|
301
|
+
|
|
302
|
+
// Last resort: semantic element selectors for well-known field names
|
|
303
|
+
// (e.g. title -> <h1>/<title>) so common fields resolve without hints.
|
|
304
|
+
if (!(key in extracted)) {
|
|
305
|
+
const semanticSelectors = SEMANTIC_FIELD_SELECTORS[key.toLowerCase()];
|
|
306
|
+
if (semanticSelectors) {
|
|
307
|
+
for (const sel of semanticSelectors) {
|
|
308
|
+
const el = $(sel);
|
|
309
|
+
if (el.length === 0) continue;
|
|
310
|
+
if (isArrayField && el.length > 1) {
|
|
311
|
+
const values = el.map((_, item) => $(item).text().trim()).get().filter(Boolean);
|
|
312
|
+
if (values.length > 0) {
|
|
313
|
+
extracted[key] = values;
|
|
314
|
+
fieldsFound++;
|
|
315
|
+
break;
|
|
316
|
+
}
|
|
317
|
+
} else {
|
|
318
|
+
const rawValue = el.first().text().trim();
|
|
319
|
+
if (rawValue) {
|
|
320
|
+
extracted[key] = this._coerceValue(rawValue, fieldSchema);
|
|
321
|
+
fieldsFound++;
|
|
322
|
+
break;
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
} catch (_fieldError) {
|
|
329
|
+
// Invalid selector for this key — skip the field, keep going.
|
|
330
|
+
continue;
|
|
290
331
|
}
|
|
291
332
|
}
|
|
292
333
|
|
|
@@ -340,7 +340,9 @@ async function callOllama({ model, systemMessage, userMessage, maxTokens, schema
|
|
|
340
340
|
],
|
|
341
341
|
stream: false,
|
|
342
342
|
options: { num_predict: maxTokens, temperature: 0 },
|
|
343
|
-
|
|
343
|
+
// Normalize the same way the Anthropic branch does — Ollama's `format`
|
|
344
|
+
// needs a valid JSON Schema, not a raw flat field->type-hint map.
|
|
345
|
+
format: (schema && Object.keys(schema).length > 0) ? buildInputSchema(schema) : 'json'
|
|
344
346
|
};
|
|
345
347
|
|
|
346
348
|
let response;
|
|
@@ -398,6 +400,12 @@ async function callLLM({ provider, apiKey, model, systemMessage, userMessage, ma
|
|
|
398
400
|
export class ExtractWithLlm {
|
|
399
401
|
constructor(config = {}) {
|
|
400
402
|
this.config = config;
|
|
403
|
+
this._mcpServer = null;
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
/** D1.3: Wire MCP server so the sampling fallback can reach the client. */
|
|
407
|
+
setMcpServer(mcpServer) {
|
|
408
|
+
this._mcpServer = mcpServer;
|
|
401
409
|
}
|
|
402
410
|
|
|
403
411
|
/**
|
|
@@ -479,7 +487,7 @@ export class ExtractWithLlm {
|
|
|
479
487
|
if (providerParam === 'auto' || providerParam === 'ollama') {
|
|
480
488
|
try {
|
|
481
489
|
const SamplingClient = await getSamplingClient();
|
|
482
|
-
const samplingClient = new SamplingClient();
|
|
490
|
+
const samplingClient = new SamplingClient({ mcpServer: this._mcpServer });
|
|
483
491
|
const { text: sampledText } = await samplingClient.complete(
|
|
484
492
|
`${systemMessage}\n\n${userMessage}`,
|
|
485
493
|
{ maxTokens }
|
|
@@ -4,6 +4,8 @@
|
|
|
4
4
|
*/
|
|
5
5
|
|
|
6
6
|
import { z } from 'zod';
|
|
7
|
+
import fs from 'fs/promises';
|
|
8
|
+
import path from 'path';
|
|
7
9
|
import { PDFProcessor } from '../../core/processing/PDFProcessor.js';
|
|
8
10
|
import { ContentProcessor } from '../../core/processing/ContentProcessor.js';
|
|
9
11
|
import { BrowserProcessor } from '../../core/processing/BrowserProcessor.js';
|
|
@@ -18,7 +20,6 @@ const ProcessDocumentSchema = z.object({
|
|
|
18
20
|
// PDF processing options
|
|
19
21
|
extractText: z.boolean().default(true),
|
|
20
22
|
extractMetadata: z.boolean().default(true),
|
|
21
|
-
password: z.string().optional(),
|
|
22
23
|
maxPages: z.number().min(1).max(500).default(100),
|
|
23
24
|
// C3: extract a specific 1-based, inclusive page range from a PDF
|
|
24
25
|
pageRange: z.object({
|
|
@@ -148,6 +149,9 @@ export class ProcessDocumentTool {
|
|
|
148
149
|
if (sourceType.includes('pdf')) {
|
|
149
150
|
result.documentType = 'pdf';
|
|
150
151
|
await this.processPDFDocument(result, source, sourceType, options);
|
|
152
|
+
} else if (sourceType === 'file') {
|
|
153
|
+
result.documentType = 'file';
|
|
154
|
+
await this.processLocalFileDocument(result, source, options);
|
|
151
155
|
} else {
|
|
152
156
|
result.documentType = 'web';
|
|
153
157
|
await this.processWebDocument(result, source, options);
|
|
@@ -200,7 +204,6 @@ export class ProcessDocumentTool {
|
|
|
200
204
|
options: {
|
|
201
205
|
extractText: options.extractText,
|
|
202
206
|
extractMetadata: options.extractMetadata,
|
|
203
|
-
password: options.password,
|
|
204
207
|
maxPages: options.maxPages,
|
|
205
208
|
...(options.pageRange ? { pageRange: options.pageRange } : {})
|
|
206
209
|
}
|
|
@@ -293,10 +296,52 @@ export class ProcessDocumentTool {
|
|
|
293
296
|
|
|
294
297
|
result.title = pageTitle;
|
|
295
298
|
|
|
299
|
+
await this.processFetchedHtml(result, html, source, options);
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* Process a local non-PDF file (sourceType 'file'): read it from disk and
|
|
304
|
+
* run it through the same content-processing pipeline used for web pages.
|
|
305
|
+
* @param {Object} result - Result object to populate
|
|
306
|
+
* @param {string} source - Local file path
|
|
307
|
+
* @param {Object} options - Processing options
|
|
308
|
+
* @returns {Promise<void>}
|
|
309
|
+
*/
|
|
310
|
+
async processLocalFileDocument(result, source, options) {
|
|
311
|
+
const resolvedPath = path.resolve(source);
|
|
312
|
+
let content;
|
|
313
|
+
try {
|
|
314
|
+
content = await fs.readFile(resolvedPath, 'utf-8');
|
|
315
|
+
} catch (error) {
|
|
316
|
+
throw new Error(`Failed to read local file: ${error.message}`);
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
const isHtml = /\.html?$/i.test(resolvedPath) || /<html[\s>]/i.test(content.slice(0, 1000));
|
|
320
|
+
const html = isHtml
|
|
321
|
+
? content
|
|
322
|
+
: `<html><body><pre>${content.replace(/[&<>]/g, c => ({ '&': '&', '<': '<', '>': '>' }[c]))}</pre></body></html>`;
|
|
323
|
+
|
|
324
|
+
result.title = isHtml ? this.extractTitleFromHTML(content) : null;
|
|
325
|
+
|
|
326
|
+
// No `url` to pass here — ContentProcessor's url field is a validated
|
|
327
|
+
// absolute URL, and a filesystem path doesn't qualify.
|
|
328
|
+
await this.processFetchedHtml(result, html, undefined, options);
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
/**
|
|
332
|
+
* Run the shared content-processing pipeline (Readability/boilerplate
|
|
333
|
+
* fallback, metadata, structured data) over already-fetched HTML.
|
|
334
|
+
* @param {Object} result - Result object to populate
|
|
335
|
+
* @param {string} html - HTML content
|
|
336
|
+
* @param {string|undefined} url - Source URL (omitted for local files)
|
|
337
|
+
* @param {Object} options - Processing options
|
|
338
|
+
* @returns {Promise<void>}
|
|
339
|
+
*/
|
|
340
|
+
async processFetchedHtml(result, html, url, options) {
|
|
296
341
|
// Step 2: Process content with ContentProcessor
|
|
297
342
|
const processingResult = await this.contentProcessor.processContent({
|
|
298
343
|
html,
|
|
299
|
-
url
|
|
344
|
+
url,
|
|
300
345
|
options: {
|
|
301
346
|
extractStructuredData: options.extractStructuredData,
|
|
302
347
|
calculateReadabilityScore: true,
|
|
@@ -449,13 +494,16 @@ export class ProcessDocumentTool {
|
|
|
449
494
|
* @returns {Promise<boolean>} - Whether JavaScript is needed
|
|
450
495
|
*/
|
|
451
496
|
async shouldUseJavaScript(url) {
|
|
497
|
+
// Strip the fragment first: an ordinary document anchor (e.g. #install)
|
|
498
|
+
// isn't a signal for client-side routing, and anchoring the path pattern
|
|
499
|
+
// to full segments avoids false positives like "/apple" or "/spaces".
|
|
500
|
+
const urlWithoutFragment = url.split('#')[0];
|
|
452
501
|
const jsIndicators = [
|
|
453
|
-
/\/(app|spa|dashboard|admin)/,
|
|
454
|
-
/#/,
|
|
502
|
+
/\/(app|spa|dashboard|admin)(\/|$)/,
|
|
455
503
|
/\.(js|jsx|ts|tsx)$/
|
|
456
504
|
];
|
|
457
505
|
|
|
458
|
-
return jsIndicators.some(pattern => pattern.test(
|
|
506
|
+
return jsIndicators.some(pattern => pattern.test(urlWithoutFragment));
|
|
459
507
|
}
|
|
460
508
|
|
|
461
509
|
/**
|
|
@@ -76,6 +76,12 @@ const SummarizeContentResult = z.object({
|
|
|
76
76
|
export class SummarizeContentTool {
|
|
77
77
|
constructor() {
|
|
78
78
|
this.contentAnalyzer = new ContentAnalyzer();
|
|
79
|
+
this._mcpServer = null;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** D1.3: Wire MCP server so the sampling fallback can reach the client. */
|
|
83
|
+
setMcpServer(mcpServer) {
|
|
84
|
+
this._mcpServer = mcpServer;
|
|
79
85
|
}
|
|
80
86
|
|
|
81
87
|
/**
|
|
@@ -211,7 +217,7 @@ export class SummarizeContentTool {
|
|
|
211
217
|
async _abstractiveSummaryViaSampling(text, extractiveSummary, summaryLength) {
|
|
212
218
|
try {
|
|
213
219
|
const SamplingClient = await getSamplingClient();
|
|
214
|
-
const client = new SamplingClient();
|
|
220
|
+
const client = new SamplingClient({ mcpServer: this._mcpServer });
|
|
215
221
|
|
|
216
222
|
const lengthGuide = {
|
|
217
223
|
short: '1-2 sentences',
|
|
@@ -53,11 +53,6 @@ export class GenerateLLMsTxtTool {
|
|
|
53
53
|
userAgent: options.userAgent || 'LLMs.txt-Generator/1.0',
|
|
54
54
|
...options
|
|
55
55
|
};
|
|
56
|
-
|
|
57
|
-
this.analyzer = new LLMsTxtAnalyzer({
|
|
58
|
-
timeout: this.options.timeout,
|
|
59
|
-
userAgent: this.options.userAgent
|
|
60
|
-
});
|
|
61
56
|
}
|
|
62
57
|
|
|
63
58
|
async execute(params) {
|
|
@@ -71,8 +66,15 @@ export class GenerateLLMsTxtTool {
|
|
|
71
66
|
const baseUrl = getBaseUrl(url);
|
|
72
67
|
|
|
73
68
|
// Step 1: Comprehensive Website Analysis
|
|
69
|
+
// A fresh analyzer is constructed per call — LLMsTxtAnalyzer keeps mutable
|
|
70
|
+
// per-analysis state on `this.analysis`, so a shared instance would let
|
|
71
|
+
// concurrent (or successive) calls cross-contaminate results.
|
|
74
72
|
logger.info(`Analyzing website: ${baseUrl}`);
|
|
75
|
-
const
|
|
73
|
+
const analyzer = new LLMsTxtAnalyzer({
|
|
74
|
+
timeout: this.options.timeout,
|
|
75
|
+
userAgent: this.options.userAgent
|
|
76
|
+
});
|
|
77
|
+
const analysis = await analyzer.analyzeWebsite(url, analysisOptions);
|
|
76
78
|
|
|
77
79
|
// Step 2: Generate LLMs.txt Content
|
|
78
80
|
const llmsTxtContent = this.generateLLMsTxt(analysis, outputOptions, complianceLevel);
|