crawlforge-mcp-server 4.9.0 → 5.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/CLAUDE.md +6 -5
  2. package/README.md +19 -3
  3. package/package.json +10 -12
  4. package/server.js +315 -214
  5. package/src/core/ActionExecutor.js +117 -33
  6. package/src/core/AgentOrchestrator.js +8 -2
  7. package/src/core/AuthManager.js +51 -17
  8. package/src/core/ChangeTracker.js +26 -10
  9. package/src/core/JobManager.js +9 -1
  10. package/src/core/LocalizationManager.js +19 -6
  11. package/src/core/ResearchOrchestrator.js +173 -35
  12. package/src/core/SnapshotManager.js +162 -165
  13. package/src/core/StealthBrowserManager.js +25 -3
  14. package/src/core/WebhookDispatcher.js +19 -14
  15. package/src/core/analysis/ContentAnalyzer.js +52 -7
  16. package/src/core/crawlers/BFSCrawler.js +27 -3
  17. package/src/core/processing/BrowserProcessor.js +19 -1
  18. package/src/core/processing/PDFProcessor.js +129 -65
  19. package/src/core/queue/QueueManager.js +3 -2
  20. package/src/schemas/toolOutputSchemas.js +269 -0
  21. package/src/server/auth/oauth.js +37 -7
  22. package/src/server/specHygiene.js +192 -0
  23. package/src/server/taskSupport.js +233 -0
  24. package/src/server/toolFilter.js +98 -0
  25. package/src/server/transports/streamableHttp.js +148 -11
  26. package/src/server/withAuth.js +11 -4
  27. package/src/skills/agent-skills/crawlforge-getting-started/SKILL.md +15 -0
  28. package/src/tools/advanced/ScrapeWithActionsTool.js +43 -52
  29. package/src/tools/advanced/batchScrape/index.js +128 -27
  30. package/src/tools/advanced/batchScrape/worker.js +55 -5
  31. package/src/tools/advanced/scrapeWithActions/recorder.js +3 -0
  32. package/src/tools/basic/_fetch.js +125 -70
  33. package/src/tools/basic/extractLinks.js +14 -12
  34. package/src/tools/basic/scrapeStructured.js +21 -4
  35. package/src/tools/crawl/crawlDeep.js +110 -48
  36. package/src/tools/crawl/mapSite.js +25 -6
  37. package/src/tools/extract/_fetchAndParse.js +98 -1
  38. package/src/tools/extract/extractContent.js +7 -4
  39. package/src/tools/extract/extractStructured.js +125 -84
  40. package/src/tools/extract/extractWithLlm.js +10 -2
  41. package/src/tools/extract/processDocument.js +54 -6
  42. package/src/tools/extract/summarizeContent.js +7 -1
  43. package/src/tools/llmstxt/generateLLMsTxt.js +8 -6
  44. package/src/tools/research/deepResearch.js +51 -31
  45. package/src/tools/scrape/_brandingExtractor.js +49 -11
  46. package/src/tools/scrape/unifiedScrape.js +27 -17
  47. package/src/tools/search/providers/searxng.js +5 -1
  48. package/src/tools/search/ranking/ResultDeduplicator.js +9 -1
  49. package/src/tools/search/ranking/ResultRanker.js +17 -2
  50. package/src/tools/search/searchWeb.js +31 -14
  51. package/src/tools/search/serpRank.js +23 -0
  52. package/src/tools/templates/TemplateRegistry.js +7 -1
  53. package/src/tools/tracking/trackChanges/index.js +87 -26
  54. package/src/tools/tracking/trackChanges/schema.js +2 -2
  55. package/src/utils/CircuitBreaker.js +11 -9
  56. package/src/utils/contentUtils.js +66 -53
  57. package/src/utils/secretMask.js +1 -1
  58. package/src/utils/sitemapParser.js +11 -9
  59. package/src/utils/ssrfGuard.js +212 -40
  60. package/src/utils/urlNormalizer.js +2 -2
@@ -12,10 +12,90 @@
12
12
 
13
13
  import { load } from 'cheerio';
14
14
  import { safeFetch } from '../../utils/ssrfGuard.js';
15
+ import { config } from '../../constants/config.js';
15
16
 
16
17
  const DEFAULT_USER_AGENT = 'Mozilla/5.0 (compatible; CrawlForge-MCP/3.0)';
17
18
  const DEFAULT_TIMEOUT_MS = 15000;
18
19
 
20
+ /**
21
+ * Read a response body as text while enforcing config.fetch.maxBodySize —
22
+ * the same cap _fetch.js applies to basic tools, so a large/hostile response
23
+ * (multi-hundred-MB file, endpoint streaming zeros) can't be buffered whole
24
+ * into a JS string and handed to cheerio/JSDOM/Turndown downstream.
25
+ * Content-Length is checked up front; actual bytes read are counted as a
26
+ * backstop for servers that omit or lie about it.
27
+ * @param {Response} response
28
+ * @returns {Promise<string>}
29
+ */
30
+ async function readTextWithSizeCap(response) {
31
+ const maxBodySize = config.fetch.maxBodySize;
32
+
33
+ const contentLengthHeader = response.headers?.get?.('content-length') ?? null;
34
+ if (contentLengthHeader !== null) {
35
+ const declared = parseInt(contentLengthHeader, 10);
36
+ if (!isNaN(declared) && declared > maxBodySize) {
37
+ throw new Error(
38
+ `Response body too large: Content-Length ${declared} exceeds limit of ${maxBodySize} bytes`
39
+ );
40
+ }
41
+ }
42
+
43
+ // Responses without a ReadableStream body (test mocks, already-buffered
44
+ // responses) fall back to the native .text() with no additional guard.
45
+ if (!response.body || typeof response.body.getReader !== 'function') {
46
+ return response.text();
47
+ }
48
+
49
+ const reader = response.body.getReader();
50
+ const chunks = [];
51
+ let totalBytes = 0;
52
+
53
+ while (true) {
54
+ const { done, value } = await reader.read();
55
+ if (done) break;
56
+ totalBytes += value.byteLength;
57
+ if (totalBytes > maxBodySize) {
58
+ reader.cancel();
59
+ throw new Error(
60
+ `Response body too large: exceeded limit of ${maxBodySize} bytes`
61
+ );
62
+ }
63
+ chunks.push(value);
64
+ }
65
+
66
+ const mergedBytes = new Uint8Array(totalBytes);
67
+ let offset = 0;
68
+ for (const chunk of chunks) {
69
+ mergedBytes.set(chunk, offset);
70
+ offset += chunk.byteLength;
71
+ }
72
+ return new TextDecoder().decode(mergedBytes);
73
+ }
74
+
75
+ /**
76
+ * Classify a Content-Type header for the purposes of HTML parsing.
77
+ * Missing header is treated as 'html' (permissive default — many servers,
78
+ * and most test doubles, omit it for what is genuinely HTML).
79
+ * @param {string|null} contentType
80
+ * @returns {'html'|'text'|'binary'}
81
+ */
82
+ function classifyContentType(contentType) {
83
+ if (!contentType) return 'html';
84
+ const type = contentType.split(';')[0].trim().toLowerCase();
85
+ if (
86
+ type === 'text/html' ||
87
+ type === 'application/xhtml+xml' ||
88
+ type === 'application/xml' ||
89
+ type === 'text/xml' ||
90
+ type.endsWith('+xml') ||
91
+ type.startsWith('text/')
92
+ ) {
93
+ return type === 'text/plain' ? 'text' : 'html';
94
+ }
95
+ if (type === 'application/json') return 'text';
96
+ return 'binary';
97
+ }
98
+
19
99
  /**
20
100
  * Fetch a URL and return parsed HTML via Cheerio.
21
101
  *
@@ -45,7 +125,24 @@ export async function fetchAndParse(url, options = {}) {
45
125
  throw new Error(`HTTP ${response.status}: ${response.statusText}`);
46
126
  }
47
127
 
48
- const html = await response.text();
128
+ const contentType = response.headers?.get?.('content-type') || null;
129
+ const classification = classifyContentType(contentType);
130
+
131
+ if (classification === 'binary') {
132
+ throw new Error(
133
+ `Unsupported content type "${contentType}" — this looks like binary content, not HTML/text. Use process_document for PDFs/documents/binary files.`
134
+ );
135
+ }
136
+
137
+ const html = await readTextWithSizeCap(response);
138
+
139
+ // text/plain and application/json aren't markup: running them through the
140
+ // HTML parser risks misinterpreting substrings (e.g. a "<script>" value
141
+ // inside a JSON string) as real tags and stripping/mangling content.
142
+ if (classification === 'text') {
143
+ return { html, $: load(''), textContent: html.trim(), finalUrl: response.url };
144
+ }
145
+
49
146
  const $ = load(html);
50
147
 
51
148
  if (stripTags.length > 0) {
@@ -298,14 +298,17 @@ export class ExtractContentTool {
298
298
  * @returns {Promise<boolean>} - Whether JavaScript is needed
299
299
  */
300
300
  async shouldUseJavaScript(url) {
301
- // Simple heuristics for determining if JavaScript is needed
301
+ // Simple heuristics for determining if JavaScript is needed. Strip the
302
+ // fragment first: an ordinary document anchor (e.g. #install) isn't a
303
+ // signal for client-side routing, and anchoring the path pattern to full
304
+ // segments avoids false positives like "/apple" or "/spaces".
305
+ const urlWithoutFragment = url.split('#')[0];
302
306
  const jsIndicators = [
303
- /\/(app|spa|dashboard|admin)/,
304
- /#/,
307
+ /\/(app|spa|dashboard|admin)(\/|$)/,
305
308
  /\.(js|jsx|ts|tsx)$/
306
309
  ];
307
310
 
308
- return jsIndicators.some(pattern => pattern.test(url));
311
+ return jsIndicators.some(pattern => pattern.test(urlWithoutFragment));
309
312
  }
310
313
 
311
314
  /**
@@ -109,10 +109,13 @@ export class ExtractStructuredTool {
109
109
  // Step 3: Try LLM extraction first
110
110
  let extractionResult = null;
111
111
  let extractionMethod = 'llm';
112
+ let llmErrorMessage = null;
113
+ let llmAvailable = false;
112
114
 
113
115
  try {
114
116
  const llm = this._ensureLLMManager(llmConfig || {});
115
- if (llm.isAvailable()) {
117
+ llmAvailable = llm.isAvailable();
118
+ if (llmAvailable) {
116
119
  extractionResult = await llm.extractStructured(textContent, schema, {
117
120
  prompt: prompt || '',
118
121
  maxContentLength: 6000
@@ -120,12 +123,36 @@ export class ExtractStructuredTool {
120
123
  extractionMethod = 'llm';
121
124
  }
122
125
  } catch (llmError) {
123
- // LLM failed — will fall through to CSS fallback
126
+ // LLM failed — will fall through to CSS fallback. Keep the message so
127
+ // callers can tell "LLM broken" apart from "no LLM configured".
124
128
  extractionResult = null;
129
+ llmErrorMessage = llmError.message;
125
130
  }
126
131
 
127
132
  // Step 4: CSS selector fallback if LLM unavailable or failed
128
133
  if (!extractionResult && fallbackToSelectors !== false) {
134
+ // D1.4: no LLM configured and the schema demands more than 3 required
135
+ // fields — confirm before running the lower-fidelity CSS fallback.
136
+ const requiredCount = (schema.required || []).length;
137
+ if (!llmAvailable && requiredCount > 3) {
138
+ const proceed = await this._elicitation.confirm(
139
+ `No LLM provider is configured and the requested schema has ${requiredCount} required fields. ` +
140
+ `extract_structured will fall back to lower-fidelity CSS selector extraction, which may miss required fields.`,
141
+ { url, required_fields: requiredCount }
142
+ );
143
+ if (!proceed) {
144
+ return {
145
+ url,
146
+ data: {},
147
+ extraction_method: 'none',
148
+ confidence: 0,
149
+ schema_used: schema,
150
+ processingTime: Date.now() - startTime,
151
+ error: 'Extraction cancelled by user (elicitation declined).',
152
+ validation: { valid: false, errors: ['Extraction cancelled by user (elicitation declined).'] }
153
+ };
154
+ }
155
+ }
129
156
  extractionResult = this._cssExtraction($, schema, selectorHints || {});
130
157
  extractionMethod = 'css_fallback';
131
158
  }
@@ -140,6 +167,11 @@ export class ExtractStructuredTool {
140
167
  // Step 6: Calculate confidence
141
168
  const confidence = this._calculateConfidence(extractionResult, extractionMethod);
142
169
 
170
+ const extractionNotes = extractionResult.extractionNotes || [];
171
+ if (llmErrorMessage) {
172
+ extractionNotes.push(`LLM extraction failed: ${llmErrorMessage}`);
173
+ }
174
+
143
175
  return {
144
176
  url,
145
177
  data: extractionResult.data || {},
@@ -151,7 +183,7 @@ export class ExtractStructuredTool {
151
183
  valid: extractionResult.valid || false,
152
184
  errors: extractionResult.validationErrors || []
153
185
  },
154
- extractionNotes: extractionResult.extractionNotes || []
186
+ extractionNotes
155
187
  };
156
188
 
157
189
  } catch (error) {
@@ -178,98 +210,77 @@ export class ExtractStructuredTool {
178
210
  let fieldsFound = 0;
179
211
 
180
212
  for (const [key, fieldSchema] of Object.entries(properties)) {
181
- const isArrayField = fieldSchema.type === 'array';
182
-
183
- // Use explicit selector hint if provided
184
- const selector = selectorHints[key];
185
- if (selector) {
186
- const els = $(selector);
187
- if (els.length > 0) {
188
- if (isArrayField || els.length > 1) {
189
- const values = els.map((_, el) => $(el).text().trim()).get().filter(Boolean);
190
- if (values.length > 0) {
191
- extracted[key] = values;
192
- fieldsFound++;
193
- continue;
194
- }
195
- } else {
196
- const rawValue = els.first().text().trim();
197
- if (rawValue) {
198
- extracted[key] = this._coerceValue(rawValue, fieldSchema);
199
- fieldsFound++;
200
- continue;
213
+ // Schema keys become CSS selector fragments below (class/id/data-attr).
214
+ // A key with spaces/parens/quotes produces an invalid selector that
215
+ // throws ("Attribute selector didn't terminate") — catch that per-field
216
+ // so one bad key can't discard extraction results for every other field.
217
+ try {
218
+ const isArrayField = fieldSchema.type === 'array';
219
+
220
+ // Use explicit selector hint if provided
221
+ const selector = selectorHints[key];
222
+ if (selector) {
223
+ const els = $(selector);
224
+ if (els.length > 0) {
225
+ if (isArrayField || els.length > 1) {
226
+ const values = els.map((_, el) => $(el).text().trim()).get().filter(Boolean);
227
+ if (values.length > 0) {
228
+ extracted[key] = values;
229
+ fieldsFound++;
230
+ continue;
231
+ }
232
+ } else {
233
+ const rawValue = els.first().text().trim();
234
+ if (rawValue) {
235
+ extracted[key] = this._coerceValue(rawValue, fieldSchema);
236
+ fieldsFound++;
237
+ continue;
238
+ }
201
239
  }
202
240
  }
203
241
  }
204
- }
205
242
 
206
- // For array fields: detect ul/ol > li patterns before meta/common selectors
207
- if (isArrayField) {
208
- const listSelectors = [
209
- `ul.${key} > li`, `ol.${key} > li`,
210
- `#${key} > li`, `[data-${key}] > li`,
211
- `ul[class*="${key}"] > li`, `ol[class*="${key}"] > li`
212
- ];
213
- let listValues = null;
214
- for (const lsel of listSelectors) {
215
- const items = $(lsel);
216
- if (items.length > 0) {
217
- listValues = items.map((_, el) => $(el).text().trim()).get().filter(Boolean);
218
- break;
243
+ // For array fields: detect ul/ol > li patterns before meta/common selectors
244
+ if (isArrayField) {
245
+ const listSelectors = [
246
+ `ul.${key} > li`, `ol.${key} > li`,
247
+ `#${key} > li`, `[data-${key}] > li`,
248
+ `ul[class*="${key}"] > li`, `ol[class*="${key}"] > li`
249
+ ];
250
+ let listValues = null;
251
+ for (const lsel of listSelectors) {
252
+ const items = $(lsel);
253
+ if (items.length > 0) {
254
+ listValues = items.map((_, el) => $(el).text().trim()).get().filter(Boolean);
255
+ break;
256
+ }
257
+ }
258
+ if (listValues && listValues.length > 0) {
259
+ extracted[key] = listValues;
260
+ fieldsFound++;
261
+ continue;
219
262
  }
220
263
  }
221
- if (listValues && listValues.length > 0) {
222
- extracted[key] = listValues;
264
+
265
+ // Try common patterns: meta tags, headings, semantic elements
266
+ const metaContent = $(`meta[name="${key}"], meta[property="${key}"], meta[property="og:${key}"]`).attr('content');
267
+ if (metaContent) {
268
+ extracted[key] = this._coerceValue(metaContent, fieldSchema);
223
269
  fieldsFound++;
224
270
  continue;
225
271
  }
226
- }
227
-
228
- // Try common patterns: meta tags, headings, semantic elements
229
- const metaContent = $(`meta[name="${key}"], meta[property="${key}"], meta[property="og:${key}"]`).attr('content');
230
- if (metaContent) {
231
- extracted[key] = this._coerceValue(metaContent, fieldSchema);
232
- fieldsFound++;
233
- continue;
234
- }
235
272
 
236
- // Try matching by common selectors based on field name
237
- const commonSelectors = [
238
- `[itemprop="${key}"]`,
239
- `[data-${key}]`,
240
- `.${key}`,
241
- `#${key}`
242
- ];
243
-
244
- for (const sel of commonSelectors) {
245
- const el = $(sel);
246
- if (el.length > 0) {
247
- if (isArrayField && el.length > 1) {
248
- const values = el.map((_, item) => $(item).text().trim()).get().filter(Boolean);
249
- if (values.length > 0) {
250
- extracted[key] = values;
251
- fieldsFound++;
252
- break;
253
- }
254
- } else {
255
- const rawValue = el.first().text().trim();
256
- if (rawValue) {
257
- extracted[key] = this._coerceValue(rawValue, fieldSchema);
258
- fieldsFound++;
259
- break;
260
- }
261
- }
262
- }
263
- }
273
+ // Try matching by common selectors based on field name
274
+ const commonSelectors = [
275
+ `[itemprop="${key}"]`,
276
+ `[data-${key}]`,
277
+ `.${key}`,
278
+ `#${key}`
279
+ ];
264
280
 
265
- // Last resort: semantic element selectors for well-known field names
266
- // (e.g. title -> <h1>/<title>) so common fields resolve without hints.
267
- if (!(key in extracted)) {
268
- const semanticSelectors = SEMANTIC_FIELD_SELECTORS[key.toLowerCase()];
269
- if (semanticSelectors) {
270
- for (const sel of semanticSelectors) {
271
- const el = $(sel);
272
- if (el.length === 0) continue;
281
+ for (const sel of commonSelectors) {
282
+ const el = $(sel);
283
+ if (el.length > 0) {
273
284
  if (isArrayField && el.length > 1) {
274
285
  const values = el.map((_, item) => $(item).text().trim()).get().filter(Boolean);
275
286
  if (values.length > 0) {
@@ -287,6 +298,36 @@ export class ExtractStructuredTool {
287
298
  }
288
299
  }
289
300
  }
301
+
302
+ // Last resort: semantic element selectors for well-known field names
303
+ // (e.g. title -> <h1>/<title>) so common fields resolve without hints.
304
+ if (!(key in extracted)) {
305
+ const semanticSelectors = SEMANTIC_FIELD_SELECTORS[key.toLowerCase()];
306
+ if (semanticSelectors) {
307
+ for (const sel of semanticSelectors) {
308
+ const el = $(sel);
309
+ if (el.length === 0) continue;
310
+ if (isArrayField && el.length > 1) {
311
+ const values = el.map((_, item) => $(item).text().trim()).get().filter(Boolean);
312
+ if (values.length > 0) {
313
+ extracted[key] = values;
314
+ fieldsFound++;
315
+ break;
316
+ }
317
+ } else {
318
+ const rawValue = el.first().text().trim();
319
+ if (rawValue) {
320
+ extracted[key] = this._coerceValue(rawValue, fieldSchema);
321
+ fieldsFound++;
322
+ break;
323
+ }
324
+ }
325
+ }
326
+ }
327
+ }
328
+ } catch (_fieldError) {
329
+ // Invalid selector for this key — skip the field, keep going.
330
+ continue;
290
331
  }
291
332
  }
292
333
 
@@ -340,7 +340,9 @@ async function callOllama({ model, systemMessage, userMessage, maxTokens, schema
340
340
  ],
341
341
  stream: false,
342
342
  options: { num_predict: maxTokens, temperature: 0 },
343
- format: (schema && Object.keys(schema).length > 0) ? schema : 'json'
343
+ // Normalize the same way the Anthropic branch does — Ollama's `format`
344
+ // needs a valid JSON Schema, not a raw flat field->type-hint map.
345
+ format: (schema && Object.keys(schema).length > 0) ? buildInputSchema(schema) : 'json'
344
346
  };
345
347
 
346
348
  let response;
@@ -398,6 +400,12 @@ async function callLLM({ provider, apiKey, model, systemMessage, userMessage, ma
398
400
  export class ExtractWithLlm {
399
401
  constructor(config = {}) {
400
402
  this.config = config;
403
+ this._mcpServer = null;
404
+ }
405
+
406
+ /** D1.3: Wire MCP server so the sampling fallback can reach the client. */
407
+ setMcpServer(mcpServer) {
408
+ this._mcpServer = mcpServer;
401
409
  }
402
410
 
403
411
  /**
@@ -479,7 +487,7 @@ export class ExtractWithLlm {
479
487
  if (providerParam === 'auto' || providerParam === 'ollama') {
480
488
  try {
481
489
  const SamplingClient = await getSamplingClient();
482
- const samplingClient = new SamplingClient();
490
+ const samplingClient = new SamplingClient({ mcpServer: this._mcpServer });
483
491
  const { text: sampledText } = await samplingClient.complete(
484
492
  `${systemMessage}\n\n${userMessage}`,
485
493
  { maxTokens }
@@ -4,6 +4,8 @@
4
4
  */
5
5
 
6
6
  import { z } from 'zod';
7
+ import fs from 'fs/promises';
8
+ import path from 'path';
7
9
  import { PDFProcessor } from '../../core/processing/PDFProcessor.js';
8
10
  import { ContentProcessor } from '../../core/processing/ContentProcessor.js';
9
11
  import { BrowserProcessor } from '../../core/processing/BrowserProcessor.js';
@@ -18,7 +20,6 @@ const ProcessDocumentSchema = z.object({
18
20
  // PDF processing options
19
21
  extractText: z.boolean().default(true),
20
22
  extractMetadata: z.boolean().default(true),
21
- password: z.string().optional(),
22
23
  maxPages: z.number().min(1).max(500).default(100),
23
24
  // C3: extract a specific 1-based, inclusive page range from a PDF
24
25
  pageRange: z.object({
@@ -148,6 +149,9 @@ export class ProcessDocumentTool {
148
149
  if (sourceType.includes('pdf')) {
149
150
  result.documentType = 'pdf';
150
151
  await this.processPDFDocument(result, source, sourceType, options);
152
+ } else if (sourceType === 'file') {
153
+ result.documentType = 'file';
154
+ await this.processLocalFileDocument(result, source, options);
151
155
  } else {
152
156
  result.documentType = 'web';
153
157
  await this.processWebDocument(result, source, options);
@@ -200,7 +204,6 @@ export class ProcessDocumentTool {
200
204
  options: {
201
205
  extractText: options.extractText,
202
206
  extractMetadata: options.extractMetadata,
203
- password: options.password,
204
207
  maxPages: options.maxPages,
205
208
  ...(options.pageRange ? { pageRange: options.pageRange } : {})
206
209
  }
@@ -293,10 +296,52 @@ export class ProcessDocumentTool {
293
296
 
294
297
  result.title = pageTitle;
295
298
 
299
+ await this.processFetchedHtml(result, html, source, options);
300
+ }
301
+
302
+ /**
303
+ * Process a local non-PDF file (sourceType 'file'): read it from disk and
304
+ * run it through the same content-processing pipeline used for web pages.
305
+ * @param {Object} result - Result object to populate
306
+ * @param {string} source - Local file path
307
+ * @param {Object} options - Processing options
308
+ * @returns {Promise<void>}
309
+ */
310
+ async processLocalFileDocument(result, source, options) {
311
+ const resolvedPath = path.resolve(source);
312
+ let content;
313
+ try {
314
+ content = await fs.readFile(resolvedPath, 'utf-8');
315
+ } catch (error) {
316
+ throw new Error(`Failed to read local file: ${error.message}`);
317
+ }
318
+
319
+ const isHtml = /\.html?$/i.test(resolvedPath) || /<html[\s>]/i.test(content.slice(0, 1000));
320
+ const html = isHtml
321
+ ? content
322
+ : `<html><body><pre>${content.replace(/[&<>]/g, c => ({ '&': '&amp;', '<': '&lt;', '>': '&gt;' }[c]))}</pre></body></html>`;
323
+
324
+ result.title = isHtml ? this.extractTitleFromHTML(content) : null;
325
+
326
+ // No `url` to pass here — ContentProcessor's url field is a validated
327
+ // absolute URL, and a filesystem path doesn't qualify.
328
+ await this.processFetchedHtml(result, html, undefined, options);
329
+ }
330
+
331
+ /**
332
+ * Run the shared content-processing pipeline (Readability/boilerplate
333
+ * fallback, metadata, structured data) over already-fetched HTML.
334
+ * @param {Object} result - Result object to populate
335
+ * @param {string} html - HTML content
336
+ * @param {string|undefined} url - Source URL (omitted for local files)
337
+ * @param {Object} options - Processing options
338
+ * @returns {Promise<void>}
339
+ */
340
+ async processFetchedHtml(result, html, url, options) {
296
341
  // Step 2: Process content with ContentProcessor
297
342
  const processingResult = await this.contentProcessor.processContent({
298
343
  html,
299
- url: source,
344
+ url,
300
345
  options: {
301
346
  extractStructuredData: options.extractStructuredData,
302
347
  calculateReadabilityScore: true,
@@ -449,13 +494,16 @@ export class ProcessDocumentTool {
449
494
  * @returns {Promise<boolean>} - Whether JavaScript is needed
450
495
  */
451
496
  async shouldUseJavaScript(url) {
497
+ // Strip the fragment first: an ordinary document anchor (e.g. #install)
498
+ // isn't a signal for client-side routing, and anchoring the path pattern
499
+ // to full segments avoids false positives like "/apple" or "/spaces".
500
+ const urlWithoutFragment = url.split('#')[0];
452
501
  const jsIndicators = [
453
- /\/(app|spa|dashboard|admin)/,
454
- /#/,
502
+ /\/(app|spa|dashboard|admin)(\/|$)/,
455
503
  /\.(js|jsx|ts|tsx)$/
456
504
  ];
457
505
 
458
- return jsIndicators.some(pattern => pattern.test(url));
506
+ return jsIndicators.some(pattern => pattern.test(urlWithoutFragment));
459
507
  }
460
508
 
461
509
  /**
@@ -76,6 +76,12 @@ const SummarizeContentResult = z.object({
76
76
  export class SummarizeContentTool {
77
77
  constructor() {
78
78
  this.contentAnalyzer = new ContentAnalyzer();
79
+ this._mcpServer = null;
80
+ }
81
+
82
+ /** D1.3: Wire MCP server so the sampling fallback can reach the client. */
83
+ setMcpServer(mcpServer) {
84
+ this._mcpServer = mcpServer;
79
85
  }
80
86
 
81
87
  /**
@@ -211,7 +217,7 @@ export class SummarizeContentTool {
211
217
  async _abstractiveSummaryViaSampling(text, extractiveSummary, summaryLength) {
212
218
  try {
213
219
  const SamplingClient = await getSamplingClient();
214
- const client = new SamplingClient();
220
+ const client = new SamplingClient({ mcpServer: this._mcpServer });
215
221
 
216
222
  const lengthGuide = {
217
223
  short: '1-2 sentences',
@@ -53,11 +53,6 @@ export class GenerateLLMsTxtTool {
53
53
  userAgent: options.userAgent || 'LLMs.txt-Generator/1.0',
54
54
  ...options
55
55
  };
56
-
57
- this.analyzer = new LLMsTxtAnalyzer({
58
- timeout: this.options.timeout,
59
- userAgent: this.options.userAgent
60
- });
61
56
  }
62
57
 
63
58
  async execute(params) {
@@ -71,8 +66,15 @@ export class GenerateLLMsTxtTool {
71
66
  const baseUrl = getBaseUrl(url);
72
67
 
73
68
  // Step 1: Comprehensive Website Analysis
69
+ // A fresh analyzer is constructed per call — LLMsTxtAnalyzer keeps mutable
70
+ // per-analysis state on `this.analysis`, so a shared instance would let
71
+ // concurrent (or successive) calls cross-contaminate results.
74
72
  logger.info(`Analyzing website: ${baseUrl}`);
75
- const analysis = await this.analyzer.analyzeWebsite(url, analysisOptions);
73
+ const analyzer = new LLMsTxtAnalyzer({
74
+ timeout: this.options.timeout,
75
+ userAgent: this.options.userAgent
76
+ });
77
+ const analysis = await analyzer.analyzeWebsite(url, analysisOptions);
76
78
 
77
79
  // Step 2: Generate LLMs.txt Content
78
80
  const llmsTxtContent = this.generateLLMsTxt(analysis, outputOptions, complianceLevel);