crawlforge-mcp-server 4.10.0 → 5.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CLAUDE.md +6 -5
  2. package/README.md +19 -3
  3. package/package.json +10 -12
  4. package/server.js +298 -212
  5. package/src/cli/commands/init.js +11 -5
  6. package/src/cli/commands/stealth.js +3 -2
  7. package/src/core/ActionExecutor.js +117 -33
  8. package/src/core/AgentOrchestrator.js +8 -2
  9. package/src/core/AuthManager.js +51 -17
  10. package/src/core/ChangeTracker.js +26 -10
  11. package/src/core/JobManager.js +9 -1
  12. package/src/core/LocalizationManager.js +19 -6
  13. package/src/core/MonitorScheduler.js +13 -4
  14. package/src/core/ResearchOrchestrator.js +173 -35
  15. package/src/core/SnapshotManager.js +167 -165
  16. package/src/core/StealthBrowserManager.js +25 -3
  17. package/src/core/WebhookDispatcher.js +19 -14
  18. package/src/core/analysis/ContentAnalyzer.js +52 -7
  19. package/src/core/crawlers/BFSCrawler.js +46 -15
  20. package/src/core/processing/BrowserProcessor.js +19 -1
  21. package/src/core/processing/PDFProcessor.js +129 -65
  22. package/src/core/queue/QueueManager.js +3 -2
  23. package/src/schemas/toolOutputSchemas.js +269 -0
  24. package/src/server/auth/oauth.js +37 -7
  25. package/src/server/specHygiene.js +192 -0
  26. package/src/server/taskSupport.js +233 -0
  27. package/src/server/toolFilter.js +98 -0
  28. package/src/server/transports/streamableHttp.js +148 -11
  29. package/src/server/withAuth.js +11 -4
  30. package/src/tools/advanced/ScrapeWithActionsTool.js +43 -52
  31. package/src/tools/advanced/batchScrape/index.js +128 -27
  32. package/src/tools/advanced/batchScrape/worker.js +55 -5
  33. package/src/tools/advanced/scrapeWithActions/recorder.js +3 -0
  34. package/src/tools/basic/_fetch.js +125 -70
  35. package/src/tools/basic/extractLinks.js +14 -12
  36. package/src/tools/basic/scrapeStructured.js +21 -4
  37. package/src/tools/crawl/crawlDeep.js +110 -48
  38. package/src/tools/crawl/mapSite.js +25 -6
  39. package/src/tools/extract/_fetchAndParse.js +98 -1
  40. package/src/tools/extract/extractContent.js +7 -4
  41. package/src/tools/extract/extractStructured.js +125 -84
  42. package/src/tools/extract/extractWithLlm.js +10 -2
  43. package/src/tools/extract/processDocument.js +54 -6
  44. package/src/tools/extract/summarizeContent.js +7 -1
  45. package/src/tools/llmstxt/generateLLMsTxt.js +11 -6
  46. package/src/tools/research/deepResearch.js +51 -31
  47. package/src/tools/scrape/_brandingExtractor.js +49 -11
  48. package/src/tools/scrape/unifiedScrape.js +27 -17
  49. package/src/tools/search/providers/searxng.js +5 -1
  50. package/src/tools/search/ranking/ResultDeduplicator.js +9 -1
  51. package/src/tools/search/ranking/ResultRanker.js +17 -2
  52. package/src/tools/search/searchWeb.js +31 -14
  53. package/src/tools/templates/TemplateRegistry.js +7 -1
  54. package/src/tools/tracking/trackChanges/index.js +123 -29
  55. package/src/tools/tracking/trackChanges/schema.js +2 -2
  56. package/src/utils/CircuitBreaker.js +11 -9
  57. package/src/utils/contentUtils.js +66 -53
  58. package/src/utils/secretMask.js +1 -1
  59. package/src/utils/sitemapParser.js +11 -9
  60. package/src/utils/ssrfGuard.js +212 -40
  61. package/src/utils/urlNormalizer.js +2 -2
@@ -3,7 +3,6 @@
3
3
  * Uses multiple NLP libraries for comprehensive content analysis
4
4
  */
5
5
 
6
- import { SummarizerManager } from 'node-summarizer';
7
6
  import { franc, francAll } from 'franc';
8
7
  import nlp from 'compromise';
9
8
  import { z } from 'zod';
@@ -179,7 +178,6 @@ const LANGUAGE_NAMES = {
179
178
 
180
179
  export class ContentAnalyzer {
181
180
  constructor() {
182
- this.summarizer = new SummarizerManager();
183
181
  this.defaultOptions = {
184
182
  summarize: true,
185
183
  detectLanguage: true,
@@ -324,7 +322,7 @@ export class ContentAnalyzer {
324
322
  .map(([code, score]) => ({
325
323
  code,
326
324
  name: LANGUAGE_NAMES[code] || code,
327
- confidence: Math.round((1 - score) * 100) / 100
325
+ confidence: Math.round(score * 100) / 100
328
326
  }));
329
327
 
330
328
  return {
@@ -379,11 +377,15 @@ export class ContentAnalyzer {
379
377
  targetSentences = Math.min(targetSentences, sentences.length);
380
378
 
381
379
  let summarySentences;
382
-
380
+
383
381
  if (options.summaryType === 'extractive') {
384
- // Use node-summarizer for extractive summarization
385
- const summary = await this.summarizer.getSummaryByRanking(text, targetSentences);
386
- summarySentences = splitSentences(summary);
382
+ // Extractive summarization via word-frequency + position scoring
383
+ // (Luhn-style salience), tokenized with compromise. Selects the
384
+ // top-N sentences and restores original document order.
385
+ summarySentences = this.createExtractiveSummary(sentences, targetSentences);
386
+ if (!summarySentences || summarySentences.length === 0) {
387
+ throw new Error('Extractive summarization returned no sentences');
388
+ }
387
389
  } else {
388
390
  // Simple abstractive approach (for demonstration)
389
391
  summarySentences = await this.createAbstractiveSummary(text, targetSentences);
@@ -444,6 +446,49 @@ export class ContentAnalyzer {
444
446
  .map(item => item.sentence.trim());
445
447
  }
446
448
 
449
+ /**
450
+ * Create extractive summary by scoring pre-split sentences via word
451
+ * frequency (Luhn-style salience), tokenized with compromise, plus a small
452
+ * positional bonus for leading/closing sentences. Selects the top N and
453
+ * restores original document order.
454
+ * @param {string[]} sentences - Sentences in original document order
455
+ * @param {number} targetSentences - Number of sentences to select
456
+ * @returns {string[]} - Selected sentences, restored to original order
457
+ */
458
+ createExtractiveSummary(sentences, targetSentences) {
459
+ if (targetSentences >= sentences.length) {
460
+ return sentences.slice();
461
+ }
462
+
463
+ // Word-frequency table (stop words excluded) drives sentence salience.
464
+ const freq = {};
465
+ const sentenceWords = sentences.map(sentence => {
466
+ const words = nlp(sentence).terms().out('array')
467
+ .map(w => w.toLowerCase().replace(/[^a-z0-9]/g, ''))
468
+ .filter(w => w.length > 2 && !this.isStopWord(w));
469
+ words.forEach(w => { freq[w] = (freq[w] || 0) + 1; });
470
+ return words;
471
+ });
472
+
473
+ const maxFreq = Math.max(1, ...Object.values(freq));
474
+
475
+ const scored = sentences.map((sentence, index) => {
476
+ const words = sentenceWords[index];
477
+ const wordScore = words.length > 0
478
+ ? words.reduce((sum, w) => sum + freq[w] / maxFreq, 0) / words.length
479
+ : 0;
480
+ // Leading/closing sentences tend to carry more salience in prose.
481
+ const positionScore = (index === 0 || index === sentences.length - 1) ? 0.15 : 0;
482
+ return { sentence, index, score: wordScore + positionScore };
483
+ });
484
+
485
+ return scored
486
+ .sort((a, b) => b.score - a.score)
487
+ .slice(0, targetSentences)
488
+ .sort((a, b) => a.index - b.index)
489
+ .map(item => item.sentence);
490
+ }
491
+
447
492
  /**
448
493
  * Extract topics from text
449
494
  * @param {string} text - Text to analyze
@@ -168,7 +168,13 @@ export class BFSCrawler {
168
168
  }
169
169
  }
170
170
 
171
- // Mark as visited
171
+ // Mark as visited. Re-check the cap and dedupe first: the checks at the
172
+ // top of this task ran before the awaited robots lookup, so concurrent
173
+ // queue tasks may have filled the budget (or claimed this URL) since.
174
+ // This block is synchronous, so the cap is exact.
175
+ if (this.visited.size >= this.maxPages || this.visited.has(normalizedUrl)) {
176
+ return;
177
+ }
172
178
  this.visited.add(normalizedUrl);
173
179
 
174
180
  try {
@@ -204,11 +210,15 @@ export class BFSCrawler {
204
210
 
205
211
  // Process links for analysis
206
212
  if (this.enableLinkAnalysis && this.linkAnalyzer && pageData.links) {
213
+ // Parse the page once and reuse for every link; re-parsing per link is
214
+ // O(links × page size) and starves the event loop on link-dense pages.
215
+ const $page = pageData.originalHtml ? load(pageData.originalHtml) : null;
216
+ const pageBodyText = $page ? $page('body').text() : '';
207
217
  for (const link of pageData.links) {
208
218
  const absoluteUrl = this.resolveUrl(link, normalizedUrl);
209
219
  if (absoluteUrl) {
210
220
  // Extract anchor text and context from link
211
- const linkMetadata = this.extractLinkMetadata(link, pageData.originalHtml, normalizedUrl);
221
+ const linkMetadata = this.extractLinkMetadata(link, $page, pageBodyText);
212
222
  this.linkAnalyzer.addLink(normalizedUrl, absoluteUrl, linkMetadata);
213
223
  }
214
224
  }
@@ -238,7 +248,19 @@ export class BFSCrawler {
238
248
 
239
249
  const absoluteUrl = this.resolveUrl(link, normalizedUrl);
240
250
  if (absoluteUrl && !this.visited.has(absoluteUrl)) {
241
- await this.queue.add(() => this.processUrl(absoluteUrl, depth + 1));
251
+ // Not awaited: this task already holds a queue slot, so awaiting a
252
+ // child would keep that slot pinned for the rest of the recursive
253
+ // crawl (starving other tasks when concurrency <= depth, and making
254
+ // the per-task queue timeout measure the whole crawl instead of one
255
+ // page). crawl() waits for everything via queue.onIdle() instead.
256
+ this.queue.add(() => this.processUrl(absoluteUrl, depth + 1)).catch(error => {
257
+ this.errors.push({
258
+ url: absoluteUrl,
259
+ depth: depth + 1,
260
+ error: error.message,
261
+ timestamp: new Date().toISOString()
262
+ });
263
+ });
242
264
  }
243
265
  }
244
266
  }
@@ -254,7 +276,7 @@ export class BFSCrawler {
254
276
 
255
277
  async fetchPage(url) {
256
278
  const controller = new AbortController();
257
- const timeoutId = setTimeout(() => controller.abort(), this.timeout);
279
+ let timeoutId = setTimeout(() => controller.abort(), this.timeout);
258
280
 
259
281
  try {
260
282
  // Get domain-specific headers and timeout
@@ -282,7 +304,7 @@ export class BFSCrawler {
282
304
  // Update timeout if different
283
305
  if (effectiveTimeout !== this.timeout) {
284
306
  clearTimeout(timeoutId);
285
- setTimeout(() => controller.abort(), effectiveTimeout);
307
+ timeoutId = setTimeout(() => controller.abort(), effectiveTimeout);
286
308
  }
287
309
 
288
310
  const response = await safeFetch(url, {
@@ -445,6 +467,18 @@ export class BFSCrawler {
445
467
  this.queue.pause();
446
468
  }
447
469
 
470
+ /**
471
+ * Release resources held by this crawler instance (cache cleanup/monitoring
472
+ * timers). Must be called by the owner once crawling is complete — unref()
473
+ * on the timers keeps the process from hanging, but the instance itself
474
+ * stays reachable (and its cached pages retained) until destroy() runs.
475
+ */
476
+ destroy() {
477
+ if (this.cache && typeof this.cache.destroy === 'function') {
478
+ this.cache.destroy();
479
+ }
480
+ }
481
+
448
482
  /**
449
483
  * Get the domain filter instance
450
484
  * @returns {DomainFilter} Current domain filter
@@ -524,19 +558,18 @@ export class BFSCrawler {
524
558
  }
525
559
 
526
560
  /**
527
- * Extract link metadata from HTML
561
+ * Extract link metadata from a pre-parsed page
528
562
  * @param {string} href - The href attribute value
529
- * @param {string} html - Original HTML content
530
- * @param {string} baseUrl - Base URL for context
563
+ * @param {Object} $ - Cheerio document for the page (parsed once per page)
564
+ * @param {string} bodyText - Pre-computed body text of the page
531
565
  * @returns {Object} Link metadata
532
566
  */
533
- extractLinkMetadata(href, html, baseUrl) {
534
- if (!html) return {};
567
+ extractLinkMetadata(href, $, bodyText = '') {
568
+ if (!$) return {};
535
569
 
536
570
  try {
537
- const $ = load(html);
538
571
  const linkElement = $(`a[href="${href}"]`).first();
539
-
572
+
540
573
  if (linkElement.length === 0) {
541
574
  return { href };
542
575
  }
@@ -545,10 +578,8 @@ export class BFSCrawler {
545
578
  const title = linkElement.attr('title');
546
579
  const rel = linkElement.attr('rel');
547
580
  const className = linkElement.attr('class');
548
-
581
+
549
582
  // Get surrounding context (up to 100 characters before and after)
550
- const linkHtml = linkElement.prop('outerHTML');
551
- const bodyText = $('body').text();
552
583
  const linkTextIndex = bodyText.indexOf(anchorText);
553
584
  let context = '';
554
585
 
@@ -201,8 +201,17 @@ export class BrowserProcessor {
201
201
  });
202
202
 
203
203
  } finally {
204
- // Always close the page
204
+ // Always close the page. Also close the owning context for
205
+ // non-stealth pages — createPage() gives each call its own
206
+ // dedicated BrowserContext that is never tracked/closed elsewhere,
207
+ // so leaving it open here leaks it until server shutdown. Stealth
208
+ // contexts are pooled/reused (see StealthBrowserManager /
209
+ // activeContexts) and must not be closed here.
210
+ const ctx = !processingOptions.stealthMode?.enabled ? page.context() : null;
205
211
  await page.close();
212
+ if (ctx) {
213
+ await ctx.close();
214
+ }
206
215
  }
207
216
 
208
217
  return result;
@@ -864,6 +873,15 @@ export class BrowserProcessor {
864
873
  this.humanBehaviorSimulator.resetStats();
865
874
  this.humanBehaviorSimulator = null;
866
875
  }
876
+
877
+ // Tear down the LocalizationManager this processor created — its
878
+ // health-check setInterval timers would otherwise keep firing for the
879
+ // process lifetime after this processor is discarded. The instance is
880
+ // kept (not nulled): it's constructor-created with no lazy re-init, and
881
+ // localizeBrowserContext() must keep working if the processor is reused.
882
+ if (this.localizationManager) {
883
+ await this.localizationManager.cleanup();
884
+ }
867
885
  }
868
886
 
869
887
  /**
@@ -7,6 +7,8 @@
7
7
  import { z } from 'zod';
8
8
  import fs from 'fs/promises';
9
9
  import path from 'path';
10
+ import { safeFetch } from '../../utils/ssrfGuard.js';
11
+ import { config } from '../../constants/config.js';
10
12
 
11
13
  const PDFProcessorSchema = z.object({
12
14
  source: z.string().min(1),
@@ -14,8 +16,9 @@ const PDFProcessorSchema = z.object({
14
16
  options: z.object({
15
17
  extractMetadata: z.boolean().default(true),
16
18
  extractText: z.boolean().default(true),
17
- password: z.string().optional(),
18
19
  maxPages: z.number().min(1).max(1000).default(100),
20
+ // For decrypting password-protected PDFs (pdf-parse 2.x / pdfjs-dist honors this).
21
+ password: z.string().optional(),
19
22
  // C3: true page-range extraction (1-based, inclusive). When set, only the
20
23
  // text from pages [start..end] is returned.
21
24
  pageRange: z.object({
@@ -101,70 +104,79 @@ export class PDFProcessor {
101
104
  return result;
102
105
  }
103
106
 
104
- // C3: when a page range is requested, capture per-page text so we can
105
- // return exactly pages [start..end] (pdf-parse otherwise concatenates the
106
- // whole document and its `max` option only caps the *upper* page bound).
107
+ // C3: page-range extraction (1-based, inclusive) — pdf-parse 2.x's
108
+ // getText({ partial: [...] }) parses and returns exactly the requested
109
+ // pages, so no manual per-page capture is needed here anymore.
107
110
  const pageRange = processingOptions.pageRange;
108
- const capturedPages = [];
109
-
110
- // Parse PDF with options
111
- const parseOptions = {
112
- ...processingOptions.parseOptions,
113
- max: processingOptions.maxPages
114
- };
115
-
116
- // If extracting a range, raise `max` to at least the requested end page
117
- // and install a pagerender that records each page's text.
118
- if (pageRange) {
119
- if (pageRange.end) {
120
- parseOptions.max = Math.max(parseOptions.max, pageRange.end);
121
- } else {
122
- parseOptions.max = processingOptions.maxPages;
123
- }
124
- parseOptions.pagerender = (pageData) => this._renderPage(pageData, capturedPages);
125
- }
126
111
 
127
- if (processingOptions.password) {
128
- parseOptions.password = processingOptions.password;
129
- }
112
+ // Dynamic import to avoid initialization issues
113
+ const { PDFParse, PasswordException } = await import('pdf-parse');
114
+ const parser = new PDFParse({
115
+ data: pdfBuffer,
116
+ ...(processingOptions.password ? { password: processingOptions.password } : {})
117
+ });
130
118
 
131
- let pdfData;
132
119
  try {
133
- // Dynamic import to avoid initialization issues
134
- const pdfParse = (await import('pdf-parse')).default;
135
- pdfData = await pdfParse(pdfBuffer, parseOptions);
136
- } catch (error) {
137
- result.error = `PDF parsing failed: ${error.message}`;
138
- result.processingTime = Date.now() - startTime;
139
- return result;
140
- }
120
+ let info;
121
+ try {
122
+ info = await parser.getInfo();
123
+ } catch (error) {
124
+ if (error instanceof PasswordException) {
125
+ result.error = `PDF parsing failed: password required or incorrect (${error.message})`;
126
+ } else {
127
+ result.error = `PDF parsing failed: ${error.message}`;
128
+ }
129
+ result.processingTime = Date.now() - startTime;
130
+ return result;
131
+ }
141
132
 
142
- // Extract text content
143
- if (processingOptions.extractText) {
144
- if (pageRange) {
145
- const start = pageRange.start || 1;
146
- const end = pageRange.end || capturedPages.length;
147
- const slice = capturedPages.slice(start - 1, end);
148
- result.text = this.cleanPDFText(slice.join('\n\n'));
149
- result.extractedPages = { start, end, count: slice.length };
150
- } else {
151
- result.text = this.cleanPDFText(pdfData.text);
133
+ const totalPages = info.total || 0;
134
+ // pdf-parse 2.x has no equivalent of v1's disableCombineTextItems; only
135
+ // normalizeWhitespace maps onto a v2 ParseParameters field (inverted).
136
+ const disableNormalization = !processingOptions.parseOptions?.normalizeWhitespace;
137
+
138
+ // Extract text content
139
+ if (processingOptions.extractText) {
140
+ if (pageRange) {
141
+ const start = pageRange.start || 1;
142
+ // C3: a start past the last page means the requested range
143
+ // doesn't exist in this PDF — report that explicitly instead of
144
+ // silently returning success:true with empty text.
145
+ if (start > totalPages) {
146
+ result.error = `Requested page range starts at page ${start}, but the PDF only has ${totalPages} page(s).`;
147
+ result.processingTime = Date.now() - startTime;
148
+ return result;
149
+ }
150
+ const end = Math.min(pageRange.end || processingOptions.maxPages, totalPages);
151
+ const pageNumbers = [];
152
+ for (let n = start; n <= end; n++) pageNumbers.push(n);
153
+
154
+ const textResult = await parser.getText({ partial: pageNumbers, disableNormalization });
155
+ const slice = textResult.pages.map(p => p.text);
156
+ result.text = this.cleanPDFText(slice.join('\n\n'));
157
+ result.extractedPages = { start, end, count: slice.length };
158
+ } else {
159
+ const textResult = await parser.getText({ first: processingOptions.maxPages, disableNormalization });
160
+ result.text = this.cleanPDFText(textResult.pages.map(p => p.text).join('\n\n'));
161
+ }
152
162
  }
153
- }
154
163
 
155
- // Extract metadata
156
- if (processingOptions.extractMetadata) {
157
- result.metadata = this.extractPDFMetadata(pdfData);
158
- }
164
+ // Extract metadata
165
+ if (processingOptions.extractMetadata) {
166
+ result.metadata = this.extractPDFMetadata(info);
167
+ }
159
168
 
160
- // Set page count
161
- result.pageCount = pdfData.numpages || 0;
169
+ // Set page count
170
+ result.pageCount = totalPages;
162
171
 
163
- // Calculate processing time
164
- result.processingTime = Date.now() - startTime;
165
- result.success = true;
172
+ // Calculate processing time
173
+ result.processingTime = Date.now() - startTime;
174
+ result.success = true;
166
175
 
167
- return result;
176
+ return result;
177
+ } finally {
178
+ await parser.destroy().catch(() => {});
179
+ }
168
180
 
169
181
  } catch (error) {
170
182
  return {
@@ -205,11 +217,14 @@ export class PDFProcessor {
205
217
  */
206
218
  async downloadPDFFromURL(url) {
207
219
  try {
208
- const response = await fetch(url, {
220
+ // `timeout` is not a fetch init option — undici/Node fetch silently
221
+ // ignores unknown properties, so only `signal` actually enforces a
222
+ // deadline here.
223
+ const response = await safeFetch(url, {
209
224
  headers: {
210
225
  'User-Agent': 'Mozilla/5.0 (compatible; MCP-WebScraper/2.0; PDF-Processor)'
211
226
  },
212
- timeout: 30000
227
+ signal: AbortSignal.timeout(30000)
213
228
  });
214
229
 
215
230
  if (!response.ok) {
@@ -221,14 +236,61 @@ export class PDFProcessor {
221
236
  console.warn(`Warning: Content-Type is ${contentType}, expected PDF`);
222
237
  }
223
238
 
224
- const arrayBuffer = await response.arrayBuffer();
225
- return Buffer.from(arrayBuffer);
239
+ return await this.readBodyWithSizeCap(response);
226
240
 
227
241
  } catch (error) {
242
+ // AbortSignal.timeout() aborts with a TimeoutError-named DOMException
243
+ // (not AbortError — that name is only used for a plain controller.abort()).
244
+ if (error.name === 'TimeoutError' || error.name === 'AbortError') {
245
+ throw new Error('Failed to download PDF from URL: request timeout after 30000ms');
246
+ }
228
247
  throw new Error(`Failed to download PDF from URL: ${error.message}`);
229
248
  }
230
249
  }
231
250
 
251
+ /**
252
+ * Read a response body into a Buffer while enforcing config.fetch.maxBodySize
253
+ * (Content-Length pre-check, then a streaming byte-count backstop for
254
+ * servers that omit or lie about it), so a stalling or multi-GB response
255
+ * can't hang the request indefinitely or OOM the process before pdf-parse's
256
+ * own maxPages cap ever applies.
257
+ * @param {Response} response
258
+ * @returns {Promise<Buffer>}
259
+ */
260
+ async readBodyWithSizeCap(response) {
261
+ const maxBodySize = config.fetch.maxBodySize;
262
+
263
+ const contentLengthHeader = response.headers?.get?.('content-length') ?? null;
264
+ if (contentLengthHeader !== null) {
265
+ const declared = parseInt(contentLengthHeader, 10);
266
+ if (!isNaN(declared) && declared > maxBodySize) {
267
+ throw new Error(`PDF too large: Content-Length ${declared} exceeds limit of ${maxBodySize} bytes`);
268
+ }
269
+ }
270
+
271
+ if (!response.body || typeof response.body.getReader !== 'function') {
272
+ const arrayBuffer = await response.arrayBuffer();
273
+ return Buffer.from(arrayBuffer);
274
+ }
275
+
276
+ const reader = response.body.getReader();
277
+ const chunks = [];
278
+ let totalBytes = 0;
279
+
280
+ while (true) {
281
+ const { done, value } = await reader.read();
282
+ if (done) break;
283
+ totalBytes += value.byteLength;
284
+ if (totalBytes > maxBodySize) {
285
+ reader.cancel();
286
+ throw new Error(`PDF too large: exceeded limit of ${maxBodySize} bytes`);
287
+ }
288
+ chunks.push(value);
289
+ }
290
+
291
+ return Buffer.concat(chunks, totalBytes);
292
+ }
293
+
232
294
  /**
233
295
  * Read PDF from local file
234
296
  * @param {string} filePath - Local file path
@@ -260,12 +322,12 @@ export class PDFProcessor {
260
322
 
261
323
  /**
262
324
  * Extract and format PDF metadata
263
- * @param {Object} pdfData - Parsed PDF data from pdf-parse
325
+ * @param {Object} infoResult - InfoResult from pdf-parse's PDFParse#getInfo()
264
326
  * @returns {Object} - Formatted metadata
265
327
  */
266
- extractPDFMetadata(pdfData) {
267
- const info = pdfData.info || {};
268
- const metadata = pdfData.metadata || {};
328
+ extractPDFMetadata(infoResult) {
329
+ const info = infoResult.info || {};
330
+ const metadata = infoResult.metadata || {};
269
331
 
270
332
  return {
271
333
  title: this.cleanMetadataValue(info.Title || metadata.title),
@@ -276,8 +338,10 @@ export class PDFProcessor {
276
338
  creationDate: this.formatPDFDate(info.CreationDate || metadata.creationDate),
277
339
  modificationDate: this.formatPDFDate(info.ModDate || metadata.modificationDate),
278
340
  format: this.cleanMetadataValue(info.Format || metadata.format),
279
- pages: pdfData.numpages || null,
280
- encrypted: info.IsEncrypted || false,
341
+ pages: infoResult.total || null,
342
+ // pdfjs-dist's Info dictionary has no `IsEncrypted` flag; it reports the
343
+ // security filter name (e.g. "Standard") when the doc is encrypted, null otherwise.
344
+ encrypted: !!info.EncryptFilterName,
281
345
  linearized: info.IsLinearized || false,
282
346
  pdfVersion: this.cleanMetadataValue(info.PDFFormatVersion || metadata.pdfVersion)
283
347
  };
@@ -9,12 +9,13 @@ export class QueueManager {
9
9
  timeout = 30000
10
10
  } = options;
11
11
 
12
+ // p-queue v9 removed `throwOnTimeout` — timeouts always throw now
13
+ // (this was already the effective behavior since it was set to true).
12
14
  this.queue = new PQueue({
13
15
  concurrency,
14
16
  interval,
15
17
  intervalCap,
16
- timeout,
17
- throwOnTimeout: true
18
+ timeout
18
19
  });
19
20
 
20
21
  this.stats = {