crawlforge-mcp-server 4.10.0 → 5.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CLAUDE.md +6 -5
  2. package/README.md +19 -3
  3. package/package.json +10 -12
  4. package/server.js +298 -212
  5. package/src/cli/commands/init.js +11 -5
  6. package/src/cli/commands/stealth.js +3 -2
  7. package/src/core/ActionExecutor.js +117 -33
  8. package/src/core/AgentOrchestrator.js +8 -2
  9. package/src/core/AuthManager.js +51 -17
  10. package/src/core/ChangeTracker.js +26 -10
  11. package/src/core/JobManager.js +9 -1
  12. package/src/core/LocalizationManager.js +19 -6
  13. package/src/core/MonitorScheduler.js +13 -4
  14. package/src/core/ResearchOrchestrator.js +173 -35
  15. package/src/core/SnapshotManager.js +167 -165
  16. package/src/core/StealthBrowserManager.js +25 -3
  17. package/src/core/WebhookDispatcher.js +19 -14
  18. package/src/core/analysis/ContentAnalyzer.js +52 -7
  19. package/src/core/crawlers/BFSCrawler.js +46 -15
  20. package/src/core/processing/BrowserProcessor.js +19 -1
  21. package/src/core/processing/PDFProcessor.js +129 -65
  22. package/src/core/queue/QueueManager.js +3 -2
  23. package/src/schemas/toolOutputSchemas.js +269 -0
  24. package/src/server/auth/oauth.js +37 -7
  25. package/src/server/specHygiene.js +192 -0
  26. package/src/server/taskSupport.js +233 -0
  27. package/src/server/toolFilter.js +98 -0
  28. package/src/server/transports/streamableHttp.js +148 -11
  29. package/src/server/withAuth.js +11 -4
  30. package/src/tools/advanced/ScrapeWithActionsTool.js +43 -52
  31. package/src/tools/advanced/batchScrape/index.js +128 -27
  32. package/src/tools/advanced/batchScrape/worker.js +55 -5
  33. package/src/tools/advanced/scrapeWithActions/recorder.js +3 -0
  34. package/src/tools/basic/_fetch.js +125 -70
  35. package/src/tools/basic/extractLinks.js +14 -12
  36. package/src/tools/basic/scrapeStructured.js +21 -4
  37. package/src/tools/crawl/crawlDeep.js +110 -48
  38. package/src/tools/crawl/mapSite.js +25 -6
  39. package/src/tools/extract/_fetchAndParse.js +98 -1
  40. package/src/tools/extract/extractContent.js +7 -4
  41. package/src/tools/extract/extractStructured.js +125 -84
  42. package/src/tools/extract/extractWithLlm.js +10 -2
  43. package/src/tools/extract/processDocument.js +54 -6
  44. package/src/tools/extract/summarizeContent.js +7 -1
  45. package/src/tools/llmstxt/generateLLMsTxt.js +11 -6
  46. package/src/tools/research/deepResearch.js +51 -31
  47. package/src/tools/scrape/_brandingExtractor.js +49 -11
  48. package/src/tools/scrape/unifiedScrape.js +27 -17
  49. package/src/tools/search/providers/searxng.js +5 -1
  50. package/src/tools/search/ranking/ResultDeduplicator.js +9 -1
  51. package/src/tools/search/ranking/ResultRanker.js +17 -2
  52. package/src/tools/search/searchWeb.js +31 -14
  53. package/src/tools/templates/TemplateRegistry.js +7 -1
  54. package/src/tools/tracking/trackChanges/index.js +123 -29
  55. package/src/tools/tracking/trackChanges/schema.js +2 -2
  56. package/src/utils/CircuitBreaker.js +11 -9
  57. package/src/utils/contentUtils.js +66 -53
  58. package/src/utils/secretMask.js +1 -1
  59. package/src/utils/sitemapParser.js +11 -9
  60. package/src/utils/ssrfGuard.js +212 -40
  61. package/src/utils/urlNormalizer.js +2 -2
@@ -14,6 +14,8 @@
14
14
  */
15
15
 
16
16
  import { EventEmitter } from 'events';
17
+ import os from 'os';
18
+ import path from 'path';
17
19
  import ChangeTracker from '../../../core/ChangeTracker.js';
18
20
  import SnapshotManager from '../../../core/SnapshotManager.js';
19
21
  import CacheManager from '../../../core/cache/CacheManager.js';
@@ -31,7 +33,12 @@ export class TrackChangesTool extends EventEmitter {
31
33
  this.options = {
32
34
  cacheEnabled: true,
33
35
  cacheTTL: 3600000,
34
- snapshotStorageDir: './snapshots',
36
+ // Rooted in a stable, non-cwd-dependent base (~/.crawlforge — same
37
+ // convention as ~/.crawlforge/config.json) rather than process.cwd(),
38
+ // since MCP clients (e.g. Claude Desktop) may launch the server with a
39
+ // cwd the process cannot write to (e.g. '/'), which previously made
40
+ // every snapshot write fail silently.
41
+ snapshotStorageDir: path.join(os.homedir(), '.crawlforge', 'snapshots'),
35
42
  enableRealTimeMonitoring: true,
36
43
  maxConcurrentMonitors: 50,
37
44
  defaultPollingInterval: 300000,
@@ -64,7 +71,15 @@ export class TrackChangesTool extends EventEmitter {
64
71
  this.monitorStore = new MonitorStore({ storageDir: this.options.monitorStorageDir || './monitors' });
65
72
  this.scheduler = new MonitorScheduler({ tool: this, store: this.monitorStore });
66
73
 
67
- this.initialize();
74
+ // Wired synchronously (no I/O) so no 'error' event emitted by
75
+ // changeTracker/snapshotManager during the lazy initialize() below (see
76
+ // ensureInitialized()) can ever be emitted before a listener exists.
77
+ this._setupEventHandlers();
78
+
79
+ // Not started here — construction stays synchronous and side-effect
80
+ // free. ensureInitialized() lazily creates and memoizes this on first
81
+ // real use (called from execute()/startScheduler()/runDueOnce()).
82
+ this._initPromise = null;
68
83
  }
69
84
 
70
85
  /** Wire the MCP server so the goal-judge can use SamplingClient (Ollama-first). */
@@ -74,6 +89,7 @@ export class TrackChangesTool extends EventEmitter {
74
89
 
75
90
  /** Start the in-process scheduler (called once, by the server). */
76
91
  async startScheduler() {
92
+ await this.ensureInitialized();
77
93
  if (this._mcpServer && !this.scheduler.samplingClient) {
78
94
  try {
79
95
  const { SamplingClient } = await import('../../../core/SamplingClient.js');
@@ -87,6 +103,7 @@ export class TrackChangesTool extends EventEmitter {
87
103
 
88
104
  /** Fire every due monitor once and exit (the external-cron one-shot path). */
89
105
  async runDueOnce() {
106
+ await this.ensureInitialized();
90
107
  if (this._mcpServer && !this.scheduler.samplingClient) {
91
108
  try {
92
109
  const { SamplingClient } = await import('../../../core/SamplingClient.js');
@@ -96,24 +113,33 @@ export class TrackChangesTool extends EventEmitter {
96
113
  return this.scheduler.runDueOnce();
97
114
  }
98
115
 
99
- async initialize() {
100
- try {
101
- await this.snapshotManager.initialize();
102
- this._setupEventHandlers();
103
- this.emit('initialized');
104
- } catch (error) {
105
- this.emit('error', { operation: 'initialize', error: error.message });
106
- throw error;
116
+ /**
117
+ * Lazily runs (and memoizes) storage initialization, awaited by execute()
118
+ * and the other top-level entry points below. Replaces the old pattern of
119
+ * firing initialize() unawaited from the constructor, which could turn a
120
+ * snapshot-directory failure into an opaque unhandled rejection.
121
+ */
122
+ async ensureInitialized() {
123
+ if (!this._initPromise) {
124
+ this._initPromise = this.snapshotManager.ensureInitialized().then(() => {
125
+ this.emit('initialized');
126
+ });
107
127
  }
128
+ return this._initPromise;
108
129
  }
109
130
 
110
131
  _setupEventHandlers() {
111
132
  this.changeTracker.on('changeDetected', async (changeRecord) => {
112
- if (changeRecord.significance !== 'none') {
133
+ // changeRecord.details is diff analysis and carries no page content —
134
+ // storing `details.current || ''` wrote an empty junk snapshot on every
135
+ // significant change (compareWithBaseline already snapshots the real
136
+ // current content). Only store if a future record ever carries content.
137
+ const current = changeRecord.details?.current;
138
+ if (changeRecord.significance !== 'none' && current) {
113
139
  try {
114
140
  await this.snapshotManager.storeSnapshot(
115
141
  changeRecord.url,
116
- changeRecord.details.current || '',
142
+ current,
117
143
  { changes: changeRecord.details, significance: changeRecord.significance, changeType: changeRecord.changeType }
118
144
  );
119
145
  } catch (error) {
@@ -129,6 +155,8 @@ export class TrackChangesTool extends EventEmitter {
129
155
 
130
156
  async execute(params) {
131
157
  try {
158
+ await this.ensureInitialized();
159
+
132
160
  const validated = TrackChangesSchema.parse(params);
133
161
  const { operation } = validated;
134
162
 
@@ -169,7 +197,7 @@ export class TrackChangesTool extends EventEmitter {
169
197
  const baseline = await this.changeTracker.createBaseline(url, sourceContent, trackingOptions);
170
198
  let snapshotInfo = null;
171
199
  if (enableSnapshots) {
172
- snapshotInfo = await this.snapshotManager.storeSnapshot(url, sourceContent, { ...fetchMeta, baseline: true, trackingOptions });
200
+ snapshotInfo = await this.snapshotManager.storeSnapshot(url, sourceContent, { ...fetchMeta, baseline: true, trackingOptions }, { enableCompression: storageOptions.compressionEnabled });
173
201
  }
174
202
 
175
203
  return {
@@ -186,9 +214,34 @@ export class TrackChangesTool extends EventEmitter {
186
214
  };
187
215
  }
188
216
 
217
+ // Rebuild the in-memory baseline from the newest persisted snapshot, so
218
+ // compare works across processes (fresh CLI runs, server restarts). Same
219
+ // fail-soft mechanism as MonitorScheduler._ensureBaseline: no usable
220
+ // snapshot means compare still reports "No baseline" for a genuine first run.
221
+ async rehydrateBaseline(url, trackingOptions = {}) {
222
+ if (this.changeTracker?.snapshots?.has(url)) return;
223
+ try {
224
+ // limit > 1: existing stores contain empty junk snapshots (from the old
225
+ // changeDetected handler) that can tie on timestamp with the real one —
226
+ // take the newest snapshot that actually has content.
227
+ const q = await this.snapshotManager.querySnapshots({ url, limit: 5, includeContent: true });
228
+ for (const snap of q?.snapshots ?? []) {
229
+ let content = snap?.content;
230
+ if (Buffer.isBuffer(content)) content = content.toString('utf8');
231
+ if (content && typeof content === 'string') {
232
+ await this.changeTracker.createBaseline(url, content, trackingOptions);
233
+ return;
234
+ }
235
+ }
236
+ } catch {
237
+ /* no usable snapshot — caller's compare will surface "No baseline" */
238
+ }
239
+ }
240
+
189
241
  async compareWithBaseline(params) {
190
242
  const { url, content, html, trackingOptions, storageOptions = {}, notificationOptions } = params;
191
243
  const enableSnapshots = storageOptions.enableSnapshots !== false;
244
+ await this.rehydrateBaseline(url, trackingOptions);
192
245
 
193
246
  let currentContent = content || html;
194
247
  let fetchMeta = {};
@@ -199,13 +252,13 @@ export class TrackChangesTool extends EventEmitter {
199
252
  }
200
253
  if (!currentContent || typeof currentContent !== 'string') throw new Error('Invalid content');
201
254
 
202
- const comparisonResult = await this.changeTracker.compareWithBaseline(url, currentContent, trackingOptions);
255
+ const comparisonResult = await this.changeTracker.compareWithBaseline(url, currentContent, trackingOptions, storageOptions);
203
256
 
204
257
  let snapshotInfo = null;
205
258
  if (comparisonResult.hasChanges && enableSnapshots) {
206
259
  snapshotInfo = await this.snapshotManager.storeSnapshot(url, currentContent, {
207
260
  ...fetchMeta, changes: comparisonResult.summary, significance: comparisonResult.significance
208
- });
261
+ }, { enableCompression: storageOptions.compressionEnabled });
209
262
  }
210
263
 
211
264
  if (comparisonResult.hasChanges && notificationOptions) {
@@ -339,7 +392,10 @@ export class TrackChangesTool extends EventEmitter {
339
392
  const monitorId = scheduledMonitorOptions?.monitorId;
340
393
  if (monitorId) {
341
394
  const result = await this.scheduler.stopMonitor(monitorId);
342
- return { success: true, operation: 'stop_scheduled_monitor', monitorId, stopped: result.stopped, timestamp: Date.now() };
395
+ if (!result.stopped) {
396
+ return { success: false, operation: 'stop_scheduled_monitor', monitorId, stopped: false, error: `No scheduled monitor found with id ${monitorId}`, timestamp: Date.now() };
397
+ }
398
+ return { success: true, operation: 'stop_scheduled_monitor', monitorId, stopped: true, timestamp: Date.now() };
343
399
  }
344
400
  if (!url) throw new Error('stop_scheduled_monitor requires a url or scheduledMonitorOptions.monitorId');
345
401
  const result = await this.scheduler.stopByUrl(url);
@@ -421,6 +477,17 @@ export class TrackChangesTool extends EventEmitter {
421
477
  }
422
478
 
423
479
  async shutdown() {
480
+ // ensureInitialized() is lazy (see above) and awaits
481
+ // this.snapshotManager.ensureInitialized(), which is itself lazy. If a
482
+ // caller triggered initialization and then immediately calls shutdown(),
483
+ // we must wait for that in-flight init to finish first — otherwise its
484
+ // cleanup timer could start after snapshotManager.shutdown() already
485
+ // tried to stop it, leaking a live timer. If nothing ever triggered
486
+ // initialization, _initPromise is still null and there's nothing to
487
+ // wait for.
488
+ if (this._initPromise) {
489
+ await this._initPromise.catch(() => {});
490
+ }
424
491
  this.stopAllMonitoring();
425
492
  this.scheduler?.stopAll();
426
493
  await this.snapshotManager.shutdown();
@@ -469,17 +536,44 @@ export class TrackChangesTool extends EventEmitter {
469
536
 
470
537
  export default TrackChangesTool;
471
538
 
472
- // Singleton instance — kept for backward-compat with any code that imports it directly
473
- export const trackChangesTool = new TrackChangesTool();
474
- trackChangesTool.name = 'track_changes';
475
- trackChangesTool.validateParameters = (params) => TrackChangesSchema.parse(params);
476
- trackChangesTool.description = 'Track and analyze content changes with baseline capture, comparison, and monitoring capabilities';
477
- trackChangesTool.inputSchema = {
478
- type: 'object',
479
- properties: {
480
- url: { type: 'string', description: 'URL to track for changes' },
481
- operation: { type: 'string', description: 'Operation to perform: create_baseline, compare, monitor, get_history, get_stats' },
482
- content: { type: 'string', description: 'Content to analyze or compare' }
539
+ // Singleton instance — kept for backward-compat with any code that imports it
540
+ // directly. Built lazily, on first property access, rather than eagerly at
541
+ // module-import time: the server (server.js) constructs and owns its own
542
+ // TrackChangesTool instance, so an eager `new TrackChangesTool()` here spun
543
+ // up a second, never-shut-down ChangeTracker/SnapshotManager/CacheManager/
544
+ // MonitorStore/MonitorScheduler (and touched the filesystem) merely by
545
+ // importing this module. Nothing in this codebase currently imports this
546
+ // named export directly, but it's preserved for backward-compat.
547
+ let _singleton = null;
548
+ function _getTrackChangesToolSingleton() {
549
+ if (!_singleton) {
550
+ _singleton = new TrackChangesTool();
551
+ _singleton.name = 'track_changes';
552
+ _singleton.validateParameters = (params) => TrackChangesSchema.parse(params);
553
+ _singleton.description = 'Track and analyze content changes with baseline capture, comparison, and monitoring capabilities';
554
+ _singleton.inputSchema = {
555
+ type: 'object',
556
+ properties: {
557
+ url: { type: 'string', description: 'URL to track for changes' },
558
+ operation: { type: 'string', description: 'Operation to perform: create_baseline, compare, monitor, get_history, get_stats' },
559
+ content: { type: 'string', description: 'Content to analyze or compare' }
560
+ },
561
+ required: ['url']
562
+ };
563
+ }
564
+ return _singleton;
565
+ }
566
+
567
+ export const trackChangesTool = new Proxy({}, {
568
+ get(_target, prop) {
569
+ const instance = _getTrackChangesToolSingleton();
570
+ const value = Reflect.get(instance, prop, instance);
571
+ return typeof value === 'function' ? value.bind(instance) : value;
483
572
  },
484
- required: ['url']
485
- };
573
+ set(_target, prop, value) {
574
+ return Reflect.set(_getTrackChangesToolSingleton(), prop, value);
575
+ },
576
+ has(_target, prop) {
577
+ return Reflect.has(_getTrackChangesToolSingleton(), prop);
578
+ }
579
+ });
@@ -56,7 +56,7 @@ export const TrackChangesSchema = z.object({
56
56
  enableWebhook: z.boolean().default(false),
57
57
  webhookUrl: z.string().url().optional(),
58
58
  webhookSecret: z.string().optional()
59
- }).optional(),
59
+ }).optional().default({}),
60
60
 
61
61
  storageOptions: z.object({
62
62
  enableSnapshots: z.boolean().default(true),
@@ -73,7 +73,7 @@ export const TrackChangesSchema = z.object({
73
73
  endTime: z.number().optional(),
74
74
  includeContent: z.boolean().default(false),
75
75
  significanceFilter: z.enum(['all', 'minor', 'moderate', 'major', 'critical']).optional()
76
- }).optional(),
76
+ }).optional().default({}),
77
77
 
78
78
  notificationOptions: z.object({
79
79
  email: z.object({
@@ -26,9 +26,11 @@ export class CircuitBreaker {
26
26
  this.monitoringWindow = monitoringWindow;
27
27
  this.errorThresholdPercentage = errorThresholdPercentage;
28
28
  this.minimumThroughput = minimumThroughput;
29
- this.onStateChange = onStateChange;
30
- this.onFailure = onFailure;
31
- this.onSuccess = onSuccess;
29
+ // Stored under distinct names so they don't shadow the onStateChange/
30
+ // onFailure/onSuccess prototype methods below (execute() calls those).
31
+ this.onStateChangeCallback = onStateChange;
32
+ this.onFailureCallback = onFailure;
33
+ this.onSuccessCallback = onSuccess;
32
34
  this.name = name;
33
35
 
34
36
  // Circuit state per service endpoint
@@ -138,8 +140,8 @@ export class CircuitBreaker {
138
140
  }
139
141
 
140
142
  // Call success callback
141
- if (this.onSuccess) {
142
- this.onSuccess(serviceId, duration);
143
+ if (this.onSuccessCallback) {
144
+ this.onSuccessCallback(serviceId, duration);
143
145
  }
144
146
  }
145
147
 
@@ -166,8 +168,8 @@ export class CircuitBreaker {
166
168
  }
167
169
 
168
170
  // Call failure callback
169
- if (this.onFailure) {
170
- this.onFailure(serviceId, error, duration);
171
+ if (this.onFailureCallback) {
172
+ this.onFailureCallback(serviceId, error, duration);
171
173
  }
172
174
  }
173
175
 
@@ -221,8 +223,8 @@ export class CircuitBreaker {
221
223
  }
222
224
 
223
225
  // Call state change callback
224
- if (this.onStateChange) {
225
- this.onStateChange(serviceId, oldState, newState, circuit);
226
+ if (this.onStateChangeCallback) {
227
+ this.onStateChangeCallback(serviceId, oldState, newState, circuit);
226
228
  }
227
229
 
228
230
  // Start health monitoring for open circuits
@@ -93,61 +93,74 @@ export class HTMLCleaner {
93
93
 
94
94
  let text = '';
95
95
 
96
- $('body').find('*').each((_, element) => {
97
- const $element = $(element);
98
- const tagName = element.tagName.toLowerCase();
99
-
100
- switch (tagName) {
101
- case 'p':
102
- case 'div':
103
- if (extractOptions.preserveParagraphs) {
104
- text += '\n\n' + $element.text().trim();
105
- } else {
106
- text += ' ' + $element.text().trim();
107
- }
108
- break;
109
- case 'br':
110
- if (extractOptions.preserveLineBreaks) {
111
- text += '\n';
112
- }
113
- break;
114
- case 'h1':
115
- case 'h2':
116
- case 'h3':
117
- case 'h4':
118
- case 'h5':
119
- case 'h6':
120
- text += '\n\n' + $element.text().trim().toUpperCase() + '\n';
121
- break;
122
- case 'a':
123
- if (extractOptions.includeLinks) {
124
- const href = $element.attr('href');
125
- const linkText = $element.text().trim();
126
- text += ` ${linkText}${href ? ` (${href})` : ''}`;
127
- } else {
128
- text += ' ' + $element.text().trim();
129
- }
130
- break;
131
- case 'img':
132
- if (extractOptions.includeImageAlt) {
133
- const alt = $element.attr('alt');
134
- if (alt) {
135
- text += ` [Image: ${alt}]`;
96
+ // Walk direct children recursively (rather than $('body').find('*'),
97
+ // which flattens every descendant) so a block element's text isn't
98
+ // captured once via $element.text() (which includes nested content) and
99
+ // then again when the walker separately visits its nested elements.
100
+ function walk($el) {
101
+ $el.contents().each((_, node) => {
102
+ if (node.type === 'text') {
103
+ const value = (node.data || '').trim();
104
+ if (value) text += ' ' + value;
105
+ return;
106
+ }
107
+ if (node.type !== 'tag') return;
108
+
109
+ const $node = $(node);
110
+ const tagName = node.tagName.toLowerCase();
111
+
112
+ switch (tagName) {
113
+ case 'p':
114
+ case 'div':
115
+ text += extractOptions.preserveParagraphs ? '\n\n' : ' ';
116
+ walk($node);
117
+ break;
118
+ case 'br':
119
+ if (extractOptions.preserveLineBreaks) {
120
+ text += '\n';
136
121
  }
137
- }
138
- break;
139
- case 'li':
140
- text += '\n• ' + $element.text().trim();
141
- break;
142
- default:
143
- // For other elements, just extract text
144
- if ($element.children().length === 0) {
145
- text += ' ' + $element.text().trim();
146
- }
147
- }
148
- });
122
+ break;
123
+ case 'h1':
124
+ case 'h2':
125
+ case 'h3':
126
+ case 'h4':
127
+ case 'h5':
128
+ case 'h6':
129
+ text += '\n\n' + $node.text().trim().toUpperCase() + '\n';
130
+ break;
131
+ case 'a':
132
+ if (extractOptions.includeLinks) {
133
+ const href = $node.attr('href');
134
+ const linkText = $node.text().trim();
135
+ text += ` ${linkText}${href ? ` (${href})` : ''}`;
136
+ } else {
137
+ walk($node);
138
+ }
139
+ break;
140
+ case 'img':
141
+ if (extractOptions.includeImageAlt) {
142
+ const alt = $node.attr('alt');
143
+ if (alt) {
144
+ text += ` [Image: ${alt}]`;
145
+ }
146
+ }
147
+ break;
148
+ case 'li':
149
+ text += '\n• ';
150
+ walk($node);
151
+ break;
152
+ default:
153
+ walk($node);
154
+ }
155
+ });
156
+ }
157
+
158
+ walk($('body'));
149
159
 
150
- return text.replace(/\s+/g, ' ').replace(/\n\s+/g, '\n').trim();
160
+ // Collapse horizontal whitespace only — collapsing all whitespace
161
+ // (including newlines) here would erase the line breaks/paragraphs the
162
+ // options above were just asked to preserve.
163
+ return text.replace(/[ \t]+/g, ' ').replace(/[ \t]*\n[ \t]*/g, '\n').trim();
151
164
  }
152
165
  }
153
166
 
@@ -6,7 +6,7 @@
6
6
  * logger.error('fetch failed', maskSecrets({ apiKey, url, error }));
7
7
  */
8
8
 
9
- const SECRET_KEYS_RE = /api[_-]?key|apikey|x-api-key|password|passwd|secret|token|authorization|auth|credential|private[_-]?key|access[_-]?key|proxy_url|proxyurl/i;
9
+ const SECRET_KEYS_RE = /api[_-]?key|apikey|x-api-key|password|passwd|secret|token|authorization|auth|credential|private[_-]?key|access[_-]?key|proxy_url|proxyurl|cookie/i;
10
10
 
11
11
  const MASK = '[REDACTED]';
12
12
  const PARTIAL_MASK_LEN = 4; // show last N chars of long secrets
@@ -392,19 +392,21 @@ export class SitemapParser {
392
392
  return null;
393
393
  }
394
394
 
395
- const contentType = response.headers.get('content-type') || '';
396
- const contentEncoding = response.headers.get('content-encoding') || '';
397
-
395
+ // fetch (undici) transparently decompresses a gzip Content-Encoding
396
+ // while leaving the response header intact, so that header can't be
397
+ // trusted to decide whether the body still needs gunzipping. Sniff the
398
+ // actual bytes instead: a real gzip payload starts with the 0x1f 0x8b
399
+ // magic number regardless of what the headers claim.
400
+ const buffer = Buffer.from(await response.arrayBuffer());
401
+ const isGzipped = buffer.length >= 2 && buffer[0] === 0x1f && buffer[1] === 0x8b;
402
+
398
403
  let content;
399
-
400
- // Handle compressed content
401
- if (url.endsWith('.gz') || contentEncoding.includes('gzip')) {
402
- const buffer = await response.arrayBuffer();
403
- const decompressed = await gunzip(Buffer.from(buffer));
404
+ if (isGzipped) {
405
+ const decompressed = await gunzip(buffer);
404
406
  content = decompressed.toString('utf8');
405
407
  this.stats.compressionSavings += buffer.byteLength - decompressed.length;
406
408
  } else {
407
- content = await response.text();
409
+ content = buffer.toString('utf8');
408
410
  }
409
411
 
410
412
  return content;