crawlforge-extractors 1.6.3 → 1.6.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "crawlforge-extractors",
3
- "version": "1.6.3",
3
+ "version": "1.6.4",
4
4
  "description": "Extraction logic shared by the CrawlForge MCP server and REST API — scrape templates, charset-correct capped body reading, structural fingerprinting, and embedded-state extraction. One implementation, so the two surfaces cannot drift apart.",
5
5
  "type": "module",
6
6
  "main": "./index.js",
package/src/body.js CHANGED
@@ -26,6 +26,8 @@ export class BodyTooLargeError extends Error {
26
26
  * @param {Uint8Array} bytes
27
27
  * @returns {string}
28
28
  */
29
+ export const META_CHARSET_SNIFF_BYTES = 8192;
30
+
29
31
  export function detectCharset(response, bytes) {
30
32
  const contentType = response.headers?.get?.('content-type') || '';
31
33
  const headerMatch = /charset=["']?([\w-]+)/i.exec(contentType);
@@ -33,10 +35,14 @@ export function detectCharset(response, bytes) {
33
35
  return headerMatch[1].trim().toLowerCase();
34
36
  }
35
37
 
36
- // <meta charset> tags must appear within the first 1024 bytes per the
37
- // HTML5 spec's prescan algorithm; ASCII-range bytes decode identically
38
- // under latin1 regardless of the document's real encoding.
39
- const sniffLength = Math.min(bytes.byteLength, 1024);
38
+ // The HTML5 prescan algorithm requires a <meta charset> within the first
39
+ // 1024 bytes, but real pages break the rule: vector.co.jp/magazine/softnews
40
+ // (Shift_JIS, no charset in the Content-Type header) declares it at byte
41
+ // 1293, behind a comment block, and every browser still decodes it
42
+ // correctly. Sniff a full 8 KB, which covers every <head> seen in the
43
+ // wild without decoding the whole body twice. ASCII-range bytes decode
44
+ // identically under latin1 regardless of the document's real encoding.
45
+ const sniffLength = Math.min(bytes.byteLength, META_CHARSET_SNIFF_BYTES);
40
46
  const sniffText = new TextDecoder('latin1').decode(bytes.subarray(0, sniffLength));
41
47
  const metaMatch =
42
48
  /<meta[^>]+charset=["']?([\w-]+)/i.exec(sniffText) ||
@@ -23,6 +23,10 @@ const STATE_VARIABLES = [
23
23
  { name: 'nuxt', variable: '__NUXT__' },
24
24
  { name: 'apollo_state', variable: '__APOLLO_STATE__' },
25
25
  { name: 'initial_state', variable: '__INITIAL_STATE__' },
26
+ // tumblr.com spells it with three underscores and assigns it in bracket
27
+ // notation: window['___INITIAL_STATE___'] = {...} (R17, 2026-09-04). Same
28
+ // key as the two-underscore form; the first one found on a page wins.
29
+ { name: 'initial_state', variable: '___INITIAL_STATE___' },
26
30
  { name: 'preloaded_state', variable: '__PRELOADED_STATE__' },
27
31
  // nytimes.com ships its whole front page as window.__preloadedData; with
28
32
  // only the four names above, a 1.1 MB page surfaced nothing but its
@@ -334,8 +338,13 @@ export function extractEmbeddedState(rawHtml) {
334
338
  }
335
339
 
336
340
  for (const { name, variable } of STATE_VARIABLES) {
341
+ if (data[name] !== undefined) continue;
342
+ // Dot or bracket notation on window/self/globalThis, or a bare/var
343
+ // assignment; \b keeps MY__INITIAL_STATE__ from matching __INITIAL_STATE__.
337
344
  const assignment = html.match(
338
- new RegExp(`(?:window|self|globalThis)?\\.?\\b${variable}\\s*=\\s*`)
345
+ new RegExp(
346
+ `(?:(?:window|self|globalThis)\\s*\\[\\s*(['"])${variable}\\1\\s*\\]|(?:(?:window|self|globalThis)\\.)?\\b${variable})\\s*=\\s*`
347
+ )
339
348
  );
340
349
  if (!assignment) continue;
341
350
 
package/src/templates.js CHANGED
@@ -270,6 +270,19 @@ function amazonByline($) {
270
270
  // continues into "(Author) Format: Hardcover".
271
271
  if (/^by\s/i.test(raw)) return contributor || tidy(raw.replace(/^by\s+/i, '').split('(')[0]);
272
272
 
273
+ // Every other marketplace phrases the book byline in its own language —
274
+ // "Engelska utgåvan av George Orwell (Författare)", "Wydanie: Angielski
275
+ // George Orwell (Autor)", "Édition en Anglais de George Orwell (Auteur)" —
276
+ // and none starts with "by", so the chrome came back whole on amazon.se,
277
+ // .pl, .com.be and .com.tr (R17, 2026-09-04). The author's own link
278
+ // carries the bare name on every marketplace.
279
+ const authorLink = tidy($('#bylineInfo .author a, #bylineInfo a.contributorNameID').first().text());
280
+ if (authorLink && /\(/.test(raw)) return authorLink;
281
+
282
+ // "Marke: Sony", "Marca: Sony", "Marque : Sony" — a localised brand label.
283
+ const labelled = raw.match(/^[\p{L}\s]{2,20}?\s?:\s*(.+)$/u);
284
+ if (labelled && !/\(/.test(raw)) return tidy(labelled[1]);
285
+
273
286
  return raw;
274
287
  }
275
288
 
@@ -356,7 +369,7 @@ function amazonPrice($) {
356
369
  */
357
370
  function amazonCurrency($, price) {
358
371
  if (!price) return null;
359
- const code = price.match(/(USD|EUR|GBP|INR|JPY|CAD|AUD|MXN|BRL|SGD|AED|SAR|PLN|TRY)(?![A-Z])/);
372
+ const code = price.match(/(USD|EUR|GBP|INR|JPY|CAD|AUD|MXN|BRL|SGD|AED|SAR|EGP|PLN|TRY|SEK)(?![A-Z])/);
360
373
  if (code) return code[1];
361
374
  if (price.includes('₹')) return 'INR';
362
375
  if (price.includes('€')) return 'EUR';
@@ -364,9 +377,16 @@ function amazonCurrency($, price) {
364
377
  if (price.includes('¥') || price.includes('¥')) return 'JPY';
365
378
  if (price.includes('R$')) return 'BRL';
366
379
  if (price.includes('zł')) return 'PLN';
367
- if (price.includes('₺')) return 'TRY';
380
+ // amazon.com.tr writes "460,67TL" (R17, 2026-09-04); the lira sign is rarer.
381
+ if (price.includes('₺') || /(?<![A-Za-z])TL(?![A-Za-z])/.test(price)) return 'TRY';
382
+ // Arabic-script marketplaces: dirham, Egyptian pound, riyal.
383
+ if (/د\.?إ/.test(price)) return 'AED';
384
+ if (/ج\.?م/.test(price)) return 'EGP';
385
+ if (/ر\.?س|﷼/.test(price)) return 'SAR';
386
+ const host = (attr($, 'link[rel="canonical"]', 'href') || '').match(/^https?:\/\/([^/]+)/)?.[1] || '';
387
+ // "114,30kr" on amazon.se — the only Amazon marketplace priced in kronor.
388
+ if (/(?<![A-Za-z])kr(?![A-Za-z])/i.test(price)) return /\.se$/.test(host) ? 'SEK' : null;
368
389
  if (price.includes('$')) {
369
- const host = (attr($, 'link[rel="canonical"]', 'href') || '').match(/^https?:\/\/([^/]+)/)?.[1] || '';
370
390
  if (/\.ca$/.test(host)) return 'CAD';
371
391
  if (/\.com\.au$/.test(host)) return 'AUD';
372
392
  if (/\.com\.mx$/.test(host)) return 'MXN';
@@ -376,6 +396,15 @@ function amazonCurrency($, price) {
376
396
  return null;
377
397
  }
378
398
 
399
+ function hnAbsoluteUrl(href) {
400
+ if (!href) return href;
401
+ try {
402
+ return new URL(href, 'https://news.ycombinator.com/').href;
403
+ } catch {
404
+ return href;
405
+ }
406
+ }
407
+
379
408
  function hnCommentCount(label) {
380
409
  const value = (label || '').trim();
381
410
  if (!value) return null;
@@ -858,7 +887,9 @@ export const TEMPLATES = [
858
887
  stories.push({
859
888
  id: $row.attr('id'),
860
889
  title: $titleLink.text().trim(),
861
- url: safeHref($titleLink.attr('href')),
890
+ // Text posts (Ask HN, Show HN without a link) carry a relative
891
+ // "item?id=…" href; resolve it so every story url is absolute.
892
+ url: safeHref(hnAbsoluteUrl($titleLink.attr('href'))),
862
893
  site: $row.find('.sitebit a').text().trim() || null,
863
894
  // "1 point" on a fresh story and "3 points" on the rest — strip both.
864
895
  score: $score.text().replace(/\s*points?$/, '').trim() || null,