crawlforge-extractors 1.6.3 → 1.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/body.js +10 -4
- package/src/embeddedState.js +10 -1
- package/src/templates.js +35 -4
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "crawlforge-extractors",
|
|
3
|
-
"version": "1.6.
|
|
3
|
+
"version": "1.6.4",
|
|
4
4
|
"description": "Extraction logic shared by the CrawlForge MCP server and REST API — scrape templates, charset-correct capped body reading, structural fingerprinting, and embedded-state extraction. One implementation, so the two surfaces cannot drift apart.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "./index.js",
|
package/src/body.js
CHANGED
|
@@ -26,6 +26,8 @@ export class BodyTooLargeError extends Error {
|
|
|
26
26
|
* @param {Uint8Array} bytes
|
|
27
27
|
* @returns {string}
|
|
28
28
|
*/
|
|
29
|
+
export const META_CHARSET_SNIFF_BYTES = 8192;
|
|
30
|
+
|
|
29
31
|
export function detectCharset(response, bytes) {
|
|
30
32
|
const contentType = response.headers?.get?.('content-type') || '';
|
|
31
33
|
const headerMatch = /charset=["']?([\w-]+)/i.exec(contentType);
|
|
@@ -33,10 +35,14 @@ export function detectCharset(response, bytes) {
|
|
|
33
35
|
return headerMatch[1].trim().toLowerCase();
|
|
34
36
|
}
|
|
35
37
|
|
|
36
|
-
// <meta charset>
|
|
37
|
-
//
|
|
38
|
-
//
|
|
39
|
-
|
|
38
|
+
// The HTML5 prescan algorithm requires a <meta charset> within the first
|
|
39
|
+
// 1024 bytes, but real pages break the rule: vector.co.jp/magazine/softnews
|
|
40
|
+
// (Shift_JIS, no charset in the Content-Type header) declares it at byte
|
|
41
|
+
// 1293, behind a comment block, and every browser still decodes it
|
|
42
|
+
// correctly. Sniff a full 8 KB, which covers every <head> seen in the
|
|
43
|
+
// wild without decoding the whole body twice. ASCII-range bytes decode
|
|
44
|
+
// identically under latin1 regardless of the document's real encoding.
|
|
45
|
+
const sniffLength = Math.min(bytes.byteLength, META_CHARSET_SNIFF_BYTES);
|
|
40
46
|
const sniffText = new TextDecoder('latin1').decode(bytes.subarray(0, sniffLength));
|
|
41
47
|
const metaMatch =
|
|
42
48
|
/<meta[^>]+charset=["']?([\w-]+)/i.exec(sniffText) ||
|
package/src/embeddedState.js
CHANGED
|
@@ -23,6 +23,10 @@ const STATE_VARIABLES = [
|
|
|
23
23
|
{ name: 'nuxt', variable: '__NUXT__' },
|
|
24
24
|
{ name: 'apollo_state', variable: '__APOLLO_STATE__' },
|
|
25
25
|
{ name: 'initial_state', variable: '__INITIAL_STATE__' },
|
|
26
|
+
// tumblr.com spells it with three underscores and assigns it in bracket
|
|
27
|
+
// notation: window['___INITIAL_STATE___'] = {...} (R17, 2026-09-04). Same
|
|
28
|
+
// key as the two-underscore form; the first one found on a page wins.
|
|
29
|
+
{ name: 'initial_state', variable: '___INITIAL_STATE___' },
|
|
26
30
|
{ name: 'preloaded_state', variable: '__PRELOADED_STATE__' },
|
|
27
31
|
// nytimes.com ships its whole front page as window.__preloadedData; with
|
|
28
32
|
// only the four names above, a 1.1 MB page surfaced nothing but its
|
|
@@ -334,8 +338,13 @@ export function extractEmbeddedState(rawHtml) {
|
|
|
334
338
|
}
|
|
335
339
|
|
|
336
340
|
for (const { name, variable } of STATE_VARIABLES) {
|
|
341
|
+
if (data[name] !== undefined) continue;
|
|
342
|
+
// Dot or bracket notation on window/self/globalThis, or a bare/var
|
|
343
|
+
// assignment; \b keeps MY__INITIAL_STATE__ from matching __INITIAL_STATE__.
|
|
337
344
|
const assignment = html.match(
|
|
338
|
-
new RegExp(
|
|
345
|
+
new RegExp(
|
|
346
|
+
`(?:(?:window|self|globalThis)\\s*\\[\\s*(['"])${variable}\\1\\s*\\]|(?:(?:window|self|globalThis)\\.)?\\b${variable})\\s*=\\s*`
|
|
347
|
+
)
|
|
339
348
|
);
|
|
340
349
|
if (!assignment) continue;
|
|
341
350
|
|
package/src/templates.js
CHANGED
|
@@ -270,6 +270,19 @@ function amazonByline($) {
|
|
|
270
270
|
// continues into "(Author) Format: Hardcover".
|
|
271
271
|
if (/^by\s/i.test(raw)) return contributor || tidy(raw.replace(/^by\s+/i, '').split('(')[0]);
|
|
272
272
|
|
|
273
|
+
// Every other marketplace phrases the book byline in its own language —
|
|
274
|
+
// "Engelska utgåvan av George Orwell (Författare)", "Wydanie: Angielski
|
|
275
|
+
// George Orwell (Autor)", "Édition en Anglais de George Orwell (Auteur)" —
|
|
276
|
+
// and none starts with "by", so the chrome came back whole on amazon.se,
|
|
277
|
+
// .pl, .com.be and .com.tr (R17, 2026-09-04). The author's own link
|
|
278
|
+
// carries the bare name on every marketplace.
|
|
279
|
+
const authorLink = tidy($('#bylineInfo .author a, #bylineInfo a.contributorNameID').first().text());
|
|
280
|
+
if (authorLink && /\(/.test(raw)) return authorLink;
|
|
281
|
+
|
|
282
|
+
// "Marke: Sony", "Marca: Sony", "Marque : Sony" — a localised brand label.
|
|
283
|
+
const labelled = raw.match(/^[\p{L}\s]{2,20}?\s?:\s*(.+)$/u);
|
|
284
|
+
if (labelled && !/\(/.test(raw)) return tidy(labelled[1]);
|
|
285
|
+
|
|
273
286
|
return raw;
|
|
274
287
|
}
|
|
275
288
|
|
|
@@ -356,7 +369,7 @@ function amazonPrice($) {
|
|
|
356
369
|
*/
|
|
357
370
|
function amazonCurrency($, price) {
|
|
358
371
|
if (!price) return null;
|
|
359
|
-
const code = price.match(/(USD|EUR|GBP|INR|JPY|CAD|AUD|MXN|BRL|SGD|AED|SAR|PLN|TRY)(?![A-Z])/);
|
|
372
|
+
const code = price.match(/(USD|EUR|GBP|INR|JPY|CAD|AUD|MXN|BRL|SGD|AED|SAR|EGP|PLN|TRY|SEK)(?![A-Z])/);
|
|
360
373
|
if (code) return code[1];
|
|
361
374
|
if (price.includes('₹')) return 'INR';
|
|
362
375
|
if (price.includes('€')) return 'EUR';
|
|
@@ -364,9 +377,16 @@ function amazonCurrency($, price) {
|
|
|
364
377
|
if (price.includes('¥') || price.includes('¥')) return 'JPY';
|
|
365
378
|
if (price.includes('R$')) return 'BRL';
|
|
366
379
|
if (price.includes('zł')) return 'PLN';
|
|
367
|
-
|
|
380
|
+
// amazon.com.tr writes "460,67TL" (R17, 2026-09-04); the lira sign is rarer.
|
|
381
|
+
if (price.includes('₺') || /(?<![A-Za-z])TL(?![A-Za-z])/.test(price)) return 'TRY';
|
|
382
|
+
// Arabic-script marketplaces: dirham, Egyptian pound, riyal.
|
|
383
|
+
if (/د\.?إ/.test(price)) return 'AED';
|
|
384
|
+
if (/ج\.?م/.test(price)) return 'EGP';
|
|
385
|
+
if (/ر\.?س|﷼/.test(price)) return 'SAR';
|
|
386
|
+
const host = (attr($, 'link[rel="canonical"]', 'href') || '').match(/^https?:\/\/([^/]+)/)?.[1] || '';
|
|
387
|
+
// "114,30kr" on amazon.se — the only Amazon marketplace priced in kronor.
|
|
388
|
+
if (/(?<![A-Za-z])kr(?![A-Za-z])/i.test(price)) return /\.se$/.test(host) ? 'SEK' : null;
|
|
368
389
|
if (price.includes('$')) {
|
|
369
|
-
const host = (attr($, 'link[rel="canonical"]', 'href') || '').match(/^https?:\/\/([^/]+)/)?.[1] || '';
|
|
370
390
|
if (/\.ca$/.test(host)) return 'CAD';
|
|
371
391
|
if (/\.com\.au$/.test(host)) return 'AUD';
|
|
372
392
|
if (/\.com\.mx$/.test(host)) return 'MXN';
|
|
@@ -376,6 +396,15 @@ function amazonCurrency($, price) {
|
|
|
376
396
|
return null;
|
|
377
397
|
}
|
|
378
398
|
|
|
399
|
+
function hnAbsoluteUrl(href) {
|
|
400
|
+
if (!href) return href;
|
|
401
|
+
try {
|
|
402
|
+
return new URL(href, 'https://news.ycombinator.com/').href;
|
|
403
|
+
} catch {
|
|
404
|
+
return href;
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
379
408
|
function hnCommentCount(label) {
|
|
380
409
|
const value = (label || '').trim();
|
|
381
410
|
if (!value) return null;
|
|
@@ -858,7 +887,9 @@ export const TEMPLATES = [
|
|
|
858
887
|
stories.push({
|
|
859
888
|
id: $row.attr('id'),
|
|
860
889
|
title: $titleLink.text().trim(),
|
|
861
|
-
|
|
890
|
+
// Text posts (Ask HN, Show HN without a link) carry a relative
|
|
891
|
+
// "item?id=…" href; resolve it so every story url is absolute.
|
|
892
|
+
url: safeHref(hnAbsoluteUrl($titleLink.attr('href'))),
|
|
862
893
|
site: $row.find('.sitebit a').text().trim() || null,
|
|
863
894
|
// "1 point" on a fresh story and "3 points" on the rest — strip both.
|
|
864
895
|
score: $score.text().replace(/\s*points?$/, '').trim() || null,
|