@opencraw/core 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/dist/index.esm.js +2648 -447
- package/dist/src/api-steps/extract-from-document.use-case.d.ts +16 -2
- package/dist/src/api-steps/index.d.ts +1 -1
- package/dist/src/api-steps/send-request.use-case.d.ts +2 -2
- package/dist/src/captcha/captcha-budget.model.d.ts +21 -0
- package/dist/src/captcha/captcha-detection.client.d.ts +28 -0
- package/dist/src/captcha/captcha-guard.use-case.d.ts +64 -0
- package/dist/src/captcha/captcha-solver-registry.store.d.ts +19 -0
- package/dist/src/captcha/captcha-solver.contract.d.ts +47 -0
- package/dist/src/captcha/captcha.error.d.ts +13 -0
- package/dist/src/captcha/index.d.ts +10 -0
- package/dist/src/captcha/resolve-captcha.use-case.d.ts +40 -0
- package/dist/src/crawl-events/crawl-event.contract.d.ts +41 -0
- package/dist/src/crawl-execution/bootstrap-session.use-case.d.ts +5 -2
- package/dist/src/crawl-execution/crawl-options.config.d.ts +3 -0
- package/dist/src/crawl-execution/crawl-report.model.d.ts +8 -0
- package/dist/src/crawl-execution/create-crawler.use-case.d.ts +2 -1
- package/dist/src/crawl-execution/run-input-recipe.use-case.d.ts +3 -0
- package/dist/src/deck-document/deck-document.model.d.ts +58 -0
- package/dist/src/deck-document/deck-table.algorithm.d.ts +35 -0
- package/dist/src/deck-document/index.d.ts +6 -0
- package/dist/src/deck-document/read-pptx.client.d.ts +16 -0
- package/dist/src/extraction-scope/extraction-scope.model.d.ts +4 -1
- package/dist/src/http-session/http-response.contract.d.ts +14 -1
- package/dist/src/http-session/text-decoding.algorithm.d.ts +35 -0
- package/dist/src/index.d.ts +11 -1
- package/dist/src/markdown-document/index.d.ts +3 -0
- package/dist/src/markdown-document/read-markdown.client.d.ts +29 -0
- package/dist/src/pdf-document/index.d.ts +7 -0
- package/dist/src/pdf-document/pdf-document.model.d.ts +51 -0
- package/dist/src/pdf-document/pdf-table.algorithm.d.ts +43 -0
- package/dist/src/pdf-document/read-pdf.client.d.ts +18 -0
- package/dist/src/pdf-document/row-assembly.algorithm.d.ts +26 -0
- package/dist/src/recipe-loading/recipe-binding.validator.d.ts +3 -1
- package/dist/src/recipe-schema/index.d.ts +2 -2
- package/dist/src/recipe-schema/input-recipe.contract.d.ts +33 -3
- package/dist/src/recipe-schema/recipe-kind.enum.d.ts +7 -3
- package/dist/src/recipe-schema/step.contract.d.ts +52 -2
- package/dist/src/selection/index.d.ts +1 -1
- package/dist/src/selection/json-text.algorithm.d.ts +30 -3
- package/dist/src/web-steps/run-web-step.use-case.d.ts +10 -2
- package/dist/src/workbook-document/csv-parser.algorithm.d.ts +26 -0
- package/dist/src/workbook-document/csv-workbook.mapper.d.ts +24 -0
- package/dist/src/workbook-document/grid-table.algorithm.d.ts +53 -0
- package/dist/src/workbook-document/html-tables.mapper.d.ts +14 -0
- package/dist/src/workbook-document/index.d.ts +9 -0
- package/dist/src/workbook-document/read-xlsx.client.d.ts +18 -0
- package/dist/src/workbook-document/workbook-document.model.d.ts +51 -0
- package/dist/src/yaml-document/index.d.ts +3 -0
- package/dist/src/yaml-document/read-yaml.client.d.ts +25 -0
- package/package.json +17 -2
package/dist/index.esm.js
CHANGED
|
@@ -7,6 +7,7 @@ import { createWriteStream } from 'node:fs';
|
|
|
7
7
|
import { once } from 'node:events';
|
|
8
8
|
import { JSONPath } from 'jsonpath-plus';
|
|
9
9
|
import { load } from 'cheerio';
|
|
10
|
+
import { fileURLToPath } from 'node:url';
|
|
10
11
|
|
|
11
12
|
/** Resource types a page may skip loading, to save bandwidth on per-GB proxies. */
|
|
12
13
|
const BLOCKABLE_RESOURCES = ['image', 'media', 'font', 'stylesheet', 'script', 'texttrack', 'xhr', 'fetch', 'eventsource', 'websocket', 'manifest', 'other'];
|
|
@@ -1247,6 +1248,928 @@ class BrowserClient {
|
|
|
1247
1248
|
}
|
|
1248
1249
|
}
|
|
1249
1250
|
|
|
1251
|
+
/**
|
|
1252
|
+
* Runs a body once per item of a list (`over`), or once per live element
|
|
1253
|
+
* matching `selector`, each in a fresh child scope with the item bound under
|
|
1254
|
+
* `as`; emits a record per iteration when asked.
|
|
1255
|
+
*
|
|
1256
|
+
* With a concurrent gate, iterations run as permits allow and records come
|
|
1257
|
+
* out in completion order; without one, in list order.
|
|
1258
|
+
*
|
|
1259
|
+
* @param step - The forEach step.
|
|
1260
|
+
* @param scope - The scope the list lives in.
|
|
1261
|
+
* @param walk - Runs a step list; also carries the emit callback.
|
|
1262
|
+
* @returns `stop` when the crawl reached its record limit.
|
|
1263
|
+
*/
|
|
1264
|
+
async function runForEach(step, scope, walk) {
|
|
1265
|
+
const items = await itemsOf$1(step, scope, walk);
|
|
1266
|
+
const gate = walk.gate;
|
|
1267
|
+
if (gate?.concurrent === true) return runPooled(step, scope, walk, items, gate);
|
|
1268
|
+
for (const item of items) {
|
|
1269
|
+
if ((await runIteration(step, scope, walk, item)) === 'stop') return 'stop';
|
|
1270
|
+
}
|
|
1271
|
+
return 'continue';
|
|
1272
|
+
}
|
|
1273
|
+
async function runIteration(step, scope, walk, item, overrides) {
|
|
1274
|
+
const child = scope.child();
|
|
1275
|
+
child.set(step.as, item);
|
|
1276
|
+
const outcome = await walk.runSteps(step.steps, child, `${walk.path}.steps`, overrides);
|
|
1277
|
+
if (outcome === 'stop' || step.emit === undefined) return outcome;
|
|
1278
|
+
return walk.onEmit(child, step.emit === true ? undefined : step.emit.output);
|
|
1279
|
+
}
|
|
1280
|
+
/**
|
|
1281
|
+
* Starts iterations as the gate hands out permits. A `stop` or a failure stops
|
|
1282
|
+
* new iterations; the ones in flight finish first, so the runner is never
|
|
1283
|
+
* disposed under them. The first failure is rethrown afterwards.
|
|
1284
|
+
*/
|
|
1285
|
+
async function runPooled(step, scope, walk, items, gate) {
|
|
1286
|
+
let stopped = false;
|
|
1287
|
+
let failure;
|
|
1288
|
+
const tasks = [];
|
|
1289
|
+
const overrides = {
|
|
1290
|
+
gate: gate.nested()
|
|
1291
|
+
};
|
|
1292
|
+
const iterate = async (item, release) => {
|
|
1293
|
+
try {
|
|
1294
|
+
if ((await runIteration(step, scope, walk, item, overrides)) === 'stop') stopped = true;
|
|
1295
|
+
} catch (error) {
|
|
1296
|
+
failure ??= {
|
|
1297
|
+
error
|
|
1298
|
+
};
|
|
1299
|
+
} finally {
|
|
1300
|
+
release();
|
|
1301
|
+
}
|
|
1302
|
+
};
|
|
1303
|
+
for (const item of items) {
|
|
1304
|
+
if (stopped || failure !== undefined) break;
|
|
1305
|
+
const release = await gate.acquire();
|
|
1306
|
+
if (stopped || failure !== undefined) {
|
|
1307
|
+
release();
|
|
1308
|
+
break;
|
|
1309
|
+
}
|
|
1310
|
+
tasks.push(iterate(item, release));
|
|
1311
|
+
}
|
|
1312
|
+
await Promise.allSettled(tasks);
|
|
1313
|
+
if (failure !== undefined) throw failure.error;
|
|
1314
|
+
return stopped ? 'stop' : 'continue';
|
|
1315
|
+
}
|
|
1316
|
+
async function itemsOf$1(step, scope, walk) {
|
|
1317
|
+
if (step.selector !== undefined) {
|
|
1318
|
+
if (walk.runner.elements === undefined) throw new Error('forEach over selector iterates live elements and needs a browser; this recipe runs in api mode');
|
|
1319
|
+
return walk.runner.elements(renderText(step.selector, path => scope.lookup(path)), scope);
|
|
1320
|
+
}
|
|
1321
|
+
const list = scope.get(step.over ?? '');
|
|
1322
|
+
return Array.isArray(list) ? list : list === undefined || list === null ? [] : [list];
|
|
1323
|
+
}
|
|
1324
|
+
|
|
1325
|
+
/**
|
|
1326
|
+
* Runs a body once per page, each in a fresh child scope, then asks the runner
|
|
1327
|
+
* for the next page until there is none, `until` renders truthy, or `maxPages`
|
|
1328
|
+
* is reached.
|
|
1329
|
+
*
|
|
1330
|
+
* The runner reports the visit of each new page; this only steers.
|
|
1331
|
+
*
|
|
1332
|
+
* @param step - The paginate step.
|
|
1333
|
+
* @param scope - The scope to page in; its page URL advances with each page.
|
|
1334
|
+
* @param walk - Runs a step list; carries the runner and the emit callback.
|
|
1335
|
+
* @returns `stop` when the crawl reached its record limit.
|
|
1336
|
+
*/
|
|
1337
|
+
async function runPaginate(step, scope, walk) {
|
|
1338
|
+
let number = scope.pageState?.number ?? 1;
|
|
1339
|
+
let bound;
|
|
1340
|
+
for (let count = 1;; count += 1) {
|
|
1341
|
+
const page = scope.child();
|
|
1342
|
+
page.setPage({
|
|
1343
|
+
number
|
|
1344
|
+
});
|
|
1345
|
+
if (bound !== undefined) page.set(bound.name, bound.value);
|
|
1346
|
+
const outcome = await walk.runSteps(step.steps, page, `${walk.path}.steps`);
|
|
1347
|
+
if (outcome === 'stop') return 'stop';
|
|
1348
|
+
if (step.until !== undefined && isTruthy(render(step.until, path => page.lookup(path)))) break;
|
|
1349
|
+
if (step.maxPages !== undefined && count >= step.maxPages) break;
|
|
1350
|
+
const next = await walk.runner.nextPage(step.next, page);
|
|
1351
|
+
if (next === null) break;
|
|
1352
|
+
number += 1;
|
|
1353
|
+
if (next.kind === 'url') {
|
|
1354
|
+
scope.setPage({
|
|
1355
|
+
url: next.url,
|
|
1356
|
+
number
|
|
1357
|
+
});
|
|
1358
|
+
bound = undefined;
|
|
1359
|
+
} else {
|
|
1360
|
+
scope.setPage({
|
|
1361
|
+
number
|
|
1362
|
+
});
|
|
1363
|
+
bound = {
|
|
1364
|
+
name: next.name,
|
|
1365
|
+
value: next.value
|
|
1366
|
+
};
|
|
1367
|
+
}
|
|
1368
|
+
}
|
|
1369
|
+
return 'continue';
|
|
1370
|
+
}
|
|
1371
|
+
|
|
1372
|
+
const FAIL = {
|
|
1373
|
+
policy: 'fail'
|
|
1374
|
+
};
|
|
1375
|
+
/**
|
|
1376
|
+
* The policy for a failing step: the step's own, else the recipe's, else `fail`.
|
|
1377
|
+
*
|
|
1378
|
+
* @param step - The step that failed.
|
|
1379
|
+
* @param recipe - Its recipe.
|
|
1380
|
+
* @returns The policy to apply.
|
|
1381
|
+
*/
|
|
1382
|
+
function resolveErrorPolicy(step, recipe) {
|
|
1383
|
+
return step.onError ?? recipe.onError ?? FAIL;
|
|
1384
|
+
}
|
|
1385
|
+
/**
|
|
1386
|
+
* How long to wait before a retry: linear backoff.
|
|
1387
|
+
*
|
|
1388
|
+
* @param policy - A retry policy.
|
|
1389
|
+
* @param attempt - The attempt about to be made, starting at 2.
|
|
1390
|
+
* @returns Milliseconds.
|
|
1391
|
+
*/
|
|
1392
|
+
function backoffFor(policy, attempt) {
|
|
1393
|
+
return (policy.backoffMs ?? 0) * (attempt - 1);
|
|
1394
|
+
}
|
|
1395
|
+
function sleep(ms) {
|
|
1396
|
+
return ms <= 0 ? Promise.resolve() : new Promise(resolve => setTimeout(resolve, ms));
|
|
1397
|
+
}
|
|
1398
|
+
|
|
1399
|
+
/** A response the recipe's `session.blockedWhen` rule (or the default one) says is the site refusing the crawl. */
|
|
1400
|
+
class BlockedError extends Error {
|
|
1401
|
+
url;
|
|
1402
|
+
status;
|
|
1403
|
+
reason;
|
|
1404
|
+
name = 'BlockedError';
|
|
1405
|
+
constructor(url, status, reason) {
|
|
1406
|
+
super(`blocked at ${url}: ${reason}`);
|
|
1407
|
+
this.url = url;
|
|
1408
|
+
this.status = status;
|
|
1409
|
+
this.reason = reason;
|
|
1410
|
+
}
|
|
1411
|
+
}
|
|
1412
|
+
|
|
1413
|
+
/** A step that failed under the `fail` policy: the recipe stops here. */
|
|
1414
|
+
class StepFailure extends Error {
|
|
1415
|
+
stepPath;
|
|
1416
|
+
stepType;
|
|
1417
|
+
name = 'StepFailure';
|
|
1418
|
+
constructor(stepPath, stepType, cause) {
|
|
1419
|
+
super(`step ${stepPath} (${stepType}) failed: ${cause instanceof Error ? cause.message : String(cause)}`, {
|
|
1420
|
+
cause
|
|
1421
|
+
});
|
|
1422
|
+
this.stepPath = stepPath;
|
|
1423
|
+
this.stepType = stepType;
|
|
1424
|
+
}
|
|
1425
|
+
}
|
|
1426
|
+
/** A single `extract` that matched nothing. */
|
|
1427
|
+
class NoMatchError extends Error {
|
|
1428
|
+
selector;
|
|
1429
|
+
name = 'NoMatchError';
|
|
1430
|
+
constructor(selector) {
|
|
1431
|
+
super(`no match for ${selector}`);
|
|
1432
|
+
this.selector = selector;
|
|
1433
|
+
}
|
|
1434
|
+
}
|
|
1435
|
+
|
|
1436
|
+
/**
|
|
1437
|
+
* Walks a step list in order. Control flow (`forEach`, `if`, `paginate`, `emit`,
|
|
1438
|
+
* `set`, `hook`, `when`, error policies) is handled here; leaf steps go to the
|
|
1439
|
+
* runner. Mode-agnostic: the same walk drives a browser page or an HTTP context.
|
|
1440
|
+
*
|
|
1441
|
+
* @param steps - The steps.
|
|
1442
|
+
* @param scope - The scope to run in.
|
|
1443
|
+
* @param options - Recipe, runner, hooks, events and the emit callback.
|
|
1444
|
+
* @param path - Where these steps are, for messages and events.
|
|
1445
|
+
* @returns `stop` when the emit callback asked to stop.
|
|
1446
|
+
* @throws StepFailure when a step fails under the `fail` policy.
|
|
1447
|
+
*/
|
|
1448
|
+
async function runSteps(steps, scope, options, path = 'steps') {
|
|
1449
|
+
for (const [index, step] of steps.entries()) {
|
|
1450
|
+
if (step.when !== undefined && !isTruthy(render(step.when, lookupIn(scope)))) continue;
|
|
1451
|
+
const at = `${path}.${index}`;
|
|
1452
|
+
const walk = {
|
|
1453
|
+
...options,
|
|
1454
|
+
path: at,
|
|
1455
|
+
runSteps: (inner, innerScope, innerPath, overrides) => runSteps(inner, innerScope, {
|
|
1456
|
+
...options,
|
|
1457
|
+
...overrides
|
|
1458
|
+
}, innerPath)
|
|
1459
|
+
};
|
|
1460
|
+
const outcome = await runWithPolicy(step, scope, walk);
|
|
1461
|
+
if (outcome === 'stop') return 'stop';
|
|
1462
|
+
}
|
|
1463
|
+
return 'continue';
|
|
1464
|
+
}
|
|
1465
|
+
async function runWithPolicy(step, scope, walk) {
|
|
1466
|
+
const policy = resolveErrorPolicy(step, walk.recipe);
|
|
1467
|
+
const attempts = policy.policy === 'retry' ? policy.attempts : 1;
|
|
1468
|
+
let attempt = 1;
|
|
1469
|
+
for (;;) {
|
|
1470
|
+
const started = Date.now();
|
|
1471
|
+
walk.events.emit({
|
|
1472
|
+
type: 'step:start',
|
|
1473
|
+
recipeId: walk.recipe.id,
|
|
1474
|
+
stepType: step.type,
|
|
1475
|
+
stepId: step.id,
|
|
1476
|
+
path: walk.path
|
|
1477
|
+
});
|
|
1478
|
+
try {
|
|
1479
|
+
const outcome = await runOne(step, scope, walk);
|
|
1480
|
+
walk.events.emit({
|
|
1481
|
+
type: 'step:finish',
|
|
1482
|
+
recipeId: walk.recipe.id,
|
|
1483
|
+
stepType: step.type,
|
|
1484
|
+
stepId: step.id,
|
|
1485
|
+
path: walk.path,
|
|
1486
|
+
durationMs: Date.now() - started
|
|
1487
|
+
});
|
|
1488
|
+
return outcome;
|
|
1489
|
+
} catch (error) {
|
|
1490
|
+
if (error instanceof StepFailure) throw error;
|
|
1491
|
+
// A block the runner can rotate away from is retried on the new access, without spending a retry attempt.
|
|
1492
|
+
if (error instanceof BlockedError && walk.runner.rotate !== undefined && (await walk.runner.rotate(error))) continue;
|
|
1493
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
1494
|
+
if (policy.policy === 'retry' && attempt < attempts) {
|
|
1495
|
+
attempt += 1;
|
|
1496
|
+
walk.events.emit({
|
|
1497
|
+
type: 'step:retry',
|
|
1498
|
+
recipeId: walk.recipe.id,
|
|
1499
|
+
stepType: step.type,
|
|
1500
|
+
stepId: step.id,
|
|
1501
|
+
path: walk.path,
|
|
1502
|
+
attempt,
|
|
1503
|
+
error: message
|
|
1504
|
+
});
|
|
1505
|
+
await sleep(backoffFor(policy, attempt));
|
|
1506
|
+
continue;
|
|
1507
|
+
}
|
|
1508
|
+
if (policy.policy === 'skip') {
|
|
1509
|
+
walk.events.emit({
|
|
1510
|
+
type: 'step:skip',
|
|
1511
|
+
recipeId: walk.recipe.id,
|
|
1512
|
+
stepType: step.type,
|
|
1513
|
+
stepId: step.id,
|
|
1514
|
+
path: walk.path,
|
|
1515
|
+
error: message
|
|
1516
|
+
});
|
|
1517
|
+
return 'continue';
|
|
1518
|
+
}
|
|
1519
|
+
throw new StepFailure(walk.path, step.type, error);
|
|
1520
|
+
}
|
|
1521
|
+
}
|
|
1522
|
+
}
|
|
1523
|
+
async function runOne(step, scope, walk) {
|
|
1524
|
+
switch (step.type) {
|
|
1525
|
+
case 'forEach':
|
|
1526
|
+
{
|
|
1527
|
+
return runForEach(step, scope, walk);
|
|
1528
|
+
}
|
|
1529
|
+
case 'if':
|
|
1530
|
+
{
|
|
1531
|
+
const branch = isTruthy(render(step.test, lookupIn(scope))) ? 'then' : 'else';
|
|
1532
|
+
walk.events.emit({
|
|
1533
|
+
type: 'step:branch',
|
|
1534
|
+
recipeId: walk.recipe.id,
|
|
1535
|
+
path: walk.path,
|
|
1536
|
+
branch
|
|
1537
|
+
});
|
|
1538
|
+
const chosen = branch === 'then' ? step.steps : step.else ?? [];
|
|
1539
|
+
return walk.runSteps(chosen, scope, `${walk.path}.${branch === 'then' ? 'steps' : 'else'}`);
|
|
1540
|
+
}
|
|
1541
|
+
case 'paginate':
|
|
1542
|
+
{
|
|
1543
|
+
return runPaginate(step, scope, walk);
|
|
1544
|
+
}
|
|
1545
|
+
case 'emit':
|
|
1546
|
+
{
|
|
1547
|
+
return walk.onEmit(scope, step.output);
|
|
1548
|
+
}
|
|
1549
|
+
case 'set':
|
|
1550
|
+
{
|
|
1551
|
+
if (step.id !== undefined) scope.set(step.id, typeof step.value === 'string' ? render(step.value, lookupIn(scope)) : step.value);
|
|
1552
|
+
return 'continue';
|
|
1553
|
+
}
|
|
1554
|
+
case 'collect':
|
|
1555
|
+
{
|
|
1556
|
+
const value = typeof step.value === 'string' ? render(step.value, lookupIn(scope)) : step.value;
|
|
1557
|
+
if (value !== undefined) scope.append(step.into, Array.isArray(value) ? value : [value]);
|
|
1558
|
+
return 'continue';
|
|
1559
|
+
}
|
|
1560
|
+
case 'hook':
|
|
1561
|
+
{
|
|
1562
|
+
const hook = walk.hooks.resolve(step.name);
|
|
1563
|
+
const result = await hook(undefined, renderArgs(step.args ?? {}, scope), {
|
|
1564
|
+
recipeId: walk.recipe.id,
|
|
1565
|
+
scope: scope.snapshot(),
|
|
1566
|
+
log: logThrough(walk)
|
|
1567
|
+
});
|
|
1568
|
+
if (step.id !== undefined) scope.set(step.id, result);
|
|
1569
|
+
return 'continue';
|
|
1570
|
+
}
|
|
1571
|
+
default:
|
|
1572
|
+
{
|
|
1573
|
+
await walk.runner.runLeaf(step, scope);
|
|
1574
|
+
return 'continue';
|
|
1575
|
+
}
|
|
1576
|
+
}
|
|
1577
|
+
}
|
|
1578
|
+
function lookupIn(scope) {
|
|
1579
|
+
return path => scope.lookup(path);
|
|
1580
|
+
}
|
|
1581
|
+
function renderArgs(args, scope) {
|
|
1582
|
+
return Object.fromEntries(Object.entries(args).map(([name, value]) => [name, typeof value === 'string' ? render(value, lookupIn(scope)) : value]));
|
|
1583
|
+
}
|
|
1584
|
+
function logThrough(walk) {
|
|
1585
|
+
return (level, message, meta) => {
|
|
1586
|
+
walk.events.emit({
|
|
1587
|
+
type: level === 'error' ? 'error' : 'warning',
|
|
1588
|
+
recipeId: walk.recipe.id,
|
|
1589
|
+
message: `[${level}] ${message}`,
|
|
1590
|
+
meta
|
|
1591
|
+
});
|
|
1592
|
+
};
|
|
1593
|
+
}
|
|
1594
|
+
|
|
1595
|
+
/**
|
|
1596
|
+
* What bounds a recipe run: how many `forEach` iterations may be in flight and
|
|
1597
|
+
* how close together requests may start. One gate per recipe run, shared by
|
|
1598
|
+
* every loop in it, so nested loops never multiply the limit.
|
|
1599
|
+
*
|
|
1600
|
+
* Permits go to the outermost concurrent loop: a loop that runs inside an
|
|
1601
|
+
* iteration already holding a permit runs its body sequentially (see `nested`),
|
|
1602
|
+
* which keeps the total at `permits` and cannot deadlock.
|
|
1603
|
+
*/
|
|
1604
|
+
class RunGate {
|
|
1605
|
+
permits;
|
|
1606
|
+
minIntervalMs;
|
|
1607
|
+
shared;
|
|
1608
|
+
inFlight = 0;
|
|
1609
|
+
waiting = [];
|
|
1610
|
+
lastStart = -Infinity;
|
|
1611
|
+
/**
|
|
1612
|
+
* @param permits - Iterations allowed in flight; 1 is sequential.
|
|
1613
|
+
* @param minIntervalMs - Minimum time between two request starts across the run.
|
|
1614
|
+
* @param shared - The throttle state to share (internal: `nested` gates keep their parent's).
|
|
1615
|
+
*/
|
|
1616
|
+
constructor(permits, minIntervalMs, shared) {
|
|
1617
|
+
this.permits = permits;
|
|
1618
|
+
this.minIntervalMs = minIntervalMs;
|
|
1619
|
+
this.shared = shared;
|
|
1620
|
+
}
|
|
1621
|
+
/** Whether this gate lets more than one iteration run at once. */
|
|
1622
|
+
get concurrent() {
|
|
1623
|
+
return this.permits > 1;
|
|
1624
|
+
}
|
|
1625
|
+
/**
|
|
1626
|
+
* Takes a permit, waiting for one when all are in flight.
|
|
1627
|
+
*
|
|
1628
|
+
* @returns The release; call it exactly once, when the iteration ends.
|
|
1629
|
+
*/
|
|
1630
|
+
async acquire() {
|
|
1631
|
+
if (this.inFlight >= this.permits) await new Promise(resolve => {
|
|
1632
|
+
this.waiting.push(resolve);
|
|
1633
|
+
});
|
|
1634
|
+
this.inFlight += 1;
|
|
1635
|
+
let released = false;
|
|
1636
|
+
return () => {
|
|
1637
|
+
if (released) return;
|
|
1638
|
+
released = true;
|
|
1639
|
+
this.inFlight -= 1;
|
|
1640
|
+
this.waiting.shift()?.();
|
|
1641
|
+
};
|
|
1642
|
+
}
|
|
1643
|
+
/**
|
|
1644
|
+
* Waits until a request may start: `minIntervalMs` after the previous start,
|
|
1645
|
+
* whichever loop started it. Returns at once when the interval has passed.
|
|
1646
|
+
*/
|
|
1647
|
+
async throttle() {
|
|
1648
|
+
const state = this.shared ?? this;
|
|
1649
|
+
if (state.minIntervalMs <= 0) return;
|
|
1650
|
+
const now = Date.now();
|
|
1651
|
+
const at = Math.max(now, state.lastStart + state.minIntervalMs);
|
|
1652
|
+
state.lastStart = at;
|
|
1653
|
+
await sleep(at - now);
|
|
1654
|
+
}
|
|
1655
|
+
/** The gate for a body running inside an iteration that holds a permit: sequential, same throttle. */
|
|
1656
|
+
nested() {
|
|
1657
|
+
return new RunGate(1, this.minIntervalMs, this.shared ?? this);
|
|
1658
|
+
}
|
|
1659
|
+
}
|
|
1660
|
+
|
|
1661
|
+
/** A block unless a recipe says otherwise: forbidden, rate limited, or an AWS WAF challenge (IMDb answers 202 with it). */
|
|
1662
|
+
const DEFAULT_BLOCK_RULE = {
|
|
1663
|
+
status: [403, 429],
|
|
1664
|
+
header: {
|
|
1665
|
+
'x-amzn-waf-action': 'challenge'
|
|
1666
|
+
}
|
|
1667
|
+
};
|
|
1668
|
+
/**
|
|
1669
|
+
* Whether a response is a block.
|
|
1670
|
+
*
|
|
1671
|
+
* @param response - What came back.
|
|
1672
|
+
* @param rule - The recipe's `session.blockedWhen`; `DEFAULT_BLOCK_RULE` when omitted.
|
|
1673
|
+
* @returns The error to throw, or `undefined` when the response is not a block.
|
|
1674
|
+
*/
|
|
1675
|
+
async function detectBlock(response, rule = DEFAULT_BLOCK_RULE) {
|
|
1676
|
+
if (rule.status?.includes(response.status) === true) return new BlockedError(response.url, response.status, `HTTP ${response.status}`);
|
|
1677
|
+
const headers = Object.entries(rule.header ?? {});
|
|
1678
|
+
for (const [name, pattern] of headers) {
|
|
1679
|
+
const value = response.headers[name.toLowerCase()];
|
|
1680
|
+
if (value !== undefined && new RegExp(pattern, 'i').test(value)) return new BlockedError(response.url, response.status, `${name.toLowerCase()}: ${value}`);
|
|
1681
|
+
}
|
|
1682
|
+
if (rule.text !== undefined && response.text !== undefined) {
|
|
1683
|
+
let body = '';
|
|
1684
|
+
try {
|
|
1685
|
+
body = await response.text();
|
|
1686
|
+
} catch {
|
|
1687
|
+
// a body that cannot be read (a redirect, a download) cannot match
|
|
1688
|
+
}
|
|
1689
|
+
const pattern = new RegExp(rule.text, 'i');
|
|
1690
|
+
if (pattern.test(body)) return new BlockedError(response.url, response.status, `body matches /${rule.text}/i`);
|
|
1691
|
+
}
|
|
1692
|
+
return undefined;
|
|
1693
|
+
}
|
|
1694
|
+
|
|
1695
|
+
/**
|
|
1696
|
+
* A challenge the engine could not get past: the solver failed, the page did
|
|
1697
|
+
* not confirm it, or the budget ran out. It is a block, so a recipe with
|
|
1698
|
+
* `onBlock.rotate` retries the step on a new access lease (a new IP often
|
|
1699
|
+
* means an easier challenge, or none), then the step's error policy applies.
|
|
1700
|
+
*/
|
|
1701
|
+
class CaptchaError extends BlockedError {
|
|
1702
|
+
kind;
|
|
1703
|
+
attempts;
|
|
1704
|
+
constructor(url, kind, attempts, reason) {
|
|
1705
|
+
super(url, 0, `captcha (${kind}) ${reason}`);
|
|
1706
|
+
this.kind = kind;
|
|
1707
|
+
this.attempts = attempts;
|
|
1708
|
+
}
|
|
1709
|
+
}
|
|
1710
|
+
|
|
1711
|
+
/** The captcha solvers a crawler was given, by name. */
|
|
1712
|
+
class CaptchaSolverRegistry {
|
|
1713
|
+
solvers = new Map();
|
|
1714
|
+
order = [];
|
|
1715
|
+
constructor(solvers = []) {
|
|
1716
|
+
for (const solver of solvers) {
|
|
1717
|
+
if (this.solvers.has(solver.name)) throw new Error(`two captcha solvers are named "${solver.name}"`);
|
|
1718
|
+
this.solvers.set(solver.name, solver);
|
|
1719
|
+
this.order.push(solver.name);
|
|
1720
|
+
}
|
|
1721
|
+
}
|
|
1722
|
+
/** The registered names. */
|
|
1723
|
+
get names() {
|
|
1724
|
+
return [...this.order];
|
|
1725
|
+
}
|
|
1726
|
+
has(name) {
|
|
1727
|
+
return this.solvers.has(name);
|
|
1728
|
+
}
|
|
1729
|
+
/**
|
|
1730
|
+
* The solver of that name.
|
|
1731
|
+
*
|
|
1732
|
+
* @param name - As a recipe names it.
|
|
1733
|
+
* @returns The solver.
|
|
1734
|
+
* @throws Error naming what is registered when it is not.
|
|
1735
|
+
*/
|
|
1736
|
+
resolve(name) {
|
|
1737
|
+
const solver = this.solvers.get(name);
|
|
1738
|
+
if (solver === undefined) throw new Error(`captcha solver "${name}" is not registered (registered: ${this.names.join(', ') || 'none'}); give it to createCrawler({ captchaSolvers }) or export it from the plugins module`);
|
|
1739
|
+
return solver;
|
|
1740
|
+
}
|
|
1741
|
+
}
|
|
1742
|
+
|
|
1743
|
+
/** Default solves a recipe run may spend. */
|
|
1744
|
+
const DEFAULT_MAX_SOLVES = 10;
|
|
1745
|
+
/**
|
|
1746
|
+
* The solves a recipe run may still spend. Every solve costs money: a detector
|
|
1747
|
+
* that matches the wrong element would drain a balance without it. Shared by
|
|
1748
|
+
* every runner of the run, rotations included.
|
|
1749
|
+
*/
|
|
1750
|
+
class CaptchaBudget {
|
|
1751
|
+
max;
|
|
1752
|
+
used = 0;
|
|
1753
|
+
constructor(max = DEFAULT_MAX_SOLVES) {
|
|
1754
|
+
this.max = max;
|
|
1755
|
+
}
|
|
1756
|
+
/** Solves spent so far. */
|
|
1757
|
+
get spent() {
|
|
1758
|
+
return this.used;
|
|
1759
|
+
}
|
|
1760
|
+
/**
|
|
1761
|
+
* Spends one solve.
|
|
1762
|
+
*
|
|
1763
|
+
* @returns Whether one was left.
|
|
1764
|
+
*/
|
|
1765
|
+
take() {
|
|
1766
|
+
if (this.used >= this.max) return false;
|
|
1767
|
+
this.used += 1;
|
|
1768
|
+
return true;
|
|
1769
|
+
}
|
|
1770
|
+
}
|
|
1771
|
+
|
|
1772
|
+
/** The widgets detection looks for when a recipe names none: reCAPTCHA v2, hCaptcha and Turnstile, as a container or as their iframe. */
|
|
1773
|
+
const DEFAULT_CAPTCHA_SELECTOR = ['.g-recaptcha', 'iframe[src*="recaptcha/api2/anchor"]', 'iframe[src*="recaptcha/enterprise/anchor"]', '.h-captcha', 'iframe[src*="hcaptcha.com"]', '.cf-turnstile', 'iframe[src*="challenges.cloudflare.com"]'].join(', ');
|
|
1774
|
+
/** reCAPTCHA v3 runs without a widget: its script is loaded with the site key as `render`. */
|
|
1775
|
+
const RECAPTCHA_V3_SCRIPT = 'script[src*="recaptcha/api.js?render="], script[src*="recaptcha/enterprise.js?render="]';
|
|
1776
|
+
/** Matches past this many are not looked at: a page does not show more challenges than that. */
|
|
1777
|
+
const MAX_CANDIDATES = 10;
|
|
1778
|
+
/**
|
|
1779
|
+
* The first visible challenge on the page, if any. An element counts only
|
|
1780
|
+
* when visible: sites keep hidden widgets around after a solve, and an
|
|
1781
|
+
* invisible reCAPTCHA shows nothing until it challenges.
|
|
1782
|
+
*
|
|
1783
|
+
* @param page - The live page.
|
|
1784
|
+
* @param selector - Where challenges are; `DEFAULT_CAPTCHA_SELECTOR` when omitted.
|
|
1785
|
+
* @param options - `v3`: also report a reCAPTCHA v3 script (a `captcha` step asks for it; the automatic checks do not, since v3 never blocks a page by itself).
|
|
1786
|
+
* @returns The challenge, or `undefined`.
|
|
1787
|
+
*/
|
|
1788
|
+
async function detectChallenge(page, selector = DEFAULT_CAPTCHA_SELECTOR, options = {}) {
|
|
1789
|
+
const matches = page.locator(selector);
|
|
1790
|
+
const count = Math.min(await matches.count(), MAX_CANDIDATES);
|
|
1791
|
+
for (let index = 0; index < count; index += 1) {
|
|
1792
|
+
const element = matches.nth(index);
|
|
1793
|
+
if (!(await element.isVisible())) continue;
|
|
1794
|
+
const facts = await element.evaluate(readWidget);
|
|
1795
|
+
return challengeOf(facts, page.url(), `${selector} >> nth=${index}`);
|
|
1796
|
+
}
|
|
1797
|
+
if (options.v3 !== true) return undefined;
|
|
1798
|
+
const script = page.locator(RECAPTCHA_V3_SCRIPT).first();
|
|
1799
|
+
if ((await script.count()) === 0) return undefined;
|
|
1800
|
+
const src = (await script.getAttribute('src')) ?? '';
|
|
1801
|
+
const siteKey = new URL(src, page.url()).searchParams.get('render') ?? undefined;
|
|
1802
|
+
return {
|
|
1803
|
+
kind: 'recaptcha-v3',
|
|
1804
|
+
url: page.url(),
|
|
1805
|
+
...(siteKey !== undefined && siteKey !== 'explicit' && {
|
|
1806
|
+
siteKey
|
|
1807
|
+
})
|
|
1808
|
+
};
|
|
1809
|
+
}
|
|
1810
|
+
/**
|
|
1811
|
+
* Whether the page is clear of challenges, tolerating a page that is
|
|
1812
|
+
* navigating (a solve often submits a form): an evaluation cut short by the
|
|
1813
|
+
* navigation counts as not clear yet.
|
|
1814
|
+
*
|
|
1815
|
+
* @param page - The live page.
|
|
1816
|
+
* @param selector - Where challenges are.
|
|
1817
|
+
* @returns Whether no challenge is visible.
|
|
1818
|
+
*/
|
|
1819
|
+
async function isClear(page, selector) {
|
|
1820
|
+
try {
|
|
1821
|
+
return (await detectChallenge(page, selector)) === undefined;
|
|
1822
|
+
} catch {
|
|
1823
|
+
return false;
|
|
1824
|
+
}
|
|
1825
|
+
}
|
|
1826
|
+
/** Runs inside the page. Keep it self-contained; it is serialised. */
|
|
1827
|
+
function readWidget(element) {
|
|
1828
|
+
const source = element.getAttribute('src') ?? element.querySelector('iframe')?.getAttribute('src') ?? '';
|
|
1829
|
+
return {
|
|
1830
|
+
tag: element.tagName.toLowerCase(),
|
|
1831
|
+
className: element.getAttribute('class') ?? '',
|
|
1832
|
+
src: source,
|
|
1833
|
+
siteKey: element.dataset.sitekey,
|
|
1834
|
+
action: element.dataset.action
|
|
1835
|
+
};
|
|
1836
|
+
}
|
|
1837
|
+
function challengeOf(facts, url, selector) {
|
|
1838
|
+
const siteKey = facts.siteKey ?? siteKeyIn(facts.src, url);
|
|
1839
|
+
return {
|
|
1840
|
+
kind: kindOf(facts),
|
|
1841
|
+
url,
|
|
1842
|
+
selector,
|
|
1843
|
+
...(siteKey !== undefined && {
|
|
1844
|
+
siteKey
|
|
1845
|
+
}),
|
|
1846
|
+
...(facts.action !== undefined && {
|
|
1847
|
+
action: facts.action
|
|
1848
|
+
})
|
|
1849
|
+
};
|
|
1850
|
+
}
|
|
1851
|
+
function kindOf(facts) {
|
|
1852
|
+
const hint = `${facts.className} ${facts.src}`.toLowerCase();
|
|
1853
|
+
if (hint.includes('recaptcha')) return 'recaptcha-v2';
|
|
1854
|
+
if (hint.includes('hcaptcha')) return 'hcaptcha';
|
|
1855
|
+
if (hint.includes('turnstile') || hint.includes('challenges.cloudflare.com')) return 'turnstile';
|
|
1856
|
+
return facts.tag === 'img' || facts.tag === 'canvas' ? 'image' : 'unknown';
|
|
1857
|
+
}
|
|
1858
|
+
/** A widget iframe carries its site key in the query: `k` (reCAPTCHA) or `sitekey` (hCaptcha). */
|
|
1859
|
+
function siteKeyIn(src, base) {
|
|
1860
|
+
if (src === '') return undefined;
|
|
1861
|
+
try {
|
|
1862
|
+
const parameters = new URL(src, base).searchParams;
|
|
1863
|
+
const hashParameters = new URLSearchParams(new URL(src, base).hash.slice(1));
|
|
1864
|
+
return parameters.get('k') ?? parameters.get('sitekey') ?? hashParameters.get('sitekey') ?? undefined;
|
|
1865
|
+
} catch {
|
|
1866
|
+
return undefined;
|
|
1867
|
+
}
|
|
1868
|
+
}
|
|
1869
|
+
|
|
1870
|
+
/** Default solves tried per challenge. */
|
|
1871
|
+
const DEFAULT_CAPTCHA_ATTEMPTS = 3;
|
|
1872
|
+
/** Default time one solve may take: token services take 10 to 60 seconds. */
|
|
1873
|
+
const DEFAULT_CAPTCHA_TIMEOUT_MS = 120_000;
|
|
1874
|
+
/** How long the page has to confirm a solve. */
|
|
1875
|
+
const VERIFY_TIMEOUT_MS = 10_000;
|
|
1876
|
+
const VERIFY_POLL_MS = 250;
|
|
1877
|
+
/**
|
|
1878
|
+
* Gets past one challenge: asks the solver, then checks the page (a solver's
|
|
1879
|
+
* `solved` is a claim; the challenge must be gone and/or the `verify` element
|
|
1880
|
+
* must appear), and tries again with what the page shows next, up to
|
|
1881
|
+
* `attempts`. Each try spends one solve of the run's budget.
|
|
1882
|
+
*
|
|
1883
|
+
* A failed try re-detects the challenge (a widget re-renders after a wrong
|
|
1884
|
+
* answer). When it is gone without the page confirming, the page is reloaded
|
|
1885
|
+
* for a fresh one; when a reload shows none, there is nothing left to solve.
|
|
1886
|
+
*
|
|
1887
|
+
* @param plan - The challenge, the solver and the limits.
|
|
1888
|
+
* @throws CaptchaError when every try failed, or the budget is spent.
|
|
1889
|
+
*/
|
|
1890
|
+
async function resolveCaptcha(plan) {
|
|
1891
|
+
const {
|
|
1892
|
+
recipeId,
|
|
1893
|
+
page,
|
|
1894
|
+
solver,
|
|
1895
|
+
events,
|
|
1896
|
+
budget
|
|
1897
|
+
} = plan;
|
|
1898
|
+
let challenge = plan.challenge;
|
|
1899
|
+
let reason = 'no attempt ran';
|
|
1900
|
+
for (let attempt = 1; attempt <= plan.attempts; attempt += 1) {
|
|
1901
|
+
if (!budget.take()) {
|
|
1902
|
+
events.emit({
|
|
1903
|
+
type: 'captcha:budget',
|
|
1904
|
+
recipeId,
|
|
1905
|
+
url: challenge.url,
|
|
1906
|
+
kind: challenge.kind,
|
|
1907
|
+
max: budget.max
|
|
1908
|
+
});
|
|
1909
|
+
throw new CaptchaError(challenge.url, challenge.kind, attempt - 1, `left unsolved: the run's ${budget.max} solves are spent (session.captcha.maxSolves)`);
|
|
1910
|
+
}
|
|
1911
|
+
events.emit({
|
|
1912
|
+
type: 'captcha:solve',
|
|
1913
|
+
recipeId,
|
|
1914
|
+
url: challenge.url,
|
|
1915
|
+
kind: challenge.kind,
|
|
1916
|
+
solver: solver.name,
|
|
1917
|
+
attempt
|
|
1918
|
+
});
|
|
1919
|
+
const started = Date.now();
|
|
1920
|
+
const outcome = await solveOnce(plan, challenge, attempt);
|
|
1921
|
+
if (outcome.status === 'solved' && (await confirmed(page, challenge, plan))) {
|
|
1922
|
+
events.emit({
|
|
1923
|
+
type: 'captcha:solved',
|
|
1924
|
+
recipeId,
|
|
1925
|
+
url: challenge.url,
|
|
1926
|
+
kind: challenge.kind,
|
|
1927
|
+
solver: solver.name,
|
|
1928
|
+
attempt,
|
|
1929
|
+
durationMs: Date.now() - started
|
|
1930
|
+
});
|
|
1931
|
+
return;
|
|
1932
|
+
}
|
|
1933
|
+
reason = outcome.status === 'failed' ? outcome.reason : 'the page still shows the challenge';
|
|
1934
|
+
events.emit({
|
|
1935
|
+
type: 'captcha:failed',
|
|
1936
|
+
recipeId,
|
|
1937
|
+
url: challenge.url,
|
|
1938
|
+
kind: challenge.kind,
|
|
1939
|
+
solver: solver.name,
|
|
1940
|
+
attempt,
|
|
1941
|
+
reason
|
|
1942
|
+
});
|
|
1943
|
+
if (attempt === plan.attempts) break;
|
|
1944
|
+
const next = await nextChallenge(page, challenge, plan.selector);
|
|
1945
|
+
if (next === undefined) return;
|
|
1946
|
+
challenge = next;
|
|
1947
|
+
}
|
|
1948
|
+
throw new CaptchaError(challenge.url, challenge.kind, plan.attempts, `not solved after ${plan.attempts} attempt${plan.attempts === 1 ? '' : 's'}: ${reason}`);
|
|
1949
|
+
}
|
|
1950
|
+
/** Runs the solver once, bounded by the timeout; a throw or a malformed answer is a failure. */
|
|
1951
|
+
async function solveOnce(plan, challenge, attempt) {
|
|
1952
|
+
const controller = new AbortController();
|
|
1953
|
+
let timer;
|
|
1954
|
+
const timeout = new Promise(resolve => {
|
|
1955
|
+
timer = setTimeout(() => {
|
|
1956
|
+
controller.abort();
|
|
1957
|
+
resolve({
|
|
1958
|
+
status: 'failed',
|
|
1959
|
+
reason: `the solver took longer than ${plan.timeoutMs} ms`
|
|
1960
|
+
});
|
|
1961
|
+
}, plan.timeoutMs);
|
|
1962
|
+
});
|
|
1963
|
+
const log = (level, message, meta) => {
|
|
1964
|
+
plan.events.emit({
|
|
1965
|
+
type: level === 'error' ? 'error' : 'warning',
|
|
1966
|
+
recipeId: plan.recipeId,
|
|
1967
|
+
message: `[${plan.solver.name}] ${message}`,
|
|
1968
|
+
meta
|
|
1969
|
+
});
|
|
1970
|
+
};
|
|
1971
|
+
const solving = (async () => {
|
|
1972
|
+
try {
|
|
1973
|
+
return outcomeOf(await plan.solver.solve(challenge, {
|
|
1974
|
+
page: plan.page,
|
|
1975
|
+
lease: plan.lease,
|
|
1976
|
+
attempt,
|
|
1977
|
+
signal: controller.signal,
|
|
1978
|
+
log
|
|
1979
|
+
}));
|
|
1980
|
+
} catch (error) {
|
|
1981
|
+
return {
|
|
1982
|
+
status: 'failed',
|
|
1983
|
+
reason: error instanceof Error ? error.message : String(error)
|
|
1984
|
+
};
|
|
1985
|
+
}
|
|
1986
|
+
})();
|
|
1987
|
+
try {
|
|
1988
|
+
return await Promise.race([solving, timeout]);
|
|
1989
|
+
} finally {
|
|
1990
|
+
clearTimeout(timer);
|
|
1991
|
+
}
|
|
1992
|
+
}
|
|
1993
|
+
function outcomeOf(value) {
|
|
1994
|
+
if (typeof value !== 'object' || value === null) return {
|
|
1995
|
+
status: 'failed',
|
|
1996
|
+
reason: 'the solver returned no outcome'
|
|
1997
|
+
};
|
|
1998
|
+
const outcome = value;
|
|
1999
|
+
if (outcome.status === 'solved') return {
|
|
2000
|
+
status: 'solved'
|
|
2001
|
+
};
|
|
2002
|
+
return {
|
|
2003
|
+
status: 'failed',
|
|
2004
|
+
reason: typeof outcome.reason === 'string' ? outcome.reason : 'the solver reported a failure'
|
|
2005
|
+
};
|
|
2006
|
+
}
|
|
2007
|
+
/**
|
|
2008
|
+
* Whether the page confirms the solve: the challenge is gone (unless
|
|
2009
|
+
* `verify.gone` is `false`, or the challenge has no widget, as with reCAPTCHA
|
|
2010
|
+
* v3) and the `verify.selector` element is visible, within ten seconds.
|
|
2011
|
+
*/
|
|
2012
|
+
async function confirmed(page, challenge, plan) {
|
|
2013
|
+
const needGone = plan.verify?.gone !== false && challenge.selector !== undefined;
|
|
2014
|
+
const shown = plan.verify?.selector;
|
|
2015
|
+
if (!needGone && shown === undefined) return true;
|
|
2016
|
+
const deadline = Date.now() + VERIFY_TIMEOUT_MS;
|
|
2017
|
+
for (;;) {
|
|
2018
|
+
const gone = !needGone || (await isClear(page, plan.selector));
|
|
2019
|
+
const visible = shown === undefined || (await isVisible(page, shown));
|
|
2020
|
+
if (gone && visible) return true;
|
|
2021
|
+
if (Date.now() >= deadline) return false;
|
|
2022
|
+
await page.waitForTimeout(VERIFY_POLL_MS);
|
|
2023
|
+
}
|
|
2024
|
+
}
|
|
2025
|
+
async function isVisible(page, selector) {
|
|
2026
|
+
try {
|
|
2027
|
+
return await page.locator(selector).first().isVisible();
|
|
2028
|
+
} catch {
|
|
2029
|
+
return false;
|
|
2030
|
+
}
|
|
2031
|
+
}
|
|
2032
|
+
/** The challenge to try next: what the page shows now, else what a reload shows, else none. A widgetless challenge is tried as it is. */
|
|
2033
|
+
async function nextChallenge(page, previous, selector) {
|
|
2034
|
+
if (previous.selector === undefined) return previous;
|
|
2035
|
+
const current = await detectChallenge(page, selector);
|
|
2036
|
+
if (current !== undefined) return current;
|
|
2037
|
+
await page.reload();
|
|
2038
|
+
return detectChallenge(page, selector);
|
|
2039
|
+
}
|
|
2040
|
+
|
|
2041
|
+
/**
|
|
2042
|
+
* Where a web runner meets captchas: after each navigation, click and key
|
|
2043
|
+
* press (`session.captcha`), on a block page (`onBlock.solve`), and at a
|
|
2044
|
+
* `captcha` step.
|
|
2045
|
+
*/
|
|
2046
|
+
class CaptchaGuard {
|
|
2047
|
+
options;
|
|
2048
|
+
constructor(options) {
|
|
2049
|
+
this.options = options;
|
|
2050
|
+
}
|
|
2051
|
+
settings() {
|
|
2052
|
+
const captcha = this.options.recipe.session?.captcha;
|
|
2053
|
+
if (captcha === undefined) return undefined;
|
|
2054
|
+
return {
|
|
2055
|
+
solver: captcha.solver,
|
|
2056
|
+
selector: captcha.detect?.selector ?? DEFAULT_CAPTCHA_SELECTOR,
|
|
2057
|
+
verify: captcha.verify,
|
|
2058
|
+
attempts: captcha.attempts ?? DEFAULT_CAPTCHA_ATTEMPTS,
|
|
2059
|
+
timeoutMs: captcha.timeoutMs ?? DEFAULT_CAPTCHA_TIMEOUT_MS
|
|
2060
|
+
};
|
|
2061
|
+
}
|
|
2062
|
+
async solve(page, challenge, settings) {
|
|
2063
|
+
const {
|
|
2064
|
+
recipe,
|
|
2065
|
+
events,
|
|
2066
|
+
solvers,
|
|
2067
|
+
budget,
|
|
2068
|
+
lease
|
|
2069
|
+
} = this.options;
|
|
2070
|
+
events.emit({
|
|
2071
|
+
type: 'captcha:detected',
|
|
2072
|
+
recipeId: recipe.id,
|
|
2073
|
+
url: challenge.url,
|
|
2074
|
+
kind: challenge.kind,
|
|
2075
|
+
siteKey: challenge.siteKey
|
|
2076
|
+
});
|
|
2077
|
+
await resolveCaptcha({
|
|
2078
|
+
recipeId: recipe.id,
|
|
2079
|
+
page,
|
|
2080
|
+
challenge,
|
|
2081
|
+
solver: solvers.resolve(settings.solver),
|
|
2082
|
+
selector: settings.selector,
|
|
2083
|
+
verify: settings.verify,
|
|
2084
|
+
attempts: settings.attempts,
|
|
2085
|
+
timeoutMs: settings.timeoutMs,
|
|
2086
|
+
budget,
|
|
2087
|
+
events,
|
|
2088
|
+
lease
|
|
2089
|
+
});
|
|
2090
|
+
}
|
|
2091
|
+
/** Whether a block page is searched for a challenge before the block counts. */
|
|
2092
|
+
get solvesBlocks() {
|
|
2093
|
+
const session = this.options.recipe.session;
|
|
2094
|
+
return session?.onBlock?.solve === true && session.captcha !== undefined;
|
|
2095
|
+
}
|
|
2096
|
+
/**
|
|
2097
|
+
* The automatic check: with `session.captcha`, solves the challenge the page
|
|
2098
|
+
* shows, if any. Without it, nothing is looked for.
|
|
2099
|
+
*
|
|
2100
|
+
* @param page - The live page.
|
|
2101
|
+
* @throws CaptchaError when the challenge could not be solved.
|
|
2102
|
+
*/
|
|
2103
|
+
async check(page) {
|
|
2104
|
+
const settings = this.settings();
|
|
2105
|
+
if (settings === undefined) return;
|
|
2106
|
+
const challenge = await detectChallenge(page, settings.selector);
|
|
2107
|
+
if (challenge !== undefined) await this.solve(page, challenge, settings);
|
|
2108
|
+
}
|
|
2109
|
+
/**
|
|
2110
|
+
* A block page under `onBlock.solve`: solves the challenge it shows. A block
|
|
2111
|
+
* without a challenge stays a block.
|
|
2112
|
+
*
|
|
2113
|
+
* @param page - The page showing the block.
|
|
2114
|
+
* @param blocked - The block.
|
|
2115
|
+
* @throws BlockedError (`blocked`) when the page shows no challenge; CaptchaError when it could not be solved.
|
|
2116
|
+
*/
|
|
2117
|
+
async solveBlock(page, blocked) {
|
|
2118
|
+
const settings = this.settings();
|
|
2119
|
+
if (settings === undefined) throw blocked;
|
|
2120
|
+
const challenge = await detectChallenge(page, settings.selector);
|
|
2121
|
+
if (challenge === undefined) throw blocked;
|
|
2122
|
+
await this.solve(page, challenge, settings);
|
|
2123
|
+
}
|
|
2124
|
+
/**
|
|
2125
|
+
* A `captcha` step: solves the challenge the page shows, reCAPTCHA v3
|
|
2126
|
+
* included; a page without one is fine.
|
|
2127
|
+
*
|
|
2128
|
+
* @param page - The live page.
|
|
2129
|
+
* @param step - The step.
|
|
2130
|
+
* @throws CaptchaError when the challenge could not be solved.
|
|
2131
|
+
*/
|
|
2132
|
+
async step(page, step) {
|
|
2133
|
+
const base = this.settings();
|
|
2134
|
+
const solver = step.solver ?? base?.solver;
|
|
2135
|
+
if (solver === undefined) throw new Error('a captcha step needs a solver: name one ("solver") or add session.captcha');
|
|
2136
|
+
const settings = {
|
|
2137
|
+
solver,
|
|
2138
|
+
selector: step.selector ?? base?.selector ?? DEFAULT_CAPTCHA_SELECTOR,
|
|
2139
|
+
verify: step.verify ?? base?.verify,
|
|
2140
|
+
attempts: step.attempts ?? base?.attempts ?? DEFAULT_CAPTCHA_ATTEMPTS,
|
|
2141
|
+
timeoutMs: step.timeoutMs ?? base?.timeoutMs ?? DEFAULT_CAPTCHA_TIMEOUT_MS
|
|
2142
|
+
};
|
|
2143
|
+
const challenge = await detectChallenge(page, settings.selector, {
|
|
2144
|
+
v3: true
|
|
2145
|
+
});
|
|
2146
|
+
if (challenge !== undefined) await this.solve(page, challenge, settings);
|
|
2147
|
+
}
|
|
2148
|
+
}
|
|
2149
|
+
/**
|
|
2150
|
+
* Every solver name a recipe uses (`session.captcha` and its `captcha`
|
|
2151
|
+
* steps, the bootstrap's included), to check them before the run starts.
|
|
2152
|
+
*
|
|
2153
|
+
* @param recipe - The input recipe.
|
|
2154
|
+
* @returns The names, without repeats.
|
|
2155
|
+
*/
|
|
2156
|
+
function captchaSolverNames(recipe) {
|
|
2157
|
+
const names = new Set();
|
|
2158
|
+
const fallback = recipe.session?.captcha?.solver;
|
|
2159
|
+
if (fallback !== undefined) names.add(fallback);
|
|
2160
|
+
const visit = steps => {
|
|
2161
|
+
for (const step of steps) {
|
|
2162
|
+
if (step.type === 'captcha') names.add(step.solver ?? fallback ?? '');
|
|
2163
|
+
if ('steps' in step) visit(step.steps);
|
|
2164
|
+
if (step.type === 'if') visit(step.else ?? []);
|
|
2165
|
+
}
|
|
2166
|
+
};
|
|
2167
|
+
visit(recipe.steps);
|
|
2168
|
+
visit(recipe.session?.bootstrap?.steps ?? []);
|
|
2169
|
+
names.delete('');
|
|
2170
|
+
return [...names];
|
|
2171
|
+
}
|
|
2172
|
+
|
|
1250
2173
|
/** Fans crawl events out to listeners. A listener that throws never breaks the crawl. */
|
|
1251
2174
|
class EventBus {
|
|
1252
2175
|
listeners = new Set();
|
|
@@ -1292,7 +2215,7 @@ function traceLine(event) {
|
|
|
1292
2215
|
}
|
|
1293
2216
|
case 'recipe:finish':
|
|
1294
2217
|
{
|
|
1295
|
-
return `■ ${event.recipeId}: ${event.emitted} emitted, ${event.rejected} rejected, ${event.duplicates} duplicates, ${event.skipped > 0 ? `${event.skipped} skipped, ` : ''}${event.pages} pages, ${event.durationMs} ms${event.error === undefined ? '' : `\n ✖ stopped: ${event.error}`}`;
|
|
2218
|
+
return `■ ${event.recipeId}: ${event.emitted} emitted, ${event.rejected} rejected, ${event.duplicates} duplicates, ${event.skipped > 0 ? `${event.skipped} skipped, ` : ''}${(event.stepsSkipped ?? 0) > 0 ? `${event.stepsSkipped} steps skipped, ` : ''}${event.pages} pages, ${event.durationMs} ms${event.error === undefined ? '' : `\n ✖ stopped: ${event.error}`}`;
|
|
1296
2219
|
}
|
|
1297
2220
|
case 'access:lease':
|
|
1298
2221
|
{
|
|
@@ -1306,6 +2229,26 @@ function traceLine(event) {
|
|
|
1306
2229
|
{
|
|
1307
2230
|
return `${indent(1)}↻ new access lease (attempt ${event.attempt})`;
|
|
1308
2231
|
}
|
|
2232
|
+
case 'captcha:detected':
|
|
2233
|
+
{
|
|
2234
|
+
return `${indent(1)}⚿ captcha ${event.kind} on ${event.url}`;
|
|
2235
|
+
}
|
|
2236
|
+
case 'captcha:solve':
|
|
2237
|
+
{
|
|
2238
|
+
return undefined;
|
|
2239
|
+
}
|
|
2240
|
+
case 'captcha:solved':
|
|
2241
|
+
{
|
|
2242
|
+
return `${indent(1)}✓ captcha solved by ${event.solver} (attempt ${event.attempt}, ${event.durationMs} ms)`;
|
|
2243
|
+
}
|
|
2244
|
+
case 'captcha:failed':
|
|
2245
|
+
{
|
|
2246
|
+
return `${indent(1)}✗ captcha attempt ${event.attempt} failed: ${event.reason}`;
|
|
2247
|
+
}
|
|
2248
|
+
case 'captcha:budget':
|
|
2249
|
+
{
|
|
2250
|
+
return `${indent(1)}⛔ captcha left unsolved: the run's ${event.max} solves are spent`;
|
|
2251
|
+
}
|
|
1309
2252
|
case 'page:visit':
|
|
1310
2253
|
{
|
|
1311
2254
|
return `${indent(1)}⇢ page ${event.number} ${event.url}${event.status === undefined || event.status >= 200 && event.status < 300 ? '' : ` [${event.status}]`}`;
|
|
@@ -1613,23 +2556,72 @@ function collapse(text) {
|
|
|
1613
2556
|
|
|
1614
2557
|
/**
|
|
1615
2558
|
* JSON that arrives as text: a `<script type="application/ld+json">` body, a
|
|
1616
|
-
* `data-*` attribute, a fetched document read as text. Sites wrap
|
|
1617
|
-
*
|
|
2559
|
+
* `data-*` attribute, a fetched document read as text. Sites wrap JSON in
|
|
2560
|
+
* things that are not JSON: comment guards around inline JSON-LD, prefixes
|
|
2561
|
+
* that stop a page from loading an API as a script, a JSONP callback, an
|
|
2562
|
+
* assignment in an inline script. Those wrappers are removed, but only after
|
|
2563
|
+
* the text failed to parse as it is, and what is left must still be strict
|
|
2564
|
+
* JSON: nothing is evaluated.
|
|
1618
2565
|
*/
|
|
1619
2566
|
/** Comment guards sites wrap inline JSON-LD in: a CDATA marker inside a block comment, or an HTML comment. */
|
|
1620
2567
|
const GUARDS = /^\s*(?:\/\*\s*<!\[CDATA\[\s*\*\/|<!\[CDATA\[|<!--)\s*|\s*(?:\/\*\s*\]\]>\s*\*\/|\]\]>|-->)\s*$/g;
|
|
2568
|
+
/** Anti-hijacking prefixes: `)]}'` (with or without a comma), `while(1);`, `for(;;);`. */
|
|
2569
|
+
const XSSI_PREFIX = /^\s*(?:\)\]\}'\s*,?|while\s*\(\s*1\s*\)\s*;|for\s*\(\s*;\s*;\s*\)\s*;)/;
|
|
2570
|
+
/** A JSONP call: `callback({...});`, the callback an identifier path. */
|
|
2571
|
+
const JSONP = /^\s*[$A-Z_][\w$]*(?:\.[$A-Z_][\w$]*)*\s*\(([\s\S]*)\)\s*(?:;\s*)?$/i;
|
|
2572
|
+
/** An assignment in an inline script: `window.__STATE__ = {...};`, with `var`, `let` or `const` or none. */
|
|
2573
|
+
const ASSIGNMENT = /^\s*(?:(?:var|let|const)\s+)?[$A-Z_a-z][\w$]*(?:\.[$A-Z_a-z][\w$]*|\[["'][^"']*["']\])*\s*=([\s\S]*)$/;
|
|
2574
|
+
/**
|
|
2575
|
+
* Parses text as JSON, or as JSON inside one of the wrappers sites put around
|
|
2576
|
+
* it: comment guards, an anti-hijacking prefix, a JSONP call, an assignment.
|
|
2577
|
+
* Valid JSON is always read as it is; a wrapper is only removed when that
|
|
2578
|
+
* fails.
|
|
2579
|
+
*
|
|
2580
|
+
* @param text - The text.
|
|
2581
|
+
* @returns The value, or the error the text as it is gave.
|
|
2582
|
+
*/
|
|
2583
|
+
function parseJsonLike(text) {
|
|
2584
|
+
const direct = parseJson$1(text);
|
|
2585
|
+
if ('value' in direct) return direct;
|
|
2586
|
+
const unguarded = text.replaceAll(GUARDS, '');
|
|
2587
|
+
const assigned = ASSIGNMENT.exec(unguarded)?.[1].trim().replace(/;$/, '');
|
|
2588
|
+
const candidates = [unguarded, unguarded.replace(XSSI_PREFIX, ''), JSONP.exec(unguarded)?.[1], assigned];
|
|
2589
|
+
for (const candidate of candidates) {
|
|
2590
|
+
if (candidate === undefined || candidate === text) continue;
|
|
2591
|
+
const parsed = parseJson$1(candidate);
|
|
2592
|
+
if ('value' in parsed) return parsed;
|
|
2593
|
+
}
|
|
2594
|
+
return direct;
|
|
2595
|
+
}
|
|
2596
|
+
/**
|
|
2597
|
+
* Parses JSON Lines (NDJSON): one JSON value per non-blank line.
|
|
2598
|
+
*
|
|
2599
|
+
* @param text - The text.
|
|
2600
|
+
* @param source - Where it came from, for the error.
|
|
2601
|
+
* @returns The values, in order.
|
|
2602
|
+
* @throws Error naming the source and the line that does not parse.
|
|
2603
|
+
*/
|
|
2604
|
+
function parseJsonLines(text, source) {
|
|
2605
|
+
const values = [];
|
|
2606
|
+
for (const [index, line] of text.split(/\r?\n/).entries()) {
|
|
2607
|
+
if (line.trim() === '') continue;
|
|
2608
|
+
const parsed = parseJson$1(line);
|
|
2609
|
+
if ('error' in parsed) throw new Error(`${source}: line ${index + 1} is not JSON (${parsed.error.message})`, {
|
|
2610
|
+
cause: parsed.error
|
|
2611
|
+
});
|
|
2612
|
+
values.push(parsed.value);
|
|
2613
|
+
}
|
|
2614
|
+
return values;
|
|
2615
|
+
}
|
|
1621
2616
|
/**
|
|
1622
|
-
* Parses text as JSON,
|
|
2617
|
+
* Parses text as JSON, wrappers removed (see {@link parseJsonLike}).
|
|
1623
2618
|
*
|
|
1624
2619
|
* @param text - The text.
|
|
1625
2620
|
* @returns The value, or `undefined` when it is not JSON.
|
|
1626
2621
|
*/
|
|
1627
2622
|
function tryParseJson(text) {
|
|
1628
|
-
|
|
1629
|
-
|
|
1630
|
-
} catch {
|
|
1631
|
-
return undefined;
|
|
1632
|
-
}
|
|
2623
|
+
const parsed = parseJsonLike(text);
|
|
2624
|
+
return 'value' in parsed ? parsed.value : undefined;
|
|
1633
2625
|
}
|
|
1634
2626
|
/**
|
|
1635
2627
|
* Parses text that must be JSON.
|
|
@@ -1653,463 +2645,1068 @@ function parseJsonText(text, id) {
|
|
|
1653
2645
|
* @returns A list of items.
|
|
1654
2646
|
*/
|
|
1655
2647
|
function dataItemsOf(value) {
|
|
1656
|
-
if (typeof value === 'string') return itemsOf
|
|
1657
|
-
if (Array.isArray(value)) return value.flatMap(entry => typeof entry === 'string' ? itemsOf
|
|
1658
|
-
return itemsOf
|
|
2648
|
+
if (typeof value === 'string') return itemsOf(tryParseJson(value));
|
|
2649
|
+
if (Array.isArray(value)) return value.flatMap(entry => typeof entry === 'string' ? itemsOf(tryParseJson(entry)) : [entry]);
|
|
2650
|
+
return itemsOf(value);
|
|
1659
2651
|
}
|
|
1660
|
-
function itemsOf
|
|
2652
|
+
function itemsOf(parsed) {
|
|
1661
2653
|
if (parsed === undefined || parsed === null) return [];
|
|
1662
2654
|
return Array.isArray(parsed) ? parsed : [parsed];
|
|
1663
2655
|
}
|
|
2656
|
+
function parseJson$1(text) {
|
|
2657
|
+
try {
|
|
2658
|
+
return {
|
|
2659
|
+
value: JSON.parse(text)
|
|
2660
|
+
};
|
|
2661
|
+
} catch (error) {
|
|
2662
|
+
return {
|
|
2663
|
+
error: error
|
|
2664
|
+
};
|
|
2665
|
+
}
|
|
2666
|
+
}
|
|
1664
2667
|
|
|
1665
2668
|
/**
|
|
1666
|
-
*
|
|
1667
|
-
*
|
|
1668
|
-
* `
|
|
2669
|
+
* The text a `regex` extract reads: per visible slide, its title, its text
|
|
2670
|
+
* boxes in reading order, its tables' rows (cells separated by a tab) and its
|
|
2671
|
+
* notes after `Notes:`; slides separated by a blank line.
|
|
1669
2672
|
*
|
|
1670
|
-
*
|
|
1671
|
-
*
|
|
2673
|
+
* @param document - The deck.
|
|
2674
|
+
* @returns The text.
|
|
2675
|
+
*/
|
|
2676
|
+
function deckText(document) {
|
|
2677
|
+
return document.slides.filter(slide => !slide.hidden).map(slide => [...slide.shapes.map(shape => shape.text), ...slide.tables.flatMap(table => table.rows.map(row => row.map(String).join('\t'))), ...(slide.notes === '' ? [] : [`Notes: ${slide.notes}`])].join('\n')).join('\n\n');
|
|
2678
|
+
}
|
|
2679
|
+
/**
|
|
2680
|
+
* Whether a value bound in scope is a read deck (so `extract … from` can take it).
|
|
1672
2681
|
*
|
|
1673
|
-
* @param
|
|
1674
|
-
* @
|
|
1675
|
-
* @param walk - Runs a step list; also carries the emit callback.
|
|
1676
|
-
* @returns `stop` when the crawl reached its record limit.
|
|
2682
|
+
* @param value - Anything.
|
|
2683
|
+
* @returns Whether it is a {@link DeckDocument}.
|
|
1677
2684
|
*/
|
|
1678
|
-
|
|
1679
|
-
|
|
1680
|
-
|
|
1681
|
-
|
|
1682
|
-
|
|
1683
|
-
|
|
2685
|
+
function isDeckDocument(value) {
|
|
2686
|
+
return typeof value === 'object' && value !== null && value.kind === 'deck' && Array.isArray(value.slides);
|
|
2687
|
+
}
|
|
2688
|
+
|
|
2689
|
+
/**
|
|
2690
|
+
* Reads a `.pptx` presentation into a deck document, through
|
|
2691
|
+
* `@opencraw/office-reader`: every slide's text boxes with their positions,
|
|
2692
|
+
* its tables with their merged cells, its charts' cached data and its notes.
|
|
2693
|
+
* The reader is imported on first use, so recipes that never read a
|
|
2694
|
+
* presentation never load it.
|
|
2695
|
+
*
|
|
2696
|
+
* @param bytes - The file.
|
|
2697
|
+
* @param source - Where it came from, for messages.
|
|
2698
|
+
* @returns The deck.
|
|
2699
|
+
* @throws Error naming the source, and saying what to do, for a file that is
|
|
2700
|
+
* not a readable presentation (a legacy `.ppt`, a password-protected file, an `.odp`…).
|
|
2701
|
+
*/
|
|
2702
|
+
async function readPptxDeck(bytes, source) {
|
|
2703
|
+
const {
|
|
2704
|
+
readPptx,
|
|
2705
|
+
OfficeReadError
|
|
2706
|
+
} = await import('@opencraw/office-reader/pptx');
|
|
2707
|
+
try {
|
|
2708
|
+
const deck = await readPptx(bytes);
|
|
2709
|
+
return {
|
|
2710
|
+
kind: 'deck',
|
|
2711
|
+
width: deck.width,
|
|
2712
|
+
height: deck.height,
|
|
2713
|
+
slides: deck.slides.map(slide => ({
|
|
2714
|
+
...slide,
|
|
2715
|
+
tables: slide.tables.map(table => ({
|
|
2716
|
+
name: table.name,
|
|
2717
|
+
rows: table.rows,
|
|
2718
|
+
merges: table.merges
|
|
2719
|
+
}))
|
|
2720
|
+
}))
|
|
2721
|
+
};
|
|
2722
|
+
} catch (error) {
|
|
2723
|
+
if (error instanceof OfficeReadError) throw new Error(`${source}: ${error.message}`, {
|
|
2724
|
+
cause: error
|
|
2725
|
+
});
|
|
2726
|
+
throw error;
|
|
1684
2727
|
}
|
|
1685
|
-
return 'continue';
|
|
1686
2728
|
}
|
|
1687
|
-
|
|
1688
|
-
|
|
1689
|
-
|
|
1690
|
-
|
|
1691
|
-
|
|
1692
|
-
|
|
2729
|
+
|
|
2730
|
+
/** Runs closer than this share of the font size join into one cell. */
|
|
2731
|
+
const JOIN_GAP = 0.35;
|
|
2732
|
+
/** Runs further apart than this share of the font size get a space between them when joined. */
|
|
2733
|
+
const SPACE_GAP = 0.1;
|
|
2734
|
+
/** Runs on baselines closer than this share of the font size are on one line. */
|
|
2735
|
+
const SAME_BASELINE = 0.2;
|
|
2736
|
+
/** A cell joins a row when this share of its height overlaps the row. */
|
|
2737
|
+
const ROW_OVERLAP = 0.4;
|
|
2738
|
+
/**
|
|
2739
|
+
* Turns a page's text runs into rows of cells, top to bottom.
|
|
2740
|
+
*
|
|
2741
|
+
* Runs on one baseline that nearly touch become one cell. Cells whose vertical
|
|
2742
|
+
* extents overlap become one row, even when their baselines differ: a table
|
|
2743
|
+
* that centres its cells vertically puts a one-line value a few points above
|
|
2744
|
+
* or below its two-line label, and a row built from equal baselines would pair
|
|
2745
|
+
* the value with the wrong label.
|
|
2746
|
+
*
|
|
2747
|
+
* @param runs - The page's text runs, in any order.
|
|
2748
|
+
* @returns The rows.
|
|
2749
|
+
*/
|
|
2750
|
+
function assembleRows(runs) {
|
|
2751
|
+
return rowsOfCells(joinCells(runs));
|
|
1693
2752
|
}
|
|
1694
2753
|
/**
|
|
1695
|
-
*
|
|
1696
|
-
*
|
|
1697
|
-
*
|
|
2754
|
+
* Groups finished cells into rows, top to bottom: cells whose vertical extents
|
|
2755
|
+
* overlap share a row. For cells that need no joining, such as a slide's text
|
|
2756
|
+
* boxes, each already a cell.
|
|
2757
|
+
*
|
|
2758
|
+
* @param cells - The cells, in any order.
|
|
2759
|
+
* @returns The rows.
|
|
1698
2760
|
*/
|
|
1699
|
-
|
|
1700
|
-
|
|
1701
|
-
|
|
1702
|
-
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
|
|
1706
|
-
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
1714
|
-
|
|
2761
|
+
function rowsOfCells(cells) {
|
|
2762
|
+
const ordered = [...cells].sort((a, b) => middle(b) - middle(a) || a.x - b.x);
|
|
2763
|
+
const rows = [];
|
|
2764
|
+
let top = 0;
|
|
2765
|
+
let bottom = 0;
|
|
2766
|
+
for (const cell of ordered) {
|
|
2767
|
+
const current = rows.at(-1);
|
|
2768
|
+
const overlap = Math.min(cell.y + cell.height, top) - Math.max(cell.y, bottom);
|
|
2769
|
+
if (current !== undefined && overlap >= ROW_OVERLAP * cell.height) {
|
|
2770
|
+
current.push(cell);
|
|
2771
|
+
top = Math.max(top, cell.y + cell.height);
|
|
2772
|
+
bottom = Math.min(bottom, cell.y);
|
|
2773
|
+
} else {
|
|
2774
|
+
rows.push([cell]);
|
|
2775
|
+
top = cell.y + cell.height;
|
|
2776
|
+
bottom = cell.y;
|
|
1715
2777
|
}
|
|
1716
|
-
}
|
|
1717
|
-
|
|
1718
|
-
|
|
1719
|
-
|
|
1720
|
-
|
|
1721
|
-
|
|
1722
|
-
|
|
2778
|
+
}
|
|
2779
|
+
return rows.map(row => rowOf(row));
|
|
2780
|
+
}
|
|
2781
|
+
function joinCells(runs) {
|
|
2782
|
+
// Whitespace runs are gaps, not text: pdf.js emits the space between two
|
|
2783
|
+
// table columns as one wide " ", which would bridge the columns.
|
|
2784
|
+
// Left to right within a line, lines top to bottom: a run may sit a fraction of a point off its neighbours' baseline.
|
|
2785
|
+
const ordered = runs.filter(run => run.text.trim() !== '').sort((a, b) => Math.abs(a.y - b.y) <= SAME_BASELINE * Math.max(a.height, b.height, 1) ? a.x - b.x : b.y - a.y);
|
|
2786
|
+
const cells = [];
|
|
2787
|
+
for (const run of ordered) {
|
|
2788
|
+
const previous = lastOf(cells, cell => Math.abs(cell.y - run.y) <= SAME_BASELINE * Math.max(cell.height, run.height, 1));
|
|
2789
|
+
const size = Math.max(run.height, previous?.height ?? 0, 1);
|
|
2790
|
+
const gap = previous === undefined ? Infinity : run.x - (previous.x + previous.width);
|
|
2791
|
+
if (previous !== undefined && gap <= JOIN_GAP * size && gap > -size) {
|
|
2792
|
+
const space = gap > SPACE_GAP * size && !/\s$/.test(previous.text) && !/^\s/.test(run.text) ? ' ' : '';
|
|
2793
|
+
previous.text += space + run.text;
|
|
2794
|
+
previous.width = run.x + run.width - previous.x;
|
|
2795
|
+
previous.height = Math.max(previous.height, run.height);
|
|
2796
|
+
} else {
|
|
2797
|
+
cells.push({
|
|
2798
|
+
...run
|
|
2799
|
+
});
|
|
1723
2800
|
}
|
|
1724
|
-
tasks.push(iterate(item, release));
|
|
1725
2801
|
}
|
|
1726
|
-
|
|
1727
|
-
|
|
1728
|
-
|
|
2802
|
+
return cells.map(cell => ({
|
|
2803
|
+
...cell,
|
|
2804
|
+
text: cell.text.trim()
|
|
2805
|
+
})).filter(cell => cell.text !== '');
|
|
1729
2806
|
}
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
|
|
2807
|
+
function rowOf(cells) {
|
|
2808
|
+
const ordered = [...cells].sort((a, b) => a.x - b.x || b.y - a.y);
|
|
2809
|
+
return {
|
|
2810
|
+
top: Math.max(...ordered.map(cell => cell.y + cell.height)),
|
|
2811
|
+
bottom: Math.min(...ordered.map(cell => cell.y)),
|
|
2812
|
+
cells: ordered,
|
|
2813
|
+
text: ordered.map(cell => cell.text).join('\t')
|
|
2814
|
+
};
|
|
2815
|
+
}
|
|
2816
|
+
function middle(cell) {
|
|
2817
|
+
return cell.y + cell.height / 2;
|
|
2818
|
+
}
|
|
2819
|
+
/** `Array#findLast`, which the es2022 library does not declare. */
|
|
2820
|
+
function lastOf(items, test) {
|
|
2821
|
+
for (let index = items.length - 1; index >= 0; index -= 1) if (test(items[index])) return items[index];
|
|
2822
|
+
return undefined;
|
|
1737
2823
|
}
|
|
1738
2824
|
|
|
2825
|
+
/** A PDF that cannot be read: not a PDF, encrypted, or with no text to read. */
|
|
2826
|
+
class PdfReadError extends Error {
|
|
2827
|
+
name = 'PdfReadError';
|
|
2828
|
+
}
|
|
1739
2829
|
/**
|
|
1740
|
-
*
|
|
1741
|
-
*
|
|
1742
|
-
* is
|
|
1743
|
-
*
|
|
1744
|
-
* The runner reports the visit of each new page; this only steers.
|
|
2830
|
+
* Reads a PDF's text layer into rows of positioned cells, with pdf.js. pdf.js
|
|
2831
|
+
* is imported on first use, so recipes that never read a PDF never load it.
|
|
2832
|
+
* Only text is read: no page is rendered, no script runs, no font is loaded.
|
|
1745
2833
|
*
|
|
1746
|
-
* @param
|
|
1747
|
-
* @param
|
|
1748
|
-
* @
|
|
1749
|
-
* @
|
|
2834
|
+
* @param bytes - The file.
|
|
2835
|
+
* @param source - Where it came from, for messages.
|
|
2836
|
+
* @returns The document.
|
|
2837
|
+
* @throws PdfReadError when the bytes are not a readable PDF, or no page has a
|
|
2838
|
+
* text layer (a scan: OCR is not supported).
|
|
1750
2839
|
*/
|
|
1751
|
-
async function
|
|
1752
|
-
|
|
1753
|
-
|
|
1754
|
-
|
|
1755
|
-
|
|
1756
|
-
|
|
1757
|
-
|
|
2840
|
+
async function readPdf(bytes, source = 'PDF') {
|
|
2841
|
+
const pdfjs = await import('pdfjs-dist/legacy/build/pdf.mjs');
|
|
2842
|
+
// pdf.js takes ownership of the buffer it is given, and refuses a Node
|
|
2843
|
+
// Buffer: hand it a plain copy.
|
|
2844
|
+
const task = pdfjs.getDocument({
|
|
2845
|
+
data: new Uint8Array(bytes),
|
|
2846
|
+
verbosity: 0,
|
|
2847
|
+
disableFontFace: true,
|
|
2848
|
+
useSystemFonts: false,
|
|
2849
|
+
stopAtErrors: true
|
|
2850
|
+
});
|
|
2851
|
+
let loaded;
|
|
2852
|
+
try {
|
|
2853
|
+
loaded = await task.promise;
|
|
2854
|
+
} catch (error) {
|
|
2855
|
+
throw new PdfReadError(`${source}: not a readable PDF (${error.message})`, {
|
|
2856
|
+
cause: error
|
|
1758
2857
|
});
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1766
|
-
|
|
1767
|
-
|
|
1768
|
-
|
|
1769
|
-
url: next.url,
|
|
1770
|
-
number
|
|
2858
|
+
}
|
|
2859
|
+
try {
|
|
2860
|
+
const pages = [];
|
|
2861
|
+
for (let number = 1; number <= loaded.numPages; number += 1) {
|
|
2862
|
+
const page = await loaded.getPage(number);
|
|
2863
|
+
const {
|
|
2864
|
+
width,
|
|
2865
|
+
height
|
|
2866
|
+
} = page.getViewport({
|
|
2867
|
+
scale: 1
|
|
1771
2868
|
});
|
|
1772
|
-
|
|
1773
|
-
|
|
1774
|
-
|
|
1775
|
-
number
|
|
2869
|
+
const content = await page.getTextContent();
|
|
2870
|
+
const runs = content.items.flatMap(item => 'str' in item ? [runOf(item)] : []);
|
|
2871
|
+
pages.push({
|
|
2872
|
+
number,
|
|
2873
|
+
width,
|
|
2874
|
+
height,
|
|
2875
|
+
rows: assembleRows(runs)
|
|
1776
2876
|
});
|
|
1777
|
-
bound = {
|
|
1778
|
-
name: next.name,
|
|
1779
|
-
value: next.value
|
|
1780
|
-
};
|
|
1781
2877
|
}
|
|
2878
|
+
if (pages.every(page => page.rows.length === 0)) throw new PdfReadError(`${source}: no page has a text layer (a scanned PDF? OCR is not supported)`);
|
|
2879
|
+
return {
|
|
2880
|
+
kind: 'pdf',
|
|
2881
|
+
pages
|
|
2882
|
+
};
|
|
2883
|
+
} finally {
|
|
2884
|
+
await task.destroy();
|
|
1782
2885
|
}
|
|
1783
|
-
|
|
2886
|
+
}
|
|
2887
|
+
function runOf(item) {
|
|
2888
|
+
const [a, b, c, d, x, y] = item.transform;
|
|
2889
|
+
return {
|
|
2890
|
+
x,
|
|
2891
|
+
y,
|
|
2892
|
+
width: item.width,
|
|
2893
|
+
height: item.height > 0 ? item.height : Math.hypot(c, d) || Math.hypot(a, b),
|
|
2894
|
+
text: item.str
|
|
2895
|
+
};
|
|
1784
2896
|
}
|
|
1785
2897
|
|
|
1786
|
-
const FAIL = {
|
|
1787
|
-
policy: 'fail'
|
|
1788
|
-
};
|
|
1789
2898
|
/**
|
|
1790
|
-
* The
|
|
2899
|
+
* The text a `regex` extract reads: one line per row (cells separated by a
|
|
2900
|
+
* tab), pages separated by a blank line.
|
|
1791
2901
|
*
|
|
1792
|
-
* @param
|
|
1793
|
-
* @
|
|
1794
|
-
* @returns The policy to apply.
|
|
2902
|
+
* @param document - The PDF.
|
|
2903
|
+
* @returns The text.
|
|
1795
2904
|
*/
|
|
1796
|
-
function
|
|
1797
|
-
return
|
|
2905
|
+
function pdfText(document) {
|
|
2906
|
+
return document.pages.map(page => page.rows.map(row => row.text).join('\n')).join('\n\n');
|
|
1798
2907
|
}
|
|
1799
2908
|
/**
|
|
1800
|
-
*
|
|
2909
|
+
* Whether a value bound in scope is a read PDF (so `extract … from` can take it).
|
|
1801
2910
|
*
|
|
1802
|
-
* @param
|
|
1803
|
-
* @
|
|
1804
|
-
* @returns Milliseconds.
|
|
2911
|
+
* @param value - Anything.
|
|
2912
|
+
* @returns Whether it is a {@link PdfDocument}.
|
|
1805
2913
|
*/
|
|
1806
|
-
function
|
|
1807
|
-
return
|
|
1808
|
-
}
|
|
1809
|
-
function sleep(ms) {
|
|
1810
|
-
return ms <= 0 ? Promise.resolve() : new Promise(resolve => setTimeout(resolve, ms));
|
|
2914
|
+
function isPdfDocument(value) {
|
|
2915
|
+
return typeof value === 'object' && value !== null && value.kind === 'pdf' && Array.isArray(value.pages);
|
|
1811
2916
|
}
|
|
1812
2917
|
|
|
1813
|
-
/**
|
|
1814
|
-
|
|
1815
|
-
|
|
1816
|
-
|
|
1817
|
-
|
|
1818
|
-
|
|
1819
|
-
|
|
1820
|
-
|
|
1821
|
-
|
|
1822
|
-
|
|
1823
|
-
|
|
2918
|
+
/**
|
|
2919
|
+
* Finds every table whose header row matches, and reads its rows by column.
|
|
2920
|
+
*
|
|
2921
|
+
* Columns come from the body, not the header: a header is often centred over
|
|
2922
|
+
* a column whose cells are left-aligned, so the header's position says little
|
|
2923
|
+
* about where the column starts. The left edges of the body cells cluster into
|
|
2924
|
+
* bands; each band belongs to the header cell that overlaps it most, or the
|
|
2925
|
+
* nearest one when none does. A header spanning two columns reads both.
|
|
2926
|
+
*
|
|
2927
|
+
* A cell wrapped over several lines (a long name, a note, a list of versions)
|
|
2928
|
+
* spreads one row over several lines, its values often centred beside it:
|
|
2929
|
+
* the lines are regrouped into rows around the lines that carry values.
|
|
2930
|
+
*
|
|
2931
|
+
* @param document - The PDF.
|
|
2932
|
+
* @param query - Which tables, and how to name their columns.
|
|
2933
|
+
* @returns The tables, in page order.
|
|
2934
|
+
*/
|
|
2935
|
+
function findTables(document, query) {
|
|
2936
|
+
const tables = [];
|
|
2937
|
+
for (const page of document.pages) {
|
|
2938
|
+
const starts = page.rows.flatMap((row, index) => query.header.test(plain$1(row)) ? [index] : []);
|
|
2939
|
+
for (const [position, start] of starts.entries()) {
|
|
2940
|
+
const body = bodyOf(page.rows.slice(start + 1, starts[position + 1] ?? page.rows.length), query.until);
|
|
2941
|
+
tables.push(readTable(page.number, page.rows[start], body, query));
|
|
2942
|
+
}
|
|
1824
2943
|
}
|
|
2944
|
+
return tables;
|
|
1825
2945
|
}
|
|
1826
|
-
|
|
1827
|
-
|
|
1828
|
-
|
|
1829
|
-
|
|
1830
|
-
|
|
1831
|
-
|
|
1832
|
-
|
|
1833
|
-
|
|
1834
|
-
|
|
1835
|
-
|
|
1836
|
-
|
|
1837
|
-
|
|
2946
|
+
/** The rows under a header, up to the first one `until` matches. */
|
|
2947
|
+
function bodyOf(rows, until) {
|
|
2948
|
+
const end = until === undefined ? -1 : rows.findIndex(row => until.test(plain$1(row)));
|
|
2949
|
+
return end === -1 ? [...rows] : rows.slice(0, end);
|
|
2950
|
+
}
|
|
2951
|
+
function readTable(page, headerRow, body, query) {
|
|
2952
|
+
const headers = headerCells(headerRow);
|
|
2953
|
+
const bands = bandsOf(body, headers);
|
|
2954
|
+
const lines = body.map(row => ({
|
|
2955
|
+
row,
|
|
2956
|
+
values: valuesOf(row, bands, headers.length)
|
|
2957
|
+
}));
|
|
2958
|
+
const groups = groupLines(lines, query.align ?? 'auto');
|
|
2959
|
+
return {
|
|
2960
|
+
page,
|
|
2961
|
+
title: headers[0]?.text ?? '',
|
|
2962
|
+
header: headers.map(header => header.text),
|
|
2963
|
+
rows: groups.map(group => named$1(joinLines(group, headers.length), headers, query.columns))
|
|
2964
|
+
};
|
|
2965
|
+
}
|
|
2966
|
+
/**
|
|
2967
|
+
* Groups the body's lines into table rows. A line anchors a row when it has a
|
|
2968
|
+
* name (the first column) and a value, or, without a name, a value in the
|
|
2969
|
+
* first value column: the middle line of a name wrapped over several lines,
|
|
2970
|
+
* the values centred beside it. Every other line (a wrapped name, a wrapped
|
|
2971
|
+
* note, a list of versions over two lines) joins an anchor:
|
|
2972
|
+
*
|
|
2973
|
+
* - `top`: the nearest anchor above it; `bottom`: the nearest below;
|
|
2974
|
+
* - `center`: a wrapped name splits evenly around its anchor (as many lines
|
|
2975
|
+
* below as above), and other lines join the nearest anchor;
|
|
2976
|
+
* - `auto`: `center` when some line carries values but no name (only a
|
|
2977
|
+
* centred table does that), else the nearest anchor, a tie going to the one
|
|
2978
|
+
* below (a wrapped cell's first line comes before its row).
|
|
2979
|
+
*/
|
|
2980
|
+
function groupLines(lines, align) {
|
|
2981
|
+
const named = line => line.values[0] !== '';
|
|
2982
|
+
const valued = line => line.values.slice(1).some(value => value !== '');
|
|
2983
|
+
let anchors = lines.filter(line => named(line) && valued(line) || !named(line) && (line.values[1] ?? '') !== '');
|
|
2984
|
+
if (anchors.length === 0) anchors = lines.filter(line => named(line));
|
|
2985
|
+
if (anchors.length === 0) return [];
|
|
2986
|
+
const groups = new Map(anchors.map(anchor => [anchor, [anchor]]));
|
|
2987
|
+
const centred = align === 'center' || align === 'auto' && anchors.some(anchor => !named(anchor));
|
|
2988
|
+
if (centred) splitNamesEvenly(lines, anchors, named, groups);
|
|
2989
|
+
const placed = new Set();
|
|
2990
|
+
for (const members of groups.values()) for (const line of members) placed.add(line);
|
|
2991
|
+
for (const line of lines) if (!placed.has(line)) groups.get(ownerOf(line, anchors, align))?.push(line);
|
|
2992
|
+
// An anchor that ended up with no name at all is a value spilling out of the row next to it.
|
|
2993
|
+
const nameless = anchors.filter(anchor => (groups.get(anchor) ?? []).every(line => !named(line)));
|
|
2994
|
+
const kept = anchors.filter(anchor => !nameless.includes(anchor));
|
|
2995
|
+
if (kept.length === 0) return [];
|
|
2996
|
+
for (const anchor of nameless) groups.get(nearest(anchor, kept))?.push(...(groups.get(anchor) ?? []));
|
|
2997
|
+
return kept.map(anchor => groups.get(anchor) ?? []);
|
|
2998
|
+
}
|
|
2999
|
+
/**
|
|
3000
|
+
* Hands the wrapped name lines between two anchors out evenly: the upper
|
|
3001
|
+
* anchor takes as many lines below it as it took above it, the lower one the
|
|
3002
|
+
* rest. Lines are in page order, top to bottom.
|
|
3003
|
+
*/
|
|
3004
|
+
function splitNamesEvenly(lines, anchors, named, groups) {
|
|
3005
|
+
const positions = anchors.map(anchor => lines.indexOf(anchor));
|
|
3006
|
+
const namesBetween = (from, to) => lines.slice(from, to).filter(line => named(line) && !groups.has(line));
|
|
3007
|
+
let above = namesBetween(0, positions[0]);
|
|
3008
|
+
groups.get(anchors[0])?.push(...above);
|
|
3009
|
+
for (const [index, anchor] of anchors.entries()) {
|
|
3010
|
+
const next = anchors[index + 1];
|
|
3011
|
+
const run = namesBetween(positions[index] + 1, next === undefined ? lines.length : positions[index + 1]);
|
|
3012
|
+
const taken = next === undefined ? run.length : Math.min(run.length, above.length);
|
|
3013
|
+
groups.get(anchor)?.push(...run.slice(0, taken));
|
|
3014
|
+
above = run.slice(taken);
|
|
3015
|
+
if (next !== undefined) groups.get(next)?.push(...above);
|
|
3016
|
+
}
|
|
3017
|
+
}
|
|
3018
|
+
function ownerOf(line, anchors, align) {
|
|
3019
|
+
if (align === 'top') return lastOf(anchors, anchor => anchor.row.bottom >= line.row.bottom) ?? anchors[0];
|
|
3020
|
+
if (align === 'bottom') return anchors.find(anchor => anchor.row.bottom <= line.row.bottom) ?? anchors.at(-1) ?? anchors[0];
|
|
3021
|
+
return nearest(line, anchors);
|
|
3022
|
+
}
|
|
3023
|
+
/** A row's text per column: its lines top to bottom, each column's pieces joined by spaces. */
|
|
3024
|
+
function joinLines(group, width) {
|
|
3025
|
+
const ordered = [...group].sort((a, b) => b.row.top - a.row.top);
|
|
3026
|
+
return Array.from({
|
|
3027
|
+
length: width
|
|
3028
|
+
}, (_, column) => ordered.map(line => line.values[column]).filter(value => value !== '').join(' '));
|
|
3029
|
+
}
|
|
3030
|
+
/** Header cells that overlap horizontally (a header on two lines) are one header. */
|
|
3031
|
+
function headerCells(row) {
|
|
3032
|
+
const merged = [];
|
|
3033
|
+
for (const cell of row.cells) {
|
|
3034
|
+
const previous = merged.at(-1);
|
|
3035
|
+
if (previous !== undefined && cell.x < previous.x + previous.width) {
|
|
3036
|
+
previous.text = `${previous.text} ${cell.text}`;
|
|
3037
|
+
previous.width = Math.max(previous.x + previous.width, cell.x + cell.width) - previous.x;
|
|
3038
|
+
} else {
|
|
3039
|
+
merged.push({
|
|
3040
|
+
...cell
|
|
3041
|
+
});
|
|
3042
|
+
}
|
|
3043
|
+
}
|
|
3044
|
+
return merged;
|
|
3045
|
+
}
|
|
3046
|
+
function bandsOf(body, headers) {
|
|
3047
|
+
const cells = body.flatMap(row => row.cells);
|
|
3048
|
+
if (cells.length === 0 || headers.length === 0) return [];
|
|
3049
|
+
const tolerance = Math.max(3, median(cells.map(cell => cell.height)) * 0.6);
|
|
3050
|
+
const edges = cells.map(cell => cell.x).sort((a, b) => a - b);
|
|
3051
|
+
const starts = [];
|
|
3052
|
+
for (const edge of edges) if (starts.length === 0 || edge - (starts.at(-1) ?? 0) > tolerance) starts.push(edge);
|
|
3053
|
+
const spans = starts.map((start, index) => {
|
|
3054
|
+
const next = starts[index + 1] ?? Infinity;
|
|
3055
|
+
// A band spans what its cells cover, not the gap up to the next band.
|
|
3056
|
+
const right = Math.max(...cells.filter(cell => cell.x >= start - 0.5 && cell.x < next - 0.5).map(cell => cell.x + cell.width));
|
|
3057
|
+
return {
|
|
3058
|
+
start,
|
|
3059
|
+
end: Math.min(next, right)
|
|
3060
|
+
};
|
|
3061
|
+
});
|
|
3062
|
+
const columns = assignColumns(spans, headers);
|
|
3063
|
+
return spans.map((span, index) => ({
|
|
3064
|
+
start: span.start,
|
|
3065
|
+
column: columns[index]
|
|
3066
|
+
}));
|
|
3067
|
+
}
|
|
3068
|
+
/**
|
|
3069
|
+
* Maps bands to headers, left to right: columns never cross, so the mapping
|
|
3070
|
+
* is monotone. Among monotone mappings it first uses as many headers as it
|
|
3071
|
+
* can (a table with as many bands as headers maps one to one), then prefers
|
|
3072
|
+
* the one where bands overlap their header most, or sit nearest to it. The
|
|
3073
|
+
* leftover choice is which neighbouring bands share a header: one header
|
|
3074
|
+
* over two columns, or a column whose cells start at two edges.
|
|
3075
|
+
*
|
|
3076
|
+
* @param spans - The bands, left to right.
|
|
3077
|
+
* @param headers - The header cells, left to right.
|
|
3078
|
+
* @returns The header index of each band.
|
|
3079
|
+
*/
|
|
3080
|
+
function assignColumns(spans, headers) {
|
|
3081
|
+
const affinity = (span, header) => {
|
|
3082
|
+
const overlap = overlapOf(header, span);
|
|
3083
|
+
return overlap > 0 ? overlap : -Math.max(0, header.x - span.end, span.start - (header.x + header.width));
|
|
3084
|
+
};
|
|
3085
|
+
const better = (a, b) => b === undefined || a.used > b.used || a.used === b.used && a.affinity > b.affinity;
|
|
3086
|
+
const table = [headers.map(header => ({
|
|
3087
|
+
used: 1,
|
|
3088
|
+
affinity: affinity(spans[0], header),
|
|
3089
|
+
previous: -1
|
|
3090
|
+
}))];
|
|
3091
|
+
for (const span of spans.slice(1)) {
|
|
3092
|
+
const last = table.at(-1) ?? [];
|
|
3093
|
+
table.push(headers.map((header, column) => {
|
|
3094
|
+
let best;
|
|
3095
|
+
for (let from = 0; from <= column; from += 1) {
|
|
3096
|
+
const candidate = {
|
|
3097
|
+
used: last[from].used + (from === column ? 0 : 1),
|
|
3098
|
+
affinity: last[from].affinity + affinity(span, header),
|
|
3099
|
+
previous: from
|
|
3100
|
+
};
|
|
3101
|
+
if (better(candidate, best)) best = candidate;
|
|
3102
|
+
}
|
|
3103
|
+
return best ?? {
|
|
3104
|
+
used: 0,
|
|
3105
|
+
affinity: -Infinity,
|
|
3106
|
+
previous: 0
|
|
3107
|
+
};
|
|
3108
|
+
}));
|
|
3109
|
+
}
|
|
3110
|
+
const last = table.at(-1) ?? [];
|
|
3111
|
+
let column = last.reduce((best, score, index) => better(score, last[best]) ? index : best, 0);
|
|
3112
|
+
const columns = [];
|
|
3113
|
+
for (let index = table.length - 1; index >= 0; index -= 1) {
|
|
3114
|
+
columns.unshift(column);
|
|
3115
|
+
column = table[index][column].previous;
|
|
3116
|
+
}
|
|
3117
|
+
return columns;
|
|
3118
|
+
}
|
|
3119
|
+
function valuesOf(row, bands, width) {
|
|
3120
|
+
const values = Array.from({
|
|
3121
|
+
length: width
|
|
3122
|
+
}, () => '');
|
|
3123
|
+
const ordered = [...row.cells].sort((a, b) => b.y - a.y || a.x - b.x);
|
|
3124
|
+
for (const cell of ordered) {
|
|
3125
|
+
const band = lastOf(bands, candidate => candidate.start <= cell.x + 0.5) ?? bands[0];
|
|
3126
|
+
if (band === undefined) continue;
|
|
3127
|
+
values[band.column] = joinText(values[band.column], cell.text);
|
|
3128
|
+
}
|
|
3129
|
+
return values;
|
|
3130
|
+
}
|
|
3131
|
+
function named$1(values, headers, columns) {
|
|
3132
|
+
if (columns === undefined) return Object.fromEntries(headers.map((header, index) => [header.text, values[index]]));
|
|
3133
|
+
const record = {};
|
|
3134
|
+
for (const [key, pattern] of Object.entries(columns)) {
|
|
3135
|
+
const index = headers.findIndex(header => pattern.test(header.text));
|
|
3136
|
+
if (index !== -1) record[key] = values[index];
|
|
1838
3137
|
}
|
|
3138
|
+
return record;
|
|
1839
3139
|
}
|
|
1840
|
-
/**
|
|
1841
|
-
|
|
1842
|
-
|
|
1843
|
-
|
|
1844
|
-
|
|
1845
|
-
|
|
1846
|
-
|
|
3140
|
+
/** Distances closer than this, in points, are a tie. */
|
|
3141
|
+
const TIE = 1;
|
|
3142
|
+
/**
|
|
3143
|
+
* The anchor a line belongs to: the nearest by vertical gap. On a tie (evenly
|
|
3144
|
+
* spaced lines) the anchor below wins: text reads top down, so a wrapped
|
|
3145
|
+
* cell's first line comes before the row it belongs to.
|
|
3146
|
+
*/
|
|
3147
|
+
function nearest(line, candidates) {
|
|
3148
|
+
const row = line.row;
|
|
3149
|
+
const gap = candidate => Math.max(0, candidate.bottom - row.top, row.bottom - candidate.top);
|
|
3150
|
+
const [first, ...rest] = candidates;
|
|
3151
|
+
if (first === undefined) throw new Error('no row to attach a line to');
|
|
3152
|
+
let best = first;
|
|
3153
|
+
for (const candidate of rest) {
|
|
3154
|
+
const difference = gap(candidate.row) - gap(best.row);
|
|
3155
|
+
if (difference < -TIE || Math.abs(difference) <= TIE && candidate.row.bottom < best.row.bottom) best = candidate;
|
|
1847
3156
|
}
|
|
3157
|
+
return best;
|
|
3158
|
+
}
|
|
3159
|
+
function overlapOf(header, band) {
|
|
3160
|
+
return Math.max(0, Math.min(header.x + header.width, band.end) - Math.max(header.x, band.start));
|
|
3161
|
+
}
|
|
3162
|
+
function plain$1(row) {
|
|
3163
|
+
return row.cells.map(cell => cell.text).join(' ');
|
|
3164
|
+
}
|
|
3165
|
+
function joinText(first, second) {
|
|
3166
|
+
return first === '' ? second : second === '' ? first : `${first} ${second}`;
|
|
3167
|
+
}
|
|
3168
|
+
function median(values) {
|
|
3169
|
+
const ordered = [...values].sort((a, b) => a - b);
|
|
3170
|
+
return ordered[Math.floor(ordered.length / 2)] ?? 0;
|
|
1848
3171
|
}
|
|
1849
3172
|
|
|
1850
3173
|
/**
|
|
1851
|
-
*
|
|
1852
|
-
*
|
|
1853
|
-
* runner. Mode-agnostic: the same walk drives a browser page or an HTTP context.
|
|
3174
|
+
* The text a `regex` extract reads: the visible rows of the visible sheets,
|
|
3175
|
+
* cells separated by a tab, sheets separated by a blank line.
|
|
1854
3176
|
*
|
|
1855
|
-
* @param
|
|
1856
|
-
* @
|
|
1857
|
-
* @param options - Recipe, runner, hooks, events and the emit callback.
|
|
1858
|
-
* @param path - Where these steps are, for messages and events.
|
|
1859
|
-
* @returns `stop` when the emit callback asked to stop.
|
|
1860
|
-
* @throws StepFailure when a step fails under the `fail` policy.
|
|
3177
|
+
* @param document - The workbook.
|
|
3178
|
+
* @returns The text.
|
|
1861
3179
|
*/
|
|
1862
|
-
|
|
1863
|
-
|
|
1864
|
-
|
|
1865
|
-
|
|
1866
|
-
|
|
1867
|
-
|
|
1868
|
-
|
|
1869
|
-
|
|
1870
|
-
|
|
1871
|
-
|
|
1872
|
-
|
|
3180
|
+
function workbookText(document) {
|
|
3181
|
+
return document.sheets.filter(sheet => sheet.hidden !== true).map(sheet => visibleRows(sheet).map(row => row.map(String).join('\t')).join('\n')).join('\n\n');
|
|
3182
|
+
}
|
|
3183
|
+
/**
|
|
3184
|
+
* Whether a value bound in scope is a read workbook (so `extract … from` can take it).
|
|
3185
|
+
*
|
|
3186
|
+
* @param value - Anything.
|
|
3187
|
+
* @returns Whether it is a {@link WorkbookDocument}.
|
|
3188
|
+
*/
|
|
3189
|
+
function isWorkbookDocument(value) {
|
|
3190
|
+
return typeof value === 'object' && value !== null && value.kind === 'workbook' && Array.isArray(value.sheets);
|
|
3191
|
+
}
|
|
3192
|
+
function visibleRows(sheet) {
|
|
3193
|
+
if (sheet.hiddenRows === undefined || sheet.hiddenRows.length === 0) return sheet.rows;
|
|
3194
|
+
const hidden = new Set(sheet.hiddenRows);
|
|
3195
|
+
return sheet.rows.filter((_row, index) => !hidden.has(index));
|
|
3196
|
+
}
|
|
3197
|
+
|
|
3198
|
+
/** The delimiters detection chooses between, in order of preference on a tie. */
|
|
3199
|
+
const CSV_DELIMITERS = [',', ';', '\t', '|'];
|
|
3200
|
+
/** How much of a file delimiter detection looks at. */
|
|
3201
|
+
const SAMPLE_CHARS = 64 * 1024;
|
|
3202
|
+
/** How many lines of the sample delimiter detection scores. */
|
|
3203
|
+
const SAMPLE_LINES = 100;
|
|
3204
|
+
/**
|
|
3205
|
+
* Parses CSV text (RFC 4180, tolerant): a field in double quotes may hold the
|
|
3206
|
+
* delimiter, line breaks and `""` for a quote; a quote inside an unquoted
|
|
3207
|
+
* field is taken literally (`1.0 Hybrid "Cross"`); CRLF, LF and CR all end a
|
|
3208
|
+
* record. Rows are kept as read: ragged rows stay ragged, nothing is trimmed.
|
|
3209
|
+
*
|
|
3210
|
+
* @param text - The decoded file.
|
|
3211
|
+
* @param delimiter - One character.
|
|
3212
|
+
* @returns The rows; a trailing empty line adds no row.
|
|
3213
|
+
*/
|
|
3214
|
+
function parseCsv(text, delimiter) {
|
|
3215
|
+
const rows = [];
|
|
3216
|
+
let row = [];
|
|
3217
|
+
let field = '';
|
|
3218
|
+
let index = 0;
|
|
3219
|
+
while (index < text.length) {
|
|
3220
|
+
const char = text[index];
|
|
3221
|
+
if (char === '"' && field === '') {
|
|
3222
|
+
const quoted = readQuoted(text, index + 1);
|
|
3223
|
+
field = quoted.value;
|
|
3224
|
+
index = quoted.next;
|
|
3225
|
+
} else if (char === delimiter) {
|
|
3226
|
+
row.push(field);
|
|
3227
|
+
field = '';
|
|
3228
|
+
index += 1;
|
|
3229
|
+
} else if (char === '\n' || char === '\r') {
|
|
3230
|
+
row.push(field);
|
|
3231
|
+
rows.push(row);
|
|
3232
|
+
row = [];
|
|
3233
|
+
field = '';
|
|
3234
|
+
index += char === '\r' && text[index + 1] === '\n' ? 2 : 1;
|
|
3235
|
+
} else {
|
|
3236
|
+
const end = plainEnd(text, index + 1, delimiter);
|
|
3237
|
+
field += text.slice(index, end);
|
|
3238
|
+
index = end;
|
|
3239
|
+
}
|
|
3240
|
+
}
|
|
3241
|
+
if (field !== '' || row.length > 0) {
|
|
3242
|
+
row.push(field);
|
|
3243
|
+
rows.push(row);
|
|
3244
|
+
}
|
|
3245
|
+
return rows;
|
|
3246
|
+
}
|
|
3247
|
+
/**
|
|
3248
|
+
* Chooses the delimiter of a CSV: the candidate whose field count (above one)
|
|
3249
|
+
* is the most consistent over the first lines, so a title line or two above
|
|
3250
|
+
* the header does not mislead it. A file of one column gets `,`.
|
|
3251
|
+
*
|
|
3252
|
+
* `;` with decimal commas (`Panda;15.950,00`), the usual European export,
|
|
3253
|
+
* scores `;`: a comma split gives rows of uneven width.
|
|
3254
|
+
*
|
|
3255
|
+
* @param text - The decoded file.
|
|
3256
|
+
* @returns The delimiter.
|
|
3257
|
+
*/
|
|
3258
|
+
function detectDelimiter(text) {
|
|
3259
|
+
const sample = text.slice(0, SAMPLE_CHARS);
|
|
3260
|
+
const truncated = text.length > SAMPLE_CHARS;
|
|
3261
|
+
let best = {
|
|
3262
|
+
delimiter: ',',
|
|
3263
|
+
score: 0
|
|
3264
|
+
};
|
|
3265
|
+
for (const delimiter of CSV_DELIMITERS) {
|
|
3266
|
+
const rows = parseCsv(sample, delimiter).slice(0, SAMPLE_LINES).filter(row => row.length > 1 || row[0] !== '');
|
|
3267
|
+
// The sample may cut the last line short.
|
|
3268
|
+
if (truncated && rows.length > 1) rows.pop();
|
|
3269
|
+
const score = consistency(rows);
|
|
3270
|
+
if (score > best.score) best = {
|
|
3271
|
+
delimiter,
|
|
3272
|
+
score
|
|
1873
3273
|
};
|
|
1874
|
-
const outcome = await runWithPolicy(step, scope, walk);
|
|
1875
|
-
if (outcome === 'stop') return 'stop';
|
|
1876
3274
|
}
|
|
1877
|
-
return
|
|
3275
|
+
return best.delimiter;
|
|
1878
3276
|
}
|
|
1879
|
-
|
|
1880
|
-
|
|
1881
|
-
const
|
|
1882
|
-
|
|
1883
|
-
|
|
1884
|
-
|
|
1885
|
-
|
|
1886
|
-
|
|
1887
|
-
|
|
1888
|
-
stepType: step.type,
|
|
1889
|
-
stepId: step.id,
|
|
1890
|
-
path: walk.path
|
|
1891
|
-
});
|
|
1892
|
-
try {
|
|
1893
|
-
const outcome = await runOne(step, scope, walk);
|
|
1894
|
-
walk.events.emit({
|
|
1895
|
-
type: 'step:finish',
|
|
1896
|
-
recipeId: walk.recipe.id,
|
|
1897
|
-
stepType: step.type,
|
|
1898
|
-
stepId: step.id,
|
|
1899
|
-
path: walk.path,
|
|
1900
|
-
durationMs: Date.now() - started
|
|
1901
|
-
});
|
|
1902
|
-
return outcome;
|
|
1903
|
-
} catch (error) {
|
|
1904
|
-
if (error instanceof StepFailure) throw error;
|
|
1905
|
-
// A block the runner can rotate away from is retried on the new access, without spending a retry attempt.
|
|
1906
|
-
if (error instanceof BlockedError && walk.runner.rotate !== undefined && (await walk.runner.rotate(error))) continue;
|
|
1907
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
1908
|
-
if (policy.policy === 'retry' && attempt < attempts) {
|
|
1909
|
-
attempt += 1;
|
|
1910
|
-
walk.events.emit({
|
|
1911
|
-
type: 'step:retry',
|
|
1912
|
-
recipeId: walk.recipe.id,
|
|
1913
|
-
stepType: step.type,
|
|
1914
|
-
stepId: step.id,
|
|
1915
|
-
path: walk.path,
|
|
1916
|
-
attempt,
|
|
1917
|
-
error: message
|
|
1918
|
-
});
|
|
1919
|
-
await sleep(backoffFor(policy, attempt));
|
|
1920
|
-
continue;
|
|
1921
|
-
}
|
|
1922
|
-
if (policy.policy === 'skip') {
|
|
1923
|
-
walk.events.emit({
|
|
1924
|
-
type: 'step:skip',
|
|
1925
|
-
recipeId: walk.recipe.id,
|
|
1926
|
-
stepType: step.type,
|
|
1927
|
-
stepId: step.id,
|
|
1928
|
-
path: walk.path,
|
|
1929
|
-
error: message
|
|
1930
|
-
});
|
|
1931
|
-
return 'continue';
|
|
1932
|
-
}
|
|
1933
|
-
throw new StepFailure(walk.path, step.type, error);
|
|
3277
|
+
/** The share of rows with the most common width above one, with that width breaking ties; 0 when no row splits. */
|
|
3278
|
+
function consistency(rows) {
|
|
3279
|
+
const counts = new Map();
|
|
3280
|
+
for (const row of rows) counts.set(row.length, (counts.get(row.length) ?? 0) + 1);
|
|
3281
|
+
let width = 1;
|
|
3282
|
+
let agreeing = 0;
|
|
3283
|
+
for (const [length, count] of counts) {
|
|
3284
|
+
if (!(length > 1 && (count > agreeing || count === agreeing && length > width))) {
|
|
3285
|
+
continue;
|
|
1934
3286
|
}
|
|
3287
|
+
width = length;
|
|
3288
|
+
agreeing = count;
|
|
1935
3289
|
}
|
|
3290
|
+
return width > 1 ? agreeing / rows.length * 1000 + width : 0;
|
|
1936
3291
|
}
|
|
1937
|
-
|
|
1938
|
-
|
|
1939
|
-
|
|
1940
|
-
|
|
1941
|
-
|
|
1942
|
-
|
|
1943
|
-
|
|
1944
|
-
|
|
1945
|
-
|
|
1946
|
-
|
|
1947
|
-
|
|
1948
|
-
|
|
1949
|
-
|
|
1950
|
-
|
|
1951
|
-
|
|
1952
|
-
|
|
1953
|
-
|
|
1954
|
-
}
|
|
1955
|
-
case 'paginate':
|
|
1956
|
-
{
|
|
1957
|
-
return runPaginate(step, scope, walk);
|
|
1958
|
-
}
|
|
1959
|
-
case 'emit':
|
|
1960
|
-
{
|
|
1961
|
-
return walk.onEmit(scope, step.output);
|
|
1962
|
-
}
|
|
1963
|
-
case 'set':
|
|
1964
|
-
{
|
|
1965
|
-
if (step.id !== undefined) scope.set(step.id, typeof step.value === 'string' ? render(step.value, lookupIn(scope)) : step.value);
|
|
1966
|
-
return 'continue';
|
|
1967
|
-
}
|
|
1968
|
-
case 'collect':
|
|
1969
|
-
{
|
|
1970
|
-
const value = typeof step.value === 'string' ? render(step.value, lookupIn(scope)) : step.value;
|
|
1971
|
-
if (value !== undefined) scope.append(step.into, Array.isArray(value) ? value : [value]);
|
|
1972
|
-
return 'continue';
|
|
1973
|
-
}
|
|
1974
|
-
case 'hook':
|
|
1975
|
-
{
|
|
1976
|
-
const hook = walk.hooks.resolve(step.name);
|
|
1977
|
-
const result = await hook(undefined, renderArgs(step.args ?? {}, scope), {
|
|
1978
|
-
recipeId: walk.recipe.id,
|
|
1979
|
-
scope: scope.snapshot(),
|
|
1980
|
-
log: logThrough(walk)
|
|
1981
|
-
});
|
|
1982
|
-
if (step.id !== undefined) scope.set(step.id, result);
|
|
1983
|
-
return 'continue';
|
|
1984
|
-
}
|
|
1985
|
-
default:
|
|
1986
|
-
{
|
|
1987
|
-
await walk.runner.runLeaf(step, scope);
|
|
1988
|
-
return 'continue';
|
|
1989
|
-
}
|
|
3292
|
+
/** Reads a quoted field starting after its opening quote. */
|
|
3293
|
+
function readQuoted(text, start) {
|
|
3294
|
+
let value = '';
|
|
3295
|
+
let index = start;
|
|
3296
|
+
for (;;) {
|
|
3297
|
+
const quote = text.indexOf('"', index);
|
|
3298
|
+
if (quote === -1) return {
|
|
3299
|
+
value: value + text.slice(index),
|
|
3300
|
+
next: text.length
|
|
3301
|
+
};
|
|
3302
|
+
value += text.slice(index, quote);
|
|
3303
|
+
if (text[quote + 1] !== '"') return {
|
|
3304
|
+
value,
|
|
3305
|
+
next: quote + 1
|
|
3306
|
+
};
|
|
3307
|
+
value += '"';
|
|
3308
|
+
index = quote + 2;
|
|
1990
3309
|
}
|
|
1991
3310
|
}
|
|
1992
|
-
|
|
1993
|
-
|
|
1994
|
-
|
|
1995
|
-
|
|
1996
|
-
|
|
3311
|
+
/** Where a run of plain characters (no delimiter, no line break) ends. */
|
|
3312
|
+
function plainEnd(text, start, delimiter) {
|
|
3313
|
+
const stops = new Set([delimiter, '\n', '\r']);
|
|
3314
|
+
let index = start;
|
|
3315
|
+
while (index < text.length) {
|
|
3316
|
+
if (stops.has(text[index])) break;
|
|
3317
|
+
index += 1;
|
|
3318
|
+
}
|
|
3319
|
+
return index;
|
|
1997
3320
|
}
|
|
1998
|
-
|
|
1999
|
-
|
|
2000
|
-
|
|
2001
|
-
|
|
2002
|
-
|
|
2003
|
-
|
|
2004
|
-
|
|
2005
|
-
|
|
3321
|
+
|
|
3322
|
+
/**
|
|
3323
|
+
* Reads decoded CSV text into a workbook of one sheet, named after the file.
|
|
3324
|
+
*
|
|
3325
|
+
* @param text - The decoded file.
|
|
3326
|
+
* @param options - The sheet name, the encoding it was decoded from (for a
|
|
3327
|
+
* probe to report) and a delimiter; without one it is detected.
|
|
3328
|
+
* @returns The workbook.
|
|
3329
|
+
* @throws Error when the delimiter given is not one character.
|
|
3330
|
+
*/
|
|
3331
|
+
function csvWorkbook(text, options) {
|
|
3332
|
+
if (options.delimiter !== undefined && [...options.delimiter].length !== 1) throw new Error(`a CSV delimiter is one character; got "${options.delimiter}"`);
|
|
3333
|
+
const delimiter = options.delimiter ?? detectDelimiter(text);
|
|
3334
|
+
return {
|
|
3335
|
+
kind: 'workbook',
|
|
3336
|
+
sheets: [{
|
|
3337
|
+
name: options.name,
|
|
3338
|
+
rows: parseCsv(text, delimiter)
|
|
3339
|
+
}],
|
|
3340
|
+
csv: {
|
|
3341
|
+
encoding: options.encoding,
|
|
3342
|
+
delimiter
|
|
3343
|
+
}
|
|
2006
3344
|
};
|
|
2007
3345
|
}
|
|
2008
|
-
|
|
2009
3346
|
/**
|
|
2010
|
-
*
|
|
2011
|
-
*
|
|
2012
|
-
* every loop in it, so nested loops never multiply the limit.
|
|
3347
|
+
* The name a CSV's sheet takes: the file name without its extension
|
|
3348
|
+
* (`…/prezzo_alle_8.csv` → `prezzo_alle_8`), else `csv`.
|
|
2013
3349
|
*
|
|
2014
|
-
*
|
|
2015
|
-
*
|
|
2016
|
-
* which keeps the total at `permits` and cannot deadlock.
|
|
3350
|
+
* @param url - Where the file came from.
|
|
3351
|
+
* @returns The name.
|
|
2017
3352
|
*/
|
|
2018
|
-
|
|
2019
|
-
|
|
2020
|
-
|
|
2021
|
-
|
|
2022
|
-
|
|
2023
|
-
|
|
2024
|
-
lastStart = -Infinity;
|
|
2025
|
-
/**
|
|
2026
|
-
* @param permits - Iterations allowed in flight; 1 is sequential.
|
|
2027
|
-
* @param minIntervalMs - Minimum time between two request starts across the run.
|
|
2028
|
-
* @param shared - The throttle state to share (internal: `nested` gates keep their parent's).
|
|
2029
|
-
*/
|
|
2030
|
-
constructor(permits, minIntervalMs, shared) {
|
|
2031
|
-
this.permits = permits;
|
|
2032
|
-
this.minIntervalMs = minIntervalMs;
|
|
2033
|
-
this.shared = shared;
|
|
2034
|
-
}
|
|
2035
|
-
/** Whether this gate lets more than one iteration run at once. */
|
|
2036
|
-
get concurrent() {
|
|
2037
|
-
return this.permits > 1;
|
|
2038
|
-
}
|
|
2039
|
-
/**
|
|
2040
|
-
* Takes a permit, waiting for one when all are in flight.
|
|
2041
|
-
*
|
|
2042
|
-
* @returns The release; call it exactly once, when the iteration ends.
|
|
2043
|
-
*/
|
|
2044
|
-
async acquire() {
|
|
2045
|
-
if (this.inFlight >= this.permits) await new Promise(resolve => {
|
|
2046
|
-
this.waiting.push(resolve);
|
|
2047
|
-
});
|
|
2048
|
-
this.inFlight += 1;
|
|
2049
|
-
let released = false;
|
|
2050
|
-
return () => {
|
|
2051
|
-
if (released) return;
|
|
2052
|
-
released = true;
|
|
2053
|
-
this.inFlight -= 1;
|
|
2054
|
-
this.waiting.shift()?.();
|
|
2055
|
-
};
|
|
3353
|
+
function sheetNameOf(url) {
|
|
3354
|
+
let path;
|
|
3355
|
+
try {
|
|
3356
|
+
path = new URL(url).pathname;
|
|
3357
|
+
} catch {
|
|
3358
|
+
path = url;
|
|
2056
3359
|
}
|
|
2057
|
-
|
|
2058
|
-
|
|
2059
|
-
|
|
2060
|
-
|
|
2061
|
-
|
|
2062
|
-
|
|
2063
|
-
if (state.minIntervalMs <= 0) return;
|
|
2064
|
-
const now = Date.now();
|
|
2065
|
-
const at = Math.max(now, state.lastStart + state.minIntervalMs);
|
|
2066
|
-
state.lastStart = at;
|
|
2067
|
-
await sleep(at - now);
|
|
3360
|
+
const file = path.split('/').at(-1) ?? '';
|
|
3361
|
+
let name;
|
|
3362
|
+
try {
|
|
3363
|
+
name = decodeURIComponent(file);
|
|
3364
|
+
} catch {
|
|
3365
|
+
name = file;
|
|
2068
3366
|
}
|
|
2069
|
-
|
|
2070
|
-
|
|
2071
|
-
|
|
3367
|
+
name = name.replace(/\.[^.]*$/, '');
|
|
3368
|
+
return name === '' ? 'csv' : name;
|
|
3369
|
+
}
|
|
3370
|
+
|
|
3371
|
+
/**
|
|
3372
|
+
* Reads an `.xlsx` workbook into a workbook document, through
|
|
3373
|
+
* `@opencraw/office-reader`: every worksheet's cells, with hidden sheets,
|
|
3374
|
+
* hidden rows and merged ranges. Numbers and booleans keep their type (a
|
|
3375
|
+
* cell's `13955.625` is unambiguous; as text, a locale guess could read it as
|
|
3376
|
+
* thirteen million), dates become ISO text, errors their text, empty cells
|
|
3377
|
+
* `''`. Formulas give their cached value. The reader is imported on first use,
|
|
3378
|
+
* so recipes that never read a spreadsheet never load it.
|
|
3379
|
+
*
|
|
3380
|
+
* @param bytes - The file.
|
|
3381
|
+
* @param source - Where it came from, for messages.
|
|
3382
|
+
* @returns The workbook.
|
|
3383
|
+
* @throws Error naming the source, and saying what to do, for a file that is
|
|
3384
|
+
* not a readable workbook (a legacy `.xls`, a password-protected file, an `.ods`…).
|
|
3385
|
+
*/
|
|
3386
|
+
async function readXlsxWorkbook(bytes, source) {
|
|
3387
|
+
const {
|
|
3388
|
+
readXlsx,
|
|
3389
|
+
OfficeReadError
|
|
3390
|
+
} = await import('@opencraw/office-reader/xlsx');
|
|
3391
|
+
try {
|
|
3392
|
+
const book = await readXlsx(bytes);
|
|
3393
|
+
return {
|
|
3394
|
+
kind: 'workbook',
|
|
3395
|
+
sheets: book.sheets.map(sheet => ({
|
|
3396
|
+
name: sheet.name,
|
|
3397
|
+
rows: sheet.rows.map(row => row.map(cell => workbookCell(cell))),
|
|
3398
|
+
hidden: sheet.hidden,
|
|
3399
|
+
hiddenRows: sheet.hiddenRows,
|
|
3400
|
+
merges: sheet.merges
|
|
3401
|
+
}))
|
|
3402
|
+
};
|
|
3403
|
+
} catch (error) {
|
|
3404
|
+
if (error instanceof OfficeReadError) throw new Error(`${source}: ${error.message}`, {
|
|
3405
|
+
cause: error
|
|
3406
|
+
});
|
|
3407
|
+
throw error;
|
|
2072
3408
|
}
|
|
2073
3409
|
}
|
|
3410
|
+
/** A typed spreadsheet value as a workbook cell. */
|
|
3411
|
+
function workbookCell(value) {
|
|
3412
|
+
if (value === null) return '';
|
|
3413
|
+
if (value instanceof Date) return isoText(value);
|
|
3414
|
+
if (typeof value === 'object') return value.error;
|
|
3415
|
+
return value;
|
|
3416
|
+
}
|
|
3417
|
+
/** A date as ISO text: the day alone at midnight, the time alone for a time of day (Excel's day zero, 1899), both otherwise. */
|
|
3418
|
+
function isoText(date) {
|
|
3419
|
+
const iso = date.toISOString();
|
|
3420
|
+
if (date.getUTCFullYear() < 1900) return iso.slice(11, 19);
|
|
3421
|
+
return iso.slice(11, 19) === '00:00:00' ? iso.slice(0, 10) : iso.slice(0, 19);
|
|
3422
|
+
}
|
|
2074
3423
|
|
|
2075
|
-
/**
|
|
2076
|
-
|
|
2077
|
-
|
|
2078
|
-
|
|
2079
|
-
|
|
3424
|
+
/**
|
|
3425
|
+
* Finds every table whose header row matches, in every sheet the query
|
|
3426
|
+
* selects, and reads its rows by column. Unlike a PDF, a grid needs no
|
|
3427
|
+
* geometry: column *i* of a row belongs to header *i*.
|
|
3428
|
+
*
|
|
3429
|
+
* Merged ranges are filled first (the file stores their value in the top-left
|
|
3430
|
+
* cell only), so a brand merged down its models' rows reads on every row, and
|
|
3431
|
+
* a group header merged across its sub-columns names each of them. Empty rows
|
|
3432
|
+
* are skipped.
|
|
3433
|
+
*
|
|
3434
|
+
* @param document - The workbook.
|
|
3435
|
+
* @param query - Which tables, and how to name their columns.
|
|
3436
|
+
* @returns The tables, sheet by sheet, top to bottom.
|
|
3437
|
+
*/
|
|
3438
|
+
function findGridTables(document, query) {
|
|
3439
|
+
const tables = [];
|
|
3440
|
+
for (const sheet of document.sheets) {
|
|
3441
|
+
if (sheet.hidden === true && query.includeHidden !== true) continue;
|
|
3442
|
+
if (query.sheet !== undefined && !query.sheet.test(sheet.name)) continue;
|
|
3443
|
+
tables.push(...sheetTables(sheet, query));
|
|
2080
3444
|
}
|
|
2081
|
-
|
|
3445
|
+
return tables;
|
|
3446
|
+
}
|
|
2082
3447
|
/**
|
|
2083
|
-
*
|
|
3448
|
+
* Fills blank cells in the given columns with the value of the row above,
|
|
3449
|
+
* within one table: pivot exports write a group's name on its first row only.
|
|
2084
3450
|
*
|
|
2085
|
-
* @param
|
|
2086
|
-
* @param
|
|
2087
|
-
* @returns The
|
|
3451
|
+
* @param rows - The table's rows, in order.
|
|
3452
|
+
* @param keys - The columns to fill.
|
|
3453
|
+
* @returns The rows, filled (new objects; the input is not changed).
|
|
2088
3454
|
*/
|
|
2089
|
-
|
|
2090
|
-
|
|
2091
|
-
|
|
2092
|
-
|
|
2093
|
-
|
|
2094
|
-
|
|
3455
|
+
function fillDown(rows, keys) {
|
|
3456
|
+
const last = new Map();
|
|
3457
|
+
return rows.map(row => {
|
|
3458
|
+
const filled = {
|
|
3459
|
+
...row
|
|
3460
|
+
};
|
|
3461
|
+
for (const key of keys) {
|
|
3462
|
+
const value = filled[key];
|
|
3463
|
+
if (value === undefined || value === '') {
|
|
3464
|
+
const above = last.get(key);
|
|
3465
|
+
if (above !== undefined) filled[key] = above;
|
|
3466
|
+
} else {
|
|
3467
|
+
last.set(key, value);
|
|
3468
|
+
}
|
|
3469
|
+
}
|
|
3470
|
+
return filled;
|
|
3471
|
+
});
|
|
3472
|
+
}
|
|
3473
|
+
function sheetTables(sheet, query) {
|
|
3474
|
+
const hidden = new Set(query.includeHidden === true ? [] : sheet.hiddenRows);
|
|
3475
|
+
const visible = sheet.rows.flatMap((_row, index) => hidden.has(index) ? [] : [index]);
|
|
3476
|
+
const grid = filledGrid(sheet);
|
|
3477
|
+
const plainRows = new Map(visible.map(index => [index, plain(sheet.rows[index])]));
|
|
3478
|
+
const starts = visible.filter(index => plainRows.get(index) !== '' && query.header.test(plainRows.get(index) ?? ''));
|
|
3479
|
+
const headerRows = query.headerRows ?? 1;
|
|
3480
|
+
return starts.map((start, position) => {
|
|
3481
|
+
const at = visible.indexOf(start);
|
|
3482
|
+
const headerIndexes = visible.slice(at, at + headerRows);
|
|
3483
|
+
const next = starts[position + 1] ?? Infinity;
|
|
3484
|
+
const body = [];
|
|
3485
|
+
const below = visible.slice(at + headerRows);
|
|
3486
|
+
for (const index of below) {
|
|
3487
|
+
if (index >= next) break;
|
|
3488
|
+
if (query.until?.test(plainRows.get(index) ?? '') === true) break;
|
|
3489
|
+
if (plainRows.get(index) !== '') body.push(index);
|
|
3490
|
+
}
|
|
3491
|
+
const columns = columnsOf(grid, headerIndexes, body);
|
|
3492
|
+
const rows = body.map(index => named(grid[index] ?? [], columns, query.columns));
|
|
3493
|
+
return {
|
|
3494
|
+
sheet: sheet.name,
|
|
3495
|
+
title: sheet.rows[start].map(text => clean(text)).find(text => text !== '') ?? '',
|
|
3496
|
+
header: columns.map(column => column.key),
|
|
3497
|
+
rows: query.fillDown === undefined ? rows : fillDown(rows, query.fillDown)
|
|
3498
|
+
};
|
|
3499
|
+
});
|
|
3500
|
+
}
|
|
3501
|
+
/**
|
|
3502
|
+
* The table's columns: each column's key joins the distinct texts its header
|
|
3503
|
+
* rows hold (`Insgesamt` over `August 2026` gives `Insgesamt August 2026`). A
|
|
3504
|
+
* column with no header text but data below is keyed by its letter (`A`); one
|
|
3505
|
+
* with neither is dropped. A key seen before gets a counter (`Price 2`).
|
|
3506
|
+
*/
|
|
3507
|
+
function columnsOf(grid, headerIndexes, body) {
|
|
3508
|
+
const width = Math.max(0, ...[...headerIndexes, ...body].map(index => grid[index]?.length ?? 0));
|
|
3509
|
+
const seen = new Map();
|
|
3510
|
+
const columns = [];
|
|
3511
|
+
for (let index = 0; index < width; index += 1) {
|
|
3512
|
+
const parts = [];
|
|
3513
|
+
for (const row of headerIndexes) {
|
|
3514
|
+
const text = clean(grid[row]?.[index] ?? '');
|
|
3515
|
+
if (text !== '' && !parts.includes(text)) parts.push(text);
|
|
3516
|
+
}
|
|
3517
|
+
const hasData = body.some(row => clean(grid[row]?.[index] ?? '') !== '');
|
|
3518
|
+
if (!hasData && parts.length === 0) continue;
|
|
3519
|
+
const base = parts.length === 0 ? columnLetter(index) : parts.join(' ');
|
|
3520
|
+
const count = (seen.get(base) ?? 0) + 1;
|
|
3521
|
+
seen.set(base, count);
|
|
3522
|
+
columns.push({
|
|
3523
|
+
index,
|
|
3524
|
+
key: count === 1 ? base : `${base} ${count}`
|
|
3525
|
+
});
|
|
2095
3526
|
}
|
|
2096
|
-
|
|
2097
|
-
|
|
2098
|
-
|
|
2099
|
-
|
|
2100
|
-
|
|
2101
|
-
|
|
3527
|
+
return columns;
|
|
3528
|
+
}
|
|
3529
|
+
function named(row, columns, patterns) {
|
|
3530
|
+
const value = column => {
|
|
3531
|
+
const cell = row[column.index] ?? '';
|
|
3532
|
+
return typeof cell === 'string' ? cell.trim() : cell;
|
|
3533
|
+
};
|
|
3534
|
+
if (patterns === undefined) return Object.fromEntries(columns.map(column => [column.key, value(column)]));
|
|
3535
|
+
const record = {};
|
|
3536
|
+
for (const [key, pattern] of Object.entries(patterns)) {
|
|
3537
|
+
const column = columns.find(candidate => pattern.test(candidate.key));
|
|
3538
|
+
if (column !== undefined) record[key] = value(column);
|
|
3539
|
+
}
|
|
3540
|
+
return record;
|
|
3541
|
+
}
|
|
3542
|
+
/** The sheet's rows with every merged range's value copied into the cells it covers. */
|
|
3543
|
+
function filledGrid(sheet) {
|
|
3544
|
+
if (sheet.merges === undefined || sheet.merges.length === 0) return sheet.rows;
|
|
3545
|
+
const grid = sheet.rows.map(row => [...row]);
|
|
3546
|
+
for (const reference of sheet.merges) {
|
|
3547
|
+
const range = rangeOf(reference);
|
|
3548
|
+
if (range === undefined) continue;
|
|
3549
|
+
const value = sheet.rows[range.top]?.[range.left] ?? '';
|
|
3550
|
+
for (let row = range.top; row <= range.bottom; row += 1) {
|
|
3551
|
+
grid[row] ??= [];
|
|
3552
|
+
for (let column = range.left; column <= range.right; column += 1) grid[row][column] = value;
|
|
2102
3553
|
}
|
|
2103
|
-
const pattern = new RegExp(rule.text, 'i');
|
|
2104
|
-
if (pattern.test(body)) return new BlockedError(response.url, response.status, `body matches /${rule.text}/i`);
|
|
2105
3554
|
}
|
|
2106
|
-
return
|
|
3555
|
+
return grid;
|
|
3556
|
+
}
|
|
3557
|
+
/** `B10:B13` as 0-based bounds; `undefined` for anything else. */
|
|
3558
|
+
function rangeOf(reference) {
|
|
3559
|
+
const [from, to = from] = reference.split(':', 2);
|
|
3560
|
+
const start = cellOf(from);
|
|
3561
|
+
const end = cellOf(to);
|
|
3562
|
+
if (start === undefined || end === undefined) return undefined;
|
|
3563
|
+
return {
|
|
3564
|
+
top: Math.min(start.row, end.row),
|
|
3565
|
+
left: Math.min(start.column, end.column),
|
|
3566
|
+
bottom: Math.max(start.row, end.row),
|
|
3567
|
+
right: Math.max(start.column, end.column)
|
|
3568
|
+
};
|
|
3569
|
+
}
|
|
3570
|
+
function cellOf(reference) {
|
|
3571
|
+
const match = /^\$?([A-Z]+)\$?(\d+)$/i.exec(reference.trim());
|
|
3572
|
+
if (match === null) return undefined;
|
|
3573
|
+
const letters = match[1].toUpperCase();
|
|
3574
|
+
let column = 0;
|
|
3575
|
+
for (const char of letters) column = column * 26 + (char.codePointAt(0) ?? 64) - 64;
|
|
3576
|
+
return {
|
|
3577
|
+
row: Number(match[2]) - 1,
|
|
3578
|
+
column: column - 1
|
|
3579
|
+
};
|
|
3580
|
+
}
|
|
3581
|
+
/** `0` → `A`, `25` → `Z`, `26` → `AA`. */
|
|
3582
|
+
function columnLetter(index) {
|
|
3583
|
+
let letters = '';
|
|
3584
|
+
for (let rest = index + 1; rest > 0; rest = Math.floor((rest - 1) / 26)) letters = String.fromCodePoint(65 + (rest - 1) % 26) + letters;
|
|
3585
|
+
return letters;
|
|
3586
|
+
}
|
|
3587
|
+
/** A row as a header pattern sees it: its non-empty cells, whitespace collapsed, joined by spaces. */
|
|
3588
|
+
function plain(row) {
|
|
3589
|
+
return (row ?? []).map(text => clean(text)).filter(text => text !== '').join(' ');
|
|
3590
|
+
}
|
|
3591
|
+
function clean(cell) {
|
|
3592
|
+
return String(cell).replaceAll(/\s+/g, ' ').trim();
|
|
3593
|
+
}
|
|
3594
|
+
|
|
3595
|
+
/**
|
|
3596
|
+
* Every `<table>` of an HTML document as a sheet (`table 1`, `table 2`…, in
|
|
3597
|
+
* document order), so the workbook table reader works on web pages and
|
|
3598
|
+
* rendered Markdown: rows in order (`thead`, `tbody`, `tfoot` alike), `th` and
|
|
3599
|
+
* `td` alike, cell text with whitespace collapsed, `colspan` and `rowspan` as
|
|
3600
|
+
* merged ranges. A table inside a table is a sheet of its own, and its rows
|
|
3601
|
+
* are not its parent's.
|
|
3602
|
+
*
|
|
3603
|
+
* @param html - The document.
|
|
3604
|
+
* @returns The tables.
|
|
3605
|
+
*/
|
|
3606
|
+
function htmlTableSheets(html) {
|
|
3607
|
+
const $ = load(html);
|
|
3608
|
+
const tables = $('table').get();
|
|
3609
|
+
return tables.map((table, index) => {
|
|
3610
|
+
const all = $(table).find('tr').get();
|
|
3611
|
+
const rows = all.filter(row => $(row).closest('table').get(0) === table);
|
|
3612
|
+
const grid = [];
|
|
3613
|
+
const merges = [];
|
|
3614
|
+
for (const [rowIndex, row] of rows.entries()) {
|
|
3615
|
+
grid[rowIndex] ??= [];
|
|
3616
|
+
let column = 0;
|
|
3617
|
+
const cells = $(row).children('th, td').get();
|
|
3618
|
+
for (const cell of cells) {
|
|
3619
|
+
while (grid[rowIndex][column] !== undefined) column += 1;
|
|
3620
|
+
const columnSpan = span($(cell).attr('colspan'));
|
|
3621
|
+
const rowSpan = span($(cell).attr('rowspan'));
|
|
3622
|
+
for (let down = 0; down < rowSpan; down += 1) {
|
|
3623
|
+
grid[rowIndex + down] ??= [];
|
|
3624
|
+
for (let across = 0; across < columnSpan; across += 1) grid[rowIndex + down][column + across] = '';
|
|
3625
|
+
}
|
|
3626
|
+
grid[rowIndex][column] = $(cell).text().replaceAll(/\s+/g, ' ').trim();
|
|
3627
|
+
if (columnSpan > 1 || rowSpan > 1) merges.push(`${letter(column)}${rowIndex + 1}:${letter(column + columnSpan - 1)}${rowIndex + rowSpan}`);
|
|
3628
|
+
column += columnSpan;
|
|
3629
|
+
}
|
|
3630
|
+
}
|
|
3631
|
+
return {
|
|
3632
|
+
name: `table ${index + 1}`,
|
|
3633
|
+
rows: grid.slice(0, rows.length).map(row => Array.from(row, cell => cell ?? '')),
|
|
3634
|
+
merges
|
|
3635
|
+
};
|
|
3636
|
+
});
|
|
3637
|
+
}
|
|
3638
|
+
function span(value) {
|
|
3639
|
+
const number = Math.trunc(Number(value ?? '1'));
|
|
3640
|
+
return Number.isFinite(number) && number > 0 ? Math.min(number, 1000) : 1;
|
|
3641
|
+
}
|
|
3642
|
+
function letter(index) {
|
|
3643
|
+
let letters = '';
|
|
3644
|
+
for (let rest = index + 1; rest > 0; rest = Math.floor((rest - 1) / 26)) letters = String.fromCodePoint(65 + (rest - 1) % 26) + letters;
|
|
3645
|
+
return letters;
|
|
3646
|
+
}
|
|
3647
|
+
|
|
3648
|
+
/**
|
|
3649
|
+
* Finds tables in a deck: native tables through the workbook table reader
|
|
3650
|
+
* (merged cells filled, a header over several rows joined), or, with
|
|
3651
|
+
* `shapes`, text boxes laid out as a table through the PDF table reader (a box
|
|
3652
|
+
* is a cell, boxes whose heights overlap a row, columns from where the body's
|
|
3653
|
+
* boxes start). Hidden slides are skipped unless `includeHidden`.
|
|
3654
|
+
*
|
|
3655
|
+
* @param document - The deck.
|
|
3656
|
+
* @param query - Which tables, on which slides, and how to name their columns.
|
|
3657
|
+
* @returns The tables, slide by slide.
|
|
3658
|
+
*/
|
|
3659
|
+
function findDeckTables(document, query) {
|
|
3660
|
+
const slides = document.slides.filter(slide => (query.includeHidden === true || !slide.hidden) && (query.slide === undefined || query.slide.test(slide.title ?? '')));
|
|
3661
|
+
if (query.shapes === true) return shapeTables(document, slides, query);
|
|
3662
|
+
return slides.flatMap(slide => findGridTables({
|
|
3663
|
+
sheets: slide.tables
|
|
3664
|
+
}, query).map(table => ({
|
|
3665
|
+
slide: slide.number,
|
|
3666
|
+
slideTitle: slide.title ?? '',
|
|
3667
|
+
title: table.title,
|
|
3668
|
+
header: table.header,
|
|
3669
|
+
rows: table.rows
|
|
3670
|
+
})));
|
|
3671
|
+
}
|
|
3672
|
+
/** Text boxes as a PDF of one page per slide, y flipped (PDF counts from the bottom), each box one cell. */
|
|
3673
|
+
function shapeTables(document, slides, query) {
|
|
3674
|
+
const pdf = {
|
|
3675
|
+
pages: slides.map(slide => ({
|
|
3676
|
+
number: slide.number,
|
|
3677
|
+
width: document.width,
|
|
3678
|
+
height: document.height,
|
|
3679
|
+
rows: rowsOfCells(slide.shapes.map(shape => ({
|
|
3680
|
+
x: shape.x,
|
|
3681
|
+
y: document.height - shape.y - shape.height,
|
|
3682
|
+
width: shape.width,
|
|
3683
|
+
height: shape.height,
|
|
3684
|
+
text: shape.text.replaceAll(/\s+/g, ' ').trim()
|
|
3685
|
+
})))
|
|
3686
|
+
}))
|
|
3687
|
+
};
|
|
3688
|
+
const titles = new Map(slides.map(slide => [slide.number, slide.title ?? '']));
|
|
3689
|
+
return findTables(pdf, {
|
|
3690
|
+
header: query.header,
|
|
3691
|
+
until: query.until,
|
|
3692
|
+
columns: query.columns,
|
|
3693
|
+
align: query.align
|
|
3694
|
+
}).map(table => ({
|
|
3695
|
+
slide: table.page,
|
|
3696
|
+
slideTitle: titles.get(table.page) ?? '',
|
|
3697
|
+
title: table.title,
|
|
3698
|
+
header: table.header,
|
|
3699
|
+
rows: query.fillDown === undefined ? table.rows : fillDown(table.rows, query.fillDown)
|
|
3700
|
+
}));
|
|
2107
3701
|
}
|
|
2108
3702
|
|
|
2109
3703
|
/**
|
|
2110
3704
|
* Runs an `extract` step against a static document: the value bound under
|
|
2111
3705
|
* `from`, else the scope's current document. `css` reads HTML, `jsonpath`
|
|
2112
|
-
* reads JSON
|
|
3706
|
+
* reads JSON (or a read PDF, workbook or deck as data), `table` reads the
|
|
3707
|
+
* tables of a PDF, a workbook (a spreadsheet, a CSV), a deck (a presentation)
|
|
3708
|
+
* or HTML (its `<table>`s), `regex` reads any document as text; `xpath` needs
|
|
3709
|
+
* a live page and is refused here.
|
|
2113
3710
|
*
|
|
2114
3711
|
* A `jsonpath` extract whose `from` is text parses that text as JSON, and a
|
|
2115
3712
|
* list of texts (every `<script type="application/ld+json">` of a page) becomes
|
|
@@ -2128,13 +3725,18 @@ function extractFromDocument(step, scope) {
|
|
|
2128
3725
|
switch (step.kind) {
|
|
2129
3726
|
case 'jsonpath':
|
|
2130
3727
|
{
|
|
2131
|
-
if (document.kind
|
|
2132
|
-
values = selectJson(document.data, selector).map(node => takeFromJson(node, take));
|
|
3728
|
+
if (document.kind === 'html' || document.kind === 'text') throw new Error(`jsonpath needs a JSON document; the current document is ${document.kind}`);
|
|
3729
|
+
values = selectJson(document.kind === 'json' ? document.data : document, selector).map(node => takeFromJson(node, take));
|
|
3730
|
+
break;
|
|
3731
|
+
}
|
|
3732
|
+
case 'table':
|
|
3733
|
+
{
|
|
3734
|
+
values = readTables(document, step, selector);
|
|
2133
3735
|
break;
|
|
2134
3736
|
}
|
|
2135
3737
|
case 'css':
|
|
2136
3738
|
{
|
|
2137
|
-
if (document.kind !== 'html') throw new Error(`css needs an HTML document; the current document is ${document.kind}`);
|
|
3739
|
+
if (document.kind !== 'html') throw new Error(`css needs an HTML document; the current document is ${document.kind}${['workbook', 'pdf', 'deck'].includes(document.kind) ? ' (read it with kind "table", "regex" or "jsonpath")' : ''}`);
|
|
2138
3740
|
values = selectHtml(document.html, selector).map(match => takeFromHtml(match, take));
|
|
2139
3741
|
break;
|
|
2140
3742
|
}
|
|
@@ -2166,10 +3768,98 @@ function extractFromDocument(step, scope) {
|
|
|
2166
3768
|
function renderSelector(selector, scope) {
|
|
2167
3769
|
return hasPlaceholder(selector) ? renderText(selector, path => scope.lookup(path)) : selector;
|
|
2168
3770
|
}
|
|
2169
|
-
/**
|
|
3771
|
+
/**
|
|
3772
|
+
* The tables a `table` extract finds in a document: a PDF's, a workbook's, a
|
|
3773
|
+
* deck's, or an HTML document's `<table>`s (a fetched page, rendered Markdown,
|
|
3774
|
+
* a live page's content).
|
|
3775
|
+
*
|
|
3776
|
+
* @param document - The document.
|
|
3777
|
+
* @param step - The extract step.
|
|
3778
|
+
* @param scope - Where its selector renders.
|
|
3779
|
+
* @returns The tables.
|
|
3780
|
+
*/
|
|
3781
|
+
function tablesIn(document, step, scope) {
|
|
3782
|
+
return readTables(document, step, renderSelector(step.selector, scope));
|
|
3783
|
+
}
|
|
3784
|
+
function readTables(document, step, selector) {
|
|
3785
|
+
const query = tableQuery(step, selector);
|
|
3786
|
+
if (document.kind === 'html') {
|
|
3787
|
+
refuseOptions(step, ['sheet', 'slide', 'shapes'], 'workbooks and decks', 'HTML');
|
|
3788
|
+
return findGridTables({
|
|
3789
|
+
sheets: htmlTableSheets(document.html)
|
|
3790
|
+
}, {
|
|
3791
|
+
...query,
|
|
3792
|
+
headerRows: step.headerRows,
|
|
3793
|
+
fillDown: step.fillDown
|
|
3794
|
+
});
|
|
3795
|
+
}
|
|
3796
|
+
if (document.kind === 'workbook') {
|
|
3797
|
+
refuseOptions(step, ['slide', 'shapes'], 'decks (presentations)', 'a workbook');
|
|
3798
|
+
return findGridTables(document, {
|
|
3799
|
+
...query,
|
|
3800
|
+
sheet: optionalPattern(step.sheet, 'sheet'),
|
|
3801
|
+
headerRows: step.headerRows,
|
|
3802
|
+
fillDown: step.fillDown,
|
|
3803
|
+
includeHidden: step.includeHidden
|
|
3804
|
+
});
|
|
3805
|
+
}
|
|
3806
|
+
if (document.kind === 'deck') {
|
|
3807
|
+
refuseOptions(step, ['sheet'], 'workbooks (spreadsheets, CSV)', 'a deck');
|
|
3808
|
+
return findDeckTables(document, {
|
|
3809
|
+
...query,
|
|
3810
|
+
slide: optionalPattern(step.slide, 'slide'),
|
|
3811
|
+
shapes: step.shapes,
|
|
3812
|
+
headerRows: step.headerRows,
|
|
3813
|
+
fillDown: step.fillDown,
|
|
3814
|
+
includeHidden: step.includeHidden
|
|
3815
|
+
});
|
|
3816
|
+
}
|
|
3817
|
+
if (document.kind !== 'pdf') throw new Error(`table reads a PDF, a workbook (a spreadsheet, a CSV), a deck (a presentation) or HTML tables; the current document is ${document.kind} (request it with "as": "pdf", "csv", "xlsx", "pptx" or "html")`);
|
|
3818
|
+
refuseOptions(step, ['sheet', 'headerRows', 'includeHidden', 'slide', 'shapes'], 'workbooks and decks', 'a PDF');
|
|
3819
|
+
const tables = findTables(document, query);
|
|
3820
|
+
return step.fillDown === undefined ? tables : tables.map(table => ({
|
|
3821
|
+
...table,
|
|
3822
|
+
rows: fillDown(table.rows, step.fillDown ?? [])
|
|
3823
|
+
}));
|
|
3824
|
+
}
|
|
3825
|
+
/**
|
|
3826
|
+
* A table extract's query: the selector matches the header row, the other
|
|
3827
|
+
* patterns come from the step; all case-insensitive, since PDFs and
|
|
3828
|
+
* spreadsheets capitalise headings freely.
|
|
3829
|
+
*/
|
|
3830
|
+
function tableQuery(step, selector) {
|
|
3831
|
+
const columns = step.columns === undefined ? undefined : Object.fromEntries(Object.entries(step.columns).map(([key, pattern]) => [key, patternOf(pattern, `columns.${key}`)]));
|
|
3832
|
+
return {
|
|
3833
|
+
header: patternOf(selector, 'selector'),
|
|
3834
|
+
until: optionalPattern(step.until, 'until'),
|
|
3835
|
+
columns,
|
|
3836
|
+
align: step.align
|
|
3837
|
+
};
|
|
3838
|
+
}
|
|
3839
|
+
function refuseOptions(step, options, reads, current) {
|
|
3840
|
+
for (const option of options) {
|
|
3841
|
+
if (step[option] !== undefined) throw new Error(`"${option}" reads ${reads}; the current document is ${current}`);
|
|
3842
|
+
}
|
|
3843
|
+
}
|
|
3844
|
+
function optionalPattern(source, where) {
|
|
3845
|
+
return source === undefined ? undefined : patternOf(source, where);
|
|
3846
|
+
}
|
|
3847
|
+
function patternOf(source, where) {
|
|
3848
|
+
try {
|
|
3849
|
+
return new RegExp(source, 'i');
|
|
3850
|
+
} catch (error) {
|
|
3851
|
+
throw new Error(`${where}: invalid pattern ${source} (${error.message})`, {
|
|
3852
|
+
cause: error
|
|
3853
|
+
});
|
|
3854
|
+
}
|
|
3855
|
+
}
|
|
3856
|
+
/** The text a regex extract reads: markup, text, a PDF's or a workbook's rows, or JSON re-serialised (a list of texts joined by newlines). */
|
|
2170
3857
|
function textOf$1(document) {
|
|
2171
3858
|
if (document.kind === 'html') return document.html;
|
|
2172
3859
|
if (document.kind === 'text') return document.text;
|
|
3860
|
+
if (document.kind === 'pdf') return pdfText(document);
|
|
3861
|
+
if (document.kind === 'workbook') return workbookText(document);
|
|
3862
|
+
if (document.kind === 'deck') return deckText(document);
|
|
2173
3863
|
if (Array.isArray(document.data) && document.data.every(entry => typeof entry === 'string')) return document.data.join('\n');
|
|
2174
3864
|
return typeof document.data === 'string' ? document.data : JSON.stringify(document.data);
|
|
2175
3865
|
}
|
|
@@ -2181,6 +3871,8 @@ function documentFor(step, scope) {
|
|
|
2181
3871
|
}
|
|
2182
3872
|
const source = scope.get(step.from);
|
|
2183
3873
|
if (source === undefined) throw new Error(`"${step.from}" is not bound`);
|
|
3874
|
+
if (isPdfDocument(source) || isWorkbookDocument(source) || isDeckDocument(source)) return source;
|
|
3875
|
+
if (step.kind === 'table') throw new Error(`"${step.from}" is not a PDF, a workbook or a deck; request it with "as": "pdf", "csv", "xlsx" or "pptx"`);
|
|
2184
3876
|
if (step.kind === 'regex') {
|
|
2185
3877
|
if (typeof source === 'string') return {
|
|
2186
3878
|
kind: 'text',
|
|
@@ -2216,6 +3908,135 @@ function documentFor(step, scope) {
|
|
|
2216
3908
|
};
|
|
2217
3909
|
}
|
|
2218
3910
|
|
|
3911
|
+
/** Aliases one document may expand: a "billion laughs" document needs far more. */
|
|
3912
|
+
const MAX_ALIASES = 100;
|
|
3913
|
+
/**
|
|
3914
|
+
* Parses YAML with the `yaml` package, imported on first use. The version is
|
|
3915
|
+
* pinned to YAML 1.2 (core schema) whatever the document declares: under a
|
|
3916
|
+
* `%YAML 1.1` directive, `NO` would read as `false` and `0123` as octal `83`.
|
|
3917
|
+
* Merge keys (`<<: *base`) are applied, duplicate keys are an error, aliases
|
|
3918
|
+
* are capped, and custom tags never build values: nothing in the text runs.
|
|
3919
|
+
*
|
|
3920
|
+
* @param text - The YAML.
|
|
3921
|
+
* @param source - Where it came from, for messages.
|
|
3922
|
+
* @param scalars - `typed` (default), or `text` to keep every scalar as written (`0123` stays `"0123"`).
|
|
3923
|
+
* @returns The data, the number of documents and the warnings.
|
|
3924
|
+
* @throws Error naming the source, with the line and column, for YAML that does not parse.
|
|
3925
|
+
*/
|
|
3926
|
+
async function readYaml(text, source, scalars = 'typed') {
|
|
3927
|
+
const {
|
|
3928
|
+
parseAllDocuments
|
|
3929
|
+
} = await import('yaml');
|
|
3930
|
+
const parsed = parseAllDocuments(text, {
|
|
3931
|
+
version: '1.2',
|
|
3932
|
+
schema: scalars === 'text' ? 'failsafe' : 'core',
|
|
3933
|
+
merge: true,
|
|
3934
|
+
uniqueKeys: true,
|
|
3935
|
+
prettyErrors: true
|
|
3936
|
+
});
|
|
3937
|
+
const documents = Array.isArray(parsed) ? parsed : [parsed];
|
|
3938
|
+
const warnings = [];
|
|
3939
|
+
const values = [];
|
|
3940
|
+
for (const document of documents) {
|
|
3941
|
+
const [error] = document.errors;
|
|
3942
|
+
if (error !== undefined) throw new Error(`${source}: not YAML (${error.message.split('\n', 1)[0]})`, {
|
|
3943
|
+
cause: error
|
|
3944
|
+
});
|
|
3945
|
+
warnings.push(...document.warnings.map(warning => warning.message.split('\n', 1)[0]));
|
|
3946
|
+
try {
|
|
3947
|
+
values.push(document.toJS({
|
|
3948
|
+
maxAliasCount: MAX_ALIASES
|
|
3949
|
+
}));
|
|
3950
|
+
} catch (error) {
|
|
3951
|
+
throw new Error(`${source}: ${error.message}`, {
|
|
3952
|
+
cause: error
|
|
3953
|
+
});
|
|
3954
|
+
}
|
|
3955
|
+
}
|
|
3956
|
+
return {
|
|
3957
|
+
data: values.length === 1 ? values[0] : values,
|
|
3958
|
+
documents: values.length,
|
|
3959
|
+
warnings
|
|
3960
|
+
};
|
|
3961
|
+
}
|
|
3962
|
+
|
|
3963
|
+
/** A leading `---` block of YAML. */
|
|
3964
|
+
const FRONT_MATTER = /^---[ \t]*\r?\n([\s\S]*?)\r?\n---[ \t]*(?:\r?\n|$)/;
|
|
3965
|
+
/**
|
|
3966
|
+
* Renders Markdown (GitHub-flavoured: tables, task lists, strikethrough,
|
|
3967
|
+
* autolinks) to HTML with `marked`, imported on first use, so every `css`
|
|
3968
|
+
* selector works on it:
|
|
3969
|
+
*
|
|
3970
|
+
* - each heading and everything up to the next heading of the same or a higher
|
|
3971
|
+
* level is wrapped in `<section data-heading="…" data-level="…">`, sections
|
|
3972
|
+
* nesting, so "the table under *Prezzi*" is one selector;
|
|
3973
|
+
* - headings get slug ids (`<h2 id="prezzi">`);
|
|
3974
|
+
* - a leading `---` YAML block is parsed (YAML 1.2, as `as: "yaml"` reads it)
|
|
3975
|
+
* and put in the head as `<script type="application/json" data-front-matter>`.
|
|
3976
|
+
*
|
|
3977
|
+
* Raw HTML in the Markdown is kept: it is data, parsed by cheerio, never run.
|
|
3978
|
+
*
|
|
3979
|
+
* @param text - The Markdown.
|
|
3980
|
+
* @param source - Where it came from, for messages.
|
|
3981
|
+
* @returns The HTML, the front matter's data and the YAML parser's warnings.
|
|
3982
|
+
* @throws Error naming the source when the front matter is not YAML.
|
|
3983
|
+
*/
|
|
3984
|
+
async function readMarkdown(text, source) {
|
|
3985
|
+
const matter = FRONT_MATTER.exec(text);
|
|
3986
|
+
const body = matter === null ? text : text.slice(matter[0].length);
|
|
3987
|
+
const front = matter === null ? undefined : await readYaml(matter[1], `${source} front matter`);
|
|
3988
|
+
const {
|
|
3989
|
+
marked
|
|
3990
|
+
} = await import('marked');
|
|
3991
|
+
const rendered = marked.parse(body, {
|
|
3992
|
+
gfm: true,
|
|
3993
|
+
async: false
|
|
3994
|
+
});
|
|
3995
|
+
const sections = sectioned(rendered);
|
|
3996
|
+
const head = front === undefined ? '' : `<script type="application/json" data-front-matter>${JSON.stringify(front.data ?? null).replaceAll('<', '<')}</script>`;
|
|
3997
|
+
return {
|
|
3998
|
+
html: `<!doctype html><html><head>${head}</head><body>${sections}</body></html>`,
|
|
3999
|
+
frontMatter: front?.data,
|
|
4000
|
+
warnings: front?.warnings ?? []
|
|
4001
|
+
};
|
|
4002
|
+
}
|
|
4003
|
+
/** Wraps each heading and what follows it, up to the next heading of the same or a higher level, in a section. */
|
|
4004
|
+
function sectioned(html) {
|
|
4005
|
+
const $ = load(html, null, false);
|
|
4006
|
+
const open = [];
|
|
4007
|
+
const used = new Map();
|
|
4008
|
+
let out = '';
|
|
4009
|
+
const nodes = $.root().contents().toArray();
|
|
4010
|
+
for (const node of nodes) {
|
|
4011
|
+
const level = node.type === 'tag' ? /^h([1-6])$/.exec(node.name)?.[1] : undefined;
|
|
4012
|
+
if (level === undefined) {
|
|
4013
|
+
out += $.html(node);
|
|
4014
|
+
continue;
|
|
4015
|
+
}
|
|
4016
|
+
const depth = Number(level);
|
|
4017
|
+
while (open.length > 0 && (open.at(-1) ?? 0) >= depth) {
|
|
4018
|
+
open.pop();
|
|
4019
|
+
out += '</section>';
|
|
4020
|
+
}
|
|
4021
|
+
const heading = $(node);
|
|
4022
|
+
const text = heading.text().replaceAll(/\s+/g, ' ').trim();
|
|
4023
|
+
heading.attr('id', uniqueSlug(text, used));
|
|
4024
|
+
out += `<section data-heading="${escapeAttribute(text)}" data-level="${depth}">${$.html(node)}`;
|
|
4025
|
+
open.push(depth);
|
|
4026
|
+
}
|
|
4027
|
+
return out + '</section>'.repeat(open.length);
|
|
4028
|
+
}
|
|
4029
|
+
/** GitHub's heading ids: lower case, spaces to hyphens, punctuation dropped, a counter for repeats. */
|
|
4030
|
+
function uniqueSlug(text, used) {
|
|
4031
|
+
const slug = text.toLowerCase().replaceAll(/[^\p{L}\p{N}\s-]/gu, '').trim().replaceAll(/\s/g, '-');
|
|
4032
|
+
const count = used.get(slug) ?? 0;
|
|
4033
|
+
used.set(slug, count + 1);
|
|
4034
|
+
return count === 0 ? slug : `${slug}-${count}`;
|
|
4035
|
+
}
|
|
4036
|
+
function escapeAttribute(text) {
|
|
4037
|
+
return text.replaceAll('&', '&').replaceAll('"', '"').replaceAll('<', '<');
|
|
4038
|
+
}
|
|
4039
|
+
|
|
2219
4040
|
/** A response with a 4xx or 5xx status. */
|
|
2220
4041
|
class HttpError extends Error {
|
|
2221
4042
|
status;
|
|
@@ -2232,6 +4053,82 @@ class HttpError extends Error {
|
|
|
2232
4053
|
}
|
|
2233
4054
|
}
|
|
2234
4055
|
|
|
4056
|
+
/**
|
|
4057
|
+
* Decodes a body, in this order: a byte-order mark (UTF-8, UTF-16 LE/BE; Excel's
|
|
4058
|
+
* "Unicode text" export is UTF-16 LE), the encoding a recipe asks for, the
|
|
4059
|
+
* charset the server declares, strict UTF-8, and Windows-1252 (a superset of
|
|
4060
|
+
* Latin-1) for text that is not UTF-8.
|
|
4061
|
+
*
|
|
4062
|
+
* The Windows-1252 fallback is only taken when the text holds no valid UTF-8
|
|
4063
|
+
* beyond ASCII: a UTF-8 page with one stray byte keeps its accents, with a
|
|
4064
|
+
* replacement character for the stray byte, instead of turning every accent
|
|
4065
|
+
* into mojibake.
|
|
4066
|
+
*
|
|
4067
|
+
* @param bytes - The body.
|
|
4068
|
+
* @param options - `encoding`: the recipe's choice, a WHATWG label (wins over
|
|
4069
|
+
* the charset, not over a BOM); `charset`: from the content type (ignored when
|
|
4070
|
+
* not a known label).
|
|
4071
|
+
* @returns The text, without its BOM, and the encoding used.
|
|
4072
|
+
* @throws Error when `encoding` is not a known label.
|
|
4073
|
+
*/
|
|
4074
|
+
function decodeText(bytes, options = {}) {
|
|
4075
|
+
const bom = bomOf(bytes);
|
|
4076
|
+
if (bom !== undefined) return decodeWith(new TextDecoder(bom), bytes);
|
|
4077
|
+
if (options.encoding !== undefined) {
|
|
4078
|
+
const decoder = decoderFor(options.encoding);
|
|
4079
|
+
if (decoder === undefined) throw new Error(`"${options.encoding}" is not an encoding this runtime knows (try utf8, windows-1252, iso-8859-15, utf-16le, shift_jis…)`);
|
|
4080
|
+
return decodeWith(decoder, bytes);
|
|
4081
|
+
}
|
|
4082
|
+
const declared = options.charset === undefined ? undefined : decoderFor(options.charset);
|
|
4083
|
+
if (declared !== undefined) return decodeWith(declared, bytes);
|
|
4084
|
+
try {
|
|
4085
|
+
return decodeWith(new TextDecoder('utf-8', {
|
|
4086
|
+
fatal: true
|
|
4087
|
+
}), bytes);
|
|
4088
|
+
} catch {
|
|
4089
|
+
const lenient = decodeWith(new TextDecoder('utf-8'), bytes);
|
|
4090
|
+
return hasNonAsciiText(lenient.text) ? lenient : decodeWith(new TextDecoder('windows-1252'), bytes);
|
|
4091
|
+
}
|
|
4092
|
+
}
|
|
4093
|
+
/** Decodes, naming the encoding by its canonical WHATWG name (`utf-8`, `windows-1252`, `utf-16le`). */
|
|
4094
|
+
function decodeWith(decoder, bytes) {
|
|
4095
|
+
return {
|
|
4096
|
+
text: decoder.decode(bytes),
|
|
4097
|
+
encoding: decoder.encoding
|
|
4098
|
+
};
|
|
4099
|
+
}
|
|
4100
|
+
/**
|
|
4101
|
+
* The charset a content type declares (`text/csv; charset=ISO-8859-1`).
|
|
4102
|
+
*
|
|
4103
|
+
* @param contentType - The header value.
|
|
4104
|
+
* @returns The charset, or `undefined`.
|
|
4105
|
+
*/
|
|
4106
|
+
function charsetOf(contentType) {
|
|
4107
|
+
const match = /;\s*charset\s*=\s*"?([^";\s]+)"?/i.exec(contentType);
|
|
4108
|
+
return match?.[1];
|
|
4109
|
+
}
|
|
4110
|
+
function bomOf(bytes) {
|
|
4111
|
+
if (bytes[0] === 0xEF && bytes[1] === 0xBB && bytes[2] === 0xBF) return 'utf8';
|
|
4112
|
+
if (bytes[0] === 0xFF && bytes[1] === 0xFE) return 'utf-16le';
|
|
4113
|
+
if (bytes[0] === 0xFE && bytes[1] === 0xFF) return 'utf-16be';
|
|
4114
|
+
return undefined;
|
|
4115
|
+
}
|
|
4116
|
+
/** Whether the text holds a character beyond ASCII other than the replacement character: valid UTF-8 was seen. */
|
|
4117
|
+
function hasNonAsciiText(text) {
|
|
4118
|
+
for (const char of text) {
|
|
4119
|
+
const code = char.codePointAt(0) ?? 0;
|
|
4120
|
+
if (code !== 0xFF_FD && code > 0x7F) return true;
|
|
4121
|
+
}
|
|
4122
|
+
return false;
|
|
4123
|
+
}
|
|
4124
|
+
function decoderFor(label) {
|
|
4125
|
+
try {
|
|
4126
|
+
return new TextDecoder(label.trim());
|
|
4127
|
+
} catch {
|
|
4128
|
+
return undefined;
|
|
4129
|
+
}
|
|
4130
|
+
}
|
|
4131
|
+
|
|
2235
4132
|
/**
|
|
2236
4133
|
* HTTP through Playwright's request context: cookies, redirects and storage
|
|
2237
4134
|
* state behave exactly as they do in the browser, so a session captured by a
|
|
@@ -2263,6 +4160,7 @@ class HttpClient {
|
|
|
2263
4160
|
* @throws HttpError for a 4xx or 5xx status.
|
|
2264
4161
|
*/
|
|
2265
4162
|
async send(httpRequest) {
|
|
4163
|
+
if (httpRequest.url.startsWith('file:')) return readLocalFile(httpRequest);
|
|
2266
4164
|
const response = await this.context.fetch(httpRequest.url, {
|
|
2267
4165
|
method: httpRequest.method ?? (httpRequest.body === undefined ? 'GET' : 'POST'),
|
|
2268
4166
|
params: httpRequest.query,
|
|
@@ -2270,12 +4168,20 @@ class HttpClient {
|
|
|
2270
4168
|
data: httpRequest.body,
|
|
2271
4169
|
timeout: httpRequest.timeoutMs ?? this.timeoutMs
|
|
2272
4170
|
});
|
|
2273
|
-
const
|
|
4171
|
+
const {
|
|
4172
|
+
body,
|
|
4173
|
+
warnings,
|
|
4174
|
+
format
|
|
4175
|
+
} = await readBody(response, httpRequest);
|
|
2274
4176
|
const result = {
|
|
2275
4177
|
status: response.status(),
|
|
2276
4178
|
url: response.url(),
|
|
2277
4179
|
headers: response.headers(),
|
|
2278
|
-
body
|
|
4180
|
+
body,
|
|
4181
|
+
format,
|
|
4182
|
+
...(warnings.length > 0 && {
|
|
4183
|
+
warnings
|
|
4184
|
+
})
|
|
2279
4185
|
};
|
|
2280
4186
|
if (response.status() >= 400) throw new HttpError(response.status(), response.url(), body, response.headers());
|
|
2281
4187
|
return result;
|
|
@@ -2288,22 +4194,107 @@ class HttpClient {
|
|
|
2288
4194
|
return this.context.dispose();
|
|
2289
4195
|
}
|
|
2290
4196
|
}
|
|
2291
|
-
async function readBody(response,
|
|
2292
|
-
const
|
|
2293
|
-
const
|
|
2294
|
-
|
|
2295
|
-
|
|
2296
|
-
|
|
4197
|
+
async function readBody(response, httpRequest) {
|
|
4198
|
+
const contentType = response.headers()['content-type'] ?? '';
|
|
4199
|
+
const format = httpRequest.as ?? formatFromContentType(contentType);
|
|
4200
|
+
return parseBody(format, await response.body(), response.url(), {
|
|
4201
|
+
...httpRequest,
|
|
4202
|
+
charset: charsetOf(contentType)
|
|
4203
|
+
});
|
|
4204
|
+
}
|
|
4205
|
+
/**
|
|
4206
|
+
* A `file:` URL, read from disk: a PDF, spreadsheet, presentation, CSV, YAML
|
|
4207
|
+
* or JSON a recipe gets from a folder instead of a server. The format is `as`,
|
|
4208
|
+
* else the file extension.
|
|
4209
|
+
*/
|
|
4210
|
+
async function readLocalFile(httpRequest) {
|
|
4211
|
+
const path = fileURLToPath(httpRequest.url);
|
|
4212
|
+
const bytes = await readFile(path);
|
|
4213
|
+
const {
|
|
4214
|
+
body,
|
|
4215
|
+
warnings,
|
|
4216
|
+
format
|
|
4217
|
+
} = await parseBody(httpRequest.as ?? formatFromExtension(extname(path)), bytes, httpRequest.url, httpRequest);
|
|
4218
|
+
return {
|
|
4219
|
+
status: 200,
|
|
4220
|
+
url: httpRequest.url,
|
|
4221
|
+
headers: {},
|
|
4222
|
+
body,
|
|
4223
|
+
format,
|
|
4224
|
+
...(warnings.length > 0 && {
|
|
4225
|
+
warnings
|
|
4226
|
+
})
|
|
4227
|
+
};
|
|
4228
|
+
}
|
|
4229
|
+
async function parseBody(format, bytes, url, reading) {
|
|
4230
|
+
if (format === 'yaml') {
|
|
4231
|
+
const {
|
|
4232
|
+
text
|
|
4233
|
+
} = decodeText(bytes, reading);
|
|
4234
|
+
const {
|
|
4235
|
+
data,
|
|
4236
|
+
warnings
|
|
4237
|
+
} = await readYaml(text, url, reading.scalars);
|
|
4238
|
+
return {
|
|
4239
|
+
body: {
|
|
2297
4240
|
kind: 'json',
|
|
2298
|
-
data
|
|
2299
|
-
}
|
|
2300
|
-
|
|
2301
|
-
|
|
2302
|
-
|
|
2303
|
-
|
|
2304
|
-
|
|
4241
|
+
data
|
|
4242
|
+
},
|
|
4243
|
+
warnings,
|
|
4244
|
+
format
|
|
4245
|
+
};
|
|
4246
|
+
}
|
|
4247
|
+
if (format === 'markdown') {
|
|
4248
|
+
const {
|
|
4249
|
+
text
|
|
4250
|
+
} = decodeText(bytes, reading);
|
|
4251
|
+
const {
|
|
4252
|
+
html,
|
|
4253
|
+
warnings
|
|
4254
|
+
} = await readMarkdown(text, url);
|
|
4255
|
+
return {
|
|
4256
|
+
body: {
|
|
4257
|
+
kind: 'html',
|
|
4258
|
+
html
|
|
4259
|
+
},
|
|
4260
|
+
warnings,
|
|
4261
|
+
format
|
|
4262
|
+
};
|
|
4263
|
+
}
|
|
4264
|
+
return {
|
|
4265
|
+
body: await parseFormat(format, bytes, url, reading),
|
|
4266
|
+
warnings: [],
|
|
4267
|
+
format
|
|
4268
|
+
};
|
|
4269
|
+
}
|
|
4270
|
+
async function parseFormat(format, bytes, url, reading) {
|
|
4271
|
+
if (format === 'pdf') return readPdf(bytes, url);
|
|
4272
|
+
if (format === 'xlsx') return readXlsxWorkbook(bytes, url);
|
|
4273
|
+
if (format === 'pptx') return readPptxDeck(bytes, url);
|
|
4274
|
+
const {
|
|
4275
|
+
text,
|
|
4276
|
+
encoding
|
|
4277
|
+
} = decodeText(bytes, reading);
|
|
4278
|
+
if (format === 'csv') return csvWorkbook(text, {
|
|
4279
|
+
name: sheetNameOf(url),
|
|
4280
|
+
encoding,
|
|
4281
|
+
delimiter: reading.delimiter
|
|
4282
|
+
});
|
|
4283
|
+
if (format === 'jsonl') return {
|
|
4284
|
+
kind: 'json',
|
|
4285
|
+
data: parseJsonLines(text, url)
|
|
4286
|
+
};
|
|
4287
|
+
if (format === 'json') {
|
|
4288
|
+
const parsed = parseJsonLike(text);
|
|
4289
|
+
if ('error' in parsed) throw new Error(`${url}: body is not JSON (${parsed.error.message})${looksLikeJsonLines(text) ? '; it looks like JSON Lines: read it with "as": "jsonl"' : ''}`, {
|
|
4290
|
+
cause: parsed.error
|
|
4291
|
+
});
|
|
4292
|
+
return {
|
|
4293
|
+
kind: 'json',
|
|
4294
|
+
data: parsed.value
|
|
4295
|
+
};
|
|
2305
4296
|
}
|
|
2306
|
-
return
|
|
4297
|
+
return format === 'html' ? {
|
|
2307
4298
|
kind: 'html',
|
|
2308
4299
|
html: text
|
|
2309
4300
|
} : {
|
|
@@ -2311,12 +4302,55 @@ async function readBody(response, as) {
|
|
|
2311
4302
|
text
|
|
2312
4303
|
};
|
|
2313
4304
|
}
|
|
2314
|
-
|
|
2315
|
-
|
|
4305
|
+
/** Several lines, the first of them JSON on its own. */
|
|
4306
|
+
function looksLikeJsonLines(text) {
|
|
4307
|
+
const lines = text.split(/\r?\n/).filter(line => line.trim() !== '');
|
|
4308
|
+
if (lines.length < 2) return false;
|
|
4309
|
+
const first = parseJsonLike(lines[0]);
|
|
4310
|
+
return 'value' in first;
|
|
4311
|
+
}
|
|
4312
|
+
function formatFromContentType(contentType) {
|
|
4313
|
+
const type = contentType.toLowerCase().split(';', 1)[0].trim();
|
|
4314
|
+
if (CSV_TYPES.has(type)) return 'csv';
|
|
4315
|
+
if (JSON_LINES_TYPES.has(type)) return 'jsonl';
|
|
4316
|
+
// A legacy .xls or .ppt goes to the Office reader too, which says what to do with it.
|
|
4317
|
+
if (type.includes('spreadsheetml') || type.startsWith('application/vnd.ms-excel')) return 'xlsx';
|
|
4318
|
+
if (type.includes('presentationml') || type.startsWith('application/vnd.ms-powerpoint')) return 'pptx';
|
|
4319
|
+
if (YAML_TYPES.has(type)) return 'yaml';
|
|
4320
|
+
if (type === 'text/markdown' || type === 'text/x-markdown') return 'markdown';
|
|
2316
4321
|
if (type.includes('json')) return 'json';
|
|
4322
|
+
if (type.includes('pdf')) return 'pdf';
|
|
2317
4323
|
if (type.includes('html') || type.includes('xml')) return 'html';
|
|
2318
4324
|
return 'text';
|
|
2319
4325
|
}
|
|
4326
|
+
const JSON_LINES_TYPES = new Set(['application/x-ndjson', 'application/ndjson', 'application/jsonl', 'application/x-jsonlines', 'application/jsonlines']);
|
|
4327
|
+
const YAML_TYPES = new Set(['application/yaml', 'application/x-yaml', 'text/yaml', 'text/x-yaml']);
|
|
4328
|
+
const CSV_TYPES = new Set(['text/csv', 'application/csv', 'text/x-csv', 'application/x-csv', 'text/comma-separated-values', 'text/tab-separated-values']);
|
|
4329
|
+
function formatFromExtension(extension) {
|
|
4330
|
+
const formats = {
|
|
4331
|
+
'.json': 'json',
|
|
4332
|
+
'.jsonl': 'jsonl',
|
|
4333
|
+
'.ndjson': 'jsonl',
|
|
4334
|
+
'.pdf': 'pdf',
|
|
4335
|
+
'.csv': 'csv',
|
|
4336
|
+
'.tsv': 'csv',
|
|
4337
|
+
'.xlsx': 'xlsx',
|
|
4338
|
+
'.xlsm': 'xlsx',
|
|
4339
|
+
'.xls': 'xlsx',
|
|
4340
|
+
'.pptx': 'pptx',
|
|
4341
|
+
'.pptm': 'pptx',
|
|
4342
|
+
'.ppsx': 'pptx',
|
|
4343
|
+
'.ppt': 'pptx',
|
|
4344
|
+
'.yaml': 'yaml',
|
|
4345
|
+
'.yml': 'yaml',
|
|
4346
|
+
'.md': 'markdown',
|
|
4347
|
+
'.markdown': 'markdown',
|
|
4348
|
+
'.html': 'html',
|
|
4349
|
+
'.htm': 'html',
|
|
4350
|
+
'.xml': 'html'
|
|
4351
|
+
};
|
|
4352
|
+
return formats[extension.toLowerCase()] ?? 'text';
|
|
4353
|
+
}
|
|
2320
4354
|
|
|
2321
4355
|
/**
|
|
2322
4356
|
* Sends a `request` step: renders its templates, waits for the gate's throttle,
|
|
@@ -2329,7 +4363,7 @@ function kindFromContentType(contentType) {
|
|
|
2329
4363
|
* @param recipe - The recipe: its limits, block rule and id.
|
|
2330
4364
|
* @param gate - Spaces request starts by `delayMs`.
|
|
2331
4365
|
* @param events - Where to report the visit.
|
|
2332
|
-
* @throws BlockedError when the response is a block
|
|
4366
|
+
* @throws BlockedError when the response is a block, or a captcha page under `session.captcha`; HttpError for any other 4xx/5xx.
|
|
2333
4367
|
*/
|
|
2334
4368
|
async function sendRequest(step, scope, client, recipe, gate, events) {
|
|
2335
4369
|
const lookup = path => scope.lookup(path);
|
|
@@ -2344,6 +4378,9 @@ async function sendRequest(step, scope, client, recipe, gate, events) {
|
|
|
2344
4378
|
headers: step.headers === undefined ? undefined : renderMap(step.headers, lookup),
|
|
2345
4379
|
body: renderDeep(step.body, lookup),
|
|
2346
4380
|
as: step.as,
|
|
4381
|
+
encoding: step.encoding,
|
|
4382
|
+
delimiter: step.delimiter,
|
|
4383
|
+
scalars: step.scalars,
|
|
2347
4384
|
timeoutMs: recipe.limits?.timeoutMs
|
|
2348
4385
|
});
|
|
2349
4386
|
} catch (error) {
|
|
@@ -2369,6 +4406,15 @@ async function sendRequest(step, scope, client, recipe, gate, events) {
|
|
|
2369
4406
|
number: scope.pageState?.number ?? 1,
|
|
2370
4407
|
status: response.status
|
|
2371
4408
|
});
|
|
4409
|
+
const warnings = response.warnings ?? [];
|
|
4410
|
+
for (const warning of warnings) events.emit({
|
|
4411
|
+
type: 'warning',
|
|
4412
|
+
recipeId: recipe.id,
|
|
4413
|
+
message: `${response.url}: ${warning}`,
|
|
4414
|
+
meta: {
|
|
4415
|
+
url: response.url
|
|
4416
|
+
}
|
|
4417
|
+
});
|
|
2372
4418
|
const blocked = await detectBlock({
|
|
2373
4419
|
url: response.url,
|
|
2374
4420
|
status: response.status,
|
|
@@ -2376,14 +4422,22 @@ async function sendRequest(step, scope, client, recipe, gate, events) {
|
|
|
2376
4422
|
text: async () => bodyText(response.body)
|
|
2377
4423
|
}, recipe.session?.blockedWhen);
|
|
2378
4424
|
if (blocked !== undefined) throw blocked;
|
|
4425
|
+
if (recipe.session?.captcha !== undefined && response.body.kind === 'html' && CAPTCHA_MARKUP.test(response.body.html)) {
|
|
4426
|
+
throw new BlockedError(response.url, response.status, 'the page shows a captcha, which is solved on a live page: run this recipe in web mode, or get past it in session.bootstrap');
|
|
4427
|
+
}
|
|
2379
4428
|
scope.setPage({
|
|
2380
4429
|
url: response.url,
|
|
2381
4430
|
document: response.body
|
|
2382
4431
|
});
|
|
2383
4432
|
if (step.id !== undefined) scope.set(step.id, documentValue(response.body));
|
|
2384
4433
|
}
|
|
4434
|
+
/** The class names of the widgets `session.captcha` solves. Checked only when a recipe declares it. */
|
|
4435
|
+
const CAPTCHA_MARKUP = /\b(?:g-recaptcha|h-captcha|cf-turnstile)\b/;
|
|
2385
4436
|
function bodyText(body) {
|
|
2386
4437
|
if (body.kind === 'json') return JSON.stringify(body.data);
|
|
4438
|
+
if (body.kind === 'pdf') return pdfText(body);
|
|
4439
|
+
if (body.kind === 'workbook') return workbookText(body);
|
|
4440
|
+
if (body.kind === 'deck') return deckText(body);
|
|
2387
4441
|
return body.kind === 'html' ? body.html : body.text;
|
|
2388
4442
|
}
|
|
2389
4443
|
function renderMap(map, lookup) {
|
|
@@ -2404,10 +4458,11 @@ function resolveUrl(target, base) {
|
|
|
2404
4458
|
throw new Error(`"${target}" is not a URL${base === undefined || base === '' ? ' and no page is known to resolve it against' : ` and cannot be resolved against ${base}`}`);
|
|
2405
4459
|
}
|
|
2406
4460
|
}
|
|
2407
|
-
/** What a step id holds for a document: parsed JSON, or the markup / text. */
|
|
4461
|
+
/** What a step id holds for a document: parsed JSON, the read PDF, workbook or deck, or the markup / text. */
|
|
2408
4462
|
function documentValue(body) {
|
|
2409
4463
|
if (body.kind === 'json') return body.data;
|
|
2410
|
-
|
|
4464
|
+
if (body.kind === 'html') return body.html;
|
|
4465
|
+
return body.kind === 'text' ? body.text : body;
|
|
2411
4466
|
}
|
|
2412
4467
|
|
|
2413
4468
|
/** Runs api-mode leaf steps against an HTTP sender. */
|
|
@@ -3489,10 +5544,10 @@ async function extractFromPage(step, page, scope) {
|
|
|
3489
5544
|
extractFromDocument(step, scope);
|
|
3490
5545
|
return;
|
|
3491
5546
|
}
|
|
3492
|
-
const
|
|
3493
|
-
|
|
3494
|
-
|
|
3495
|
-
|
|
5547
|
+
const values = step.kind === 'table' ? tablesIn({
|
|
5548
|
+
kind: 'html',
|
|
5549
|
+
html: await page.content()
|
|
5550
|
+
}, step, scope) : await readPage(step, page, scope);
|
|
3496
5551
|
if (step.many === true) {
|
|
3497
5552
|
if (step.id !== undefined) scope.set(step.id, values);
|
|
3498
5553
|
return;
|
|
@@ -3500,6 +5555,13 @@ async function extractFromPage(step, page, scope) {
|
|
|
3500
5555
|
if (values.length === 0) throw new NoMatchError(step.selector);
|
|
3501
5556
|
if (step.id !== undefined) scope.set(step.id, values[0]);
|
|
3502
5557
|
}
|
|
5558
|
+
/** A css, xpath or regex extract on the live page. */
|
|
5559
|
+
async function readPage(step, page, scope) {
|
|
5560
|
+
const rendered = renderSelector(step.selector, scope);
|
|
5561
|
+
const take = step.take ?? 'text';
|
|
5562
|
+
const raw = step.kind === 'regex' ? selectRegex(await page.content(), rendered) : await page.locator(step.kind === 'xpath' ? `xpath=${rendered}` : rendered).evaluateAll(readAll, take);
|
|
5563
|
+
return take === 'text' ? raw.map(value => typeof value === 'string' ? collapse(value) : value) : raw;
|
|
5564
|
+
}
|
|
3503
5565
|
/** Runs inside the page: one value per matched element. Keep it self-contained; it is serialised. */
|
|
3504
5566
|
function readAll(elements, take) {
|
|
3505
5567
|
return elements.map(element => {
|
|
@@ -3685,18 +5747,26 @@ async function navigate(step, page, scope, recipe, gate, events) {
|
|
|
3685
5747
|
}
|
|
3686
5748
|
|
|
3687
5749
|
const NEXT_LINK_TIMEOUT_MS = 2000;
|
|
3688
|
-
/**
|
|
5750
|
+
/** Steps after which a page may show a new captcha (`session.captcha`). */
|
|
5751
|
+
const CHALLENGING_STEPS = new Set(['click', 'press']);
|
|
5752
|
+
/**
|
|
5753
|
+
* Runs web-mode leaf steps on a browser page. With a captcha guard, a page a
|
|
5754
|
+
* navigation, click or key press leads to is checked for a challenge, solved
|
|
5755
|
+
* before the next step runs.
|
|
5756
|
+
*/
|
|
3689
5757
|
class WebStepRunner {
|
|
3690
5758
|
session;
|
|
3691
5759
|
recipe;
|
|
3692
5760
|
events;
|
|
3693
5761
|
gate;
|
|
5762
|
+
captcha;
|
|
3694
5763
|
page;
|
|
3695
|
-
constructor(session, recipe, events, gate = new RunGate(1, recipe.limits?.delayMs ?? 0)) {
|
|
5764
|
+
constructor(session, recipe, events, gate = new RunGate(1, recipe.limits?.delayMs ?? 0), captcha) {
|
|
3696
5765
|
this.session = session;
|
|
3697
5766
|
this.recipe = recipe;
|
|
3698
5767
|
this.events = events;
|
|
3699
5768
|
this.gate = gate;
|
|
5769
|
+
this.captcha = captcha;
|
|
3700
5770
|
this.page = session.page;
|
|
3701
5771
|
}
|
|
3702
5772
|
/** Clicks and key presses can navigate; keep `page.url` honest after every leaf step. */
|
|
@@ -3706,11 +5776,29 @@ class WebStepRunner {
|
|
|
3706
5776
|
url
|
|
3707
5777
|
});
|
|
3708
5778
|
}
|
|
5779
|
+
/** Navigates; a block page showing a captcha is solved under `onBlock.solve`, and a page reached is checked for one. */
|
|
5780
|
+
async visit(step, scope) {
|
|
5781
|
+
try {
|
|
5782
|
+
await navigate(step, this.page, scope, this.recipe, this.gate, this.events);
|
|
5783
|
+
} catch (error) {
|
|
5784
|
+
if (!(error instanceof BlockedError) || this.captcha?.solvesBlocks !== true) throw error;
|
|
5785
|
+
await this.captcha.solveBlock(this.page, error);
|
|
5786
|
+
return;
|
|
5787
|
+
}
|
|
5788
|
+
await this.captcha?.check(this.page);
|
|
5789
|
+
}
|
|
3709
5790
|
async runLeaf(step, scope) {
|
|
3710
5791
|
switch (step.type) {
|
|
3711
5792
|
case 'goto':
|
|
3712
5793
|
{
|
|
3713
|
-
|
|
5794
|
+
await this.visit(step, scope);
|
|
5795
|
+
break;
|
|
5796
|
+
}
|
|
5797
|
+
case 'captcha':
|
|
5798
|
+
{
|
|
5799
|
+
if (this.captcha === undefined) throw new Error('a captcha step needs a crawler with captcha solvers');
|
|
5800
|
+
await this.captcha.step(this.page, step);
|
|
5801
|
+
break;
|
|
3714
5802
|
}
|
|
3715
5803
|
case 'click':
|
|
3716
5804
|
{
|
|
@@ -3762,6 +5850,7 @@ class WebStepRunner {
|
|
|
3762
5850
|
throw new Error(`"${step.type}" is an api step; this recipe runs in web mode`);
|
|
3763
5851
|
}
|
|
3764
5852
|
}
|
|
5853
|
+
if (CHALLENGING_STEPS.has(step.type)) await this.captcha?.check(this.page);
|
|
3765
5854
|
this.trackUrl(scope);
|
|
3766
5855
|
}
|
|
3767
5856
|
async nextPage(next, scope) {
|
|
@@ -3769,9 +5858,10 @@ class WebStepRunner {
|
|
|
3769
5858
|
if ('url' in next) {
|
|
3770
5859
|
const target = renderText(next.url, path => scope.lookup(path));
|
|
3771
5860
|
if (target === '') return null;
|
|
3772
|
-
await
|
|
5861
|
+
await this.visit({
|
|
5862
|
+
type: 'goto',
|
|
3773
5863
|
url: target
|
|
3774
|
-
},
|
|
5864
|
+
}, scope);
|
|
3775
5865
|
return {
|
|
3776
5866
|
kind: 'url',
|
|
3777
5867
|
url: this.page.url()
|
|
@@ -3793,6 +5883,7 @@ class WebStepRunner {
|
|
|
3793
5883
|
url: this.page.url(),
|
|
3794
5884
|
number: (scope.pageState?.number ?? 1) + 1
|
|
3795
5885
|
});
|
|
5886
|
+
await this.captcha?.check(this.page);
|
|
3796
5887
|
return {
|
|
3797
5888
|
kind: 'url',
|
|
3798
5889
|
url: this.page.url()
|
|
@@ -3837,9 +5928,10 @@ function accessOptions(lease, headers) {
|
|
|
3837
5928
|
* @param recipe - The input recipe.
|
|
3838
5929
|
* @param deps - Browser, hooks, events.
|
|
3839
5930
|
* @param lease - The recipe run's access; direct when omitted.
|
|
5931
|
+
* @param captcha - Solves the bootstrap's captchas (a login form's).
|
|
3840
5932
|
* @returns The state, or `undefined` when the recipe declares none.
|
|
3841
5933
|
*/
|
|
3842
|
-
async function resolveStorageState(recipe, deps, lease) {
|
|
5934
|
+
async function resolveStorageState(recipe, deps, lease, captcha) {
|
|
3843
5935
|
const saved = await readSavedState(recipe, deps);
|
|
3844
5936
|
const session = recipe.session;
|
|
3845
5937
|
if (saved !== undefined || session?.bootstrap === undefined) return saved;
|
|
@@ -3851,7 +5943,7 @@ async function resolveStorageState(recipe, deps, lease) {
|
|
|
3851
5943
|
...accessOptions(lease, session.headers)
|
|
3852
5944
|
});
|
|
3853
5945
|
try {
|
|
3854
|
-
return await runBootstrap(recipe, browserSession, deps);
|
|
5946
|
+
return await runBootstrap(recipe, browserSession, deps, captcha);
|
|
3855
5947
|
} finally {
|
|
3856
5948
|
await browserSession.close();
|
|
3857
5949
|
}
|
|
@@ -3876,15 +5968,16 @@ async function readSavedState(recipe, deps) {
|
|
|
3876
5968
|
* @param recipe - An input recipe with `session.bootstrap`.
|
|
3877
5969
|
* @param browserSession - Where the steps run.
|
|
3878
5970
|
* @param deps - Hooks, events, `storageStateDir`.
|
|
5971
|
+
* @param captcha - Solves the bootstrap's captchas.
|
|
3879
5972
|
* @returns The kept state.
|
|
3880
5973
|
*/
|
|
3881
|
-
async function runBootstrap(recipe, browserSession, deps) {
|
|
5974
|
+
async function runBootstrap(recipe, browserSession, deps, captcha) {
|
|
3882
5975
|
const bootstrap = recipe.session?.bootstrap;
|
|
3883
5976
|
if (bootstrap === undefined) return {
|
|
3884
5977
|
cookies: [],
|
|
3885
5978
|
origins: []
|
|
3886
5979
|
};
|
|
3887
|
-
const runner = new WebStepRunner(browserSession, recipe, deps.events);
|
|
5980
|
+
const runner = new WebStepRunner(browserSession, recipe, deps.events, undefined, captcha);
|
|
3888
5981
|
const scope = new ExtractionScope();
|
|
3889
5982
|
scope.set('vars', recipe.vars ?? {});
|
|
3890
5983
|
scope.set('start', {
|
|
@@ -4051,9 +6144,15 @@ async function runInputRecipe(input, output, deps) {
|
|
|
4051
6144
|
rejected: 0,
|
|
4052
6145
|
duplicates: 0,
|
|
4053
6146
|
skipped: 0,
|
|
6147
|
+
stepsSkipped: 0,
|
|
4054
6148
|
pages: 0,
|
|
4055
6149
|
durationMs: 0
|
|
4056
6150
|
};
|
|
6151
|
+
const captchas = {
|
|
6152
|
+
detected: 0,
|
|
6153
|
+
solved: 0,
|
|
6154
|
+
failed: 0
|
|
6155
|
+
};
|
|
4057
6156
|
const limits = input.limits ?? {};
|
|
4058
6157
|
// A web recipe drives one page, so only api mode runs iterations in parallel.
|
|
4059
6158
|
const gate = new RunGate(input.mode === 'web' ? 1 : limits.concurrency ?? 1, limits.delayMs ?? 0);
|
|
@@ -4061,6 +6160,10 @@ async function runInputRecipe(input, output, deps) {
|
|
|
4061
6160
|
let chain = Promise.resolve();
|
|
4062
6161
|
const unsubscribe = deps.events.subscribe(event => {
|
|
4063
6162
|
if (event.type === 'page:visit' && event.recipeId === input.id) report.pages += 1;
|
|
6163
|
+
if (event.type === 'step:skip' && event.recipeId === input.id) report.stepsSkipped += 1;
|
|
6164
|
+
if (event.type === 'captcha:detected' && event.recipeId === input.id) captchas.detected += 1;
|
|
6165
|
+
if (event.type === 'captcha:solved' && event.recipeId === input.id) captchas.solved += 1;
|
|
6166
|
+
if (event.type === 'captcha:failed' && event.recipeId === input.id) captchas.failed += 1;
|
|
4064
6167
|
});
|
|
4065
6168
|
deps.events.emit({
|
|
4066
6169
|
type: 'recipe:start',
|
|
@@ -4071,11 +6174,18 @@ async function runInputRecipe(input, output, deps) {
|
|
|
4071
6174
|
let runner;
|
|
4072
6175
|
try {
|
|
4073
6176
|
const onBlock = input.session?.onBlock;
|
|
6177
|
+
const solvers = deps.captchaSolvers ?? new CaptchaSolverRegistry();
|
|
6178
|
+
for (const name of captchaSolverNames(input)) solvers.resolve(name);
|
|
6179
|
+
const context = {
|
|
6180
|
+
gate,
|
|
6181
|
+
solvers,
|
|
6182
|
+
budget: new CaptchaBudget(input.session?.captcha?.maxSolves ?? DEFAULT_MAX_SOLVES)
|
|
6183
|
+
};
|
|
4074
6184
|
runner = await RotatingRunner.open({
|
|
4075
6185
|
recipe: input,
|
|
4076
6186
|
events: deps.events,
|
|
4077
6187
|
maxRotations: onBlock?.rotate === true ? onBlock.attempts ?? 2 : 0,
|
|
4078
|
-
open: attempt => openLeased(input, deps,
|
|
6188
|
+
open: attempt => openLeased(input, deps, context, attempt)
|
|
4079
6189
|
});
|
|
4080
6190
|
for (const point of input.start) {
|
|
4081
6191
|
const scope = new ExtractionScope();
|
|
@@ -4110,6 +6220,7 @@ async function runInputRecipe(input, output, deps) {
|
|
|
4110
6220
|
} finally {
|
|
4111
6221
|
await runner?.dispose();
|
|
4112
6222
|
unsubscribe();
|
|
6223
|
+
if (captchas.detected > 0) report.captchas = captchas;
|
|
4113
6224
|
report.durationMs = Date.now() - started;
|
|
4114
6225
|
deps.events.emit({
|
|
4115
6226
|
type: 'recipe:finish',
|
|
@@ -4118,6 +6229,7 @@ async function runInputRecipe(input, output, deps) {
|
|
|
4118
6229
|
rejected: report.rejected,
|
|
4119
6230
|
duplicates: report.duplicates,
|
|
4120
6231
|
skipped: report.skipped,
|
|
6232
|
+
stepsSkipped: report.stepsSkipped,
|
|
4121
6233
|
pages: report.pages,
|
|
4122
6234
|
durationMs: report.durationMs,
|
|
4123
6235
|
error: report.error
|
|
@@ -4234,11 +6346,11 @@ async function leaseAccess(input, deps, attempt) {
|
|
|
4234
6346
|
return lease;
|
|
4235
6347
|
}
|
|
4236
6348
|
/** A lease and a runner opened on it; the lease is released again when opening fails. */
|
|
4237
|
-
async function openLeased(input, deps,
|
|
6349
|
+
async function openLeased(input, deps, context, attempt) {
|
|
4238
6350
|
const lease = await leaseAccess(input, deps, attempt);
|
|
4239
6351
|
try {
|
|
4240
6352
|
return {
|
|
4241
|
-
runner: await openRunner(input, deps,
|
|
6353
|
+
runner: await openRunner(input, deps, context, lease),
|
|
4242
6354
|
lease
|
|
4243
6355
|
};
|
|
4244
6356
|
} catch (error) {
|
|
@@ -4246,9 +6358,19 @@ async function openLeased(input, deps, gate, attempt) {
|
|
|
4246
6358
|
throw error;
|
|
4247
6359
|
}
|
|
4248
6360
|
}
|
|
4249
|
-
async function openRunner(input, deps,
|
|
4250
|
-
|
|
4251
|
-
|
|
6361
|
+
async function openRunner(input, deps, context, lease) {
|
|
6362
|
+
const {
|
|
6363
|
+
gate
|
|
6364
|
+
} = context;
|
|
6365
|
+
const captcha = new CaptchaGuard({
|
|
6366
|
+
recipe: input,
|
|
6367
|
+
events: deps.events,
|
|
6368
|
+
solvers: context.solvers,
|
|
6369
|
+
budget: context.budget,
|
|
6370
|
+
lease
|
|
6371
|
+
});
|
|
6372
|
+
if (lease.cdp !== undefined) return openRemoteRunner(input, deps, context, lease, lease.cdp);
|
|
6373
|
+
const storageState = await resolveStorageState(input, deps, lease, captcha);
|
|
4252
6374
|
const session = input.session;
|
|
4253
6375
|
const access = accessOptions(lease, session?.headers);
|
|
4254
6376
|
if (input.mode === 'web') {
|
|
@@ -4260,7 +6382,7 @@ async function openRunner(input, deps, gate, lease) {
|
|
|
4260
6382
|
viewport: session?.viewport,
|
|
4261
6383
|
...access
|
|
4262
6384
|
});
|
|
4263
|
-
return new WebStepRunner(browserSession, input, deps.events, gate);
|
|
6385
|
+
return new WebStepRunner(browserSession, input, deps.events, gate, captcha);
|
|
4264
6386
|
}
|
|
4265
6387
|
const client = await HttpClient.open({
|
|
4266
6388
|
storageState,
|
|
@@ -4277,7 +6399,14 @@ async function openRunner(input, deps, gate, lease) {
|
|
|
4277
6399
|
* session as the crawl: providers tie the IP and fingerprint to the
|
|
4278
6400
|
* connection, so a login in one connection would not carry to another.
|
|
4279
6401
|
*/
|
|
4280
|
-
async function openRemoteRunner(input, deps,
|
|
6402
|
+
async function openRemoteRunner(input, deps, context, lease, cdp) {
|
|
6403
|
+
const captcha = new CaptchaGuard({
|
|
6404
|
+
recipe: input,
|
|
6405
|
+
events: deps.events,
|
|
6406
|
+
solvers: context.solvers,
|
|
6407
|
+
budget: context.budget,
|
|
6408
|
+
lease
|
|
6409
|
+
});
|
|
4281
6410
|
if (input.mode === 'api') throw new AccessConfigError(`recipe "${input.id}" runs in api mode, but access profile "${lease.profile}" is a remote browser; api recipes need a proxy profile`);
|
|
4282
6411
|
const session = input.session;
|
|
4283
6412
|
const storageState = await readSavedState(input, deps);
|
|
@@ -4288,12 +6417,12 @@ async function openRemoteRunner(input, deps, gate, lease, cdp) {
|
|
|
4288
6417
|
...accessOptions(lease, session?.headers)
|
|
4289
6418
|
}, input.limits?.timeoutMs);
|
|
4290
6419
|
try {
|
|
4291
|
-
if (storageState === undefined && session?.bootstrap !== undefined) await runBootstrap(input, browserSession, deps);
|
|
6420
|
+
if (storageState === undefined && session?.bootstrap !== undefined) await runBootstrap(input, browserSession, deps, captcha);
|
|
4292
6421
|
} catch (error) {
|
|
4293
6422
|
await browserSession.close();
|
|
4294
6423
|
throw error;
|
|
4295
6424
|
}
|
|
4296
|
-
return new WebStepRunner(browserSession, input, deps.events, gate);
|
|
6425
|
+
return new WebStepRunner(browserSession, input, deps.events, context.gate, captcha);
|
|
4297
6426
|
}
|
|
4298
6427
|
|
|
4299
6428
|
/**
|
|
@@ -4331,15 +6460,17 @@ async function runCrawl(set, deps, onRecipeError) {
|
|
|
4331
6460
|
* Creates a crawler. The browser is launched lazily, on the first recipe or
|
|
4332
6461
|
* bootstrap that needs it, and shared by every run until `close`.
|
|
4333
6462
|
*
|
|
4334
|
-
* @param options - Hooks, sink, events, browser settings, access, policies.
|
|
6463
|
+
* @param options - Hooks, sink, events, browser settings, access, captcha solvers, policies.
|
|
4335
6464
|
* @returns The crawler.
|
|
4336
6465
|
* @throws AccessConfigError when the access config cannot work.
|
|
6466
|
+
* @throws Error when two captcha solvers share a name.
|
|
4337
6467
|
*/
|
|
4338
6468
|
function createCrawler(options = {}) {
|
|
4339
6469
|
const sink = options.sink ?? memorySink();
|
|
4340
6470
|
if (options.resume === true && sink.has === undefined) throw new Error('resume needs a sink that can tell which keys it has (jsonLinesSink with append, memorySink, or a custom sink with `has`)');
|
|
4341
6471
|
const access = new AccessBroker(options.access, options.accessPlugins);
|
|
4342
6472
|
const hooks = new HookRegistry(options.hooks);
|
|
6473
|
+
const captchaSolvers = new CaptchaSolverRegistry(options.captchaSolvers);
|
|
4343
6474
|
const events = new EventBus(options.onEvent);
|
|
4344
6475
|
let browser;
|
|
4345
6476
|
const launch = () => {
|
|
@@ -4357,6 +6488,7 @@ function createCrawler(options = {}) {
|
|
|
4357
6488
|
resume: options.resume === true,
|
|
4358
6489
|
debug: options.debug === true,
|
|
4359
6490
|
access,
|
|
6491
|
+
captchaSolvers,
|
|
4360
6492
|
ignoreHTTPSErrors: options.browser?.ignoreHTTPSErrors
|
|
4361
6493
|
}, options.onRecipeError ?? 'continue'),
|
|
4362
6494
|
async close() {
|
|
@@ -4371,10 +6503,13 @@ function createCrawler(options = {}) {
|
|
|
4371
6503
|
|
|
4372
6504
|
/** The closed vocabularies a recipe file can use. Each is a `readonly` tuple so zod and TypeScript share it. */
|
|
4373
6505
|
const CRAWL_MODES = ['web', 'api'];
|
|
4374
|
-
const SELECTOR_KINDS = ['css', 'xpath', 'jsonpath', 'regex'];
|
|
6506
|
+
const SELECTOR_KINDS = ['css', 'xpath', 'jsonpath', 'regex', 'table'];
|
|
4375
6507
|
/** `take` also accepts `attr:<name>`, which is validated by pattern rather than listed. */
|
|
4376
6508
|
const TAKE_KINDS = ['text', 'html', 'value', 'json'];
|
|
4377
|
-
const BODY_KINDS = ['json', 'html', 'text'];
|
|
6509
|
+
const BODY_KINDS = ['json', 'jsonl', 'html', 'text', 'pdf', 'csv', 'xlsx', 'pptx', 'yaml', 'markdown'];
|
|
6510
|
+
const YAML_SCALARS = ['typed', 'text'];
|
|
6511
|
+
/** How a PDF table aligns a row's values against a cell wrapped over several lines. */
|
|
6512
|
+
const TABLE_ALIGNS = ['auto', 'top', 'center', 'bottom'];
|
|
4378
6513
|
const FIELD_TYPES = ['string', 'number', 'integer', 'boolean', 'date', 'datetime', 'currency', 'url', 'enum', 'array', 'object', 'json'];
|
|
4379
6514
|
const MISSING_POLICIES = ['fail', 'skip-record', 'null', 'default'];
|
|
4380
6515
|
const RECIPE_MISSING_POLICIES = ['fail', 'skip-record', 'null'];
|
|
@@ -4384,7 +6519,7 @@ const KEEP_KINDS = ['cookies', 'localStorage'];
|
|
|
4384
6519
|
const WAIT_UNTIL = ['load', 'domcontentloaded', 'networkidle', 'commit'];
|
|
4385
6520
|
const HTTP_METHODS = ['GET', 'POST', 'PUT', 'PATCH', 'DELETE', 'HEAD'];
|
|
4386
6521
|
/** Steps that only make sense with a live browser page. */
|
|
4387
|
-
const WEB_ONLY_STEPS = ['goto', 'click', 'fill', 'press', 'select', 'scroll', 'wait', 'evaluate', 'screenshot'];
|
|
6522
|
+
const WEB_ONLY_STEPS = ['goto', 'click', 'fill', 'press', 'select', 'scroll', 'wait', 'evaluate', 'screenshot', 'captcha'];
|
|
4388
6523
|
/** Steps that only make sense against an HTTP request context. */
|
|
4389
6524
|
const API_ONLY_STEPS = ['request'];
|
|
4390
6525
|
|
|
@@ -4549,8 +6684,18 @@ const requestStep = z.strictObject({
|
|
|
4549
6684
|
query: stringMap.optional(),
|
|
4550
6685
|
headers: stringMap.optional(),
|
|
4551
6686
|
body: z.unknown().optional(),
|
|
4552
|
-
as: z.enum(BODY_KINDS).optional()
|
|
6687
|
+
as: z.enum(BODY_KINDS).optional(),
|
|
6688
|
+
encoding: z.string().min(1).optional(),
|
|
6689
|
+
delimiter: z.string().length(1).optional(),
|
|
6690
|
+
scalars: z.enum(YAML_SCALARS).optional()
|
|
6691
|
+
}).refine(step => step.delimiter === undefined || step.as === undefined || step.as === 'csv', {
|
|
6692
|
+
message: '"delimiter" reads CSV only: drop it or set "as": "csv"',
|
|
6693
|
+
path: ['delimiter']
|
|
6694
|
+
}).refine(step => step.scalars === undefined || step.as === undefined || step.as === 'yaml', {
|
|
6695
|
+
message: '"scalars" reads YAML only: drop it or set "as": "yaml"',
|
|
6696
|
+
path: ['scalars']
|
|
4553
6697
|
});
|
|
6698
|
+
const tableOnly = ['columns', 'until', 'align', 'sheet', 'headerRows', 'fillDown', 'includeHidden', 'slide', 'shapes'];
|
|
4554
6699
|
const extractStep = z.strictObject({
|
|
4555
6700
|
...base,
|
|
4556
6701
|
type: z.literal('extract'),
|
|
@@ -4558,7 +6703,26 @@ const extractStep = z.strictObject({
|
|
|
4558
6703
|
kind: z.enum(SELECTOR_KINDS),
|
|
4559
6704
|
take: takeKindSchema.optional(),
|
|
4560
6705
|
many: z.boolean().optional(),
|
|
4561
|
-
from: stepId.optional()
|
|
6706
|
+
from: stepId.optional(),
|
|
6707
|
+
columns: stringMap.optional(),
|
|
6708
|
+
until: z.string().min(1).optional(),
|
|
6709
|
+
align: z.enum(TABLE_ALIGNS).optional(),
|
|
6710
|
+
sheet: z.string().min(1).optional(),
|
|
6711
|
+
headerRows: z.int().min(1).optional(),
|
|
6712
|
+
fillDown: z.array(z.string().min(1)).min(1).optional(),
|
|
6713
|
+
includeHidden: z.boolean().optional(),
|
|
6714
|
+
slide: z.string().min(1).optional(),
|
|
6715
|
+
shapes: z.boolean().optional()
|
|
6716
|
+
}).check(context => {
|
|
6717
|
+
if (context.value.kind === 'table') return;
|
|
6718
|
+
for (const key of tableOnly) {
|
|
6719
|
+
if (context.value[key] !== undefined) context.issues.push({
|
|
6720
|
+
code: 'custom',
|
|
6721
|
+
input: context.value,
|
|
6722
|
+
path: [key],
|
|
6723
|
+
message: `"${key}" belongs to kind "table"`
|
|
6724
|
+
});
|
|
6725
|
+
}
|
|
4562
6726
|
});
|
|
4563
6727
|
const assignStep = z.strictObject({
|
|
4564
6728
|
...base,
|
|
@@ -4582,6 +6746,19 @@ const hookStep = z.strictObject({
|
|
|
4582
6746
|
name: z.string().min(1),
|
|
4583
6747
|
args: z.record(z.string(), z.unknown()).optional()
|
|
4584
6748
|
});
|
|
6749
|
+
const captchaCheckSchema = z.strictObject({
|
|
6750
|
+
gone: z.boolean().optional(),
|
|
6751
|
+
selector: z.string().min(1).optional()
|
|
6752
|
+
});
|
|
6753
|
+
const captchaStep = z.strictObject({
|
|
6754
|
+
...base,
|
|
6755
|
+
type: z.literal('captcha'),
|
|
6756
|
+
solver: z.string().min(1).optional(),
|
|
6757
|
+
selector: z.string().min(1).optional(),
|
|
6758
|
+
verify: captchaCheckSchema.optional(),
|
|
6759
|
+
attempts: z.int().min(1).max(10).optional(),
|
|
6760
|
+
timeoutMs: z.int().min(1000).optional()
|
|
6761
|
+
});
|
|
4585
6762
|
const emitFlag = z.union([z.literal(true), z.strictObject({
|
|
4586
6763
|
output: z.string().min(1)
|
|
4587
6764
|
})]);
|
|
@@ -4610,7 +6787,7 @@ const paginateStep = z.strictObject({
|
|
|
4610
6787
|
maxPages: z.int().min(1).optional(),
|
|
4611
6788
|
steps
|
|
4612
6789
|
});
|
|
4613
|
-
const stepSchema = z.discriminatedUnion('type', [gotoStep, clickStep, fillStep, pressStep, selectStep, scrollStep, waitStep, evaluateStep, screenshotStep, requestStep, extractStep, assignStep, collectStep, emitStep, hookStep, forEachStep, ifStep, paginateStep]);
|
|
6790
|
+
const stepSchema = z.discriminatedUnion('type', [gotoStep, clickStep, fillStep, pressStep, selectStep, scrollStep, waitStep, evaluateStep, screenshotStep, requestStep, extractStep, assignStep, collectStep, emitStep, hookStep, forEachStep, ifStep, paginateStep, captchaStep]);
|
|
4614
6791
|
|
|
4615
6792
|
const args = z.record(z.string(), z.unknown());
|
|
4616
6793
|
const stringList = z.array(z.string());
|
|
@@ -4758,9 +6935,20 @@ const blockRuleSchema = z.strictObject({
|
|
|
4758
6935
|
text: regexSource.optional()
|
|
4759
6936
|
});
|
|
4760
6937
|
const blockRotationSchema = z.strictObject({
|
|
4761
|
-
rotate: z.boolean(),
|
|
6938
|
+
rotate: z.boolean().optional(),
|
|
6939
|
+
solve: z.boolean().optional(),
|
|
4762
6940
|
attempts: z.int().min(1).max(10).optional()
|
|
4763
6941
|
});
|
|
6942
|
+
const captchaSettingsSchema = z.strictObject({
|
|
6943
|
+
solver: z.string().min(1),
|
|
6944
|
+
detect: z.strictObject({
|
|
6945
|
+
selector: z.string().min(1)
|
|
6946
|
+
}).optional(),
|
|
6947
|
+
verify: captchaCheckSchema.optional(),
|
|
6948
|
+
attempts: z.int().min(1).max(10).optional(),
|
|
6949
|
+
timeoutMs: z.int().min(1000).optional(),
|
|
6950
|
+
maxSolves: z.int().nonnegative().optional()
|
|
6951
|
+
});
|
|
4764
6952
|
const sessionSpecSchema = z.strictObject({
|
|
4765
6953
|
headers: z.record(z.string(), z.string()).optional(),
|
|
4766
6954
|
cookies: z.array(cookieSchema).optional(),
|
|
@@ -4773,7 +6961,8 @@ const sessionSpecSchema = z.strictObject({
|
|
|
4773
6961
|
bootstrap: bootstrapSchema.optional(),
|
|
4774
6962
|
access: sessionAccessSchema.optional(),
|
|
4775
6963
|
blockedWhen: blockRuleSchema.optional(),
|
|
4776
|
-
onBlock: blockRotationSchema.optional()
|
|
6964
|
+
onBlock: blockRotationSchema.optional(),
|
|
6965
|
+
captcha: captchaSettingsSchema.optional()
|
|
4777
6966
|
});
|
|
4778
6967
|
const limitsSchema = z.strictObject({
|
|
4779
6968
|
maxRecords: z.int().positive().optional(),
|
|
@@ -5076,7 +7265,9 @@ const API_ONLY = new Set(API_ONLY_STEPS);
|
|
|
5076
7265
|
* - web-only steps appear only in web recipes or inside a bootstrap, api-only
|
|
5077
7266
|
* steps only in api recipes, and `next.selector` only in web mode;
|
|
5078
7267
|
* - exactly one emitting construct exists on any path (the two branches of an
|
|
5079
|
-
* `if` are separate paths)
|
|
7268
|
+
* `if` are separate paths);
|
|
7269
|
+
* - a `captcha` step, and `onBlock.solve`, have a solver: their own or
|
|
7270
|
+
* `session.captcha.solver`.
|
|
5080
7271
|
*
|
|
5081
7272
|
* @param input - A parsed input recipe.
|
|
5082
7273
|
* @param output - The parsed output recipe it names.
|
|
@@ -5094,17 +7285,22 @@ function validateBinding(input, output) {
|
|
|
5094
7285
|
const known = new Set(RESERVED);
|
|
5095
7286
|
const varNames = [input.vars ?? {}, ...input.start.map(point => point.vars ?? {})].flatMap(record => Object.keys(record));
|
|
5096
7287
|
for (const name of varNames) known.add(name);
|
|
7288
|
+
const solver = input.session?.captcha !== undefined;
|
|
5097
7289
|
walkSteps(input.steps, 'steps', input.mode, known, report, {
|
|
5098
7290
|
emitting: false,
|
|
5099
|
-
ids: new Set()
|
|
7291
|
+
ids: new Set(),
|
|
7292
|
+
solver
|
|
5100
7293
|
});
|
|
5101
7294
|
if (input.session?.bootstrap !== undefined) {
|
|
5102
7295
|
walkSteps(input.session.bootstrap.steps, 'session.bootstrap.steps', 'web', new Set(RESERVED), report, {
|
|
5103
7296
|
emitting: false,
|
|
5104
7297
|
ids: new Set(),
|
|
5105
|
-
bootstrap: true
|
|
7298
|
+
bootstrap: true,
|
|
7299
|
+
solver
|
|
5106
7300
|
});
|
|
5107
7301
|
}
|
|
7302
|
+
if (!solver && input.session?.onBlock?.solve === true) report('session.onBlock.solve', 'solving a block needs a solver: add session.captcha');
|
|
7303
|
+
if (input.mode === 'api' && input.session?.onBlock?.solve === true) report('session.onBlock.solve', 'captchas are solved on a live page; this recipe runs in api mode (solve them in session.bootstrap)');
|
|
5108
7304
|
for (const [target, rule] of Object.entries(input.mapping)) {
|
|
5109
7305
|
const field = fieldAt(output.fields, target);
|
|
5110
7306
|
if (field === undefined) {
|
|
@@ -5136,6 +7332,11 @@ function walkStep(step, at, mode, known, report, state) {
|
|
|
5136
7332
|
known.add(step.id);
|
|
5137
7333
|
}
|
|
5138
7334
|
if (step.type === 'extract' && step.from !== undefined && !known.has(step.from)) report(`${at}.from`, `"${step.from}" is not a known id`);
|
|
7335
|
+
if (mode === 'web' && state.bootstrap !== true && step.type === 'extract' && step.kind === 'table' && step.from === undefined) {
|
|
7336
|
+
const foreign = ['sheet', 'slide', 'shapes', 'align', 'includeHidden'].filter(option => step[option] !== undefined);
|
|
7337
|
+
if (foreign.length > 0) report(at, `a "table" extract on a web page reads its HTML tables; ${foreign.map(option => `"${option}"`).join(', ')} belong to PDFs, workbooks or decks (fetch one with a request step in an api recipe)`);
|
|
7338
|
+
}
|
|
7339
|
+
if (step.type === 'captcha' && step.solver === undefined && !state.solver) report(at, 'a captcha step needs a solver: name one ("solver") or add session.captcha');
|
|
5139
7340
|
if (step.type === 'collect' && !known.has(step.into)) report(`${at}.into`, `"${step.into}" is not a known id: set it to [] before the loop that collects into it`);
|
|
5140
7341
|
const nested = () => ({
|
|
5141
7342
|
...state,
|
|
@@ -5298,5 +7499,5 @@ function isSameOutput(document, output) {
|
|
|
5298
7499
|
return recipeKindOf(document.content) === 'output' && document.content.id === output.id;
|
|
5299
7500
|
}
|
|
5300
7501
|
|
|
5301
|
-
export { ACCESS_PRESETS, AccessBroker, AccessConfigError, BrowserClient, BrowserSession, HttpClient, HttpError, MappingFailedError, RecipeBindingError, RecipeSet, RecipeValidationError, RecordRejectedError, StepFailure, TransformError, UnknownHookError, accessConfigJsonSchema, accessConfigSchema, bindRecipeSet, createCrawler, inputRecipeJsonSchema, inputRecipeSchema, jsonLinesSink, loadAccessConfig, loadRecipeSet, loadRecipes, memorySink, outputRecipeJsonSchema, outputRecipeSchema, parseInputRecipe, parseOutputRecipe, readRecipeSource, traceLine, tryParseJson, validateBinding };
|
|
7502
|
+
export { ACCESS_PRESETS, AccessBroker, AccessConfigError, BrowserClient, BrowserSession, CaptchaError, DEFAULT_CAPTCHA_SELECTOR, HttpClient, HttpError, MappingFailedError, PdfReadError, RecipeBindingError, RecipeSet, RecipeValidationError, RecordRejectedError, StepFailure, TransformError, UnknownHookError, accessConfigJsonSchema, accessConfigSchema, bindRecipeSet, createCrawler, csvWorkbook, deckText, detectChallenge, detectDelimiter, fillDown, findDeckTables, findGridTables, findTables, htmlTableSheets, inputRecipeJsonSchema, inputRecipeSchema, isDeckDocument, isWorkbookDocument, jsonLinesSink, loadAccessConfig, loadRecipeSet, loadRecipes, memorySink, outputRecipeJsonSchema, outputRecipeSchema, parseCsv, parseInputRecipe, parseOutputRecipe, pdfText, readMarkdown, readPdf, readRecipeSource, readYaml, traceLine, tryParseJson, validateBinding, workbookText };
|
|
5302
7503
|
//# sourceMappingURL=index.esm.js.map
|