@nitpicker/crawler 0.11.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/archive/archive.d.ts +117 -2
- package/lib/archive/archive.js +147 -2
- package/lib/archive/cache/compute-archive-cache-key.d.ts +39 -0
- package/lib/archive/cache/compute-archive-cache-key.js +95 -0
- package/lib/archive/cache/extract-archive-to-cache.d.ts +43 -0
- package/lib/archive/cache/extract-archive-to-cache.js +309 -0
- package/lib/archive/cache/get-archive-cache-root.d.ts +20 -0
- package/lib/archive/cache/get-archive-cache-root.js +53 -0
- package/lib/archive/cache/is-archive-cache-disabled.d.ts +24 -0
- package/lib/archive/cache/is-archive-cache-disabled.js +34 -0
- package/lib/archive/cache/resolve-archive-cache-dir.d.ts +26 -0
- package/lib/archive/cache/resolve-archive-cache-dir.js +32 -0
- package/lib/archive/database.d.ts +216 -15
- package/lib/archive/database.js +1459 -938
- package/lib/archive/derive-lineage-from-parent.d.ts +37 -0
- package/lib/archive/derive-lineage-from-parent.js +42 -0
- package/lib/archive/get-failed-page-messages.d.ts +43 -0
- package/lib/archive/get-failed-page-messages.js +131 -0
- package/lib/archive/init-schema.js +153 -1
- package/lib/archive/is-inventory-source.d.ts +21 -0
- package/lib/archive/is-inventory-source.js +22 -0
- package/lib/archive/migrate-inventory-runs.d.ts +29 -0
- package/lib/archive/migrate-inventory-runs.js +52 -0
- package/lib/archive/types.d.ts +33 -0
- package/lib/classify-error-kind.d.ts +19 -0
- package/lib/classify-error-kind.js +122 -0
- package/lib/crawler/build-js-redirect-edge.d.ts +68 -0
- package/lib/crawler/build-js-redirect-edge.js +57 -0
- package/lib/crawler/build-redirect-event.d.ts +24 -0
- package/lib/crawler/build-redirect-event.js +28 -0
- package/lib/crawler/clear-dns-burned-host-cache.d.ts +6 -0
- package/lib/crawler/clear-dns-burned-host-cache.js +11 -0
- package/lib/crawler/crawler.d.ts +3 -1
- package/lib/crawler/crawler.js +655 -107
- package/lib/crawler/derive-js-redirect-target.d.ts +68 -0
- package/lib/crawler/derive-js-redirect-target.js +129 -0
- package/lib/crawler/derive-resource-source.d.ts +25 -15
- package/lib/crawler/derive-resource-source.js +28 -17
- package/lib/crawler/dns-burned-host-cache.d.ts +26 -0
- package/lib/crawler/dns-burned-host-cache.js +25 -0
- package/lib/crawler/dns-burned-host-short-circuit-counter.d.ts +13 -0
- package/lib/crawler/dns-burned-host-short-circuit-counter.js +11 -0
- package/lib/crawler/fetch-destination.d.ts +12 -4
- package/lib/crawler/fetch-destination.js +94 -16
- package/lib/crawler/is-js-redirect-error-shape.d.ts +40 -0
- package/lib/crawler/is-js-redirect-error-shape.js +53 -0
- package/lib/crawler/is-puppeteer-fallback-candidate.d.ts +16 -0
- package/lib/crawler/is-puppeteer-fallback-candidate.js +63 -0
- package/lib/crawler/link-list.d.ts +21 -1
- package/lib/crawler/link-list.js +23 -3
- package/lib/crawler/plan-sub-resource-emits.d.ts +63 -0
- package/lib/crawler/plan-sub-resource-emits.js +44 -0
- package/lib/crawler/preload-short-circuit-error.d.ts +22 -0
- package/lib/crawler/preload-short-circuit-error.js +25 -0
- package/lib/crawler/should-burn-host.d.ts +78 -0
- package/lib/crawler/should-burn-host.js +61 -0
- package/lib/crawler/should-get-fallback-on-head-failure.d.ts +38 -0
- package/lib/crawler/should-get-fallback-on-head-failure.js +46 -0
- package/lib/crawler/types.d.ts +107 -0
- package/lib/crawler-orchestrator.d.ts +13 -3
- package/lib/crawler-orchestrator.js +292 -69
- package/lib/crawler.d.ts +3 -2
- package/lib/crawler.js +3 -1
- package/lib/permanent-error-kinds.d.ts +43 -0
- package/lib/permanent-error-kinds.js +48 -0
- package/lib/types.d.ts +84 -0
- package/lib/utils/compute-file-sha256.d.ts +23 -0
- package/lib/utils/compute-file-sha256.js +55 -0
- package/lib/utils/error/emit-error-with-retry.d.ts +40 -0
- package/lib/utils/error/emit-error-with-retry.js +44 -0
- package/lib/utils/error/emit-error.d.ts +39 -0
- package/lib/utils/error/emit-error.js +41 -0
- package/package.json +11 -11
- package/lib/utils/error/error-emitter.d.ts +0 -18
- package/lib/utils/error/error-emitter.js +0 -29
package/lib/archive/database.js
CHANGED
|
@@ -1,53 +1,24 @@
|
|
|
1
|
-
var __runInitializers = (this && this.__runInitializers) || function (thisArg, initializers, value) {
|
|
2
|
-
var useValue = arguments.length > 2;
|
|
3
|
-
for (var i = 0; i < initializers.length; i++) {
|
|
4
|
-
value = useValue ? initializers[i].call(thisArg, value) : initializers[i].call(thisArg);
|
|
5
|
-
}
|
|
6
|
-
return useValue ? value : void 0;
|
|
7
|
-
};
|
|
8
|
-
var __esDecorate = (this && this.__esDecorate) || function (ctor, descriptorIn, decorators, contextIn, initializers, extraInitializers) {
|
|
9
|
-
function accept(f) { if (f !== void 0 && typeof f !== "function") throw new TypeError("Function expected"); return f; }
|
|
10
|
-
var kind = contextIn.kind, key = kind === "getter" ? "get" : kind === "setter" ? "set" : "value";
|
|
11
|
-
var target = !descriptorIn && ctor ? contextIn["static"] ? ctor : ctor.prototype : null;
|
|
12
|
-
var descriptor = descriptorIn || (target ? Object.getOwnPropertyDescriptor(target, contextIn.name) : {});
|
|
13
|
-
var _, done = false;
|
|
14
|
-
for (var i = decorators.length - 1; i >= 0; i--) {
|
|
15
|
-
var context = {};
|
|
16
|
-
for (var p in contextIn) context[p] = p === "access" ? {} : contextIn[p];
|
|
17
|
-
for (var p in contextIn.access) context.access[p] = contextIn.access[p];
|
|
18
|
-
context.addInitializer = function (f) { if (done) throw new TypeError("Cannot add initializers after decoration has completed"); extraInitializers.push(accept(f || null)); };
|
|
19
|
-
var result = (0, decorators[i])(kind === "accessor" ? { get: descriptor.get, set: descriptor.set } : descriptor[key], context);
|
|
20
|
-
if (kind === "accessor") {
|
|
21
|
-
if (result === void 0) continue;
|
|
22
|
-
if (result === null || typeof result !== "object") throw new TypeError("Object expected");
|
|
23
|
-
if (_ = accept(result.get)) descriptor.get = _;
|
|
24
|
-
if (_ = accept(result.set)) descriptor.set = _;
|
|
25
|
-
if (_ = accept(result.init)) initializers.unshift(_);
|
|
26
|
-
}
|
|
27
|
-
else if (_ = accept(result)) {
|
|
28
|
-
if (kind === "field") initializers.unshift(_);
|
|
29
|
-
else descriptor[key] = _;
|
|
30
|
-
}
|
|
31
|
-
}
|
|
32
|
-
if (target) Object.defineProperty(target, contextIn.name, descriptor);
|
|
33
|
-
done = true;
|
|
34
|
-
};
|
|
35
1
|
import { createHash } from 'node:crypto';
|
|
36
2
|
import { existsSync } from 'node:fs';
|
|
37
3
|
import path from 'node:path';
|
|
38
4
|
import { zstdCompressSync, zstdDecompressSync } from 'node:zlib';
|
|
39
5
|
import { tryParseUrl as parseUrl } from '@d-zero/shared/parse-url';
|
|
40
|
-
import {
|
|
6
|
+
import { retryCall } from '@d-zero/shared/retry';
|
|
41
7
|
import { pathComparator } from '@d-zero/shared/sort/path';
|
|
42
8
|
import { TypedAwaitEventEmitter as EventEmitter } from '@d-zero/shared/typed-await-event-emitter';
|
|
43
9
|
import knex from 'knex';
|
|
10
|
+
import { classifyErrorKind } from '../classify-error-kind.js';
|
|
44
11
|
import { findScopeEntry } from '../crawler/find-scope-entry.js';
|
|
45
12
|
import { isHtmlContentType } from '../crawler/is-html-content-type.js';
|
|
46
13
|
import { normalizeContentType } from '../crawler/normalize-content-type.js';
|
|
14
|
+
import { PERMANENT_ERROR_KINDS } from '../permanent-error-kinds.js';
|
|
47
15
|
import { eachSplitted } from '../utils/array/each-splitted.js';
|
|
48
|
-
import {
|
|
16
|
+
import { emitErrorAndRetry } from '../utils/error/emit-error-with-retry.js';
|
|
17
|
+
import { emitError } from '../utils/error/emit-error.js';
|
|
49
18
|
import { dbLog } from './debug.js';
|
|
19
|
+
import { deriveLineageFromParent } from './derive-lineage-from-parent.js';
|
|
50
20
|
import { mkdir } from './filesystem/mkdir.js';
|
|
21
|
+
import { getFailedPageMessages } from './get-failed-page-messages.js';
|
|
51
22
|
import { getJSON } from './get-json.js';
|
|
52
23
|
import { applyConnectionPragmas, initSchema } from './init-schema.js';
|
|
53
24
|
import { LibsqlDialect } from './libsql-dialect.js';
|
|
@@ -61,6 +32,7 @@ import { extractTagsForArchive } from './meta/extract-tags-for-archive.js';
|
|
|
61
32
|
import { migrateCrawlErrors } from './migrate-crawl-errors.js';
|
|
62
33
|
import { migrateHtmlBlobTables } from './migrate-html-blob-tables.js';
|
|
63
34
|
import { migrateInfoRoots } from './migrate-info-roots.js';
|
|
35
|
+
import { migrateInventoryRuns } from './migrate-inventory-runs.js';
|
|
64
36
|
import { migratePageErrors } from './migrate-page-errors.js';
|
|
65
37
|
import { migratePagesResourcesSource } from './migrate-pages-resources-source.js';
|
|
66
38
|
import { redirectTable } from './redirect-table.js';
|
|
@@ -240,195 +212,122 @@ function makeMetaResetPayload() {
|
|
|
240
212
|
/**
|
|
241
213
|
* Low-level database abstraction layer for the archive's SQLite database.
|
|
242
214
|
*
|
|
243
|
-
* Public methods that perform database queries use the
|
|
244
|
-
*
|
|
245
|
-
*
|
|
215
|
+
* Public methods that perform database queries use the `emitErrorAndRetry`
|
|
216
|
+
* HOF for automatic retry on transient failures combined with error-event
|
|
217
|
+
* propagation, or `emitError` when retry is not appropriate. The set of
|
|
218
|
+
* tables this layer manages is
|
|
246
219
|
* defined by `init-schema.ts` (the source of truth — query that file for
|
|
247
220
|
* the canonical list).
|
|
248
221
|
*
|
|
222
|
+
* **Label sync caveat**: each `emitError` / `emitErrorAndRetry` call passes
|
|
223
|
+
* the method name as a string literal (e.g. `'Database.getAnchorsOnPage'`).
|
|
224
|
+
* TypeScript cannot check that the string matches the enclosing method's
|
|
225
|
+
* real name — the two-way sync is manual. Renaming a method here **must**
|
|
226
|
+
* update the literal string too, otherwise debug logs and `RetryTimeoutError`
|
|
227
|
+
* messages will silently report the old name.
|
|
228
|
+
*
|
|
249
229
|
* Use the static {@link Database.connect} factory method to create instances.
|
|
250
230
|
* The constructor is private.
|
|
251
231
|
*/
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
_getExistingPageUrls_decorators = [ErrorEmitter()];
|
|
294
|
-
_getExistingResourceUrls_decorators = [ErrorEmitter()];
|
|
295
|
-
_getHtmlOfPageById_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
296
|
-
_getJsonLdOfPage_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
297
|
-
_getName_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
298
|
-
_getPageCount_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
299
|
-
_getPages_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
300
|
-
_getPagesWithRels_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
301
|
-
_getRedirectsForPages_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
302
|
-
_getReferrersOfPage_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
303
|
-
_getReferrersOfResource_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
304
|
-
_getResourceByUrl_decorators = [retry(retrySetting)];
|
|
305
|
-
_getResources_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
306
|
-
_getResourceUrlList_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
307
|
-
_getScrapedHtmlPageCount_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
308
|
-
_getTagsOfPage_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
309
|
-
_insertCrawlError_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
310
|
-
_insertPageError_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
311
|
-
_insertResource_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
312
|
-
_insertResourceReferrers_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
313
|
-
_recordRedirect_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
314
|
-
_repromoteExternalPages_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
315
|
-
_resetFailedPages_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
316
|
-
_setConfig_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
317
|
-
_setSkippedPage_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
318
|
-
_updateConfig_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
319
|
-
_updatePage_decorators = [ErrorEmitter(), retry(retrySetting)];
|
|
320
|
-
__esDecorate(this, null, _getAnchorsOnPage_decorators, { kind: "method", name: "getAnchorsOnPage", static: false, private: false, access: { has: obj => "getAnchorsOnPage" in obj, get: obj => obj.getAnchorsOnPage }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
321
|
-
__esDecorate(this, null, _getBaseUrl_decorators, { kind: "method", name: "getBaseUrl", static: false, private: false, access: { has: obj => "getBaseUrl" in obj, get: obj => obj.getBaseUrl }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
322
|
-
__esDecorate(this, null, _getConfig_decorators, { kind: "method", name: "getConfig", static: false, private: false, access: { has: obj => "getConfig" in obj, get: obj => obj.getConfig }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
323
|
-
__esDecorate(this, null, _getCrawlingState_decorators, { kind: "method", name: "getCrawlingState", static: false, private: false, access: { has: obj => "getCrawlingState" in obj, get: obj => obj.getCrawlingState }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
324
|
-
__esDecorate(this, null, _getExistingPageUrls_decorators, { kind: "method", name: "getExistingPageUrls", static: false, private: false, access: { has: obj => "getExistingPageUrls" in obj, get: obj => obj.getExistingPageUrls }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
325
|
-
__esDecorate(this, null, _getExistingResourceUrls_decorators, { kind: "method", name: "getExistingResourceUrls", static: false, private: false, access: { has: obj => "getExistingResourceUrls" in obj, get: obj => obj.getExistingResourceUrls }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
326
|
-
__esDecorate(this, null, _getHtmlOfPageById_decorators, { kind: "method", name: "getHtmlOfPageById", static: false, private: false, access: { has: obj => "getHtmlOfPageById" in obj, get: obj => obj.getHtmlOfPageById }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
327
|
-
__esDecorate(this, null, _getJsonLdOfPage_decorators, { kind: "method", name: "getJsonLdOfPage", static: false, private: false, access: { has: obj => "getJsonLdOfPage" in obj, get: obj => obj.getJsonLdOfPage }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
328
|
-
__esDecorate(this, null, _getName_decorators, { kind: "method", name: "getName", static: false, private: false, access: { has: obj => "getName" in obj, get: obj => obj.getName }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
329
|
-
__esDecorate(this, null, _getPageCount_decorators, { kind: "method", name: "getPageCount", static: false, private: false, access: { has: obj => "getPageCount" in obj, get: obj => obj.getPageCount }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
330
|
-
__esDecorate(this, null, _getPages_decorators, { kind: "method", name: "getPages", static: false, private: false, access: { has: obj => "getPages" in obj, get: obj => obj.getPages }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
331
|
-
__esDecorate(this, null, _getPagesWithRels_decorators, { kind: "method", name: "getPagesWithRels", static: false, private: false, access: { has: obj => "getPagesWithRels" in obj, get: obj => obj.getPagesWithRels }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
332
|
-
__esDecorate(this, null, _getRedirectsForPages_decorators, { kind: "method", name: "getRedirectsForPages", static: false, private: false, access: { has: obj => "getRedirectsForPages" in obj, get: obj => obj.getRedirectsForPages }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
333
|
-
__esDecorate(this, null, _getReferrersOfPage_decorators, { kind: "method", name: "getReferrersOfPage", static: false, private: false, access: { has: obj => "getReferrersOfPage" in obj, get: obj => obj.getReferrersOfPage }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
334
|
-
__esDecorate(this, null, _getReferrersOfResource_decorators, { kind: "method", name: "getReferrersOfResource", static: false, private: false, access: { has: obj => "getReferrersOfResource" in obj, get: obj => obj.getReferrersOfResource }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
335
|
-
__esDecorate(this, null, _getResourceByUrl_decorators, { kind: "method", name: "getResourceByUrl", static: false, private: false, access: { has: obj => "getResourceByUrl" in obj, get: obj => obj.getResourceByUrl }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
336
|
-
__esDecorate(this, null, _getResources_decorators, { kind: "method", name: "getResources", static: false, private: false, access: { has: obj => "getResources" in obj, get: obj => obj.getResources }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
337
|
-
__esDecorate(this, null, _getResourceUrlList_decorators, { kind: "method", name: "getResourceUrlList", static: false, private: false, access: { has: obj => "getResourceUrlList" in obj, get: obj => obj.getResourceUrlList }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
338
|
-
__esDecorate(this, null, _getScrapedHtmlPageCount_decorators, { kind: "method", name: "getScrapedHtmlPageCount", static: false, private: false, access: { has: obj => "getScrapedHtmlPageCount" in obj, get: obj => obj.getScrapedHtmlPageCount }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
339
|
-
__esDecorate(this, null, _getTagsOfPage_decorators, { kind: "method", name: "getTagsOfPage", static: false, private: false, access: { has: obj => "getTagsOfPage" in obj, get: obj => obj.getTagsOfPage }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
340
|
-
__esDecorate(this, null, _insertCrawlError_decorators, { kind: "method", name: "insertCrawlError", static: false, private: false, access: { has: obj => "insertCrawlError" in obj, get: obj => obj.insertCrawlError }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
341
|
-
__esDecorate(this, null, _insertPageError_decorators, { kind: "method", name: "insertPageError", static: false, private: false, access: { has: obj => "insertPageError" in obj, get: obj => obj.insertPageError }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
342
|
-
__esDecorate(this, null, _insertResource_decorators, { kind: "method", name: "insertResource", static: false, private: false, access: { has: obj => "insertResource" in obj, get: obj => obj.insertResource }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
343
|
-
__esDecorate(this, null, _insertResourceReferrers_decorators, { kind: "method", name: "insertResourceReferrers", static: false, private: false, access: { has: obj => "insertResourceReferrers" in obj, get: obj => obj.insertResourceReferrers }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
344
|
-
__esDecorate(this, null, _recordRedirect_decorators, { kind: "method", name: "recordRedirect", static: false, private: false, access: { has: obj => "recordRedirect" in obj, get: obj => obj.recordRedirect }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
345
|
-
__esDecorate(this, null, _repromoteExternalPages_decorators, { kind: "method", name: "repromoteExternalPages", static: false, private: false, access: { has: obj => "repromoteExternalPages" in obj, get: obj => obj.repromoteExternalPages }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
346
|
-
__esDecorate(this, null, _resetFailedPages_decorators, { kind: "method", name: "resetFailedPages", static: false, private: false, access: { has: obj => "resetFailedPages" in obj, get: obj => obj.resetFailedPages }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
347
|
-
__esDecorate(this, null, _setConfig_decorators, { kind: "method", name: "setConfig", static: false, private: false, access: { has: obj => "setConfig" in obj, get: obj => obj.setConfig }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
348
|
-
__esDecorate(this, null, _setSkippedPage_decorators, { kind: "method", name: "setSkippedPage", static: false, private: false, access: { has: obj => "setSkippedPage" in obj, get: obj => obj.setSkippedPage }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
349
|
-
__esDecorate(this, null, _updateConfig_decorators, { kind: "method", name: "updateConfig", static: false, private: false, access: { has: obj => "updateConfig" in obj, get: obj => obj.updateConfig }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
350
|
-
__esDecorate(this, null, _updatePage_decorators, { kind: "method", name: "updatePage", static: false, private: false, access: { has: obj => "updatePage" in obj, get: obj => obj.updatePage }, metadata: _metadata }, null, _instanceExtraInitializers);
|
|
351
|
-
if (_metadata) Object.defineProperty(this, Symbol.metadata, { enumerable: true, configurable: true, writable: true, value: _metadata });
|
|
352
|
-
}
|
|
353
|
-
/** The Knex query builder instance connected to the SQLite database. */
|
|
354
|
-
#instance = __runInitializers(this, _instanceExtraInitializers);
|
|
355
|
-
// eslint-disable-next-line no-restricted-syntax
|
|
356
|
-
constructor(options) {
|
|
357
|
-
super();
|
|
358
|
-
this.#instance = knex({
|
|
359
|
-
client: LibsqlDialect,
|
|
360
|
-
connection: {
|
|
361
|
-
filename: options.filename,
|
|
362
|
-
},
|
|
363
|
-
useNullAsDefault: true,
|
|
364
|
-
pool: {
|
|
365
|
-
acquireTimeoutMillis: 600_000,
|
|
366
|
-
},
|
|
367
|
-
});
|
|
368
|
-
}
|
|
369
|
-
/**
|
|
370
|
-
* Adds the `order` column to the `pages` table for URL sort ordering.
|
|
371
|
-
* If the column already exists, this method does nothing.
|
|
372
|
-
* @deprecated Since v0.1.x. The column is now created during table initialization.
|
|
373
|
-
* @returns The result of the schema alteration, or void if the column already exists.
|
|
374
|
-
*/
|
|
375
|
-
async addOrderField() {
|
|
376
|
-
const hasColumn = await this.#instance.schema.hasColumn('pages', 'order');
|
|
377
|
-
if (hasColumn) {
|
|
378
|
-
return;
|
|
379
|
-
}
|
|
380
|
-
return await this.#instance.schema.table('pages', (t) => {
|
|
381
|
-
t.integer('order').unsigned().nullable().defaultTo(null);
|
|
382
|
-
});
|
|
383
|
-
}
|
|
384
|
-
/**
|
|
385
|
-
* Forces a WAL checkpoint, writing all pending WAL data back to the main database file.
|
|
386
|
-
* Uses TRUNCATE mode to reset the WAL file to zero bytes after checkpointing.
|
|
387
|
-
* This ensures the database is fully self-contained in `db.sqlite` before archiving.
|
|
388
|
-
*/
|
|
389
|
-
async checkpoint() {
|
|
390
|
-
await this.#instance.raw('PRAGMA wal_checkpoint(TRUNCATE)');
|
|
391
|
-
}
|
|
392
|
-
/**
|
|
393
|
-
* Destroys the database connection, releasing all pooled resources.
|
|
394
|
-
*/
|
|
395
|
-
async destroy() {
|
|
396
|
-
await this.#instance.destroy();
|
|
232
|
+
export class Database extends EventEmitter {
|
|
233
|
+
/** The Knex query builder instance connected to the SQLite database. */
|
|
234
|
+
#instance;
|
|
235
|
+
// eslint-disable-next-line no-restricted-syntax
|
|
236
|
+
constructor(options) {
|
|
237
|
+
super();
|
|
238
|
+
// **Known caveat (libsql 0.5.x)**: passing `readonly: true` via
|
|
239
|
+
// `connection.options` is accepted by the libsql driver but is
|
|
240
|
+
// NOT enforced at the SQL layer — `CREATE TABLE` / `INSERT`
|
|
241
|
+
// against the resulting connection still succeed. The flag
|
|
242
|
+
// remains a no-op until libsql adds real read-only enforcement
|
|
243
|
+
// upstream. Read-only safety in cache mode therefore relies on:
|
|
244
|
+
//
|
|
245
|
+
// 1. `Database.#init` skipping schema init + migrations when
|
|
246
|
+
// `readOnly` is set (so no `initSchema` / `migrate*` ever
|
|
247
|
+
// writes to the shared cache directory).
|
|
248
|
+
// 2. `ArchiveAccessor.setData` rejecting writes when the
|
|
249
|
+
// `readOnly` flag is set on the accessor.
|
|
250
|
+
// 3. Code review on any future internal use of
|
|
251
|
+
// `accessor.getKnex()` — there is no driver-level guard.
|
|
252
|
+
this.#instance = knex({
|
|
253
|
+
client: LibsqlDialect,
|
|
254
|
+
connection: {
|
|
255
|
+
filename: options.filename,
|
|
256
|
+
},
|
|
257
|
+
useNullAsDefault: true,
|
|
258
|
+
pool: {
|
|
259
|
+
acquireTimeoutMillis: 600_000,
|
|
260
|
+
},
|
|
261
|
+
});
|
|
262
|
+
}
|
|
263
|
+
/**
|
|
264
|
+
* Adds the `order` column to the `pages` table for URL sort ordering.
|
|
265
|
+
* If the column already exists, this method does nothing.
|
|
266
|
+
* @deprecated Since v0.1.x. The column is now created during table initialization.
|
|
267
|
+
* @returns The result of the schema alteration, or void if the column already exists.
|
|
268
|
+
*/
|
|
269
|
+
async addOrderField() {
|
|
270
|
+
const hasColumn = await this.#instance.schema.hasColumn('pages', 'order');
|
|
271
|
+
if (hasColumn) {
|
|
272
|
+
return;
|
|
397
273
|
}
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
274
|
+
return await this.#instance.schema.table('pages', (t) => {
|
|
275
|
+
t.integer('order').unsigned().nullable().defaultTo(null);
|
|
276
|
+
});
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* Forces a WAL checkpoint, writing all pending WAL data back to the main database file.
|
|
280
|
+
* Uses TRUNCATE mode to reset the WAL file to zero bytes after checkpointing.
|
|
281
|
+
* This ensures the database is fully self-contained in `db.sqlite` before archiving.
|
|
282
|
+
*/
|
|
283
|
+
async checkpoint() {
|
|
284
|
+
await this.#instance.raw('PRAGMA wal_checkpoint(TRUNCATE)');
|
|
285
|
+
}
|
|
286
|
+
/**
|
|
287
|
+
* Destroys the database connection, releasing all pooled resources.
|
|
288
|
+
*/
|
|
289
|
+
async destroy() {
|
|
290
|
+
await this.#instance.destroy();
|
|
291
|
+
}
|
|
292
|
+
/**
|
|
293
|
+
* Retrieves all anchors (outgoing links) on a specific page.
|
|
294
|
+
* Joins the `anchors` table with the `pages` table to resolve link destinations.
|
|
295
|
+
* @param pageId - The database ID of the page whose anchors to retrieve.
|
|
296
|
+
* @returns An array of anchor records with resolved URL, title, status, and content type.
|
|
297
|
+
*/
|
|
298
|
+
async getAnchorsOnPage(pageId) {
|
|
299
|
+
return emitErrorAndRetry(this, 'Database.getAnchorsOnPage', async () => {
|
|
405
300
|
const res = await this.#instance
|
|
406
301
|
.select('pages.url', 'pages.title', 'pages.status', 'pages.statusText', 'pages.contentType', 'anchors.hash', 'anchors.textContent')
|
|
407
302
|
.from('anchors')
|
|
408
303
|
.join('pages', 'anchors.hrefId', '=', 'pages.id')
|
|
409
304
|
.where('anchors.pageId', pageId);
|
|
410
305
|
return res;
|
|
411
|
-
}
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
306
|
+
}, retrySetting);
|
|
307
|
+
}
|
|
308
|
+
/**
|
|
309
|
+
* Retrieves the base URL of the crawl session from the `info` table.
|
|
310
|
+
* @returns The base URL string.
|
|
311
|
+
* @throws {Error} If no base URL is found in the database.
|
|
312
|
+
*/
|
|
313
|
+
async getBaseUrl() {
|
|
314
|
+
return emitErrorAndRetry(this, 'Database.getBaseUrl', async () => {
|
|
418
315
|
const selected = await this.#instance.select('baseUrl').from('info');
|
|
419
316
|
if (!selected[0]) {
|
|
420
317
|
throw new Error('No baseUrl');
|
|
421
318
|
}
|
|
422
319
|
const [{ baseUrl }] = selected;
|
|
423
320
|
return baseUrl || '';
|
|
424
|
-
}
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
321
|
+
}, retrySetting);
|
|
322
|
+
}
|
|
323
|
+
/**
|
|
324
|
+
* Retrieves the full crawl configuration from the `info` table.
|
|
325
|
+
* Deserializes JSON-encoded fields (`roots`, `excludes`, `excludeKeywords`, `excludeUrls`).
|
|
326
|
+
* @returns The parsed {@link Config} object.
|
|
327
|
+
* @throws {Error} If no configuration is found in the database.
|
|
328
|
+
*/
|
|
329
|
+
async getConfig() {
|
|
330
|
+
return emitErrorAndRetry(this, 'Database.getConfig', async () => {
|
|
432
331
|
const [config] = await this.#instance.select('*').from('info');
|
|
433
332
|
if (!config) {
|
|
434
333
|
throw new Error('No config');
|
|
@@ -445,12 +344,74 @@ let Database = (() => {
|
|
|
445
344
|
delete opt.id;
|
|
446
345
|
dbLog('Table `info`: %O => %O', config, opt);
|
|
447
346
|
return opt;
|
|
448
|
-
}
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
347
|
+
}, retrySetting);
|
|
348
|
+
}
|
|
349
|
+
/**
|
|
350
|
+
* Retrieves the current crawling state by listing scraped and pending URLs.
|
|
351
|
+
*
|
|
352
|
+
* `scraped` is straightforward: every page row whose `scraped` flag is `1`
|
|
353
|
+
* — that is, every URL the crawl reached a terminal state on, including
|
|
354
|
+
* setSkippedPage / setExternalPage / outright setPage success or failure.
|
|
355
|
+
*
|
|
356
|
+
* `pending` is intentionally STRICT — not "every `scraped = 0` row".
|
|
357
|
+
* Three filters apply:
|
|
358
|
+
*
|
|
359
|
+
* 1. `scraped = 0` — work still incomplete.
|
|
360
|
+
* 2. `isExternal = 0` — only in-scope work. External URLs go through a
|
|
361
|
+
* HEAD-only path that always lands on `scraped = 1` (either setPage or
|
|
362
|
+
* setExternalPage). A row with `isExternal = 1 AND scraped = 0` is
|
|
363
|
+
* therefore a data anomaly, and resume / inventory / append have no
|
|
364
|
+
* business retrying it on the next session.
|
|
365
|
+
* 3. `EXISTS (anchor with hrefId = pages.id) OR source != 'crawled'` —
|
|
366
|
+
* the row was either discovered as an anchor destination during a
|
|
367
|
+
* previous scrape OR was explicitly tagged with a non-default
|
|
368
|
+
* source label (`'inventory-seed'`, `'inventory-discovered'`, …).
|
|
369
|
+
* Both halves of the OR represent "deliberately enqueued, expected
|
|
370
|
+
* to be processed", which is exactly what `resume` should pick up.
|
|
371
|
+
*
|
|
372
|
+
* The orphan filter targets the **predicted-discard leak** in
|
|
373
|
+
* `crawler.ts` where `shouldDiscardPredicted` returns true but no
|
|
374
|
+
* `emit('skip')` follows. Such placeholders are inserted with the
|
|
375
|
+
* DB DEFAULT `source = 'crawled'` (no caller explicitly labels
|
|
376
|
+
* them) AND have no anchor referrer (predicted URLs are
|
|
377
|
+
* synthesised from pagination patterns, never anchored from a
|
|
378
|
+
* rendered page) — both halves of the OR are therefore false and
|
|
379
|
+
* the leak is excluded.
|
|
380
|
+
*
|
|
381
|
+
* The `source != 'crawled'` clause specifically saves the
|
|
382
|
+
* `--inventory` × `--retry-failed` interaction: an inventory-seed
|
|
383
|
+
* URL came from the operator's URL list (no anchor referrer) and
|
|
384
|
+
* `resetFailedPages` puts it back at `scraped = 0`. Without this
|
|
385
|
+
* clause those legitimate retries would be dropped on resume.
|
|
386
|
+
*
|
|
387
|
+
* The defensive shape is on purpose: the data source can drift into
|
|
388
|
+
* anomalous states under interruption, but the reader must never throw
|
|
389
|
+
* or feed garbage back into the dealer. A real in-scope URL that was
|
|
390
|
+
* truly interrupted mid-crawl will always have at least one anchor
|
|
391
|
+
* referrer (otherwise the dealer would not have queued it), so the
|
|
392
|
+
* strict filter loses no legitimate pending work.
|
|
393
|
+
*
|
|
394
|
+
* Seeds passed directly to `Crawler.start()` are NOT in the strict
|
|
395
|
+
* pending set when they were never picked by the dealer — they have no
|
|
396
|
+
* DB row at all in that case (`linkList.add` is purely in-memory until
|
|
397
|
+
* `setPage` runs). A Ctrl-C between dealer pick and `setPage` likewise
|
|
398
|
+
* leaves no row to recover. Recovery of un-picked seeds is the
|
|
399
|
+
* responsibility of the caller (e.g. re-running `--inventory ./list.txt`
|
|
400
|
+
* with the same URL list).
|
|
401
|
+
*
|
|
402
|
+
* The query uses an explicit `p` alias on the `pages` table so the
|
|
403
|
+
* correlated `EXISTS` subquery can join via `whereRaw('anchors.hrefId =
|
|
404
|
+
* p.id')`. A future refactor that renames the alias must update both
|
|
405
|
+
* sites — the raw string in the subquery cannot be grep-resolved
|
|
406
|
+
* automatically. Read-only / stub viewer connections never call this
|
|
407
|
+
* method (they do not need to know about pending state), so the EXISTS
|
|
408
|
+
* shape is safe to use without the `migrate*` guards that other writer
|
|
409
|
+
* methods carry.
|
|
410
|
+
* @returns An object with `scraped` (completed URLs) and `pending` (the
|
|
411
|
+
* strict set of in-scope, anchor-referenced, unfinished URLs).
|
|
412
|
+
*/
|
|
413
|
+
async getCrawlingState() {
|
|
414
|
+
return emitErrorAndRetry(this, 'Database.getCrawlingState', async () => {
|
|
454
415
|
const ex = (r) => r.url;
|
|
455
416
|
const $scraped = await this.#instance
|
|
456
417
|
.select('url')
|
|
@@ -458,28 +419,43 @@ let Database = (() => {
|
|
|
458
419
|
.where('scraped', 1);
|
|
459
420
|
const scraped = $scraped.map(ex);
|
|
460
421
|
const $pending = await this.#instance
|
|
461
|
-
.select('url')
|
|
462
|
-
.from('pages')
|
|
463
|
-
.where('scraped', 0)
|
|
422
|
+
.select('p.url')
|
|
423
|
+
.from({ p: 'pages' })
|
|
424
|
+
.where('p.scraped', 0)
|
|
425
|
+
.where('p.isExternal', 0)
|
|
426
|
+
.where((qb) => {
|
|
427
|
+
// "Anchored OR explicitly labelled". Either side is evidence
|
|
428
|
+
// that the row was deliberately enqueued for processing —
|
|
429
|
+
// only the predicted-discard leak (DEFAULT 'crawled' + no
|
|
430
|
+
// anchor) fails both halves. The `whereExists` callback
|
|
431
|
+
// uses `select('*')` since the column list is irrelevant
|
|
432
|
+
// inside an EXISTS check; calling through `client.raw(...)`
|
|
433
|
+
// would reach a private builder field.
|
|
434
|
+
qb.whereExists(function () {
|
|
435
|
+
this.select('*').from('anchors').whereRaw('anchors.hrefId = p.id');
|
|
436
|
+
}).orWhereNot('p.source', 'crawled');
|
|
437
|
+
});
|
|
464
438
|
const pending = $pending.map(ex);
|
|
465
439
|
return {
|
|
466
440
|
scraped,
|
|
467
441
|
pending,
|
|
468
442
|
};
|
|
469
|
-
}
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
443
|
+
}, retrySetting);
|
|
444
|
+
}
|
|
445
|
+
/**
|
|
446
|
+
* Return the subset of `urls` that already exist in the `pages` table.
|
|
447
|
+
* Chunked into batches so SQLite's `IN (?, ?, …)` parameter limit
|
|
448
|
+
* (`SQLITE_MAX_VARIABLE_NUMBER`, default 999) cannot be hit even when the
|
|
449
|
+
* inventory list contains tens of thousands of URLs.
|
|
450
|
+
*
|
|
451
|
+
* Read-only — no transaction, no lock contention with the crawler write
|
|
452
|
+
* pipeline (callers run this BEFORE the `<archive>.bak` is taken and the
|
|
453
|
+
* crawl is started).
|
|
454
|
+
* @param urls - URL strings to probe (already in `withoutHashAndAuth` form).
|
|
455
|
+
* @returns URLs found in `pages`. Order is not preserved.
|
|
456
|
+
*/
|
|
457
|
+
async getExistingPageUrls(urls) {
|
|
458
|
+
return emitError(this, 'Database.getExistingPageUrls', async () => {
|
|
483
459
|
if (urls.length === 0) {
|
|
484
460
|
return [];
|
|
485
461
|
}
|
|
@@ -494,14 +470,16 @@ let Database = (() => {
|
|
|
494
470
|
}
|
|
495
471
|
});
|
|
496
472
|
return found;
|
|
497
|
-
}
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
473
|
+
});
|
|
474
|
+
}
|
|
475
|
+
/**
|
|
476
|
+
* Return the subset of `urls` that already exist in the `resources` table.
|
|
477
|
+
* See {@link Database.getExistingPageUrls} — same chunking strategy.
|
|
478
|
+
* @param urls - URL strings to probe.
|
|
479
|
+
* @returns URLs found in `resources`.
|
|
480
|
+
*/
|
|
481
|
+
async getExistingResourceUrls(urls) {
|
|
482
|
+
return emitError(this, 'Database.getExistingResourceUrls', async () => {
|
|
505
483
|
if (urls.length === 0) {
|
|
506
484
|
return [];
|
|
507
485
|
}
|
|
@@ -516,23 +494,25 @@ let Database = (() => {
|
|
|
516
494
|
}
|
|
517
495
|
});
|
|
518
496
|
return found;
|
|
519
|
-
}
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
497
|
+
});
|
|
498
|
+
}
|
|
499
|
+
/**
|
|
500
|
+
* Reads the HTML snapshot stored as a zstd-compressed BLOB for the given page.
|
|
501
|
+
*
|
|
502
|
+
* Joins `page_html_ref` → `page_html_blobs` and decompresses inline. Returns
|
|
503
|
+
* `null` when the page has no stored body (a non-HTML resource, a redirect
|
|
504
|
+
* source, a degraded render). Read works identically on read-only / stub
|
|
505
|
+
* connections — the special-cased "do we have a loose dir vs zip?" branching
|
|
506
|
+
* the previous file-backed layout required is gone.
|
|
507
|
+
*
|
|
508
|
+
* Tables `page_html_ref` and `page_html_blobs` are created by `initSchema`.
|
|
509
|
+
* Older `.nitpicker` archives that predate this migration must be passed
|
|
510
|
+
* through `scripts/migrate-to-0.10.mjs` before they can be read.
|
|
511
|
+
* @param pageId - The database ID of the page.
|
|
512
|
+
* @returns The decompressed HTML string, or `null` if no snapshot is stored.
|
|
513
|
+
*/
|
|
514
|
+
async getHtmlOfPageById(pageId) {
|
|
515
|
+
return emitErrorAndRetry(this, 'Database.getHtmlOfPageById', async () => {
|
|
536
516
|
const row = await this.#instance
|
|
537
517
|
.from('page_html_ref')
|
|
538
518
|
.join('page_html_blobs', 'page_html_ref.hash', '=', 'page_html_blobs.hash')
|
|
@@ -543,17 +523,19 @@ let Database = (() => {
|
|
|
543
523
|
return null;
|
|
544
524
|
}
|
|
545
525
|
return decodeStoredBlob(row.body, row.codec);
|
|
546
|
-
}
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
526
|
+
}, retrySetting);
|
|
527
|
+
}
|
|
528
|
+
/**
|
|
529
|
+
* Retrieves all `page_jsonld` rows for the given page id, parsed back into
|
|
530
|
+
* {@link JsonLdRow} shape (with `parsed` deserialised from its JSON column).
|
|
531
|
+
*
|
|
532
|
+
* Read-side counterpart to `#insertJsonLd`. Returns rows in insertion order
|
|
533
|
+
* by `id` so the order observed by `get-page-jsonld` matches the order the
|
|
534
|
+
* scraper saw them.
|
|
535
|
+
* @param pageId
|
|
536
|
+
*/
|
|
537
|
+
async getJsonLdOfPage(pageId) {
|
|
538
|
+
return emitErrorAndRetry(this, 'Database.getJsonLdOfPage', async () => {
|
|
557
539
|
const rows = await this.#instance
|
|
558
540
|
.select('id', 'pageId', 'kind', 'type', 'raw', 'parsed', 'parseError')
|
|
559
541
|
.from('page_jsonld')
|
|
@@ -568,35 +550,39 @@ let Database = (() => {
|
|
|
568
550
|
parsed: r.parsed === null ? null : safeParseJson(r.parsed),
|
|
569
551
|
parseError: r.parseError,
|
|
570
552
|
}));
|
|
571
|
-
}
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
553
|
+
}, retrySetting);
|
|
554
|
+
}
|
|
555
|
+
/**
|
|
556
|
+
* Returns the underlying Knex query builder instance for direct SQL access.
|
|
557
|
+
* This enables advanced queries (GROUP BY, HAVING, JOINs) at the database
|
|
558
|
+
* layer for performance with large datasets.
|
|
559
|
+
* @returns The Knex instance connected to the SQLite database.
|
|
560
|
+
*/
|
|
561
|
+
getKnex() {
|
|
562
|
+
return this.#instance;
|
|
563
|
+
}
|
|
564
|
+
/**
|
|
565
|
+
* Retrieves the crawl session name from the `info` table.
|
|
566
|
+
* @returns The name string.
|
|
567
|
+
* @throws {Error} If no name is found in the database.
|
|
568
|
+
*/
|
|
569
|
+
async getName() {
|
|
570
|
+
return emitErrorAndRetry(this, 'Database.getName', async () => {
|
|
587
571
|
const selected = await this.#instance.select('name').from('info');
|
|
588
572
|
if (!selected[0]) {
|
|
589
573
|
throw new Error('No name');
|
|
590
574
|
}
|
|
591
575
|
const [{ name }] = selected;
|
|
592
576
|
return name;
|
|
593
|
-
}
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
577
|
+
}, retrySetting);
|
|
578
|
+
}
|
|
579
|
+
/**
|
|
580
|
+
* Counts the total number of pages in the database.
|
|
581
|
+
* @returns The total page count.
|
|
582
|
+
* @throws {Error} If the count query fails.
|
|
583
|
+
*/
|
|
584
|
+
async getPageCount() {
|
|
585
|
+
return emitErrorAndRetry(this, 'Database.getPageCount', async () => {
|
|
600
586
|
const selected = await this.#instance.count('id').from('pages');
|
|
601
587
|
if (!selected[0]) {
|
|
602
588
|
throw new Error('No count');
|
|
@@ -605,15 +591,17 @@ let Database = (() => {
|
|
|
605
591
|
const count = selected[0]['count(`id`)'];
|
|
606
592
|
dbLog('Number of pages: %d', count);
|
|
607
593
|
return count;
|
|
608
|
-
}
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
594
|
+
}, retrySetting);
|
|
595
|
+
}
|
|
596
|
+
/**
|
|
597
|
+
* Retrieves pages from the database with optional filtering, pagination via offset and limit.
|
|
598
|
+
* @param filter - An optional {@link PageFilter} to narrow results by content type and origin.
|
|
599
|
+
* @param offset - The number of rows to skip. Defaults to `0`.
|
|
600
|
+
* @param limit - The maximum number of rows to return. Defaults to `100000`.
|
|
601
|
+
* @returns An array of raw {@link DB_Page} rows.
|
|
602
|
+
*/
|
|
603
|
+
async getPages(filter, offset = 0, limit = 100_000) {
|
|
604
|
+
return emitErrorAndRetry(this, 'Database.getPages', async () => {
|
|
617
605
|
const q = this.#instance.select('*').from('pages');
|
|
618
606
|
switch (filter) {
|
|
619
607
|
case 'page': {
|
|
@@ -688,15 +676,46 @@ let Database = (() => {
|
|
|
688
676
|
}
|
|
689
677
|
}
|
|
690
678
|
return q.limit(limit).offset(offset);
|
|
691
|
-
}
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
679
|
+
}, retrySetting);
|
|
680
|
+
}
|
|
681
|
+
/**
|
|
682
|
+
* Look up the `source` column of a single page by its URL key. Used by
|
|
683
|
+
* the orchestrator's `PageSourceLookup` injection so the Crawler can
|
|
684
|
+
* resolve a parent page's lineage on `--resume` / `--retry-failed`
|
|
685
|
+
* sessions, where the in-memory `inventoryMode` is no longer
|
|
686
|
+
* available but the DB still remembers what label was last persisted.
|
|
687
|
+
*
|
|
688
|
+
* Returns `undefined` when the URL has no `pages` row (e.g. a brand-new
|
|
689
|
+
* URL that has not been seen yet) so the caller can fall through to
|
|
690
|
+
* its default behaviour without distinguishing "row absent" from "row
|
|
691
|
+
* present with NULL source" — the schema's `NOT NULL DEFAULT 'crawled'`
|
|
692
|
+
* makes a NULL value impossible in practice.
|
|
693
|
+
*
|
|
694
|
+
* Read-only — no transaction, single PK-equivalent lookup on
|
|
695
|
+
* `pages.url` (a UNIQUE column), so the cost is constant per call. The
|
|
696
|
+
* Crawler calls this at most once per page render, NOT per
|
|
697
|
+
* sub-resource, so the N+1 risk does not apply.
|
|
698
|
+
* @param url - URL key in `url.withoutHashAndAuth` form.
|
|
699
|
+
* @returns The recorded `source`, or `undefined` when no row exists.
|
|
700
|
+
*/
|
|
701
|
+
async getPageSourceByUrl(url) {
|
|
702
|
+
return emitError(this, 'Database.getPageSourceByUrl', async () => {
|
|
703
|
+
const [row] = await this.#instance
|
|
704
|
+
.select('source')
|
|
705
|
+
.from('pages')
|
|
706
|
+
.where('url', url);
|
|
707
|
+
return row?.source;
|
|
708
|
+
});
|
|
709
|
+
}
|
|
710
|
+
/**
|
|
711
|
+
* Retrieves pages along with their related redirect, anchor, and referrer data.
|
|
712
|
+
* Results are ordered by the natural URL sort order. Only non-redirected pages are returned.
|
|
713
|
+
* @param offset - The number of rows to skip.
|
|
714
|
+
* @param limit - The maximum number of pages to return.
|
|
715
|
+
* @returns An object containing `pages`, `redirects`, `anchors`, and `referrers` arrays.
|
|
716
|
+
*/
|
|
717
|
+
async getPagesWithRels(offset, limit) {
|
|
718
|
+
return emitErrorAndRetry(this, 'Database.getPagesWithRels', async () => {
|
|
700
719
|
await this.addOrderField();
|
|
701
720
|
await this.setUrlOrder();
|
|
702
721
|
dbLog('Get Pages');
|
|
@@ -761,34 +780,38 @@ let Database = (() => {
|
|
|
761
780
|
anchors,
|
|
762
781
|
referrers,
|
|
763
782
|
};
|
|
764
|
-
}
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
783
|
+
}, retrySetting);
|
|
784
|
+
}
|
|
785
|
+
/**
|
|
786
|
+
* Retrieves redirect sources for the given page IDs in bulk.
|
|
787
|
+
* @param pageIds - The database IDs of the destination pages.
|
|
788
|
+
* @returns An array of {@link DB_Redirect} records mapping destination pages to their redirect sources.
|
|
789
|
+
*/
|
|
790
|
+
async getRedirectsForPages(pageIds) {
|
|
791
|
+
return emitErrorAndRetry(this, 'Database.getRedirectsForPages', async () => {
|
|
771
792
|
if (pageIds.length === 0)
|
|
772
793
|
return [];
|
|
773
794
|
return this.#instance
|
|
774
795
|
.select('redirectDestId as pageId', 'url as from', 'id as fromId')
|
|
775
796
|
.from('pages')
|
|
776
797
|
.whereIn('redirectDestId', pageIds);
|
|
777
|
-
}
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
798
|
+
}, retrySetting);
|
|
799
|
+
}
|
|
800
|
+
/**
|
|
801
|
+
* Retrieves pages that link to a specific page (incoming links / referrers).
|
|
802
|
+
*
|
|
803
|
+
* Incoming links are resolved **through redirects**: an anchor pointing at a
|
|
804
|
+
* redirect source (e.g. `http://x` that 301s to `https://x`) counts as a
|
|
805
|
+
* referrer of the redirect's final destination, not of the source. This keeps
|
|
806
|
+
* backlinks merged on the canonical page instead of splitting them across the
|
|
807
|
+
* `http`/`https` (or any redirect source/dest) pair. The resolution mirrors
|
|
808
|
+
* `redirectTable()` — `redirectDestId` is pre-flattened to the final
|
|
809
|
+
* destination, so `COALESCE(target.redirectDestId, target.id)` is a single hop.
|
|
810
|
+
* @param pageId - The database ID of the target page.
|
|
811
|
+
* @returns An array of referrer records with URL, hash, and text content.
|
|
812
|
+
*/
|
|
813
|
+
async getReferrersOfPage(pageId) {
|
|
814
|
+
return emitErrorAndRetry(this, 'Database.getReferrersOfPage', async () => {
|
|
792
815
|
const res = await this.#instance
|
|
793
816
|
.select('referrer.url',
|
|
794
817
|
// `through` / `throughId` = the URL the anchor actually pointed at (the
|
|
@@ -801,13 +824,15 @@ let Database = (() => {
|
|
|
801
824
|
.join('pages as target', 'anchors.hrefId', '=', 'target.id')
|
|
802
825
|
.whereRaw('coalesce("target"."redirectDestId", "target"."id") = ?', [pageId]);
|
|
803
826
|
return res;
|
|
804
|
-
}
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
827
|
+
}, retrySetting);
|
|
828
|
+
}
|
|
829
|
+
/**
|
|
830
|
+
* Retrieves the page URLs that reference a specific resource.
|
|
831
|
+
* @param id - The database ID of the resource.
|
|
832
|
+
* @returns An array of page URL strings that reference the resource.
|
|
833
|
+
*/
|
|
834
|
+
async getReferrersOfResource(id) {
|
|
835
|
+
return emitErrorAndRetry(this, 'Database.getReferrersOfResource', async () => {
|
|
811
836
|
const res = await this.#instance
|
|
812
837
|
.select('pages.url')
|
|
813
838
|
.from('resources-referrers')
|
|
@@ -815,61 +840,69 @@ let Database = (() => {
|
|
|
815
840
|
.join('pages', 'pages.id', '=', 'resources-referrers.pageId')
|
|
816
841
|
.where('resources.id', id);
|
|
817
842
|
return res.map((r) => r.url);
|
|
818
|
-
}
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
843
|
+
}, retrySetting);
|
|
844
|
+
}
|
|
845
|
+
/**
|
|
846
|
+
* Retrieves a single sub-resource from the `resources` table by its URL.
|
|
847
|
+
*
|
|
848
|
+
* Accepts multiple URL candidates because the stored key is the resource's
|
|
849
|
+
* `href` while callers may only know the hash-stripped form; the first match
|
|
850
|
+
* wins.
|
|
851
|
+
*
|
|
852
|
+
* Deliberately NOT wrapped with `emitError`/`emitErrorAndRetry`: the only caller (the
|
|
853
|
+
* crawler's resource-reuse hook) has a full fallback (the HEAD pre-flight),
|
|
854
|
+
* so a read failure here must not surface as a database `error` event —
|
|
855
|
+
* the orchestrator aborts the whole crawl on that event, which is the
|
|
856
|
+
* correct reaction to write failures but not to a recoverable read.
|
|
857
|
+
* @param urls - URL candidates to match against the `url` column.
|
|
858
|
+
* @returns The raw {@link DB_Resource} row, or `null` if none match.
|
|
859
|
+
*/
|
|
860
|
+
async getResourceByUrl(urls) {
|
|
861
|
+
return retryCall(async () => {
|
|
835
862
|
const res = await this.#instance
|
|
836
863
|
.select('*')
|
|
837
864
|
.from('resources')
|
|
838
865
|
.whereIn('url', [...urls])
|
|
839
866
|
.first();
|
|
840
867
|
return res ?? null;
|
|
841
|
-
}
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
868
|
+
}, { ...retrySetting, label: 'Database.getResourceByUrl' });
|
|
869
|
+
}
|
|
870
|
+
/**
|
|
871
|
+
* Retrieves all sub-resources from the `resources` table.
|
|
872
|
+
* @returns An array of raw {@link DB_Resource} rows.
|
|
873
|
+
*/
|
|
874
|
+
async getResources() {
|
|
875
|
+
return emitErrorAndRetry(this, 'Database.getResources', async () => {
|
|
847
876
|
return this.#instance.select('*').from('resources');
|
|
848
|
-
}
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
877
|
+
}, retrySetting);
|
|
878
|
+
}
|
|
879
|
+
/**
|
|
880
|
+
* Retrieves a flat list of all resource URLs from the `resources` table.
|
|
881
|
+
* @returns An array of resource URL strings.
|
|
882
|
+
*/
|
|
883
|
+
async getResourceUrlList() {
|
|
884
|
+
return emitErrorAndRetry(this, 'Database.getResourceUrlList', async () => {
|
|
854
885
|
const res = await this.#instance.select('url').from('resources');
|
|
855
886
|
return res.map((r) => r.url);
|
|
856
|
-
}
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
887
|
+
}, retrySetting);
|
|
888
|
+
}
|
|
889
|
+
/**
|
|
890
|
+
* Counts pages that were scraped as crawl targets (full HTML render).
|
|
891
|
+
*
|
|
892
|
+
* Used by the crawler to seed its `pagesScraped` counter on resume so the
|
|
893
|
+
* progress display reflects all browser-rendered HTML pages across sessions,
|
|
894
|
+
* not just the current one.
|
|
895
|
+
*
|
|
896
|
+
* "HTML page" is guaranteed by `contentType = 'text/html'`, NOT by `isTarget`
|
|
897
|
+
* alone: `isTarget` means "in-scope crawl target" and is set for in-scope
|
|
898
|
+
* non-HTML resources too (e.g. a PDF reached via the HEAD pre-flight is
|
|
899
|
+
* `isTarget = 1`). Counting those would over-report the HTML page total, so
|
|
900
|
+
* page-ness is asserted at the read layer here rather than by trusting
|
|
901
|
+
* `isTarget`.
|
|
902
|
+
* @returns The number of `text/html` rows with `isTarget = 1` and `scraped = 1`.
|
|
903
|
+
*/
|
|
904
|
+
async getScrapedHtmlPageCount() {
|
|
905
|
+
return emitErrorAndRetry(this, 'Database.getScrapedHtmlPageCount', async () => {
|
|
873
906
|
const [row] = await this.#instance
|
|
874
907
|
.from('pages')
|
|
875
908
|
.where('isTarget', 1)
|
|
@@ -877,16 +910,18 @@ let Database = (() => {
|
|
|
877
910
|
.andWhere('contentType', 'text/html')
|
|
878
911
|
.count('* as count');
|
|
879
912
|
return row ? Number(row.count) : 0;
|
|
880
|
-
}
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
913
|
+
}, retrySetting);
|
|
914
|
+
}
|
|
915
|
+
/**
|
|
916
|
+
* Retrieves all `page_tags` rows for the given page id, parsed back into
|
|
917
|
+
* {@link TagRow} shape (with `categories` and `sources` JSON columns
|
|
918
|
+
* deserialised).
|
|
919
|
+
*
|
|
920
|
+
* Read-side counterpart to `#insertTags`.
|
|
921
|
+
* @param pageId
|
|
922
|
+
*/
|
|
923
|
+
async getTagsOfPage(pageId) {
|
|
924
|
+
return emitErrorAndRetry(this, 'Database.getTagsOfPage', async () => {
|
|
890
925
|
const rows = await this.#instance
|
|
891
926
|
.select('id', 'pageId', 'provider', 'category', 'externalId', 'version', 'confidence', 'categories', 'sources')
|
|
892
927
|
.from('page_tags')
|
|
@@ -900,47 +935,148 @@ let Database = (() => {
|
|
|
900
935
|
externalId: r.externalId,
|
|
901
936
|
version: r.version,
|
|
902
937
|
confidence: r.confidence,
|
|
903
|
-
categories: r.categories === null
|
|
904
|
-
|
|
938
|
+
categories: r.categories === null
|
|
939
|
+
? []
|
|
940
|
+
: (safeParseJson(r.categories) ?? []),
|
|
941
|
+
sources: r.sources === null
|
|
942
|
+
? []
|
|
943
|
+
: (safeParseJson(r.sources) ?? []),
|
|
905
944
|
}));
|
|
906
|
-
}
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
945
|
+
}, retrySetting);
|
|
946
|
+
}
|
|
947
|
+
/**
|
|
948
|
+
* Records a crawler-level (`error` channel) failure into `crawl_errors`.
|
|
949
|
+
*
|
|
950
|
+
* Unlike {@link insertPageError} this is not tied to a scraped page: `url`
|
|
951
|
+
* may be an external link that never became a page row, or `null` for a
|
|
952
|
+
* process-level error. The cause is intentionally not stored — it is derived
|
|
953
|
+
* on read so that older archives (which only have `error.log`) and freshly
|
|
954
|
+
* captured rows classify identically.
|
|
955
|
+
* @param url - The URL the error is about, or `null` for a process-level error.
|
|
956
|
+
* @param message - The error message (one line is enough for classification).
|
|
957
|
+
* @param isExternal - Whether the URL is external to the crawl scope.
|
|
958
|
+
*/
|
|
959
|
+
async insertCrawlError(url, message, isExternal = false) {
|
|
960
|
+
return emitErrorAndRetry(this, 'Database.insertCrawlError', async () => {
|
|
920
961
|
await this.#instance('crawl_errors').insert({
|
|
921
962
|
url,
|
|
922
963
|
isExternal: isExternal ? 1 : 0,
|
|
923
964
|
message,
|
|
924
965
|
createdAt: Date.now(),
|
|
925
966
|
});
|
|
926
|
-
}
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
|
|
967
|
+
}, retrySetting);
|
|
968
|
+
}
|
|
969
|
+
/**
|
|
970
|
+
* Pre-insert inventory non-HTML URLs into `resources` as placeholder rows
|
|
971
|
+
* with `source = 'inventory-seed'` and all metadata columns NULL — the
|
|
972
|
+
* non-HTML counterpart of {@link Database.insertInventorySeeds}. Used by
|
|
973
|
+
* `CrawlerOrchestrator.inventory` so the ingestion phase commits all of
|
|
974
|
+
* its non-HTML URLs in one chunked round-trip per 500 instead of N
|
|
975
|
+
* sequential `insertResource` awaits. On a 50k-URL inventory list the
|
|
976
|
+
* old per-URL loop spent minutes inside the `.bak`-protected window;
|
|
977
|
+
* the bulk path finishes in seconds.
|
|
978
|
+
*
|
|
979
|
+
* Idempotent: `onConflict('url').ignore()` leaves existing rows untouched
|
|
980
|
+
* (the orchestrator's `getExistingResourceUrls` filter is what keeps a
|
|
981
|
+
* crawled-lineage `resources` row from being downgraded to the
|
|
982
|
+
* inventory label here).
|
|
983
|
+
*
|
|
984
|
+
* Chunked at 500 to stay well under SQLite's `SQLITE_MAX_VARIABLE_NUMBER`
|
|
985
|
+
* (default 999) — every row binds the URL plus the `responseHeaders`
|
|
986
|
+
* JSON null, so the per-chunk bound budget is well within limits.
|
|
987
|
+
* @param urls - URL strings (already in `withoutHashAndAuth` form).
|
|
988
|
+
*/
|
|
989
|
+
async insertInventoryResources(urls) {
|
|
990
|
+
return emitErrorAndRetry(this, 'Database.insertInventoryResources', async () => {
|
|
991
|
+
if (urls.length === 0) {
|
|
992
|
+
return;
|
|
993
|
+
}
|
|
994
|
+
await eachSplitted([...urls], 500, async (chunk) => {
|
|
995
|
+
await this.#instance('resources')
|
|
996
|
+
.insert(chunk.map((url) => ({
|
|
997
|
+
url,
|
|
998
|
+
isExternal: 0,
|
|
999
|
+
status: null,
|
|
1000
|
+
statusText: null,
|
|
1001
|
+
contentType: null,
|
|
1002
|
+
contentLength: null,
|
|
1003
|
+
compress: 0,
|
|
1004
|
+
cdn: 0,
|
|
1005
|
+
responseHeaders: null,
|
|
1006
|
+
source: 'inventory-seed',
|
|
1007
|
+
})))
|
|
1008
|
+
.onConflict('url')
|
|
1009
|
+
.ignore();
|
|
1010
|
+
});
|
|
1011
|
+
}, retrySetting);
|
|
1012
|
+
}
|
|
1013
|
+
/**
|
|
1014
|
+
* Pre-insert inventory HTML seeds into `pages` as `scraped = 0`,
|
|
1015
|
+
* `source = 'inventory-seed'` placeholders so the URL's existence in the
|
|
1016
|
+
* archive is **durable before the scrape phase starts**.
|
|
1017
|
+
*
|
|
1018
|
+
* Why this is the linchpin of `--inventory` Ctrl+C tolerance: HTML seeds
|
|
1019
|
+
* used to live only in the Crawler's in-memory `LinkList` until the
|
|
1020
|
+
* dealer eventually called `setPage`. A Ctrl+C / crash before that point
|
|
1021
|
+
* lost the seed without trace, and `--resume` could not recover it
|
|
1022
|
+
* because `getCrawlingState`'s strict pending set requires a `pages` row.
|
|
1023
|
+
* Pre-inserting fills exactly that gap: the strict pending set picks
|
|
1024
|
+
* these rows up via its `OR p.source != 'crawled'` clause, so
|
|
1025
|
+
* `--resume` after an interrupted inventory pass picks every seed back
|
|
1026
|
+
* up. See {@link Database.getCrawlingState} for the strict-set rationale.
|
|
1027
|
+
*
|
|
1028
|
+
* Idempotent: `onConflict('url').ignore()` keeps existing rows intact.
|
|
1029
|
+
* The {@link Database.#getIdByUrl} crawled-wins downgrade still fires
|
|
1030
|
+
* later when a crawled-lineage anchor reaches one of these seeds —
|
|
1031
|
+
* that's the right behaviour (a seed that turned out to be reachable
|
|
1032
|
+
* is not an orphan and should not retain the inventory label).
|
|
1033
|
+
*
|
|
1034
|
+
* Chunked into 500-URL batches so SQLite's bound-parameter limit
|
|
1035
|
+
* (`SQLITE_MAX_VARIABLE_NUMBER`, default 999) cannot be hit even on a
|
|
1036
|
+
* tens-of-thousands inventory list.
|
|
1037
|
+
*
|
|
1038
|
+
* Called by {@link CrawlerOrchestrator.inventory} during the
|
|
1039
|
+
* `.bak`-protected ingestion phase, so any failure here aborts the run
|
|
1040
|
+
* and restores from backup — the operator reruns from scratch.
|
|
1041
|
+
* @param urls - URL strings already in `withoutHashAndAuth` form.
|
|
1042
|
+
*/
|
|
1043
|
+
async insertInventorySeeds(urls) {
|
|
1044
|
+
return emitErrorAndRetry(this, 'Database.insertInventorySeeds', async () => {
|
|
1045
|
+
if (urls.length === 0) {
|
|
1046
|
+
return;
|
|
1047
|
+
}
|
|
1048
|
+
await eachSplitted([...urls], 500, async (chunk) => {
|
|
1049
|
+
await this.#instance('pages')
|
|
1050
|
+
.insert(chunk.map((url) => ({
|
|
1051
|
+
url,
|
|
1052
|
+
scraped: 0,
|
|
1053
|
+
isExternal: 0,
|
|
1054
|
+
isTarget: 0,
|
|
1055
|
+
source: 'inventory-seed',
|
|
1056
|
+
})))
|
|
1057
|
+
.onConflict('url')
|
|
1058
|
+
.ignore();
|
|
1059
|
+
});
|
|
1060
|
+
}, retrySetting);
|
|
1061
|
+
}
|
|
1062
|
+
/**
|
|
1063
|
+
* Records a partial scrape failure against the page identified by `url`.
|
|
1064
|
+
*
|
|
1065
|
+
* The page row is resolved (or inserted as a stub) via
|
|
1066
|
+
* {@link Database.#getIdByUrl} so the error can be recorded even before
|
|
1067
|
+
* `setPage` has run — useful when the failure fires during scraping
|
|
1068
|
+
* (e.g. mid-`scrapeStart`) and the orchestrator enqueues this write
|
|
1069
|
+
* before the success write for the same URL.
|
|
1070
|
+
*
|
|
1071
|
+
* A single page can have multiple `page_errors` rows (e.g. both
|
|
1072
|
+
* `desktop-compact` and `mobile-small` viewports failing).
|
|
1073
|
+
* @param url - URL of the page being scraped.
|
|
1074
|
+
* @param phase - Scrape phase name (typically `'retryExhausted'`).
|
|
1075
|
+
* @param message - Human-readable failure message.
|
|
1076
|
+
* @param isExternal - Whether the URL is external. Defaults to `false`.
|
|
1077
|
+
*/
|
|
1078
|
+
async insertPageError(url, phase, message, isExternal = false) {
|
|
1079
|
+
return emitErrorAndRetry(this, 'Database.insertPageError', async () => {
|
|
944
1080
|
const pageId = await this.#getIdByUrl(url, isExternal ? 1 : 0);
|
|
945
1081
|
await this.#instance('page_errors').insert({
|
|
946
1082
|
pageId,
|
|
@@ -948,19 +1084,21 @@ let Database = (() => {
|
|
|
948
1084
|
message,
|
|
949
1085
|
createdAt: Date.now(),
|
|
950
1086
|
});
|
|
951
|
-
}
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
1087
|
+
}, retrySetting);
|
|
1088
|
+
}
|
|
1089
|
+
/**
|
|
1090
|
+
* Inserts a sub-resource into the `resources` table.
|
|
1091
|
+
* Ignores duplicate URLs (uses `ON CONFLICT IGNORE`).
|
|
1092
|
+
*
|
|
1093
|
+
* The `source` provenance label is written ONLY on insert; an
|
|
1094
|
+
* `ON CONFLICT IGNORE` collision leaves an existing row's source untouched
|
|
1095
|
+
* (this is what makes a second `crawl --inventory` non-destructive — see
|
|
1096
|
+
* the inventory plan).
|
|
1097
|
+
* @param resource - The resource data to insert.
|
|
1098
|
+
* @param source - Provenance label for new rows. `undefined` leaves the DB DEFAULT (`'crawled'`).
|
|
1099
|
+
*/
|
|
1100
|
+
async insertResource(resource, source) {
|
|
1101
|
+
return emitErrorAndRetry(this, 'Database.insertResource', async () => {
|
|
964
1102
|
await this.#instance
|
|
965
1103
|
.from('resources')
|
|
966
1104
|
.insert({
|
|
@@ -979,14 +1117,16 @@ let Database = (() => {
|
|
|
979
1117
|
})
|
|
980
1118
|
.onConflict('url')
|
|
981
1119
|
.ignore();
|
|
982
|
-
}
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
1120
|
+
}, retrySetting);
|
|
1121
|
+
}
|
|
1122
|
+
/**
|
|
1123
|
+
* Inserts a referrer relationship between a resource and a page into the
|
|
1124
|
+
* `resources-referrers` table. Silently skips if the resource is not found.
|
|
1125
|
+
* @param src - The URL of the resource.
|
|
1126
|
+
* @param pageUrl - The URL of the page that references the resource.
|
|
1127
|
+
*/
|
|
1128
|
+
async insertResourceReferrers(src, pageUrl) {
|
|
1129
|
+
return emitErrorAndRetry(this, 'Database.insertResourceReferrers', async () => {
|
|
990
1130
|
const selected = await this.#instance
|
|
991
1131
|
.select('id')
|
|
992
1132
|
.from('resources')
|
|
@@ -1004,26 +1144,215 @@ let Database = (() => {
|
|
|
1004
1144
|
})
|
|
1005
1145
|
.onConflict(['resourceId', 'pageId'])
|
|
1006
1146
|
.ignore();
|
|
1007
|
-
}
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1147
|
+
}, retrySetting);
|
|
1148
|
+
}
|
|
1149
|
+
/**
|
|
1150
|
+
* Hostnames whose `crawl_errors` history is consistently DNS failures and
|
|
1151
|
+
* for which no recent 2xx-3xx page or resource is recorded — i.e. hosts
|
|
1152
|
+
* the previous crawl already proved unreachable. Returned in lower-cased
|
|
1153
|
+
* form. Used by `CrawlerOrchestrator.#preloadDnsBurnedHostCache` so the
|
|
1154
|
+
* next session short-circuits HEAD pre-flight on these hosts.
|
|
1155
|
+
*
|
|
1156
|
+
* Implementation: a coarse `LIKE` filter over `crawl_errors.message`
|
|
1157
|
+
* narrows the row set, then `classifyErrorKind` confirms `'dns'` in JS
|
|
1158
|
+
* (the regex is the single truth source — DB-side filters never narrow
|
|
1159
|
+
* it). Exclusion bags are built from a single `pages` and a single
|
|
1160
|
+
* `resources` scan: any host with a 2xx-3xx page, a 2xx-3xx resource, or
|
|
1161
|
+
* a `pages.lastCrawledAt` newer than its latest DNS error is dropped
|
|
1162
|
+
* (the host probably recovered between the failure and the last crawl).
|
|
1163
|
+
*
|
|
1164
|
+
* Returns `[]` on legacy archives that pre-date the `crawl_errors`
|
|
1165
|
+
* table — the `hasTable` guard keeps the call non-destructive.
|
|
1166
|
+
* @returns Lower-cased hostnames safe to short-circuit.
|
|
1167
|
+
*/
|
|
1168
|
+
async listDnsBurnedHostCandidates() {
|
|
1169
|
+
return emitErrorAndRetry(this, 'Database.listDnsBurnedHostCandidates', async () => {
|
|
1170
|
+
const hasCrawlErrors = await this.#instance.schema.hasTable('crawl_errors');
|
|
1171
|
+
if (!hasCrawlErrors) {
|
|
1172
|
+
return [];
|
|
1173
|
+
}
|
|
1174
|
+
// Coarse SQL filter: cheap LIKE OR-chain over `message`. The dns regex
|
|
1175
|
+
// truth source lives in `classifyErrorKind`, so we only need to feed it
|
|
1176
|
+
// rows that COULD match a DNS token. Each LIKE is anchored on a known
|
|
1177
|
+
// substring of the regex so future additions to the regex (without
|
|
1178
|
+
// matching new SQL terms) widen the JS-side filter only — never narrow it.
|
|
1179
|
+
//
|
|
1180
|
+
// `%EAI_AGAIN%` is deliberately NOT in the SQL filter: it now classifies
|
|
1181
|
+
// as `dns-transient` (local resolver hiccup), not `dns`, so it must not
|
|
1182
|
+
// reach this candidate set. The `%getaddrinfo%` term still pulls
|
|
1183
|
+
// `getaddrinfo EAI_AGAIN ...` rows but the JS-side `classifyErrorKind`
|
|
1184
|
+
// check (first-match-wins) routes them to `dns-transient` and they
|
|
1185
|
+
// silently drop out — keeping the cache focused on real NXDOMAIN.
|
|
1186
|
+
const dnsLikeRows = (await this.#instance('crawl_errors')
|
|
1187
|
+
.select('url', 'message', 'createdAt')
|
|
1188
|
+
.whereNotNull('url')
|
|
1189
|
+
.where((qb) => {
|
|
1190
|
+
qb.where('message', 'like', '%ENOTFOUND%')
|
|
1191
|
+
.orWhere('message', 'like', '%getaddrinfo%')
|
|
1192
|
+
.orWhere('message', 'like', '%ERR_NAME_NOT_RESOLVED%')
|
|
1193
|
+
.orWhere('message', 'like', '%ERR_NAME_RESOLUTION_FAILED%');
|
|
1194
|
+
}));
|
|
1195
|
+
if (dnsLikeRows.length === 0) {
|
|
1196
|
+
return [];
|
|
1197
|
+
}
|
|
1198
|
+
// Map<hostname, latestErrorCreatedAt> for hosts whose error message
|
|
1199
|
+
// confidently classifies as DNS (LIKE matched but classifyErrorKind says
|
|
1200
|
+
// e.g. `unknown` → drop).
|
|
1201
|
+
const candidateLatestErrorAt = new Map();
|
|
1202
|
+
for (const row of dnsLikeRows) {
|
|
1203
|
+
if (classifyErrorKind(row.message) !== 'dns') {
|
|
1204
|
+
continue;
|
|
1205
|
+
}
|
|
1206
|
+
let host;
|
|
1207
|
+
try {
|
|
1208
|
+
host = new URL(row.url).hostname.toLowerCase();
|
|
1209
|
+
}
|
|
1210
|
+
catch {
|
|
1211
|
+
continue;
|
|
1212
|
+
}
|
|
1213
|
+
if (!host) {
|
|
1214
|
+
continue;
|
|
1215
|
+
}
|
|
1216
|
+
const createdAt = typeof row.createdAt === 'number' ? row.createdAt : 0;
|
|
1217
|
+
const previous = candidateLatestErrorAt.get(host) ?? 0;
|
|
1218
|
+
if (createdAt > previous) {
|
|
1219
|
+
candidateLatestErrorAt.set(host, createdAt);
|
|
1220
|
+
}
|
|
1221
|
+
}
|
|
1222
|
+
if (candidateLatestErrorAt.size === 0) {
|
|
1223
|
+
return [];
|
|
1224
|
+
}
|
|
1225
|
+
// Exclusion-bag #1: pages with a 2xx-3xx status anywhere on the host.
|
|
1226
|
+
// Tracking the latest `lastCrawledAt` per host lets us additionally
|
|
1227
|
+
// drop hosts whose last successful contact post-dates the most recent
|
|
1228
|
+
// DNS error (the host probably came back after a transient outage).
|
|
1229
|
+
const pageOkRows = (await this.#instance('pages')
|
|
1230
|
+
.select('url', 'lastCrawledAt')
|
|
1231
|
+
.whereBetween('status', [200, 399]));
|
|
1232
|
+
const pageOkHosts = new Set();
|
|
1233
|
+
const latestPageOkAt = new Map();
|
|
1234
|
+
for (const row of pageOkRows) {
|
|
1235
|
+
let host;
|
|
1236
|
+
try {
|
|
1237
|
+
host = new URL(row.url).hostname.toLowerCase();
|
|
1238
|
+
}
|
|
1239
|
+
catch {
|
|
1240
|
+
continue;
|
|
1241
|
+
}
|
|
1242
|
+
pageOkHosts.add(host);
|
|
1243
|
+
if (typeof row.lastCrawledAt === 'number') {
|
|
1244
|
+
const previous = latestPageOkAt.get(host) ?? 0;
|
|
1245
|
+
if (row.lastCrawledAt > previous) {
|
|
1246
|
+
latestPageOkAt.set(host, row.lastCrawledAt);
|
|
1247
|
+
}
|
|
1248
|
+
}
|
|
1249
|
+
}
|
|
1250
|
+
// Exclusion-bag #2: non-HTML resources with a 2xx-3xx status. resources
|
|
1251
|
+
// have no timestamp column so this is presence-only.
|
|
1252
|
+
const resourceOkRows = (await this.#instance('resources')
|
|
1253
|
+
.select('url')
|
|
1254
|
+
.whereBetween('status', [200, 399]));
|
|
1255
|
+
const resourceOkHosts = new Set();
|
|
1256
|
+
for (const row of resourceOkRows) {
|
|
1257
|
+
let host;
|
|
1258
|
+
try {
|
|
1259
|
+
host = new URL(row.url).hostname.toLowerCase();
|
|
1260
|
+
}
|
|
1261
|
+
catch {
|
|
1262
|
+
continue;
|
|
1263
|
+
}
|
|
1264
|
+
resourceOkHosts.add(host);
|
|
1265
|
+
}
|
|
1266
|
+
// A candidate host is burned only if neither pages nor resources hold a
|
|
1267
|
+
// 2xx-3xx for it, AND its latest 2xx page (if any) is not newer than
|
|
1268
|
+
// the latest DNS error. The third check guards against re-burning a
|
|
1269
|
+
// host that recovered between the last DNS failure and the most recent
|
|
1270
|
+
// crawl.
|
|
1271
|
+
const burned = [];
|
|
1272
|
+
for (const [host, latestErrorAt] of candidateLatestErrorAt) {
|
|
1273
|
+
if (pageOkHosts.has(host)) {
|
|
1274
|
+
continue;
|
|
1275
|
+
}
|
|
1276
|
+
if (resourceOkHosts.has(host)) {
|
|
1277
|
+
continue;
|
|
1278
|
+
}
|
|
1279
|
+
const latestOkAt = latestPageOkAt.get(host);
|
|
1280
|
+
if (typeof latestOkAt === 'number' && latestOkAt > latestErrorAt) {
|
|
1281
|
+
continue;
|
|
1282
|
+
}
|
|
1283
|
+
burned.push(host);
|
|
1284
|
+
}
|
|
1285
|
+
return burned;
|
|
1286
|
+
}, retrySetting);
|
|
1287
|
+
}
|
|
1288
|
+
/**
|
|
1289
|
+
* Appends one row to the `inventory_runs` audit log.
|
|
1290
|
+
*
|
|
1291
|
+
* Called by {@link CrawlerOrchestrator.inventory} on every successful
|
|
1292
|
+
* `--inventory <list>` invocation so the archive carries a durable
|
|
1293
|
+
* record of which deploy list was applied when and at what scale —
|
|
1294
|
+
* the operational question "did we apply last month's list" the
|
|
1295
|
+
* archive itself can answer without consulting external bookkeeping.
|
|
1296
|
+
*
|
|
1297
|
+
* Append-only at Phase 1. There is intentionally no UPDATE path and
|
|
1298
|
+
* no UNIQUE constraint on `source_file_sha256`; two applies of the
|
|
1299
|
+
* same list each get their own row, and `Phase 3 --refresh` is where
|
|
1300
|
+
* dedupe / pre-flight against the hash will land. Field-level NULL
|
|
1301
|
+
* semantics live on {@link InventoryRunMeta}.
|
|
1302
|
+
* @param meta - The run metadata to record. Only `ran_at` is required.
|
|
1303
|
+
* @returns The autoincremented `id` of the newly-inserted row.
|
|
1304
|
+
*/
|
|
1305
|
+
async recordInventoryRun(meta) {
|
|
1306
|
+
return emitErrorAndRetry(this, 'Database.recordInventoryRun', async () => {
|
|
1307
|
+
const inserted = await this.#instance
|
|
1308
|
+
.from('inventory_runs')
|
|
1309
|
+
.insert({
|
|
1310
|
+
ran_at: meta.ran_at,
|
|
1311
|
+
list_label: meta.list_label ?? null,
|
|
1312
|
+
source_file_sha256: meta.source_file_sha256 ?? null,
|
|
1313
|
+
total_lines: meta.total_lines ?? null,
|
|
1314
|
+
new_pages: meta.new_pages ?? null,
|
|
1315
|
+
new_resources: meta.new_resources ?? null,
|
|
1316
|
+
scope_skipped: meta.scope_skipped ?? null,
|
|
1317
|
+
notes: meta.notes ?? null,
|
|
1318
|
+
})
|
|
1319
|
+
.returning('id');
|
|
1320
|
+
const id = inserted[0]?.id;
|
|
1321
|
+
if (typeof id !== 'number') {
|
|
1322
|
+
throw new TypeError('recordInventoryRun: INSERT returned no row id');
|
|
1323
|
+
}
|
|
1324
|
+
return id;
|
|
1325
|
+
}, retrySetting);
|
|
1326
|
+
}
|
|
1327
|
+
/**
|
|
1328
|
+
* Records a redirect edge (source → destination) **without** re-storing the
|
|
1329
|
+
* destination's content.
|
|
1330
|
+
*
|
|
1331
|
+
* The crawler renders a many-to-one redirect destination exactly once. For
|
|
1332
|
+
* every subsequent source URL that redirects to that already-rendered
|
|
1333
|
+
* destination, it calls this instead of {@link updatePage} (#73). Routing a
|
|
1334
|
+
* content-less HEAD result through `updatePage` would funnel it into
|
|
1335
|
+
* `#insertPage` and overwrite the destination's good title / meta with empty
|
|
1336
|
+
* values, so the dedicated edge-only path is required.
|
|
1337
|
+
*
|
|
1338
|
+
* The destination row is resolved (created on demand if a concurrent in-flight
|
|
1339
|
+
* render has not committed it yet) so the edge always points at a valid id;
|
|
1340
|
+
* the single render fills in the destination's content under that same id.
|
|
1341
|
+
* The destination's existing anchors / images are never touched here.
|
|
1342
|
+
* @param page - HEAD-resolved page data carrying the redirect chain. Its
|
|
1343
|
+
* `anchorList` / `imageList` are ignored (a redirect source owns no content).
|
|
1344
|
+
* @param source - Inventory provenance forwarded by the orchestrator
|
|
1345
|
+
* (`Archive.setRedirect` → here) for the redirect-edge fast path. Used
|
|
1346
|
+
* as the fallback when the originating URL's row does NOT yet exist in
|
|
1347
|
+
* the archive (`#73` convergence on first sight, js-redirect rescue
|
|
1348
|
+
* before any prior write). When the originating row already exists
|
|
1349
|
+
* (e.g. anchor-lineage INSERT from a prior pass), its stored `source`
|
|
1350
|
+
* takes precedence so transitive lineage is preserved across resume /
|
|
1351
|
+
* retry-failed sessions. `undefined` keeps the DB DEFAULT `'crawled'`
|
|
1352
|
+
* on a brand-new destination row.
|
|
1353
|
+
*/
|
|
1354
|
+
async recordRedirect(page, source) {
|
|
1355
|
+
return emitErrorAndRetry(this, 'Database.recordRedirect', async () => {
|
|
1027
1356
|
const { destUrl, sources } = resolveRedirectChain(page.url.withoutHashAndAuth, page.redirectPaths);
|
|
1028
1357
|
// No redirect chain (the URL is itself the already-rendered destination,
|
|
1029
1358
|
// reached both directly and via a redirect) → there is no edge to write.
|
|
@@ -1043,29 +1372,57 @@ let Database = (() => {
|
|
|
1043
1372
|
return;
|
|
1044
1373
|
}
|
|
1045
1374
|
await this.#instance.transaction(async (trx) => {
|
|
1046
|
-
|
|
1047
|
-
|
|
1375
|
+
// Pass the caller-supplied `source` straight through so a
|
|
1376
|
+
// brand-new destination row INSERTed here picks up the
|
|
1377
|
+
// inventory lineage (instead of the DB DEFAULT `'crawled'`)
|
|
1378
|
+
// when the caller is in the inventory chain — closes the
|
|
1379
|
+
// hole where `recordRedirect` was previously laundering
|
|
1380
|
+
// inventory lineage to `'crawled'` for js-redirect rescue /
|
|
1381
|
+
// #73 convergence destinations that had not yet been
|
|
1382
|
+
// rendered.
|
|
1383
|
+
const destId = await this.#getIdByUrl(destUrlObject.withoutHashAndAuth, undefined, trx, source);
|
|
1384
|
+
// Chain lineage propagates FROM the originating URL
|
|
1385
|
+
// (`page.url`), NOT from the destination. The originating
|
|
1386
|
+
// URL is what initiated the redirect chain, so its lineage
|
|
1387
|
+
// is what every intermediate hop transitively inherits.
|
|
1388
|
+
// Reading from the destination would mis-propagate in
|
|
1389
|
+
// "inventory-seed → ... → existing crawled dest" chains:
|
|
1390
|
+
// the intermediates are reached only via the inventory
|
|
1391
|
+
// chain, so they belong to the inventory chain even though
|
|
1392
|
+
// the chain happens to land on a crawled URL. The
|
|
1393
|
+
// `'crawled'` fallback arms the crawled-wins downgrade for
|
|
1394
|
+
// existing `'inventory-*'` intermediates that a crawled
|
|
1395
|
+
// chain reaches.
|
|
1396
|
+
const [originatingRow] = await trx
|
|
1397
|
+
.select('source')
|
|
1398
|
+
.from('pages')
|
|
1399
|
+
.where('url', page.url.withoutHashAndAuth);
|
|
1400
|
+
const originatingSource = originatingRow?.source ?? source;
|
|
1401
|
+
const chainLineageSource = deriveLineageFromParent(originatingSource, 'crawled');
|
|
1402
|
+
await this.#linkRedirectSources(trx, sources, destId, destUrlObject.withoutHashAndAuth, page.isExternal, chainLineageSource);
|
|
1048
1403
|
});
|
|
1049
|
-
}
|
|
1050
|
-
|
|
1051
|
-
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1055
|
-
|
|
1056
|
-
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
|
|
1404
|
+
}, retrySetting);
|
|
1405
|
+
}
|
|
1406
|
+
/**
|
|
1407
|
+
* Promote previously-external pages whose URL falls under any of the new scope
|
|
1408
|
+
* entries back to a "needs scraping" state so that the next crawl picks them up
|
|
1409
|
+
* as full internal pages.
|
|
1410
|
+
*
|
|
1411
|
+
* For each matching page:
|
|
1412
|
+
* - clears the scrape metadata (status, headers, snapshot path, etc.),
|
|
1413
|
+
* - flips `isExternal` to `0` and `scraped` to `0`,
|
|
1414
|
+
* - removes stale `anchors`, `images`, and `resources-referrers` rows so that
|
|
1415
|
+
* the re-scrape can re-insert fresh ones without duplicates.
|
|
1416
|
+
*
|
|
1417
|
+
* The page row itself is kept (id is preserved) so existing referrers via
|
|
1418
|
+
* `anchors.hrefId` remain valid. SELECT and UPDATE/DELETE statements are
|
|
1419
|
+
* chunked to stay below SQLite's `SQLITE_LIMIT_VARIABLE_NUMBER`.
|
|
1420
|
+
* @param scopes - The hostname-indexed scope map after the new roots are merged.
|
|
1421
|
+
* @param options - URL parsing options forwarded to {@link findScopeEntry}.
|
|
1422
|
+
* @returns The URLs of the pages that were promoted.
|
|
1423
|
+
*/
|
|
1424
|
+
async repromoteExternalPages(scopes, options) {
|
|
1425
|
+
return emitErrorAndRetry(this, 'Database.repromoteExternalPages', async () => {
|
|
1069
1426
|
if (scopes.size === 0) {
|
|
1070
1427
|
return [];
|
|
1071
1428
|
}
|
|
@@ -1134,36 +1491,51 @@ let Database = (() => {
|
|
|
1134
1491
|
}
|
|
1135
1492
|
dbLog('Repromoted %d external pages back to pending', promotedUrls.length);
|
|
1136
1493
|
return promotedUrls;
|
|
1137
|
-
}
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
|
|
1161
|
-
|
|
1162
|
-
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1166
|
-
|
|
1494
|
+
}, retrySetting);
|
|
1495
|
+
}
|
|
1496
|
+
/**
|
|
1497
|
+
* Reset previously-attempted pages that ended in a recoverable failure so a
|
|
1498
|
+
* follow-up crawl can re-fetch them from scratch.
|
|
1499
|
+
*
|
|
1500
|
+
* A page qualifies as a recoverable failure when it was already scraped
|
|
1501
|
+
* (`scraped = 1`), is not a redirect source (`redirectDestId IS NULL`), was
|
|
1502
|
+
* not intentionally skipped (`isSkipped` is not `1`), and one of the
|
|
1503
|
+
* following holds:
|
|
1504
|
+
*
|
|
1505
|
+
* - `status = -1` — the sentinel a hard scrape failure (network error,
|
|
1506
|
+
* timeout, browser crash) is recorded with (see `handle-scrape-error.ts`);
|
|
1507
|
+
* - `status IS NULL` — no status was ever stored for the row;
|
|
1508
|
+
* - `contentType IS NULL` — the content type could not be determined;
|
|
1509
|
+
* - `status` is in the `5xx` range — a (frequently transient) server error.
|
|
1510
|
+
*
|
|
1511
|
+
* Definitive `4xx` responses are intentionally excluded: re-fetching a 404
|
|
1512
|
+
* almost always yields the same answer.
|
|
1513
|
+
*
|
|
1514
|
+
* A second exclusion runs in JS after the SQL candidate scan: any page whose
|
|
1515
|
+
* latest recorded `page_errors` / `crawl_errors` message classifies into a
|
|
1516
|
+
* permanent {@link PERMANENT_ERROR_KINDS} kind (dns / tls / client-blocked /
|
|
1517
|
+
* parse-error / connection-refused) is left as-is rather than reset to
|
|
1518
|
+
* pending. Without this filter, `--retry-failed` never converges: NXDOMAIN
|
|
1519
|
+
* hosts, expired-cert hosts, and `ERR_BLOCKED_BY_CLIENT` ad pixels would be
|
|
1520
|
+
* reset every iteration, re-attempted, fail identically, and rejoin the
|
|
1521
|
+
* candidate pool for the next iteration. The exclusion keeps the retry
|
|
1522
|
+
* target shrinking across `--retry-failed` passes by leaving deterministic
|
|
1523
|
+
* dead-ends alone.
|
|
1524
|
+
*
|
|
1525
|
+
* Matching rows — internal and external alike — are demoted back to pending
|
|
1526
|
+
* (`scraped = 0`) and have their stale scrape metadata cleared. The page row
|
|
1527
|
+
* itself is kept (id preserved) so existing `anchors.hrefId` referrers stay
|
|
1528
|
+
* valid, and `isExternal` is left untouched so the next pass re-classifies
|
|
1529
|
+
* each page from the crawl scope. Related `anchors`, `images`,
|
|
1530
|
+
* `resources-referrers`, and `page_errors` rows are deleted so the re-scrape
|
|
1531
|
+
* can re-insert fresh data without duplicates.
|
|
1532
|
+
*
|
|
1533
|
+
* SELECT and UPDATE/DELETE statements are chunked to stay below SQLite's
|
|
1534
|
+
* `SQLITE_LIMIT_VARIABLE_NUMBER`.
|
|
1535
|
+
* @returns The URLs of the pages that were reset to pending.
|
|
1536
|
+
*/
|
|
1537
|
+
async resetFailedPages() {
|
|
1538
|
+
return emitErrorAndRetry(this, 'Database.resetFailedPages', async () => {
|
|
1167
1539
|
const candidates = await this.#instance
|
|
1168
1540
|
.select('id', 'url')
|
|
1169
1541
|
.from('pages')
|
|
@@ -1181,8 +1553,28 @@ let Database = (() => {
|
|
|
1181
1553
|
if (candidates.length === 0) {
|
|
1182
1554
|
return [];
|
|
1183
1555
|
}
|
|
1184
|
-
const
|
|
1185
|
-
const
|
|
1556
|
+
const candidateIds = candidates.map((row) => row.id);
|
|
1557
|
+
const candidateUrls = candidates.map((row) => row.url);
|
|
1558
|
+
const messages = await getFailedPageMessages(this.#instance, candidateIds, candidateUrls);
|
|
1559
|
+
// Drop candidates whose latest recorded message classifies as permanent.
|
|
1560
|
+
// An empty/absent message stays in the retry pool — we keep retrying when
|
|
1561
|
+
// we don't know it's permanent, erring on the side of investigation.
|
|
1562
|
+
const retryable = candidates.filter((row) => {
|
|
1563
|
+
const message = messages.get(row.id) ?? '';
|
|
1564
|
+
if (message === '') {
|
|
1565
|
+
return true;
|
|
1566
|
+
}
|
|
1567
|
+
return !PERMANENT_ERROR_KINDS.has(classifyErrorKind(message));
|
|
1568
|
+
});
|
|
1569
|
+
const excludedCount = candidates.length - retryable.length;
|
|
1570
|
+
if (excludedCount > 0) {
|
|
1571
|
+
dbLog('Excluded %d page(s) from retry — permanent failure kinds (dns/tls/client-blocked/parse-error/connection-refused)', excludedCount);
|
|
1572
|
+
}
|
|
1573
|
+
if (retryable.length === 0) {
|
|
1574
|
+
return [];
|
|
1575
|
+
}
|
|
1576
|
+
const ids = retryable.map((row) => row.id);
|
|
1577
|
+
const urls = retryable.map((row) => row.url);
|
|
1186
1578
|
const chunkSize = 500;
|
|
1187
1579
|
const metaReset = makeMetaResetPayload();
|
|
1188
1580
|
for (let i = 0; i < ids.length; i += chunkSize) {
|
|
@@ -1220,16 +1612,18 @@ let Database = (() => {
|
|
|
1220
1612
|
}
|
|
1221
1613
|
dbLog('Reset %d failed pages back to pending', urls.length);
|
|
1222
1614
|
return urls;
|
|
1223
|
-
}
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1229
|
-
|
|
1230
|
-
|
|
1231
|
-
|
|
1232
|
-
|
|
1615
|
+
}, retrySetting);
|
|
1616
|
+
}
|
|
1617
|
+
/**
|
|
1618
|
+
* Stores the crawl configuration in the `info` table.
|
|
1619
|
+
* Only fields in {@link INFO_COLUMN_ALLOWLIST} are forwarded — any extra
|
|
1620
|
+
* runtime-only field on the input is silently dropped so callers can splat
|
|
1621
|
+
* a wider config object without producing SQL errors. JSON-array fields
|
|
1622
|
+
* are serialized via `JSON.stringify`.
|
|
1623
|
+
* @param config - The {@link Config} object to store.
|
|
1624
|
+
*/
|
|
1625
|
+
async setConfig(config) {
|
|
1626
|
+
return emitErrorAndRetry(this, 'Database.setConfig', async () => {
|
|
1233
1627
|
const payload = {};
|
|
1234
1628
|
for (const [key, value] of Object.entries(config)) {
|
|
1235
1629
|
if (!INFO_COLUMN_ALLOWLIST.has(key)) {
|
|
@@ -1238,15 +1632,17 @@ let Database = (() => {
|
|
|
1238
1632
|
payload[key] = INFO_JSON_COLUMNS.has(key) ? JSON.stringify(value) : value;
|
|
1239
1633
|
}
|
|
1240
1634
|
return this.#instance.from('info').insert(payload);
|
|
1241
|
-
}
|
|
1242
|
-
|
|
1243
|
-
|
|
1244
|
-
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1248
|
-
|
|
1249
|
-
|
|
1635
|
+
}, retrySetting);
|
|
1636
|
+
}
|
|
1637
|
+
/**
|
|
1638
|
+
* Marks a page as skipped in the database with the given reason.
|
|
1639
|
+
* Creates the page row if it does not already exist.
|
|
1640
|
+
* @param url - The URL of the skipped page.
|
|
1641
|
+
* @param reason - The reason the page was skipped.
|
|
1642
|
+
* @param isExternal - Whether the page is on an external domain. Defaults to `false`.
|
|
1643
|
+
*/
|
|
1644
|
+
async setSkippedPage(url, reason, isExternal = false) {
|
|
1645
|
+
return emitErrorAndRetry(this, 'Database.setSkippedPage', async () => {
|
|
1250
1646
|
const pageId = await this.#getIdByUrl(url, isExternal ? 1 : 0);
|
|
1251
1647
|
await this.#instance('pages')
|
|
1252
1648
|
.where('id', pageId)
|
|
@@ -1256,48 +1652,50 @@ let Database = (() => {
|
|
|
1256
1652
|
isSkipped: 1,
|
|
1257
1653
|
skipReason: reason,
|
|
1258
1654
|
});
|
|
1655
|
+
}, retrySetting);
|
|
1656
|
+
}
|
|
1657
|
+
/**
|
|
1658
|
+
* Assigns natural URL sort order values to all internal pages.
|
|
1659
|
+
* Pages are sorted using {@link pathComparator} and assigned sequential order numbers.
|
|
1660
|
+
*/
|
|
1661
|
+
async setUrlOrder() {
|
|
1662
|
+
dbLog('Set URL Order');
|
|
1663
|
+
const res = await this.#instance
|
|
1664
|
+
.select('id', 'url')
|
|
1665
|
+
.from('pages')
|
|
1666
|
+
.where('isExternal', '=', 0);
|
|
1667
|
+
const sorted = res.toSorted((a, b) => pathComparator(a.url, b.url));
|
|
1668
|
+
// Batch update using chunked CASE statements to avoid N+1 queries
|
|
1669
|
+
const BATCH_SIZE = 500;
|
|
1670
|
+
for (let i = 0; i < sorted.length; i += BATCH_SIZE) {
|
|
1671
|
+
const batch = sorted.slice(i, i + BATCH_SIZE);
|
|
1672
|
+
const ids = batch.map((row) => row.id);
|
|
1673
|
+
const bindings = [];
|
|
1674
|
+
const cases = batch
|
|
1675
|
+
.map((row, j) => {
|
|
1676
|
+
bindings.push(row.id, i + j + 1);
|
|
1677
|
+
return 'WHEN ? THEN ?';
|
|
1678
|
+
})
|
|
1679
|
+
.join(' ');
|
|
1680
|
+
const placeholders = ids.map(() => '?').join(',');
|
|
1681
|
+
await this.#instance.raw(`UPDATE pages SET \`order\` = CASE id ${cases} END WHERE id IN (${placeholders})`, [...bindings, ...ids]);
|
|
1259
1682
|
}
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
const bindings = [];
|
|
1277
|
-
const cases = batch
|
|
1278
|
-
.map((row, j) => {
|
|
1279
|
-
bindings.push(row.id, i + j + 1);
|
|
1280
|
-
return 'WHEN ? THEN ?';
|
|
1281
|
-
})
|
|
1282
|
-
.join(' ');
|
|
1283
|
-
const placeholders = ids.map(() => '?').join(',');
|
|
1284
|
-
await this.#instance.raw(`UPDATE pages SET \`order\` = CASE id ${cases} END WHERE id IN (${placeholders})`, [...bindings, ...ids]);
|
|
1285
|
-
}
|
|
1286
|
-
}
|
|
1287
|
-
/**
|
|
1288
|
-
* Update the single row in the `info` table with a partial config patch.
|
|
1289
|
-
*
|
|
1290
|
-
* Used by the append flow to extend `roots` (and any other tweakable
|
|
1291
|
-
* field) without replacing the entire row. JSON-array fields are serialized on
|
|
1292
|
-
* the fly; primitive fields are written verbatim. Unspecified fields stay as-is.
|
|
1293
|
-
*
|
|
1294
|
-
* Unknown keys (anything outside the allow-list of `info`-table columns) are
|
|
1295
|
-
* silently dropped instead of being passed to SQL, so callers that splat a
|
|
1296
|
-
* wider runtime config (e.g. `CrawlConfig` with `cwd` / `executablePath`)
|
|
1297
|
-
* cannot accidentally trigger a "no such column" SQL error.
|
|
1298
|
-
* @param patch - Partial {@link Config} fields to overwrite. `undefined` values are skipped.
|
|
1299
|
-
*/
|
|
1300
|
-
async updateConfig(patch) {
|
|
1683
|
+
}
|
|
1684
|
+
/**
|
|
1685
|
+
* Update the single row in the `info` table with a partial config patch.
|
|
1686
|
+
*
|
|
1687
|
+
* Used by the append flow to extend `roots` (and any other tweakable
|
|
1688
|
+
* field) without replacing the entire row. JSON-array fields are serialized on
|
|
1689
|
+
* the fly; primitive fields are written verbatim. Unspecified fields stay as-is.
|
|
1690
|
+
*
|
|
1691
|
+
* Unknown keys (anything outside the allow-list of `info`-table columns) are
|
|
1692
|
+
* silently dropped instead of being passed to SQL, so callers that splat a
|
|
1693
|
+
* wider runtime config (e.g. `CrawlConfig` with `cwd` / `executablePath`)
|
|
1694
|
+
* cannot accidentally trigger a "no such column" SQL error.
|
|
1695
|
+
* @param patch - Partial {@link Config} fields to overwrite. `undefined` values are skipped.
|
|
1696
|
+
*/
|
|
1697
|
+
async updateConfig(patch) {
|
|
1698
|
+
return emitErrorAndRetry(this, 'Database.updateConfig', async () => {
|
|
1301
1699
|
const payload = {};
|
|
1302
1700
|
for (const [key, value] of Object.entries(patch)) {
|
|
1303
1701
|
if (value === undefined) {
|
|
@@ -1316,28 +1714,30 @@ let Database = (() => {
|
|
|
1316
1714
|
return;
|
|
1317
1715
|
}
|
|
1318
1716
|
await this.#instance.from('info').update(payload);
|
|
1319
|
-
}
|
|
1320
|
-
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1325
|
-
|
|
1326
|
-
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1330
|
-
|
|
1331
|
-
|
|
1332
|
-
|
|
1333
|
-
|
|
1334
|
-
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
|
|
1717
|
+
}, retrySetting);
|
|
1718
|
+
}
|
|
1719
|
+
/**
|
|
1720
|
+
* Inserts or updates a crawled page in the database, including its redirect chain,
|
|
1721
|
+
* anchors, images, and (when `writeHtml`) its compressed HTML snapshot BLOB.
|
|
1722
|
+
*
|
|
1723
|
+
* Self-redirects (where the source URL equals the destination URL after normalization)
|
|
1724
|
+
* are skipped to avoid marking a page as redirected to itself — a situation caused by
|
|
1725
|
+
* authentication challenges (e.g. Basic Auth 302) that would otherwise exclude the page
|
|
1726
|
+
* from reports via the `whereNull('redirectDestId')` filter.
|
|
1727
|
+
* @param page - The page data to store.
|
|
1728
|
+
* @param writeHtml - When `true`, this call is allowed to insert (or clear)
|
|
1729
|
+
* the page's HTML blob. `setExternalPage` passes `false` because external
|
|
1730
|
+
* metadata-only scrapes never carry HTML and must not perturb an already
|
|
1731
|
+
* stored body.
|
|
1732
|
+
* @param isTarget - Whether this page is a crawl target.
|
|
1733
|
+
* @param source - Provenance label written ONLY when the row is freshly
|
|
1734
|
+
* inserted. Existing rows keep their original `source` (this is why a
|
|
1735
|
+
* second `crawl --inventory` does not "demote" an `'inventory-seed'` row
|
|
1736
|
+
* that was discovered earlier).
|
|
1737
|
+
* @returns The database `pageId` of the inserted/updated row.
|
|
1738
|
+
*/
|
|
1739
|
+
async updatePage(page, writeHtml, isTarget, source) {
|
|
1740
|
+
return emitErrorAndRetry(this, 'Database.updatePage', async () => {
|
|
1341
1741
|
const { destUrl, sources } = resolveRedirectChain(page.url.withoutHashAndAuth, page.redirectPaths);
|
|
1342
1742
|
const destUrlObject = parseUrl(destUrl);
|
|
1343
1743
|
if (!destUrlObject) {
|
|
@@ -1359,7 +1759,24 @@ let Database = (() => {
|
|
|
1359
1759
|
if (writeHtml) {
|
|
1360
1760
|
await this.#insertJsonLd(pageId, page.meta, trx);
|
|
1361
1761
|
}
|
|
1362
|
-
|
|
1762
|
+
// Chain lineage propagates FROM the originating URL
|
|
1763
|
+
// (`page.url`), NOT from the destination. See the matching
|
|
1764
|
+
// rationale in `recordRedirect` above: intermediates are
|
|
1765
|
+
// reached transitively from the originating URL's render,
|
|
1766
|
+
// so they inherit its lineage. The `source` argument is the
|
|
1767
|
+
// authoritative origin label when inventoryMode is live;
|
|
1768
|
+
// fall through to a DB lookup of `page.url` for the resume
|
|
1769
|
+
// / retry-failed path where the call-site has no source.
|
|
1770
|
+
let originatingSource = source;
|
|
1771
|
+
if (originatingSource === undefined) {
|
|
1772
|
+
const [originatingRow] = await trx
|
|
1773
|
+
.select('source')
|
|
1774
|
+
.from('pages')
|
|
1775
|
+
.where('url', page.url.withoutHashAndAuth);
|
|
1776
|
+
originatingSource = originatingRow?.source;
|
|
1777
|
+
}
|
|
1778
|
+
const chainLineageSource = deriveLineageFromParent(originatingSource, 'crawled');
|
|
1779
|
+
await this.#linkRedirectSources(trx, sources, pageId, destUrlObject.withoutHashAndAuth, page.isExternal, chainLineageSource);
|
|
1363
1780
|
// Only insert a snapshot blob when there is actual HTML to write.
|
|
1364
1781
|
// `page.html.length > 0` is the precise signal: the scraper returns
|
|
1365
1782
|
// `html: ''` for everything that is not a rendered `text/html` document
|
|
@@ -1411,8 +1828,35 @@ let Database = (() => {
|
|
|
1411
1828
|
// duplication, but multiple distinct anchors can share the same
|
|
1412
1829
|
// hrefId/hash/textContent legitimately, so there is no natural unique
|
|
1413
1830
|
// key to enforce — replace-on-write is the correct mechanism here.)
|
|
1831
|
+
// Lineage propagation: read the current page's merged source
|
|
1832
|
+
// (post-UPDATE by `#insertPage`) so anchor placeholder rows
|
|
1833
|
+
// inherit a label that reflects the parent's chain. A
|
|
1834
|
+
// `'crawled'`-lineage parent passes `'crawled'` explicitly so the
|
|
1835
|
+
// crawled-wins downgrade in `#getIdByUrl` fires when an anchor
|
|
1836
|
+
// hits an existing `'inventory-*'` row. An inventory-lineage
|
|
1837
|
+
// parent passes `'inventory-discovered'` to label transitively-
|
|
1838
|
+
// reached URLs correctly without the orchestrator needing to
|
|
1839
|
+
// rehydrate `inventoryMode` from disk.
|
|
1840
|
+
//
|
|
1841
|
+
// Cost: one extra SELECT on `pages` per scraped page (the
|
|
1842
|
+
// `id` is a PK index lookup so it is sub-millisecond even at
|
|
1843
|
+
// 1M-row scale). The alternative — passing `mergedSource`
|
|
1844
|
+
// through from the UPDATE result — would require RETURNING
|
|
1845
|
+
// support that knex's SQLite dialect handles inconsistently;
|
|
1846
|
+
// the small per-page round-trip is the cheaper trade.
|
|
1847
|
+
const [parentRow] = await trx
|
|
1848
|
+
.select('source')
|
|
1849
|
+
.from('pages')
|
|
1850
|
+
.where('id', pageId);
|
|
1851
|
+
// `deriveLineageFromParent` collapses the three call sites
|
|
1852
|
+
// (anchor / redirect intermediate × updatePage / recordRedirect)
|
|
1853
|
+
// onto the same rule. `'crawled'` fallback (vs `undefined`)
|
|
1854
|
+
// arms the crawled-wins downgrade in `#getIdByUrl` for
|
|
1855
|
+
// existing `'inventory-*'` rows reached from a crawled
|
|
1856
|
+
// parent — see `isInventorySource` for the membership rule.
|
|
1857
|
+
const anchorLineageSource = deriveLineageFromParent(parentRow?.source, 'crawled');
|
|
1414
1858
|
const anchors = await Promise.all(page.anchorList.map(async (anchor) => {
|
|
1415
|
-
const hrefId = await this.#getIdByUrl(anchor.href.withoutHashAndAuth, anchor.isExternal ? 1 : 0, trx);
|
|
1859
|
+
const hrefId = await this.#getIdByUrl(anchor.href.withoutHashAndAuth, anchor.isExternal ? 1 : 0, trx, anchorLineageSource);
|
|
1416
1860
|
return {
|
|
1417
1861
|
pageId,
|
|
1418
1862
|
hrefId,
|
|
@@ -1440,352 +1884,429 @@ let Database = (() => {
|
|
|
1440
1884
|
}
|
|
1441
1885
|
return pageId;
|
|
1442
1886
|
});
|
|
1443
|
-
}
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1447
|
-
|
|
1448
|
-
|
|
1449
|
-
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
|
|
1456
|
-
|
|
1457
|
-
|
|
1458
|
-
|
|
1459
|
-
|
|
1460
|
-
|
|
1461
|
-
|
|
1462
|
-
|
|
1463
|
-
|
|
1464
|
-
|
|
1465
|
-
|
|
1466
|
-
|
|
1467
|
-
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1472
|
-
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
if (!insertedId) {
|
|
1478
|
-
// onConflict.ignore() returns 0 on race condition — re-select
|
|
1479
|
-
const [existing] = await qb.select('id').from('pages').where('url', url);
|
|
1480
|
-
if (existing?.id) {
|
|
1481
|
-
return existing.id;
|
|
1482
|
-
}
|
|
1483
|
-
throw new Error(`Failed to insert a new page: ${url}`);
|
|
1887
|
+
}, retrySetting);
|
|
1888
|
+
}
|
|
1889
|
+
/**
|
|
1890
|
+
* Returns the database ID for a URL, creating a new page row if needed.
|
|
1891
|
+
* Uses `ON CONFLICT IGNORE` to handle race conditions in concurrent inserts.
|
|
1892
|
+
*
|
|
1893
|
+
* `source` is written ONLY on the INSERT path — when the row already
|
|
1894
|
+
* exists, we never reach the INSERT and the existing row's `source`
|
|
1895
|
+
* stays untouched. This is what keeps a second `crawl --inventory` from
|
|
1896
|
+
* "demoting" a page that was first labelled `'inventory-seed'` back to
|
|
1897
|
+
* `'inventory-discovered'` on later passes.
|
|
1898
|
+
* @param url
|
|
1899
|
+
* @param isExternal
|
|
1900
|
+
* @param trx
|
|
1901
|
+
* @param source - Provenance label to put on the newly-inserted row. `undefined` lets the DB DEFAULT (`'crawled'`) apply.
|
|
1902
|
+
*/
|
|
1903
|
+
async #getIdByUrl(url, isExternal, trx, source) {
|
|
1904
|
+
const qb = trx ?? this.#instance;
|
|
1905
|
+
const [record] = await qb
|
|
1906
|
+
.select('id', 'source')
|
|
1907
|
+
.from('pages')
|
|
1908
|
+
.where('url', url);
|
|
1909
|
+
// Must use `?` because it may be `undefined`
|
|
1910
|
+
const pageId = record?.id ?? Number.NaN;
|
|
1911
|
+
if (Number.isFinite(pageId)) {
|
|
1912
|
+
// Crawled-wins downgrade: when a row that was previously labelled
|
|
1913
|
+
// `'inventory-seed'` or `'inventory-discovered'` is re-encountered
|
|
1914
|
+
// via a `'crawled'`-lineage anchor (the parent page is part of the
|
|
1915
|
+
// graph reachable from the original crawl roots), downgrade it to
|
|
1916
|
+
// `'crawled'`. The inventory goal is finding orphans — anything
|
|
1917
|
+
// reachable from the crawled chain is NOT an orphan and should
|
|
1918
|
+
// not retain an inventory label.
|
|
1919
|
+
if (source === 'crawled' && record?.source && record.source !== 'crawled') {
|
|
1920
|
+
await qb('pages').where('id', pageId).update({ source: 'crawled' });
|
|
1484
1921
|
}
|
|
1485
|
-
return
|
|
1922
|
+
return pageId;
|
|
1486
1923
|
}
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
|
|
1497
|
-
|
|
1498
|
-
|
|
1499
|
-
//
|
|
1500
|
-
|
|
1501
|
-
|
|
1502
|
-
|
|
1503
|
-
// Reject pre-0.10 archives before any further work. Runs for both
|
|
1504
|
-
// writer and read-only (stub viewer) connections so old
|
|
1505
|
-
// `._nitpicker-*` stubs surface a clear error instead of
|
|
1506
|
-
// dereferencing missing columns at query time. New archives (no
|
|
1507
|
-
// `info` table yet) pass through; the schema is filled in by
|
|
1508
|
-
// `initSchema` below.
|
|
1509
|
-
await assertCompatibleVersion(this.#instance);
|
|
1510
|
-
if (readOnly) {
|
|
1511
|
-
return;
|
|
1924
|
+
const insertedRows = await qb('pages')
|
|
1925
|
+
.insert({
|
|
1926
|
+
url,
|
|
1927
|
+
scraped: 0,
|
|
1928
|
+
isTarget: 0,
|
|
1929
|
+
...(isExternal != null && { isExternal }),
|
|
1930
|
+
...(source === undefined ? {} : { source }),
|
|
1931
|
+
})
|
|
1932
|
+
.onConflict('url')
|
|
1933
|
+
.ignore();
|
|
1934
|
+
const [insertedId] = insertedRows;
|
|
1935
|
+
if (!insertedId) {
|
|
1936
|
+
// onConflict.ignore() returns 0 on race condition — re-select
|
|
1937
|
+
const [existing] = await qb.select('id').from('pages').where('url', url);
|
|
1938
|
+
if (existing?.id) {
|
|
1939
|
+
return existing.id;
|
|
1512
1940
|
}
|
|
1513
|
-
|
|
1514
|
-
await migrateInfoRoots(this.#instance);
|
|
1515
|
-
await migratePageErrors(this.#instance);
|
|
1516
|
-
await migrateCrawlErrors(this.#instance);
|
|
1517
|
-
await migrateHtmlBlobTables(this.#instance);
|
|
1518
|
-
await migratePagesResourcesSource(this.#instance);
|
|
1941
|
+
throw new Error(`Failed to insert a new page: ${url}`);
|
|
1519
1942
|
}
|
|
1520
|
-
|
|
1521
|
-
|
|
1522
|
-
|
|
1523
|
-
|
|
1524
|
-
|
|
1525
|
-
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
|
|
1537
|
-
|
|
1538
|
-
|
|
1539
|
-
|
|
1540
|
-
|
|
1541
|
-
|
|
1542
|
-
|
|
1543
|
-
|
|
1544
|
-
|
|
1545
|
-
|
|
1546
|
-
|
|
1547
|
-
parseError: entry.parseError ?? null,
|
|
1548
|
-
});
|
|
1549
|
-
}
|
|
1550
|
-
for (const entry of speculationRules) {
|
|
1551
|
-
rows.push({
|
|
1552
|
-
pageId,
|
|
1553
|
-
kind: 'speculationrules',
|
|
1554
|
-
type: classifyJsonLdType(entry),
|
|
1555
|
-
raw: entry.raw,
|
|
1556
|
-
parsed: entry.parsed === undefined ? null : JSON.stringify(entry.parsed),
|
|
1557
|
-
parseError: entry.parseError ?? null,
|
|
1558
|
-
});
|
|
1559
|
-
}
|
|
1560
|
-
if (rows.length === 0)
|
|
1561
|
-
return;
|
|
1562
|
-
await trx('page_jsonld').where('pageId', pageId).delete();
|
|
1563
|
-
await eachSplitted(rows, 100, async (chunk) => {
|
|
1564
|
-
await trx('page_jsonld').insert(chunk);
|
|
1565
|
-
});
|
|
1943
|
+
return insertedId;
|
|
1944
|
+
}
|
|
1945
|
+
/**
|
|
1946
|
+
* Initializes the database schema if tables do not exist, then runs lightweight
|
|
1947
|
+
* migrations that bring older archives up to the current schema.
|
|
1948
|
+
*
|
|
1949
|
+
* Migrations are idempotent and run on every writer-side {@link Database.connect};
|
|
1950
|
+
* in read-only mode they are SKIPPED so the same DB can be opened safely
|
|
1951
|
+
* by a viewer attached to a live (or interrupted) crawl without rewriting
|
|
1952
|
+
* the user's tmpDir.
|
|
1953
|
+
* @param readOnly - When true, skip schema init + migrations.
|
|
1954
|
+
*/
|
|
1955
|
+
async #init(readOnly) {
|
|
1956
|
+
// Connection-level PRAGMAs (foreign_keys, mmap_size, …) must be
|
|
1957
|
+
// reapplied on every connect — they are not persisted across opens.
|
|
1958
|
+
// They are safe in read-only mode because they don't write to the
|
|
1959
|
+
// user's tmpDir, just configure the libsql connection.
|
|
1960
|
+
await applyConnectionPragmas(this.#instance);
|
|
1961
|
+
// Reject pre-0.10 archives before any further work. Runs for both
|
|
1962
|
+
// writer and read-only (stub viewer) connections so old
|
|
1963
|
+
// `._nitpicker-*` stubs surface a clear error instead of
|
|
1964
|
+
// dereferencing missing columns at query time. New archives (no
|
|
1965
|
+
// `info` table yet) pass through; the schema is filled in by
|
|
1966
|
+
// `initSchema` below.
|
|
1967
|
+
await assertCompatibleVersion(this.#instance);
|
|
1968
|
+
if (readOnly) {
|
|
1969
|
+
return;
|
|
1566
1970
|
}
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1571
|
-
|
|
1572
|
-
|
|
1573
|
-
|
|
1574
|
-
|
|
1575
|
-
|
|
1576
|
-
|
|
1577
|
-
|
|
1578
|
-
|
|
1579
|
-
|
|
1580
|
-
|
|
1581
|
-
|
|
1582
|
-
|
|
1583
|
-
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
|
|
1587
|
-
|
|
1588
|
-
|
|
1589
|
-
|
|
1590
|
-
|
|
1591
|
-
|
|
1592
|
-
|
|
1593
|
-
|
|
1594
|
-
|
|
1595
|
-
|
|
1596
|
-
|
|
1597
|
-
|
|
1598
|
-
|
|
1599
|
-
|
|
1600
|
-
|
|
1601
|
-
.
|
|
1602
|
-
.
|
|
1603
|
-
scraped: true,
|
|
1604
|
-
isTarget,
|
|
1605
|
-
isExternal: page.isExternal,
|
|
1606
|
-
status: page.status,
|
|
1607
|
-
statusText: page.statusText,
|
|
1608
|
-
// Canonicalize so the stored value matches the exact-string page-ness
|
|
1609
|
-
// predicate (`WHERE contentType = 'text/html'`) used by the read layer
|
|
1610
|
-
// and the case-insensitive `isHtmlContentType` used in code. Responses
|
|
1611
|
-
// are recorded verbatim upstream, so `Text/HTML` / `text/html ` can
|
|
1612
|
-
// otherwise be stored and silently misclassified.
|
|
1613
|
-
contentType: normalizeContentType(page.contentType),
|
|
1614
|
-
contentLength: page.contentLength,
|
|
1615
|
-
responseHeaders: JSON.stringify(page.responseHeaders),
|
|
1616
|
-
// Flat meta columns derived from beholder 3.0.0 nested Meta.
|
|
1617
|
-
// URL-shaped columns (canonical / og_url / og_image / amphtml / manifest /
|
|
1618
|
-
// icon_href / appleTouchIcon_href / twitter_image) are already absolutised
|
|
1619
|
-
// by `deriveFlatFromMeta` against the page URL — `find-mismatches` compares
|
|
1620
|
-
// `canonical != url` directly, so storing the raw `getAttribute('href')`
|
|
1621
|
-
// would generate false positives for sites using relative canonicals.
|
|
1622
|
-
...flat,
|
|
1623
|
-
// Denormalised aggregates: written once at scrape time so list reads
|
|
1624
|
-
// (Sheets, page-detail summary) can answer "how many JSON-LD entries?"
|
|
1625
|
-
// and "which Wappalyzer providers?" by selecting a single pages column
|
|
1626
|
-
// rather than running a GROUP BY join on every read.
|
|
1627
|
-
tag_count: denorm.tag_count,
|
|
1628
|
-
jsonld_count: denorm.jsonld_count,
|
|
1629
|
-
tags_providers_csv: denorm.tags_providers_csv,
|
|
1630
|
-
// JSON catch-all for nested Meta sub-objects not flattened above.
|
|
1631
|
-
meta_extras: JSON.stringify(extras),
|
|
1632
|
-
// Timestamps: `firstCrawledAt` is set only on first INSERT — `COALESCE`
|
|
1633
|
-
// preserves the existing value so a re-scrape (`--append`, `--retry-failed`)
|
|
1634
|
-
// does not erase the discovery time. `lastCrawledAt` is updated every
|
|
1635
|
-
// successful scrape.
|
|
1636
|
-
firstCrawledAt: qb.raw('COALESCE(firstCrawledAt, ?)', [now]),
|
|
1637
|
-
lastCrawledAt: now,
|
|
1638
|
-
isSkipped: page.isSkipped,
|
|
1639
|
-
...sourceUpdate,
|
|
1971
|
+
await initSchema(this.#instance);
|
|
1972
|
+
await migrateInfoRoots(this.#instance);
|
|
1973
|
+
await migratePageErrors(this.#instance);
|
|
1974
|
+
await migrateCrawlErrors(this.#instance);
|
|
1975
|
+
await migrateHtmlBlobTables(this.#instance);
|
|
1976
|
+
await migratePagesResourcesSource(this.#instance);
|
|
1977
|
+
await migrateInventoryRuns(this.#instance);
|
|
1978
|
+
}
|
|
1979
|
+
/**
|
|
1980
|
+
* Replaces the page's JSON-LD / SpeculationRules rows with the freshly
|
|
1981
|
+
* captured set. Called inside `updatePage`'s transaction.
|
|
1982
|
+
*
|
|
1983
|
+
* `writeHtml = false` branches (`setExternalPage`, metadata-only) skip
|
|
1984
|
+
* this entirely — JSON-LD lives inside the HTML body, so external pages
|
|
1985
|
+
* that are not rendered have no entries to write. An empty array on a
|
|
1986
|
+
* normally-rendered page is treated as a degraded re-scrape: prior rows
|
|
1987
|
+
* are kept (same `delete-only-when-replacing` invariant as `anchors` /
|
|
1988
|
+
* `images`).
|
|
1989
|
+
* @param pageId
|
|
1990
|
+
* @param meta
|
|
1991
|
+
* @param trx
|
|
1992
|
+
*/
|
|
1993
|
+
async #insertJsonLd(pageId, meta, trx) {
|
|
1994
|
+
// `??` guards tolerate the legacy "minimal meta" shape from older test
|
|
1995
|
+
// fixtures. Real beholder 3.0.0 always populates these required fields.
|
|
1996
|
+
const jsonLd = meta.jsonLd ?? [];
|
|
1997
|
+
const speculationRules = meta.speculationRules ?? [];
|
|
1998
|
+
const rows = [];
|
|
1999
|
+
for (const entry of jsonLd) {
|
|
2000
|
+
rows.push({
|
|
2001
|
+
pageId,
|
|
2002
|
+
kind: 'ld+json',
|
|
2003
|
+
type: classifyJsonLdType(entry),
|
|
2004
|
+
raw: entry.raw,
|
|
2005
|
+
parsed: entry.parsed === undefined ? null : JSON.stringify(entry.parsed),
|
|
2006
|
+
parseError: entry.parseError ?? null,
|
|
1640
2007
|
});
|
|
1641
|
-
return pageId;
|
|
1642
2008
|
}
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
* Called inside `updatePage`'s transaction unconditionally — tag
|
|
1646
|
-
* detection draws on `<script src>` / `<iframe src>` / window globals /
|
|
1647
|
-
* response headers, not the HTML body, so external pages that skip
|
|
1648
|
-
* rendering still contribute tags.
|
|
1649
|
-
*
|
|
1650
|
-
* Same empty-guard as `#insertJsonLd`: an empty array does not wipe
|
|
1651
|
-
* prior rows on a degraded re-scrape.
|
|
1652
|
-
* @param pageId
|
|
1653
|
-
* @param meta
|
|
1654
|
-
* @param trx
|
|
1655
|
-
*/
|
|
1656
|
-
async #insertTags(pageId, meta, trx) {
|
|
1657
|
-
const partial = extractTagsForArchive(meta.tags);
|
|
1658
|
-
if (partial.length === 0)
|
|
1659
|
-
return;
|
|
1660
|
-
const rows = partial.map((p) => ({
|
|
2009
|
+
for (const entry of speculationRules) {
|
|
2010
|
+
rows.push({
|
|
1661
2011
|
pageId,
|
|
1662
|
-
|
|
1663
|
-
|
|
1664
|
-
|
|
1665
|
-
|
|
1666
|
-
|
|
1667
|
-
categories: JSON.stringify(p.categories),
|
|
1668
|
-
sources: JSON.stringify(p.sources),
|
|
1669
|
-
}));
|
|
1670
|
-
await trx('page_tags').where('pageId', pageId).delete();
|
|
1671
|
-
await eachSplitted(rows, 100, async (chunk) => {
|
|
1672
|
-
await trx('page_tags').insert(chunk);
|
|
2012
|
+
kind: 'speculationrules',
|
|
2013
|
+
type: classifyJsonLdType(entry),
|
|
2014
|
+
raw: entry.raw,
|
|
2015
|
+
parsed: entry.parsed === undefined ? null : JSON.stringify(entry.parsed),
|
|
2016
|
+
parseError: entry.parseError ?? null,
|
|
1673
2017
|
});
|
|
1674
2018
|
}
|
|
1675
|
-
|
|
1676
|
-
|
|
1677
|
-
|
|
1678
|
-
|
|
1679
|
-
|
|
1680
|
-
|
|
1681
|
-
|
|
1682
|
-
|
|
1683
|
-
|
|
1684
|
-
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
|
|
1688
|
-
|
|
1689
|
-
|
|
1690
|
-
|
|
1691
|
-
|
|
1692
|
-
|
|
1693
|
-
|
|
1694
|
-
|
|
1695
|
-
|
|
1696
|
-
|
|
1697
|
-
|
|
1698
|
-
|
|
1699
|
-
|
|
1700
|
-
|
|
1701
|
-
|
|
1702
|
-
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
|
|
1706
|
-
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
2019
|
+
if (rows.length === 0)
|
|
2020
|
+
return;
|
|
2021
|
+
await trx('page_jsonld').where('pageId', pageId).delete();
|
|
2022
|
+
await eachSplitted(rows, 100, async (chunk) => {
|
|
2023
|
+
await trx('page_jsonld').insert(chunk);
|
|
2024
|
+
});
|
|
2025
|
+
}
|
|
2026
|
+
/**
|
|
2027
|
+
* Upserts page data into the `pages` table (inserts if new, updates if existing).
|
|
2028
|
+
*
|
|
2029
|
+
* `source` is intentionally NOT in the UPDATE clause — provenance is set
|
|
2030
|
+
* once at INSERT time inside `#getIdByUrl`, and existing rows keep
|
|
2031
|
+
* whatever label they were first inserted with.
|
|
2032
|
+
* @param page
|
|
2033
|
+
* @param isTarget
|
|
2034
|
+
* @param trx
|
|
2035
|
+
* @param source - Inventory provenance for the INSERT path. Ignored on UPDATE.
|
|
2036
|
+
*/
|
|
2037
|
+
async #insertPage(page, isTarget, trx, source) {
|
|
2038
|
+
const qb = trx ?? this.#instance;
|
|
2039
|
+
const pageId = await this.#getIdByUrl(page.url.withoutHashAndAuth, undefined, trx, source);
|
|
2040
|
+
const flat = deriveFlatFromMeta(page.meta, page.url.href);
|
|
2041
|
+
const denorm = computePageDenormalized(page.meta);
|
|
2042
|
+
const extras = deriveMetaExtras(page.meta);
|
|
2043
|
+
const now = Date.now();
|
|
2044
|
+
// Source priority on UPDATE: 'crawled' > 'inventory-seed' >
|
|
2045
|
+
// 'inventory-discovered'. The inventory feature exists to surface
|
|
2046
|
+
// orphans (= URLs NOT reachable from the original crawl roots).
|
|
2047
|
+
// Anything reachable via the crawled chain is therefore NOT an
|
|
2048
|
+
// orphan and must be labelled `'crawled'`, even if previously
|
|
2049
|
+
// labelled `'inventory-*'`. Within the inventory variants, the
|
|
2050
|
+
// explicit user-listed `'inventory-seed'` wins over the transitive
|
|
2051
|
+
// `'inventory-discovered'`.
|
|
2052
|
+
//
|
|
2053
|
+
// Note: in current callers, `source` only arrives as
|
|
2054
|
+
// `'inventory-seed'` / `'inventory-discovered'` / `undefined`
|
|
2055
|
+
// (`derivePageSource` never emits `'crawled'`, and outside inventory
|
|
2056
|
+
// mode `source` is `undefined` so this CASE never runs). The
|
|
2057
|
+
// `? = 'crawled'` branch is therefore reachable only via a future
|
|
2058
|
+
// call site that wants to explicitly assert a crawled lineage —
|
|
2059
|
+
// today the actual crawled-wins downgrade fires in `#getIdByUrl`'s
|
|
2060
|
+
// SELECT path when an anchor lineage `'crawled'` lands on an
|
|
2061
|
+
// existing `'inventory-*'` row. The branch is kept so the CASE
|
|
2062
|
+
// completely describes the priority lattice in one place.
|
|
2063
|
+
const sourceUpdate = source === undefined
|
|
2064
|
+
? {}
|
|
2065
|
+
: {
|
|
2066
|
+
source: qb.raw(`CASE
|
|
2067
|
+
WHEN source = 'crawled' OR ? = 'crawled' THEN 'crawled'
|
|
2068
|
+
WHEN source = 'inventory-seed' OR ? = 'inventory-seed' THEN 'inventory-seed'
|
|
2069
|
+
WHEN source = 'inventory-discovered' OR ? = 'inventory-discovered' THEN 'inventory-discovered'
|
|
2070
|
+
ELSE source
|
|
2071
|
+
END`, [source, source, source]),
|
|
2072
|
+
};
|
|
2073
|
+
await qb('pages')
|
|
2074
|
+
.where('id', pageId)
|
|
2075
|
+
.update({
|
|
2076
|
+
scraped: true,
|
|
2077
|
+
isTarget,
|
|
2078
|
+
isExternal: page.isExternal,
|
|
2079
|
+
status: page.status,
|
|
2080
|
+
statusText: page.statusText,
|
|
2081
|
+
// Canonicalize so the stored value matches the exact-string page-ness
|
|
2082
|
+
// predicate (`WHERE contentType = 'text/html'`) used by the read layer
|
|
2083
|
+
// and the case-insensitive `isHtmlContentType` used in code. Responses
|
|
2084
|
+
// are recorded verbatim upstream, so `Text/HTML` / `text/html ` can
|
|
2085
|
+
// otherwise be stored and silently misclassified.
|
|
2086
|
+
contentType: normalizeContentType(page.contentType),
|
|
2087
|
+
contentLength: page.contentLength,
|
|
2088
|
+
responseHeaders: JSON.stringify(page.responseHeaders),
|
|
2089
|
+
// Flat meta columns derived from beholder 3.0.0 nested Meta.
|
|
2090
|
+
// URL-shaped columns (canonical / og_url / og_image / amphtml / manifest /
|
|
2091
|
+
// icon_href / appleTouchIcon_href / twitter_image) are already absolutised
|
|
2092
|
+
// by `deriveFlatFromMeta` against the page URL — `find-mismatches` compares
|
|
2093
|
+
// `canonical != url` directly, so storing the raw `getAttribute('href')`
|
|
2094
|
+
// would generate false positives for sites using relative canonicals.
|
|
2095
|
+
...flat,
|
|
2096
|
+
// Denormalised aggregates: written once at scrape time so list reads
|
|
2097
|
+
// (Sheets, page-detail summary) can answer "how many JSON-LD entries?"
|
|
2098
|
+
// and "which Wappalyzer providers?" by selecting a single pages column
|
|
2099
|
+
// rather than running a GROUP BY join on every read.
|
|
2100
|
+
tag_count: denorm.tag_count,
|
|
2101
|
+
jsonld_count: denorm.jsonld_count,
|
|
2102
|
+
tags_providers_csv: denorm.tags_providers_csv,
|
|
2103
|
+
// JSON catch-all for nested Meta sub-objects not flattened above.
|
|
2104
|
+
meta_extras: JSON.stringify(extras),
|
|
2105
|
+
// Timestamps: `firstCrawledAt` is set only on first INSERT — `COALESCE`
|
|
2106
|
+
// preserves the existing value so a re-scrape (`--append`, `--retry-failed`)
|
|
2107
|
+
// does not erase the discovery time. `lastCrawledAt` is updated every
|
|
2108
|
+
// successful scrape.
|
|
2109
|
+
firstCrawledAt: qb.raw('COALESCE(firstCrawledAt, ?)', [now]),
|
|
2110
|
+
lastCrawledAt: now,
|
|
2111
|
+
isSkipped: page.isSkipped,
|
|
2112
|
+
...sourceUpdate,
|
|
2113
|
+
});
|
|
2114
|
+
return pageId;
|
|
2115
|
+
}
|
|
2116
|
+
/**
|
|
2117
|
+
* Replaces the page's Wappalyzer tag rows with the freshly captured set.
|
|
2118
|
+
* Called inside `updatePage`'s transaction unconditionally — tag
|
|
2119
|
+
* detection draws on `<script src>` / `<iframe src>` / window globals /
|
|
2120
|
+
* response headers, not the HTML body, so external pages that skip
|
|
2121
|
+
* rendering still contribute tags.
|
|
2122
|
+
*
|
|
2123
|
+
* Same empty-guard as `#insertJsonLd`: an empty array does not wipe
|
|
2124
|
+
* prior rows on a degraded re-scrape.
|
|
2125
|
+
* @param pageId
|
|
2126
|
+
* @param meta
|
|
2127
|
+
* @param trx
|
|
2128
|
+
*/
|
|
2129
|
+
async #insertTags(pageId, meta, trx) {
|
|
2130
|
+
const partial = extractTagsForArchive(meta.tags);
|
|
2131
|
+
if (partial.length === 0)
|
|
2132
|
+
return;
|
|
2133
|
+
const rows = partial.map((p) => ({
|
|
2134
|
+
pageId,
|
|
2135
|
+
provider: p.provider,
|
|
2136
|
+
category: p.category,
|
|
2137
|
+
externalId: p.externalId,
|
|
2138
|
+
version: p.version,
|
|
2139
|
+
confidence: p.confidence,
|
|
2140
|
+
categories: JSON.stringify(p.categories),
|
|
2141
|
+
sources: JSON.stringify(p.sources),
|
|
2142
|
+
}));
|
|
2143
|
+
await trx('page_tags').where('pageId', pageId).delete();
|
|
2144
|
+
await eachSplitted(rows, 100, async (chunk) => {
|
|
2145
|
+
await trx('page_tags').insert(chunk);
|
|
2146
|
+
});
|
|
2147
|
+
}
|
|
2148
|
+
/**
|
|
2149
|
+
* Points each redirect-source URL at the destination page, marking it scraped
|
|
2150
|
+
* and clearing any content it owned in a former life.
|
|
2151
|
+
*
|
|
2152
|
+
* Shared by {@link updatePage} (which also renders and stores the destination)
|
|
2153
|
+
* and {@link recordRedirect} (which only records the edge for a destination
|
|
2154
|
+
* rendered elsewhere). Self-redirects (source equal to the destination) are
|
|
2155
|
+
* skipped so a page is never marked as redirecting to itself — that would
|
|
2156
|
+
* exclude it from reports via the `whereNull('redirectDestId')` filter.
|
|
2157
|
+
* @param trx - The active transaction.
|
|
2158
|
+
* @param sources - Redirect-source URLs (normalised): the original URL plus
|
|
2159
|
+
* any intermediate hops. Empty when the page was not redirected.
|
|
2160
|
+
* @param destId - Database id of the redirect destination page.
|
|
2161
|
+
* @param destUrlNormalized - Normalised destination URL, used to detect and
|
|
2162
|
+
* skip self-redirects.
|
|
2163
|
+
* @param isExternal - Whether the sources are external to the crawl scope.
|
|
2164
|
+
* @param chainLineageSource - Lineage label propagated to each intermediate
|
|
2165
|
+
* hop's row (passed through to {@link #getIdByUrl}). Derived by the caller
|
|
2166
|
+
* from the **originating** page's source (`page.url`), not from the
|
|
2167
|
+
* destination — intermediates are reached transitively from the
|
|
2168
|
+
* originating render, so they inherit its lineage. Pass `'inventory-discovered'`
|
|
2169
|
+
* for chains rooted at inventory-seed/discovered pages so new intermediates
|
|
2170
|
+
* stay in the inventory chain; pass `'crawled'` for crawled chains so the
|
|
2171
|
+
* crawled-wins downgrade inside `#getIdByUrl` fires on existing
|
|
2172
|
+
* `'inventory-*'` intermediates a crawled chain reaches. Pass `undefined`
|
|
2173
|
+
* to fall back to the DB DEFAULT (`'crawled'`) on INSERT without
|
|
2174
|
+
* triggering the downgrade on existing rows.
|
|
2175
|
+
*/
|
|
2176
|
+
async #linkRedirectSources(trx, sources, destId, destUrlNormalized, isExternal, chainLineageSource) {
|
|
2177
|
+
for (const redirect of sources) {
|
|
2178
|
+
if (redirect === destUrlNormalized) {
|
|
2179
|
+
dbLog('Skip self-redirect: %s', redirect);
|
|
2180
|
+
continue;
|
|
1714
2181
|
}
|
|
2182
|
+
dbLog('Set redirected url: %s -> id:%d', redirect, destId);
|
|
2183
|
+
// Pass `chainLineageSource` through so a brand-new
|
|
2184
|
+
// intermediate hop INSERTed here inherits the originating
|
|
2185
|
+
// page's lineage label (inventory-discovered when the
|
|
2186
|
+
// originating chain is in the inventory chain, undefined
|
|
2187
|
+
// otherwise). The crawled-wins downgrade inside
|
|
2188
|
+
// `#getIdByUrl` still fires when this argument is `'crawled'`,
|
|
2189
|
+
// matching the anchor-lineage propagation contract — an
|
|
2190
|
+
// existing inventory-* intermediate that is later traversed
|
|
2191
|
+
// by a `'crawled'` chain gets downgraded.
|
|
2192
|
+
const redirectId = await this.#getIdByUrl(redirect, undefined, trx, chainLineageSource);
|
|
2193
|
+
await trx('pages')
|
|
2194
|
+
.where('id', redirectId)
|
|
2195
|
+
.update({
|
|
2196
|
+
scraped: 1,
|
|
2197
|
+
redirectDestId: destId,
|
|
2198
|
+
isExternal: isExternal ? 1 : 0,
|
|
2199
|
+
});
|
|
2200
|
+
// Conditional `301 Moved Permanently` stamp — applied ONLY
|
|
2201
|
+
// when the row carries no definitive status yet (NULL or
|
|
2202
|
+
// the `-1` hard-failure sentinel). HEAD pre-flight does not
|
|
2203
|
+
// retain each hop's individual status code (`redirectPaths`
|
|
2204
|
+
// is a URL[] without statuses), so the only honest answer
|
|
2205
|
+
// for an unknown-status hop is "some 3xx" — 301 is the
|
|
2206
|
+
// canonical representative.
|
|
2207
|
+
//
|
|
2208
|
+
// We deliberately do NOT overwrite an existing definitive
|
|
2209
|
+
// status (200 / 302 / 307 / etc.): a row that already
|
|
2210
|
+
// captured a concrete status from a prior direct scrape
|
|
2211
|
+
// would lose accuracy. The stamp only flips two cases:
|
|
2212
|
+
// - NULL: a placeholder row created by `#getIdByUrl`
|
|
2213
|
+
// because the URL was reached only as a redirect
|
|
2214
|
+
// target / source, never directly scraped. Without the
|
|
2215
|
+
// stamp the row is invisible on the Errors view's status
|
|
2216
|
+
// distribution.
|
|
2217
|
+
// - -1: a row that recorded a hard scrape failure (e.g. a
|
|
2218
|
+
// puppeteer goto returned null on a HTTPS→HTTP downgrade
|
|
2219
|
+
// redirect) BEFORE the chain was understood. That `-1`
|
|
2220
|
+
// then conflated "real failure" with "actually a redirect
|
|
2221
|
+
// source we now know about", polluting the `-1` bucket
|
|
2222
|
+
// AND inflating the `--retry-failed` target (via the
|
|
2223
|
+
// `whereNull('redirectDestId')` filter — the redirectDestId
|
|
2224
|
+
// update above already excludes the row from retry; this
|
|
2225
|
+
// stamp restores the visible identity).
|
|
2226
|
+
await trx('pages')
|
|
2227
|
+
.where('id', redirectId)
|
|
2228
|
+
.where((qb) => qb.whereNull('status').orWhere('status', -1))
|
|
2229
|
+
.update({ status: 301, statusText: 'Moved Permanently' });
|
|
2230
|
+
// A page that used to be scraped as content can later turn into a
|
|
2231
|
+
// redirect source. It owns no content anymore, so drop any anchors /
|
|
2232
|
+
// images it captured in its former life — otherwise they linger and
|
|
2233
|
+
// leak into referrer / incoming-link reads (which do not filter out
|
|
2234
|
+
// redirect sources).
|
|
2235
|
+
await trx('anchors').where('pageId', redirectId).delete();
|
|
2236
|
+
await trx('images').where('pageId', redirectId).delete();
|
|
1715
2237
|
}
|
|
1716
|
-
|
|
1717
|
-
|
|
1718
|
-
|
|
1719
|
-
|
|
1720
|
-
|
|
1721
|
-
|
|
1722
|
-
|
|
1723
|
-
|
|
1724
|
-
|
|
1725
|
-
|
|
1726
|
-
|
|
1727
|
-
|
|
1728
|
-
|
|
1729
|
-
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
|
|
1737
|
-
|
|
1738
|
-
|
|
1739
|
-
|
|
1740
|
-
|
|
1741
|
-
|
|
1742
|
-
|
|
1743
|
-
|
|
1744
|
-
|
|
1745
|
-
|
|
1746
|
-
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1750
|
-
|
|
1751
|
-
|
|
1752
|
-
|
|
1753
|
-
|
|
1754
|
-
|
|
1755
|
-
|
|
1756
|
-
|
|
1757
|
-
|
|
1758
|
-
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1766
|
-
|
|
1767
|
-
|
|
1768
|
-
|
|
1769
|
-
|
|
1770
|
-
|
|
1771
|
-
|
|
1772
|
-
|
|
1773
|
-
|
|
1774
|
-
|
|
1775
|
-
|
|
1776
|
-
|
|
1777
|
-
}
|
|
1778
|
-
if (!existsSync(options.filename)) {
|
|
1779
|
-
throw new Error(`Cannot open archive read-only: database file missing (${options.filename}). The source may have been removed by another process.`);
|
|
1780
|
-
}
|
|
2238
|
+
}
|
|
2239
|
+
/**
|
|
2240
|
+
* Encodes, dedups, and persists a page's HTML snapshot.
|
|
2241
|
+
*
|
|
2242
|
+
* Computes SHA-256 over the raw UTF-8 bytes, compresses them with zstd,
|
|
2243
|
+
* inserts into `page_html_blobs` only if the hash is new (so identical
|
|
2244
|
+
* bodies — 404 templates, error pages, redirect destinations — share a
|
|
2245
|
+
* single row), and then upserts `page_html_ref(page_id → hash)` so the
|
|
2246
|
+
* latest scrape always points at the right body.
|
|
2247
|
+
*
|
|
2248
|
+
* Runs entirely inside the caller's transaction; a failure here rolls
|
|
2249
|
+
* back the rest of `updatePage`, which is the desired semantics (an
|
|
2250
|
+
* archive that lost its HTML for a page would otherwise serve stale
|
|
2251
|
+
* meta against a missing body).
|
|
2252
|
+
* @param pageId - The database id of the page.
|
|
2253
|
+
* @param html - The raw HTML string (UTF-8).
|
|
2254
|
+
* @param trx - The active transaction.
|
|
2255
|
+
*/
|
|
2256
|
+
async #writePageHtmlBlob(pageId, html, trx) {
|
|
2257
|
+
const rawBytes = Buffer.from(html, 'utf8');
|
|
2258
|
+
const hash = createHash('sha256').update(rawBytes).digest();
|
|
2259
|
+
const compressed = zstdCompressSync(rawBytes);
|
|
2260
|
+
await trx('page_html_blobs')
|
|
2261
|
+
.insert({
|
|
2262
|
+
hash,
|
|
2263
|
+
body: compressed,
|
|
2264
|
+
codec: 'zstd',
|
|
2265
|
+
size_raw: rawBytes.byteLength,
|
|
2266
|
+
size_stored: compressed.byteLength,
|
|
2267
|
+
})
|
|
2268
|
+
.onConflict('hash')
|
|
2269
|
+
.ignore();
|
|
2270
|
+
// Upsert so a re-scrape's body cleanly supersedes the prior pointer.
|
|
2271
|
+
// The old blob row is intentionally left in place — a future #23 GC
|
|
2272
|
+
// pass will sweep unreachable hashes.
|
|
2273
|
+
await trx('page_html_ref')
|
|
2274
|
+
.insert({ page_id: pageId, hash })
|
|
2275
|
+
.onConflict('page_id')
|
|
2276
|
+
.merge(['hash']);
|
|
2277
|
+
}
|
|
2278
|
+
/**
|
|
2279
|
+
* Creates and initializes a new Database instance.
|
|
2280
|
+
*
|
|
2281
|
+
* **Writer mode (default)**: creates the parent directory for the
|
|
2282
|
+
* database file if needed, establishes the connection, and initializes
|
|
2283
|
+
* the schema + migrations.
|
|
2284
|
+
*
|
|
2285
|
+
* **Read-only mode** (`options.readOnly`): refuses to resurrect a
|
|
2286
|
+
* missing parent directory or db file — throws if either is absent at
|
|
2287
|
+
* the time of the call. Skips schema init and migrations entirely so
|
|
2288
|
+
* the user's tmpDir is never modified. Required by viewer / MCP
|
|
2289
|
+
* stub-mode opens, where a TOCTOU window between classification and
|
|
2290
|
+
* `connect()` could otherwise leave behind a phantom empty tmpDir.
|
|
2291
|
+
* @param options - Database connection options.
|
|
2292
|
+
* @returns A fully initialized Database instance.
|
|
2293
|
+
* @throws {Error} In read-only mode, if the parent directory or db
|
|
2294
|
+
* file does not exist when `connect()` runs.
|
|
2295
|
+
*/
|
|
2296
|
+
static async connect(options) {
|
|
2297
|
+
if (options.readOnly) {
|
|
2298
|
+
if (!existsSync(path.dirname(options.filename))) {
|
|
2299
|
+
throw new Error(`Cannot open archive read-only: parent directory disappeared (${path.dirname(options.filename)}). The source may have been removed by another process.`);
|
|
1781
2300
|
}
|
|
1782
|
-
|
|
1783
|
-
|
|
2301
|
+
if (!existsSync(options.filename)) {
|
|
2302
|
+
throw new Error(`Cannot open archive read-only: database file missing (${options.filename}). The source may have been removed by another process.`);
|
|
1784
2303
|
}
|
|
1785
|
-
const db = new Database(options);
|
|
1786
|
-
await db.#init(options.readOnly ?? false);
|
|
1787
|
-
return db;
|
|
1788
2304
|
}
|
|
1789
|
-
|
|
1790
|
-
|
|
1791
|
-
|
|
2305
|
+
else {
|
|
2306
|
+
mkdir(options.filename);
|
|
2307
|
+
}
|
|
2308
|
+
const db = new Database(options);
|
|
2309
|
+
await db.#init(options.readOnly ?? false);
|
|
2310
|
+
return db;
|
|
2311
|
+
}
|
|
2312
|
+
}
|