@nitpicker/crawler 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -4
- package/lib/archive/archive-accessor.d.ts +2 -2
- package/lib/archive/archive-accessor.js +2 -2
- package/lib/archive/archive-lock.d.ts +7 -0
- package/lib/archive/archive-lock.js +7 -0
- package/lib/archive/archive.d.ts +63 -16
- package/lib/archive/archive.js +56 -17
- package/lib/archive/create-adjunct-tables.d.ts +43 -0
- package/lib/archive/create-adjunct-tables.js +213 -0
- package/lib/archive/create-entity-tables.d.ts +173 -0
- package/lib/archive/create-entity-tables.js +318 -0
- package/lib/archive/create-progress-reporter.d.ts +30 -0
- package/lib/archive/create-progress-reporter.js +38 -0
- package/lib/archive/create-ref-tables.d.ts +35 -0
- package/lib/archive/create-ref-tables.js +188 -0
- package/lib/archive/database.d.ts +92 -345
- package/lib/archive/database.js +168 -1942
- package/lib/archive/db-ops/_shared/clear-write-ref-caches.d.ts +27 -0
- package/lib/archive/db-ops/_shared/clear-write-ref-caches.js +34 -0
- package/lib/archive/db-ops/_shared/create-write-ref-caches.d.ts +17 -0
- package/lib/archive/db-ops/_shared/create-write-ref-caches.js +26 -0
- package/lib/archive/db-ops/_shared/decode-json-ref.d.ts +17 -0
- package/lib/archive/db-ops/_shared/decode-json-ref.js +31 -0
- package/lib/archive/db-ops/_shared/load-response-headers-by-set-ids.d.ts +20 -0
- package/lib/archive/db-ops/_shared/load-response-headers-by-set-ids.js +53 -0
- package/lib/archive/db-ops/_shared/resolve-content-item-id.d.ts +61 -0
- package/lib/archive/db-ops/_shared/resolve-content-item-id.js +111 -0
- package/lib/archive/db-ops/_shared/resolve-url-or-blob.d.ts +23 -0
- package/lib/archive/db-ops/_shared/resolve-url-or-blob.js +29 -0
- package/lib/archive/db-ops/_shared/retry-setting.d.ts +16 -0
- package/lib/archive/db-ops/_shared/retry-setting.js +18 -0
- package/lib/archive/db-ops/_shared/safe-parse-json.d.ts +11 -0
- package/lib/archive/db-ops/_shared/safe-parse-json.js +18 -0
- package/lib/archive/db-ops/_shared/types.d.ts +53 -0
- package/lib/archive/db-ops/_shared/types.js +1 -0
- package/lib/archive/db-ops/_shared/upsert-blob-ref.d.ts +25 -0
- package/lib/archive/db-ops/_shared/upsert-blob-ref.js +48 -0
- package/lib/archive/db-ops/_shared/upsert-content-type-ref.d.ts +30 -0
- package/lib/archive/db-ops/_shared/upsert-content-type-ref.js +45 -0
- package/lib/archive/db-ops/_shared/upsert-json-ref.d.ts +22 -0
- package/lib/archive/db-ops/_shared/upsert-json-ref.js +41 -0
- package/lib/archive/db-ops/_shared/upsert-response-headers.d.ts +35 -0
- package/lib/archive/db-ops/_shared/upsert-response-headers.js +49 -0
- package/lib/archive/db-ops/_shared/upsert-url-ref.d.ts +39 -0
- package/lib/archive/db-ops/_shared/upsert-url-ref.js +62 -0
- package/lib/archive/db-ops/analysis/replace-analysis-violations.d.ts +28 -0
- package/lib/archive/db-ops/analysis/replace-analysis-violations.js +152 -0
- package/lib/archive/db-ops/anchors/get-anchors-on-page.d.ts +10 -0
- package/lib/archive/db-ops/anchors/get-anchors-on-page.js +21 -0
- package/lib/archive/db-ops/config/get-base-url.d.ts +8 -0
- package/lib/archive/db-ops/config/get-base-url.js +14 -0
- package/lib/archive/db-ops/config/get-config.d.ts +10 -0
- package/lib/archive/db-ops/config/get-config.js +27 -0
- package/lib/archive/db-ops/config/get-name.d.ts +8 -0
- package/lib/archive/db-ops/config/get-name.js +14 -0
- package/lib/archive/db-ops/config/info-column-allowlist.d.ts +7 -0
- package/lib/archive/db-ops/config/info-column-allowlist.js +26 -0
- package/lib/archive/db-ops/config/info-json-columns.d.ts +5 -0
- package/lib/archive/db-ops/config/info-json-columns.js +10 -0
- package/lib/archive/db-ops/config/set-config.d.ts +12 -0
- package/lib/archive/db-ops/config/set-config.js +21 -0
- package/lib/archive/db-ops/config/update-config.d.ts +17 -0
- package/lib/archive/db-ops/config/update-config.js +36 -0
- package/lib/archive/db-ops/errors/insert-crawl-error.d.ts +15 -0
- package/lib/archive/db-ops/errors/insert-crawl-error.js +21 -0
- package/lib/archive/db-ops/errors/insert-page-error.d.ts +21 -0
- package/lib/archive/db-ops/errors/insert-page-error.js +28 -0
- package/lib/archive/db-ops/errors/list-dns-burned-host-candidates.d.ts +22 -0
- package/lib/archive/db-ops/errors/list-dns-burned-host-candidates.js +141 -0
- package/lib/archive/db-ops/html/get-html-of-page-by-id.d.ts +18 -0
- package/lib/archive/db-ops/html/get-html-of-page-by-id.js +29 -0
- package/lib/archive/db-ops/inventory/record-inventory-run.d.ts +21 -0
- package/lib/archive/db-ops/inventory/record-inventory-run.js +38 -0
- package/lib/archive/db-ops/lifecycle/checkpoint.d.ts +8 -0
- package/lib/archive/db-ops/lifecycle/checkpoint.js +9 -0
- package/lib/archive/db-ops/lifecycle/destroy.d.ts +6 -0
- package/lib/archive/db-ops/lifecycle/destroy.js +7 -0
- package/lib/archive/db-ops/lifecycle/init.d.ts +22 -0
- package/lib/archive/db-ops/lifecycle/init.js +42 -0
- package/lib/archive/db-ops/meta/get-jsonld-of-page.d.ts +13 -0
- package/lib/archive/db-ops/meta/get-jsonld-of-page.js +27 -0
- package/lib/archive/db-ops/meta/get-tags-of-page.d.ts +12 -0
- package/lib/archive/db-ops/meta/get-tags-of-page.js +28 -0
- package/lib/archive/db-ops/pages/order/set-url-order.d.ts +8 -0
- package/lib/archive/db-ops/pages/order/set-url-order.js +32 -0
- package/lib/archive/db-ops/pages/read/build-page-query.d.ts +18 -0
- package/lib/archive/db-ops/pages/read/build-page-query.js +40 -0
- package/lib/archive/db-ops/pages/read/get-crawling-state.d.ts +70 -0
- package/lib/archive/db-ops/pages/read/get-crawling-state.js +98 -0
- package/lib/archive/db-ops/pages/read/get-existing-page-urls.d.ts +15 -0
- package/lib/archive/db-ops/pages/read/get-existing-page-urls.js +30 -0
- package/lib/archive/db-ops/pages/read/get-page-count.d.ts +12 -0
- package/lib/archive/db-ops/pages/read/get-page-count.js +21 -0
- package/lib/archive/db-ops/pages/read/get-page-source-by-url.d.ts +24 -0
- package/lib/archive/db-ops/pages/read/get-page-source-by-url.js +28 -0
- package/lib/archive/db-ops/pages/read/get-pages-with-rels.d.ts +38 -0
- package/lib/archive/db-ops/pages/read/get-pages-with-rels.js +107 -0
- package/lib/archive/db-ops/pages/read/get-pages.d.ts +11 -0
- package/lib/archive/db-ops/pages/read/get-pages.js +51 -0
- package/lib/archive/db-ops/pages/read/get-scraped-html-page-count.d.ts +18 -0
- package/lib/archive/db-ops/pages/read/get-scraped-html-page-count.js +25 -0
- package/lib/archive/db-ops/pages/read/reconstruct-page-rows.d.ts +31 -0
- package/lib/archive/db-ops/pages/read/reconstruct-page-rows.js +32 -0
- package/lib/archive/db-ops/pages/reset/repromote-external-pages.d.ts +24 -0
- package/lib/archive/db-ops/pages/reset/repromote-external-pages.js +93 -0
- package/lib/archive/db-ops/pages/reset/reset-failed-pages.d.ts +47 -0
- package/lib/archive/db-ops/pages/reset/reset-failed-pages.js +124 -0
- package/lib/archive/db-ops/pages/write/insert-inventory-seeds.d.ts +37 -0
- package/lib/archive/db-ops/pages/write/insert-inventory-seeds.js +72 -0
- package/lib/archive/db-ops/pages/write/insert-jsonld.d.ts +17 -0
- package/lib/archive/db-ops/pages/write/insert-jsonld.js +49 -0
- package/lib/archive/db-ops/pages/write/insert-page.d.ts +36 -0
- package/lib/archive/db-ops/pages/write/insert-page.js +208 -0
- package/lib/archive/db-ops/pages/write/insert-tags.d.ts +16 -0
- package/lib/archive/db-ops/pages/write/insert-tags.js +34 -0
- package/lib/archive/db-ops/pages/write/link-redirect-sources.d.ts +36 -0
- package/lib/archive/db-ops/pages/write/link-redirect-sources.js +93 -0
- package/lib/archive/db-ops/pages/write/record-redirect.d.ts +35 -0
- package/lib/archive/db-ops/pages/write/record-redirect.js +100 -0
- package/lib/archive/db-ops/pages/write/set-skipped-page.d.ts +13 -0
- package/lib/archive/db-ops/pages/write/set-skipped-page.js +22 -0
- package/lib/archive/db-ops/pages/write/update-page.d.ts +29 -0
- package/lib/archive/db-ops/pages/write/update-page.js +334 -0
- package/lib/archive/db-ops/pages/write/write-page-html-blob.d.ts +19 -0
- package/lib/archive/db-ops/pages/write/write-page-html-blob.js +41 -0
- package/lib/archive/db-ops/referrers/get-redirects-for-pages.d.ts +9 -0
- package/lib/archive/db-ops/referrers/get-redirects-for-pages.js +15 -0
- package/lib/archive/db-ops/referrers/get-referrers-of-page.d.ts +17 -0
- package/lib/archive/db-ops/referrers/get-referrers-of-page.js +32 -0
- package/lib/archive/db-ops/referrers/get-referrers-of-resource.d.ts +8 -0
- package/lib/archive/db-ops/referrers/get-referrers-of-resource.js +15 -0
- package/lib/archive/db-ops/resources/build-resource-query.d.ts +25 -0
- package/lib/archive/db-ops/resources/build-resource-query.js +29 -0
- package/lib/archive/db-ops/resources/get-existing-resource-urls.d.ts +9 -0
- package/lib/archive/db-ops/resources/get-existing-resource-urls.js +24 -0
- package/lib/archive/db-ops/resources/get-resource-by-url.d.ts +13 -0
- package/lib/archive/db-ops/resources/get-resource-by-url.js +22 -0
- package/lib/archive/db-ops/resources/get-resource-url-list.d.ts +9 -0
- package/lib/archive/db-ops/resources/get-resource-url-list.js +13 -0
- package/lib/archive/db-ops/resources/get-resources.d.ts +8 -0
- package/lib/archive/db-ops/resources/get-resources.js +11 -0
- package/lib/archive/db-ops/resources/insert-inventory-resources.d.ts +24 -0
- package/lib/archive/db-ops/resources/insert-inventory-resources.js +64 -0
- package/lib/archive/db-ops/resources/insert-resource-referrers.d.ts +15 -0
- package/lib/archive/db-ops/resources/insert-resource-referrers.js +54 -0
- package/lib/archive/db-ops/resources/insert-resource.d.ts +34 -0
- package/lib/archive/db-ops/resources/insert-resource.js +73 -0
- package/lib/archive/db-ops/resources/reconstruct-resource-rows.d.ts +26 -0
- package/lib/archive/db-ops/resources/reconstruct-resource-rows.js +30 -0
- package/lib/archive/decode-html-blob.d.ts +18 -0
- package/lib/archive/decode-html-blob.js +31 -0
- package/lib/archive/derive-lineage-from-parent.d.ts +1 -1
- package/lib/archive/derive-lineage-from-parent.js +1 -1
- package/lib/archive/drop-legacy-tables.d.ts +45 -0
- package/lib/archive/drop-legacy-tables.js +56 -0
- package/lib/archive/filesystem/rename.js +1 -1
- package/lib/archive/get-failed-page-messages.d.ts +5 -4
- package/lib/archive/get-failed-page-messages.js +5 -4
- package/lib/archive/init-schema.d.ts +35 -39
- package/lib/archive/init-schema.js +99 -460
- package/lib/archive/limited-page-ids.d.ts +2 -1
- package/lib/archive/limited-page-ids.js +5 -4
- package/lib/archive/meta/assert-compatible-version.d.ts +24 -3
- package/lib/archive/meta/assert-compatible-version.js +24 -3
- package/lib/archive/meta/types.d.ts +87 -1
- package/lib/archive/meta/types.js +34 -2
- package/lib/archive/migrate-entity-tables.d.ts +45 -0
- package/lib/archive/migrate-entity-tables.js +56 -0
- package/lib/archive/migrate-ref-tables.d.ts +25 -0
- package/lib/archive/migrate-ref-tables.js +38 -0
- package/lib/archive/page-meta-column-maps.d.ts +32 -0
- package/lib/archive/page-meta-column-maps.js +43 -0
- package/lib/archive/page.d.ts +6 -6
- package/lib/archive/page.js +5 -5
- package/lib/archive/peek-archive-lock.d.ts +2 -2
- package/lib/archive/peek-archive-lock.js +2 -2
- package/lib/archive/populate-entity-tables/collapse-anchor-rows.d.ts +41 -0
- package/lib/archive/populate-entity-tables/collapse-anchor-rows.js +87 -0
- package/lib/archive/populate-entity-tables/derive-dom-path.d.ts +35 -0
- package/lib/archive/populate-entity-tables/derive-dom-path.js +72 -0
- package/lib/archive/populate-entity-tables/is-blob-ref-value.d.ts +16 -0
- package/lib/archive/populate-entity-tables/is-blob-ref-value.js +19 -0
- package/lib/archive/populate-entity-tables/match-images-to-dom-paths.d.ts +66 -0
- package/lib/archive/populate-entity-tables/match-images-to-dom-paths.js +96 -0
- package/lib/archive/populate-entity-tables/populate-anchor-edges.d.ts +33 -0
- package/lib/archive/populate-entity-tables/populate-anchor-edges.js +153 -0
- package/lib/archive/populate-entity-tables/populate-content-items.d.ts +40 -0
- package/lib/archive/populate-entity-tables/populate-content-items.js +141 -0
- package/lib/archive/populate-entity-tables/populate-entities.d.ts +81 -0
- package/lib/archive/populate-entity-tables/populate-entities.js +111 -0
- package/lib/archive/populate-entity-tables/populate-image-items.d.ts +91 -0
- package/lib/archive/populate-entity-tables/populate-image-items.js +223 -0
- package/lib/archive/populate-entity-tables/populate-page-meta.d.ts +33 -0
- package/lib/archive/populate-entity-tables/populate-page-meta.js +267 -0
- package/lib/archive/populate-entity-tables/populate-resource-items.d.ts +22 -0
- package/lib/archive/populate-entity-tables/populate-resource-items.js +114 -0
- package/lib/archive/populate-entity-tables/populate-resource-ref-edges.d.ts +31 -0
- package/lib/archive/populate-entity-tables/populate-resource-ref-edges.js +33 -0
- package/lib/archive/populate-entity-tables/resolve-blob-refs.d.ts +31 -0
- package/lib/archive/populate-entity-tables/resolve-blob-refs.js +100 -0
- package/lib/archive/populate-entity-tables/resolve-content-type-refs.d.ts +22 -0
- package/lib/archive/populate-entity-tables/resolve-content-type-refs.js +27 -0
- package/lib/archive/populate-entity-tables/resolve-header-sets.d.ts +49 -0
- package/lib/archive/populate-entity-tables/resolve-header-sets.js +122 -0
- package/lib/archive/populate-entity-tables/resolve-json-refs.d.ts +25 -0
- package/lib/archive/populate-entity-tables/resolve-json-refs.js +67 -0
- package/lib/archive/populate-entity-tables/resolve-text-refs.d.ts +30 -0
- package/lib/archive/populate-entity-tables/resolve-text-refs.js +61 -0
- package/lib/archive/populate-entity-tables/resolve-url-or-blob-from-maps.d.ts +21 -0
- package/lib/archive/populate-entity-tables/resolve-url-or-blob-from-maps.js +27 -0
- package/lib/archive/populate-entity-tables/resolve-url-refs.d.ts +33 -0
- package/lib/archive/populate-entity-tables/resolve-url-refs.js +60 -0
- package/lib/archive/populate-entity-tables/test-utils/count-rows.d.ts +17 -0
- package/lib/archive/populate-entity-tables/test-utils/count-rows.js +20 -0
- package/lib/archive/populate-entity-tables/test-utils/seed-content-items.d.ts +25 -0
- package/lib/archive/populate-entity-tables/test-utils/seed-content-items.js +42 -0
- package/lib/archive/populate-entity-tables/test-utils/setup-entities-db.d.ts +23 -0
- package/lib/archive/populate-entity-tables/test-utils/setup-entities-db.js +178 -0
- package/lib/archive/populate-entity-tables/types.d.ts +157 -0
- package/lib/archive/populate-entity-tables/types.js +12 -0
- package/lib/archive/populate-entity-tables/upsert-text-refs.d.ts +38 -0
- package/lib/archive/populate-entity-tables/upsert-text-refs.js +78 -0
- package/lib/archive/populate-ref-tables/classify-content-type.d.ts +16 -0
- package/lib/archive/populate-ref-tables/classify-content-type.js +52 -0
- package/lib/archive/populate-ref-tables/compute-content-hash.d.ts +22 -0
- package/lib/archive/populate-ref-tables/compute-content-hash.js +26 -0
- package/lib/archive/populate-ref-tables/compute-header-flags.d.ts +16 -0
- package/lib/archive/populate-ref-tables/compute-header-flags.js +70 -0
- package/lib/archive/populate-ref-tables/content-type-rules.d.ts +38 -0
- package/lib/archive/populate-ref-tables/content-type-rules.js +133 -0
- package/lib/archive/populate-ref-tables/create-header-table-caches.d.ts +25 -0
- package/lib/archive/populate-ref-tables/create-header-table-caches.js +49 -0
- package/lib/archive/populate-ref-tables/data-uri-url-refs-limit.d.ts +15 -0
- package/lib/archive/populate-ref-tables/data-uri-url-refs-limit.js +15 -0
- package/lib/archive/populate-ref-tables/decode-data-uri.d.ts +21 -0
- package/lib/archive/populate-ref-tables/decode-data-uri.js +126 -0
- package/lib/archive/populate-ref-tables/decompose-header-set.d.ts +29 -0
- package/lib/archive/populate-ref-tables/decompose-header-set.js +157 -0
- package/lib/archive/populate-ref-tables/decompose-url.d.ts +25 -0
- package/lib/archive/populate-ref-tables/decompose-url.js +70 -0
- package/lib/archive/populate-ref-tables/header-stability.d.ts +19 -0
- package/lib/archive/populate-ref-tables/header-stability.js +22 -0
- package/lib/archive/populate-ref-tables/header-value-cache-key.d.ts +17 -0
- package/lib/archive/populate-ref-tables/header-value-cache-key.js +19 -0
- package/lib/archive/populate-ref-tables/normalize-mime.d.ts +24 -0
- package/lib/archive/populate-ref-tables/normalize-mime.js +36 -0
- package/lib/archive/populate-ref-tables/populate-blob-refs.d.ts +38 -0
- package/lib/archive/populate-ref-tables/populate-blob-refs.js +134 -0
- package/lib/archive/populate-ref-tables/populate-content-type-refs.d.ts +27 -0
- package/lib/archive/populate-ref-tables/populate-content-type-refs.js +70 -0
- package/lib/archive/populate-ref-tables/populate-header-tables.d.ts +35 -0
- package/lib/archive/populate-ref-tables/populate-header-tables.js +80 -0
- package/lib/archive/populate-ref-tables/populate-json-refs.d.ts +29 -0
- package/lib/archive/populate-ref-tables/populate-json-refs.js +101 -0
- package/lib/archive/populate-ref-tables/populate-refs.d.ts +51 -0
- package/lib/archive/populate-ref-tables/populate-refs.js +62 -0
- package/lib/archive/populate-ref-tables/populate-text-refs.d.ts +32 -0
- package/lib/archive/populate-ref-tables/populate-text-refs.js +133 -0
- package/lib/archive/populate-ref-tables/populate-url-refs.d.ts +28 -0
- package/lib/archive/populate-ref-tables/populate-url-refs.js +148 -0
- package/lib/archive/populate-ref-tables/test-utils/count-rows.d.ts +15 -0
- package/lib/archive/populate-ref-tables/test-utils/count-rows.js +17 -0
- package/lib/archive/populate-ref-tables/types.d.ts +197 -0
- package/lib/archive/populate-ref-tables/types.js +7 -0
- package/lib/archive/populate-ref-tables/upsert-one-header-set.d.ts +34 -0
- package/lib/archive/populate-ref-tables/upsert-one-header-set.js +208 -0
- package/lib/archive/populate-ref-tables/volatile-header-names.d.ts +20 -0
- package/lib/archive/populate-ref-tables/volatile-header-names.js +33 -0
- package/lib/archive/redirect-table.d.ts +4 -2
- package/lib/archive/redirect-table.js +15 -10
- package/lib/archive/resolve-redirect-chain.d.ts +3 -3
- package/lib/archive/resolve-redirect-chain.js +2 -2
- package/lib/archive/resource.d.ts +1 -1
- package/lib/archive/retarget-legacy-fk-tables.d.ts +47 -0
- package/lib/archive/retarget-legacy-fk-tables.js +107 -0
- package/lib/archive/test-utils/fk-parent-tables.d.ts +15 -0
- package/lib/archive/test-utils/fk-parent-tables.js +19 -0
- package/lib/archive/test-utils/seed-content-item.d.ts +35 -0
- package/lib/archive/test-utils/seed-content-item.js +42 -0
- package/lib/archive/test-utils/setup-legacy-fk-db.d.ts +33 -0
- package/lib/archive/test-utils/setup-legacy-fk-db.js +270 -0
- package/lib/archive/types.d.ts +127 -24
- package/lib/archive/verify-migration/capture-rejection.d.ts +24 -0
- package/lib/archive/verify-migration/capture-rejection.js +31 -0
- package/lib/archive/verify-migration/check-anchor-edges-count.d.ts +34 -0
- package/lib/archive/verify-migration/check-anchor-edges-count.js +72 -0
- package/lib/archive/verify-migration/check-anchor-edges-sum.d.ts +13 -0
- package/lib/archive/verify-migration/check-anchor-edges-sum.js +27 -0
- package/lib/archive/verify-migration/check-content-items-count.d.ts +16 -0
- package/lib/archive/verify-migration/check-content-items-count.js +30 -0
- package/lib/archive/verify-migration/check-content-type-preservation.d.ts +22 -0
- package/lib/archive/verify-migration/check-content-type-preservation.js +40 -0
- package/lib/archive/verify-migration/check-foreign-key-integrity.d.ts +31 -0
- package/lib/archive/verify-migration/check-foreign-key-integrity.js +47 -0
- package/lib/archive/verify-migration/check-image-items-count.d.ts +12 -0
- package/lib/archive/verify-migration/check-image-items-count.js +26 -0
- package/lib/archive/verify-migration/check-page-meta-count.d.ts +15 -0
- package/lib/archive/verify-migration/check-page-meta-count.js +31 -0
- package/lib/archive/verify-migration/check-reader-parity.d.ts +23 -0
- package/lib/archive/verify-migration/check-reader-parity.js +211 -0
- package/lib/archive/verify-migration/check-resource-items-count.d.ts +17 -0
- package/lib/archive/verify-migration/check-resource-items-count.js +33 -0
- package/lib/archive/verify-migration/check-url-round-trip.d.ts +43 -0
- package/lib/archive/verify-migration/check-url-round-trip.js +112 -0
- package/lib/archive/verify-migration/types.d.ts +70 -0
- package/lib/archive/verify-migration/types.js +63 -0
- package/lib/archive/verify-migration/verify-migration.d.ts +41 -0
- package/lib/archive/verify-migration/verify-migration.js +120 -0
- package/lib/crawler/build-redirect-event.d.ts +1 -1
- package/lib/crawler/build-redirect-event.js +1 -1
- package/lib/crawler/capture-image-dom-paths.d.ts +33 -0
- package/lib/crawler/capture-image-dom-paths.js +39 -0
- package/lib/crawler/clear-dns-burned-host-cache.d.ts +1 -1
- package/lib/crawler/clear-dns-burned-host-cache.js +1 -1
- package/lib/crawler/collect-image-dom-paths.d.ts +23 -0
- package/lib/crawler/collect-image-dom-paths.js +64 -0
- package/lib/crawler/crawler.d.ts +19 -0
- package/lib/crawler/crawler.js +40 -26
- package/lib/crawler/dns-burned-host-cache.d.ts +3 -3
- package/lib/crawler/dns-burned-host-cache.js +3 -3
- package/lib/crawler/dns-burned-host-short-circuit-counter.d.ts +2 -2
- package/lib/crawler/dns-burned-host-short-circuit-counter.js +2 -2
- package/lib/crawler/inject-scope-auth.d.ts +1 -1
- package/lib/crawler/inject-scope-auth.js +1 -1
- package/lib/crawler/normalize-content-type.d.ts +1 -1
- package/lib/crawler/normalize-content-type.js +1 -1
- package/lib/crawler/types.d.ts +3 -3
- package/lib/crawler-orchestrator.d.ts +9 -0
- package/lib/crawler-orchestrator.js +44 -28
- package/lib/crawler.d.ts +12 -0
- package/lib/crawler.js +21 -0
- package/lib/permanent-error-kinds.d.ts +1 -1
- package/lib/permanent-error-kinds.js +1 -1
- package/lib/types.d.ts +1 -1
- package/lib/utils/compute-file-sha256.d.ts +5 -4
- package/lib/utils/compute-file-sha256.js +5 -4
- package/lib/utils/error/emit-error-with-retry.d.ts +1 -1
- package/lib/utils/error/emit-error-with-retry.js +1 -1
- package/package.json +10 -10
- package/lib/archive/migrate-crawl-errors.d.ts +0 -20
- package/lib/archive/migrate-crawl-errors.js +0 -38
- package/lib/archive/migrate-html-blob-tables.d.ts +0 -24
- package/lib/archive/migrate-html-blob-tables.js +0 -53
- package/lib/archive/migrate-inventory-runs.d.ts +0 -29
- package/lib/archive/migrate-inventory-runs.js +0 -52
- package/lib/archive/migrate-page-errors.d.ts +0 -16
- package/lib/archive/migrate-page-errors.js +0 -35
- package/lib/archive/migrate-pages-resources-source.d.ts +0 -16
- package/lib/archive/migrate-pages-resources-source.js +0 -46
package/README.md
CHANGED
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
# @nitpicker/crawler
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
ヘッドレスブラウザでWebサイトをクロールし、`.nitpicker` アーカイブを生成・更新する内部パッケージです。
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
通常は [@nitpicker/cli](../cli/README.md) の `crawl` コマンドから利用します。
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
## 関連リンク
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
- [Nitpicker README](../../../README.md)
|
|
10
|
+
- [CLI crawl docs](../cli/docs/crawl.md)
|
|
11
|
+
- [ARCHITECTURE.md](../../../ARCHITECTURE.md)
|
|
10
12
|
|
|
11
13
|
## ライセンス
|
|
12
14
|
|
|
@@ -39,7 +39,7 @@ export declare class ArchiveAccessor extends EventEmitter<DatabaseEvent> {
|
|
|
39
39
|
* tmpDir), where touching the filesystem would race with — or destroy —
|
|
40
40
|
* the live crawler's working state.
|
|
41
41
|
*
|
|
42
|
-
* Subclasses that own the archive's lifecycle (notably
|
|
42
|
+
* Subclasses that own the archive's lifecycle (notably `Archive`)
|
|
43
43
|
* override this to add write/cleanup steps.
|
|
44
44
|
*
|
|
45
45
|
* **Idempotent and concurrent-safe**: the first invocation captures the
|
|
@@ -171,7 +171,7 @@ export declare class ArchiveAccessor extends EventEmitter<DatabaseEvent> {
|
|
|
171
171
|
* Retrieves a flat list of all resource URLs stored in the archive.
|
|
172
172
|
* @returns An array of resource URL strings.
|
|
173
173
|
*/
|
|
174
|
-
getResourceUrlList(): Promise<
|
|
174
|
+
getResourceUrlList(): Promise<string[]>;
|
|
175
175
|
/**
|
|
176
176
|
* Retrieves the Wappalyzer tag entries for the given page, parsed back
|
|
177
177
|
* from the `page_tags` table.
|
|
@@ -30,7 +30,7 @@ export class ArchiveAccessor extends EventEmitter {
|
|
|
30
30
|
* Promise tracking an in-progress (or completed) close. `null` means the
|
|
31
31
|
* accessor is open and idle; a settled promise means we are closed (the
|
|
32
32
|
* accessor stays "closed" even if `db.destroy()` rejected, because there
|
|
33
|
-
* is nothing safe to retry — see {@link close}).
|
|
33
|
+
* is nothing safe to retry — see {@link ArchiveAccessor.close}).
|
|
34
34
|
*/
|
|
35
35
|
#closeOnce = null;
|
|
36
36
|
/** The SQLite database instance for querying archived data. */
|
|
@@ -91,7 +91,7 @@ export class ArchiveAccessor extends EventEmitter {
|
|
|
91
91
|
* tmpDir), where touching the filesystem would race with — or destroy —
|
|
92
92
|
* the live crawler's working state.
|
|
93
93
|
*
|
|
94
|
-
* Subclasses that own the archive's lifecycle (notably
|
|
94
|
+
* Subclasses that own the archive's lifecycle (notably `Archive`)
|
|
95
95
|
* override this to add write/cleanup steps.
|
|
96
96
|
*
|
|
97
97
|
* **Idempotent and concurrent-safe**: the first invocation captures the
|
|
@@ -34,5 +34,12 @@ export declare class ArchiveLockError extends Error {
|
|
|
34
34
|
* @param tmpDir - Absolute path to the archive's temporary working directory.
|
|
35
35
|
* @returns A release function to be called when the work is done.
|
|
36
36
|
* @throws {ArchiveLockError} When the lock cannot be acquired even after a stale-lock retry.
|
|
37
|
+
* @example
|
|
38
|
+
* const releaseLock = await acquireArchiveLock(tmpDir);
|
|
39
|
+
* try {
|
|
40
|
+
* // ... exclusive work against tmpDir ...
|
|
41
|
+
* } finally {
|
|
42
|
+
* await releaseLock();
|
|
43
|
+
* }
|
|
37
44
|
*/
|
|
38
45
|
export declare function acquireArchiveLock(tmpDir: string): Promise<() => Promise<void>>;
|
|
@@ -42,6 +42,13 @@ export class ArchiveLockError extends Error {
|
|
|
42
42
|
* @param tmpDir - Absolute path to the archive's temporary working directory.
|
|
43
43
|
* @returns A release function to be called when the work is done.
|
|
44
44
|
* @throws {ArchiveLockError} When the lock cannot be acquired even after a stale-lock retry.
|
|
45
|
+
* @example
|
|
46
|
+
* const releaseLock = await acquireArchiveLock(tmpDir);
|
|
47
|
+
* try {
|
|
48
|
+
* // ... exclusive work against tmpDir ...
|
|
49
|
+
* } finally {
|
|
50
|
+
* await releaseLock();
|
|
51
|
+
* }
|
|
45
52
|
*/
|
|
46
53
|
export async function acquireArchiveLock(tmpDir) {
|
|
47
54
|
const lockPath = `${tmpDir}.lock`;
|
package/lib/archive/archive.d.ts
CHANGED
|
@@ -14,6 +14,15 @@ import { ArchiveAccessor } from './archive-accessor.js';
|
|
|
14
14
|
* Use the static factory methods ({@link Archive.create}, {@link Archive.open},
|
|
15
15
|
* {@link Archive.resume}, {@link Archive.connect}) to obtain instances.
|
|
16
16
|
* The constructor is private.
|
|
17
|
+
* @example
|
|
18
|
+
* const archive = await Archive.create({ filePath: '/path/to/site.nitpicker' });
|
|
19
|
+
* try {
|
|
20
|
+
* await archive.setConfig(config);
|
|
21
|
+
* const pageId = await archive.setPage(pageData);
|
|
22
|
+
* } finally {
|
|
23
|
+
* // Writes the `.nitpicker` tar (if absent), removes tmpDir, releases the lock.
|
|
24
|
+
* await archive.close();
|
|
25
|
+
* }
|
|
17
26
|
*/
|
|
18
27
|
export default class Archive extends ArchiveAccessor {
|
|
19
28
|
#private;
|
|
@@ -112,15 +121,14 @@ export default class Archive extends ArchiveAccessor {
|
|
|
112
121
|
* Retrieves the base URL of the crawl session from the archive database.
|
|
113
122
|
* @returns The base URL string.
|
|
114
123
|
*/
|
|
115
|
-
getUrl(): Promise<
|
|
124
|
+
getUrl(): Promise<string>;
|
|
116
125
|
/**
|
|
117
126
|
* Pre-insert inventory non-HTML URLs as `source='inventory-seed'`
|
|
118
127
|
* placeholders in the `resources` table — the non-HTML counterpart of
|
|
119
|
-
* {@link Archive.insertInventorySeeds}.
|
|
120
|
-
*
|
|
121
|
-
*
|
|
122
|
-
*
|
|
123
|
-
* to seconds).
|
|
128
|
+
* {@link Archive.insertInventorySeeds}. Rows are committed in chunked
|
|
129
|
+
* bulk inserts (500 per round-trip) rather than per-URL awaits — a
|
|
130
|
+
* per-URL loop would keep a 50k-URL inventory list inside the `.bak`
|
|
131
|
+
* window for minutes instead of seconds.
|
|
124
132
|
*
|
|
125
133
|
* Thin facade over {@link Database.insertInventoryResources}.
|
|
126
134
|
* `ExURL.href` is the storage key for `resources.url` (matches what
|
|
@@ -137,7 +145,7 @@ export default class Archive extends ArchiveAccessor {
|
|
|
137
145
|
* Ctrl+C-tolerance rationale and the `getCrawlingState` interaction.
|
|
138
146
|
*
|
|
139
147
|
* `ExURL` inputs are normalised to `withoutHashAndAuth` here so the storage
|
|
140
|
-
* key matches what
|
|
148
|
+
* key matches what `resolveContentItemId` writes for crawled rows, keeping the
|
|
141
149
|
* crawled-wins downgrade and the existing-URL filter (`getExistingPageUrls`)
|
|
142
150
|
* lookups consistent.
|
|
143
151
|
* @param urls - HTML seed URLs to pre-insert. No-op when empty.
|
|
@@ -180,6 +188,24 @@ export default class Archive extends ArchiveAccessor {
|
|
|
180
188
|
* are mutually exclusive (the first one called wins).
|
|
181
189
|
*/
|
|
182
190
|
releaseHandle(): Promise<void>;
|
|
191
|
+
/**
|
|
192
|
+
* Replaces the archive's analysis violations with a fresh SQL-backed set.
|
|
193
|
+
*
|
|
194
|
+
* Thin facade over {@link Database.replaceAnalysisViolations}; kept on
|
|
195
|
+
* `Archive` so the analyze pipeline can persist violations without
|
|
196
|
+
* reaching into the low-level database class directly.
|
|
197
|
+
* @param violations - Flat analyze violations.
|
|
198
|
+
*/
|
|
199
|
+
replaceAnalysisViolations(violations: readonly {
|
|
200
|
+
validator: string;
|
|
201
|
+
severity: string;
|
|
202
|
+
rule: string;
|
|
203
|
+
code?: string | null;
|
|
204
|
+
message: string;
|
|
205
|
+
url: string;
|
|
206
|
+
line?: number | null;
|
|
207
|
+
col?: number | null;
|
|
208
|
+
}[]): Promise<void>;
|
|
183
209
|
/**
|
|
184
210
|
* Promote previously-external pages that now fall under the (possibly extended)
|
|
185
211
|
* scope back to a pending state so that the crawler re-scrapes them as fully
|
|
@@ -287,23 +313,44 @@ export default class Archive extends ArchiveAccessor {
|
|
|
287
313
|
/** The prefix used for temporary working directories during archive operations. */
|
|
288
314
|
static TMP_DIR_PREFIX: string;
|
|
289
315
|
/**
|
|
290
|
-
* Opens a
|
|
316
|
+
* Opens a connection to an existing archive's database, defaulting to
|
|
317
|
+
* read-only.
|
|
291
318
|
*
|
|
292
|
-
* Returns an {@link ArchiveAccessor} that provides query methods
|
|
293
|
-
*
|
|
294
|
-
* in **read-only mode**: no schema migrations run, and the connection
|
|
319
|
+
* Returns an {@link ArchiveAccessor} that provides query methods. In the
|
|
320
|
+
* default read-only mode, no schema migrations run and the connection
|
|
295
321
|
* refuses to resurrect a missing parent directory or db file (so a
|
|
296
322
|
* TOCTOU window between source classification and this call cannot
|
|
297
|
-
* silently produce an empty phantom tmpDir)
|
|
323
|
+
* silently produce an empty phantom tmpDir); the returned accessor is
|
|
324
|
+
* also marked read-only so consumer-facing helpers (e.g.
|
|
325
|
+
* {@link ArchiveAccessor.getHtmlOfPage}) avoid any filesystem mutation
|
|
326
|
+
* on the user's tmpDir.
|
|
298
327
|
*
|
|
299
|
-
*
|
|
300
|
-
*
|
|
301
|
-
*
|
|
328
|
+
* `options.readOnly: false` is a narrow escape hatch for opening a
|
|
329
|
+
* second, writable connection to a `tmpDir` that {@link Archive.openCached}
|
|
330
|
+
* already extracted (and migrated) into an OS-temp cache directory —
|
|
331
|
+
* never the caller's live/interrupted crawl tmpDir, which must stay
|
|
332
|
+
* read-only. A read-only open (`Archive.openCached`/`ArchiveManager.open`)
|
|
333
|
+
* must never take this path itself — blocking or writing during what
|
|
334
|
+
* must be a read-only open is forbidden (issue #177). This escape
|
|
335
|
+
* hatch has no current production caller; any future
|
|
336
|
+
* one is responsible for its own cross-process coordination (see
|
|
337
|
+
* `acquireArchiveLock`) — this method does not acquire any lock itself.
|
|
302
338
|
* @param tmpDir - The path to the temporary directory containing the database.
|
|
303
339
|
* @param namespace - An optional namespace for scoping data access within the archive.
|
|
340
|
+
* @param options - Connection options.
|
|
341
|
+
* @param options.readOnly - Defaults to `true`. Pass `false` to obtain a
|
|
342
|
+
* writable accessor against an already-extracted cache directory.
|
|
304
343
|
* @returns An ArchiveAccessor instance for querying the archive data.
|
|
344
|
+
* @example
|
|
345
|
+
* // Default (read-only) — safe for stub mode and cache reads:
|
|
346
|
+
* const accessor = await Archive.connect(tmpDir);
|
|
347
|
+
* @example
|
|
348
|
+
* // Writable escape hatch — only against a tar-cache extraction:
|
|
349
|
+
* const writable = await Archive.connect(cacheDir, null, { readOnly: false });
|
|
305
350
|
*/
|
|
306
|
-
static connect(tmpDir: string, namespace?: string | null
|
|
351
|
+
static connect(tmpDir: string, namespace?: string | null, options?: {
|
|
352
|
+
readOnly?: boolean;
|
|
353
|
+
}): Promise<ArchiveAccessor>;
|
|
307
354
|
/**
|
|
308
355
|
* Open a `.nitpicker` archive through the read-only tar cache.
|
|
309
356
|
*
|
package/lib/archive/archive.js
CHANGED
|
@@ -27,6 +27,15 @@ import { untar } from './filesystem/untar.js';
|
|
|
27
27
|
* Use the static factory methods ({@link Archive.create}, {@link Archive.open},
|
|
28
28
|
* {@link Archive.resume}, {@link Archive.connect}) to obtain instances.
|
|
29
29
|
* The constructor is private.
|
|
30
|
+
* @example
|
|
31
|
+
* const archive = await Archive.create({ filePath: '/path/to/site.nitpicker' });
|
|
32
|
+
* try {
|
|
33
|
+
* await archive.setConfig(config);
|
|
34
|
+
* const pageId = await archive.setPage(pageData);
|
|
35
|
+
* } finally {
|
|
36
|
+
* // Writes the `.nitpicker` tar (if absent), removes tmpDir, releases the lock.
|
|
37
|
+
* await archive.close();
|
|
38
|
+
* }
|
|
30
39
|
*/
|
|
31
40
|
export default class Archive extends ArchiveAccessor {
|
|
32
41
|
/**
|
|
@@ -180,11 +189,10 @@ export default class Archive extends ArchiveAccessor {
|
|
|
180
189
|
/**
|
|
181
190
|
* Pre-insert inventory non-HTML URLs as `source='inventory-seed'`
|
|
182
191
|
* placeholders in the `resources` table — the non-HTML counterpart of
|
|
183
|
-
* {@link Archive.insertInventorySeeds}.
|
|
184
|
-
*
|
|
185
|
-
*
|
|
186
|
-
*
|
|
187
|
-
* to seconds).
|
|
192
|
+
* {@link Archive.insertInventorySeeds}. Rows are committed in chunked
|
|
193
|
+
* bulk inserts (500 per round-trip) rather than per-URL awaits — a
|
|
194
|
+
* per-URL loop would keep a 50k-URL inventory list inside the `.bak`
|
|
195
|
+
* window for minutes instead of seconds.
|
|
188
196
|
*
|
|
189
197
|
* Thin facade over {@link Database.insertInventoryResources}.
|
|
190
198
|
* `ExURL.href` is the storage key for `resources.url` (matches what
|
|
@@ -207,7 +215,7 @@ export default class Archive extends ArchiveAccessor {
|
|
|
207
215
|
* Ctrl+C-tolerance rationale and the `getCrawlingState` interaction.
|
|
208
216
|
*
|
|
209
217
|
* `ExURL` inputs are normalised to `withoutHashAndAuth` here so the storage
|
|
210
|
-
* key matches what
|
|
218
|
+
* key matches what `resolveContentItemId` writes for crawled rows, keeping the
|
|
211
219
|
* crawled-wins downgrade and the existing-URL filter (`getExistingPageUrls`)
|
|
212
220
|
* lookups consistent.
|
|
213
221
|
* @param urls - HTML seed URLs to pre-insert. No-op when empty.
|
|
@@ -267,6 +275,17 @@ export default class Archive extends ArchiveAccessor {
|
|
|
267
275
|
this.#closeOnce = this.#runReleaseHandle();
|
|
268
276
|
return this.#closeOnce;
|
|
269
277
|
}
|
|
278
|
+
/**
|
|
279
|
+
* Replaces the archive's analysis violations with a fresh SQL-backed set.
|
|
280
|
+
*
|
|
281
|
+
* Thin facade over {@link Database.replaceAnalysisViolations}; kept on
|
|
282
|
+
* `Archive` so the analyze pipeline can persist violations without
|
|
283
|
+
* reaching into the low-level database class directly.
|
|
284
|
+
* @param violations - Flat analyze violations.
|
|
285
|
+
*/
|
|
286
|
+
async replaceAnalysisViolations(violations) {
|
|
287
|
+
await this.#db.replaceAnalysisViolations(violations);
|
|
288
|
+
}
|
|
270
289
|
/**
|
|
271
290
|
* Promote previously-external pages that now fall under the (possibly extended)
|
|
272
291
|
* scope back to a pending state so that the crawler re-scrapes them as fully
|
|
@@ -452,25 +471,45 @@ export default class Archive extends ArchiveAccessor {
|
|
|
452
471
|
/** The prefix used for temporary working directories during archive operations. */
|
|
453
472
|
static TMP_DIR_PREFIX = '._nitpicker-';
|
|
454
473
|
/**
|
|
455
|
-
* Opens a
|
|
474
|
+
* Opens a connection to an existing archive's database, defaulting to
|
|
475
|
+
* read-only.
|
|
456
476
|
*
|
|
457
|
-
* Returns an {@link ArchiveAccessor} that provides query methods
|
|
458
|
-
*
|
|
459
|
-
* in **read-only mode**: no schema migrations run, and the connection
|
|
477
|
+
* Returns an {@link ArchiveAccessor} that provides query methods. In the
|
|
478
|
+
* default read-only mode, no schema migrations run and the connection
|
|
460
479
|
* refuses to resurrect a missing parent directory or db file (so a
|
|
461
480
|
* TOCTOU window between source classification and this call cannot
|
|
462
|
-
* silently produce an empty phantom tmpDir)
|
|
481
|
+
* silently produce an empty phantom tmpDir); the returned accessor is
|
|
482
|
+
* also marked read-only so consumer-facing helpers (e.g.
|
|
483
|
+
* {@link ArchiveAccessor.getHtmlOfPage}) avoid any filesystem mutation
|
|
484
|
+
* on the user's tmpDir.
|
|
463
485
|
*
|
|
464
|
-
*
|
|
465
|
-
*
|
|
466
|
-
*
|
|
486
|
+
* `options.readOnly: false` is a narrow escape hatch for opening a
|
|
487
|
+
* second, writable connection to a `tmpDir` that {@link Archive.openCached}
|
|
488
|
+
* already extracted (and migrated) into an OS-temp cache directory —
|
|
489
|
+
* never the caller's live/interrupted crawl tmpDir, which must stay
|
|
490
|
+
* read-only. A read-only open (`Archive.openCached`/`ArchiveManager.open`)
|
|
491
|
+
* must never take this path itself — blocking or writing during what
|
|
492
|
+
* must be a read-only open is forbidden (issue #177). This escape
|
|
493
|
+
* hatch has no current production caller; any future
|
|
494
|
+
* one is responsible for its own cross-process coordination (see
|
|
495
|
+
* `acquireArchiveLock`) — this method does not acquire any lock itself.
|
|
467
496
|
* @param tmpDir - The path to the temporary directory containing the database.
|
|
468
497
|
* @param namespace - An optional namespace for scoping data access within the archive.
|
|
498
|
+
* @param options - Connection options.
|
|
499
|
+
* @param options.readOnly - Defaults to `true`. Pass `false` to obtain a
|
|
500
|
+
* writable accessor against an already-extracted cache directory.
|
|
469
501
|
* @returns An ArchiveAccessor instance for querying the archive data.
|
|
502
|
+
* @example
|
|
503
|
+
* // Default (read-only) — safe for stub mode and cache reads:
|
|
504
|
+
* const accessor = await Archive.connect(tmpDir);
|
|
505
|
+
* @example
|
|
506
|
+
* // Writable escape hatch — only against a tar-cache extraction:
|
|
507
|
+
* const writable = await Archive.connect(cacheDir, null, { readOnly: false });
|
|
470
508
|
*/
|
|
471
|
-
static async connect(tmpDir, namespace = null) {
|
|
472
|
-
const
|
|
473
|
-
const
|
|
509
|
+
static async connect(tmpDir, namespace = null, options = {}) {
|
|
510
|
+
const readOnly = options.readOnly ?? true;
|
|
511
|
+
const db = await Archive.#connectDB(tmpDir, { readOnly });
|
|
512
|
+
const archive = new ArchiveAccessor(tmpDir, db, namespace, { readOnly });
|
|
474
513
|
return archive;
|
|
475
514
|
}
|
|
476
515
|
/**
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
import type { Knex } from 'knex';
|
|
2
|
+
/**
|
|
3
|
+
* Creates the adjunct tables that hang off `content_items` (plus the two
|
|
4
|
+
* standalone log tables), guarded per-table so partially-provisioned
|
|
5
|
+
* archives converge to the full set:
|
|
6
|
+
*
|
|
7
|
+
* - `page_errors` — partial scrape failures, FK → `content_items(id)`
|
|
8
|
+
* - `crawl_errors` — crawler-level error channel (no FK; the URL may be
|
|
9
|
+
* an external link that failed DNS, or null for a process-level error)
|
|
10
|
+
* - `page_tags` — Wappalyzer detections, FK → `content_items(id)`
|
|
11
|
+
* - `page_jsonld` — JSON-LD / SpeculationRules, FK → `content_items(id)`
|
|
12
|
+
* - `inventory_runs` — `--inventory` audit log (no FK; append-only)
|
|
13
|
+
* - `analysis_text_refs` + `analysis_violations` — analyze-phase findings,
|
|
14
|
+
* FK → `content_items(id)`
|
|
15
|
+
* - `page_html_blobs` + `page_html_ref` — content-addressable HTML
|
|
16
|
+
* snapshots, FK → `content_items(id)`
|
|
17
|
+
*
|
|
18
|
+
* The DDL is shared between fresh-archive provisioning ({@link initSchema}
|
|
19
|
+
* calls this right after `createEntityTables`) and the migration script
|
|
20
|
+
* (`scripts/migrate-to-0.13.mjs` calls it before retargeting FK
|
|
21
|
+
* declarations and dropping the legacy tables). Keeping the schema in one
|
|
22
|
+
* function guarantees both origin points produce identical tables — the
|
|
23
|
+
* pre-0.13 era kept per-table copies of this DDL in separate lazy-migration
|
|
24
|
+
* modules, and those copies drifted: they declared `REFERENCES pages(id)`
|
|
25
|
+
* while `initSchema` had moved on to `content_items(id)`, leaving migrated
|
|
26
|
+
* archives with stale FK targets that only `scripts/migrate-to-0.13.mjs`'s
|
|
27
|
+
* rename-copy-drop pass can now repair.
|
|
28
|
+
*
|
|
29
|
+
* Unlike `createRefTables` / `createEntityTables` (whose callers guard with
|
|
30
|
+
* a single sentinel table), each table here is guarded individually because
|
|
31
|
+
* the migration-script caller sees archives where any subset may already
|
|
32
|
+
* exist (e.g. `page_tags` from the 0.10 migration but no `inventory_runs`).
|
|
33
|
+
* Index creation stays inside each guard: an existing table keeps whatever
|
|
34
|
+
* indexes its creation path declared.
|
|
35
|
+
* @param instance - The Knex query builder instance connected to the database.
|
|
36
|
+
* @example
|
|
37
|
+
* // Must run after createEntityTables — the page-scoped tables FK into
|
|
38
|
+
* // content_items(id).
|
|
39
|
+
* await createRefTables(db);
|
|
40
|
+
* await createEntityTables(db);
|
|
41
|
+
* await createAdjunctTables(db);
|
|
42
|
+
*/
|
|
43
|
+
export declare function createAdjunctTables(instance: Knex): Promise<void>;
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Creates the adjunct tables that hang off `content_items` (plus the two
|
|
3
|
+
* standalone log tables), guarded per-table so partially-provisioned
|
|
4
|
+
* archives converge to the full set:
|
|
5
|
+
*
|
|
6
|
+
* - `page_errors` — partial scrape failures, FK → `content_items(id)`
|
|
7
|
+
* - `crawl_errors` — crawler-level error channel (no FK; the URL may be
|
|
8
|
+
* an external link that failed DNS, or null for a process-level error)
|
|
9
|
+
* - `page_tags` — Wappalyzer detections, FK → `content_items(id)`
|
|
10
|
+
* - `page_jsonld` — JSON-LD / SpeculationRules, FK → `content_items(id)`
|
|
11
|
+
* - `inventory_runs` — `--inventory` audit log (no FK; append-only)
|
|
12
|
+
* - `analysis_text_refs` + `analysis_violations` — analyze-phase findings,
|
|
13
|
+
* FK → `content_items(id)`
|
|
14
|
+
* - `page_html_blobs` + `page_html_ref` — content-addressable HTML
|
|
15
|
+
* snapshots, FK → `content_items(id)`
|
|
16
|
+
*
|
|
17
|
+
* The DDL is shared between fresh-archive provisioning ({@link initSchema}
|
|
18
|
+
* calls this right after `createEntityTables`) and the migration script
|
|
19
|
+
* (`scripts/migrate-to-0.13.mjs` calls it before retargeting FK
|
|
20
|
+
* declarations and dropping the legacy tables). Keeping the schema in one
|
|
21
|
+
* function guarantees both origin points produce identical tables — the
|
|
22
|
+
* pre-0.13 era kept per-table copies of this DDL in separate lazy-migration
|
|
23
|
+
* modules, and those copies drifted: they declared `REFERENCES pages(id)`
|
|
24
|
+
* while `initSchema` had moved on to `content_items(id)`, leaving migrated
|
|
25
|
+
* archives with stale FK targets that only `scripts/migrate-to-0.13.mjs`'s
|
|
26
|
+
* rename-copy-drop pass can now repair.
|
|
27
|
+
*
|
|
28
|
+
* Unlike `createRefTables` / `createEntityTables` (whose callers guard with
|
|
29
|
+
* a single sentinel table), each table here is guarded individually because
|
|
30
|
+
* the migration-script caller sees archives where any subset may already
|
|
31
|
+
* exist (e.g. `page_tags` from the 0.10 migration but no `inventory_runs`).
|
|
32
|
+
* Index creation stays inside each guard: an existing table keeps whatever
|
|
33
|
+
* indexes its creation path declared.
|
|
34
|
+
* @param instance - The Knex query builder instance connected to the database.
|
|
35
|
+
* @example
|
|
36
|
+
* // Must run after createEntityTables — the page-scoped tables FK into
|
|
37
|
+
* // content_items(id).
|
|
38
|
+
* await createRefTables(db);
|
|
39
|
+
* await createEntityTables(db);
|
|
40
|
+
* await createAdjunctTables(db);
|
|
41
|
+
*/
|
|
42
|
+
export async function createAdjunctTables(instance) {
|
|
43
|
+
if (!(await instance.schema.hasTable('page_errors'))) {
|
|
44
|
+
await instance.schema.createTable('page_errors', (t) => {
|
|
45
|
+
// Records partial scrape failures (e.g. a viewport switch that
|
|
46
|
+
// detaches the frame and trips beholder's @retryable into the
|
|
47
|
+
// `retryExhausted` phase). A page can have zero or more rows here
|
|
48
|
+
// in addition to its normal `content_items` entry — the page
|
|
49
|
+
// itself is considered successfully scraped, but image capture or
|
|
50
|
+
// another secondary step failed for at least one device preset.
|
|
51
|
+
t.increments('id');
|
|
52
|
+
t.integer('pageId').notNullable().unsigned().references('content_items.id');
|
|
53
|
+
t.string('phase').notNullable();
|
|
54
|
+
t.text('message').notNullable();
|
|
55
|
+
t.integer('createdAt').notNullable();
|
|
56
|
+
t.index('pageId');
|
|
57
|
+
});
|
|
58
|
+
}
|
|
59
|
+
if (!(await instance.schema.hasTable('crawl_errors'))) {
|
|
60
|
+
await instance.schema.createTable('crawl_errors', (t) => {
|
|
61
|
+
// Structured form of the crawler-level `error` channel that otherwise
|
|
62
|
+
// only lands in `error.log`. Unlike `page_errors` these are not tied to
|
|
63
|
+
// a scraped page (the URL may be an external link that failed DNS, or
|
|
64
|
+
// null for a process-level error), so there is no `pageId` FK and `url`
|
|
65
|
+
// is nullable. The cause is NOT stored — it is classified on read from
|
|
66
|
+
// `message` so older archives (which only have `error.log`) classify the
|
|
67
|
+
// same way.
|
|
68
|
+
t.increments('id');
|
|
69
|
+
t.string('url', 8190).nullable();
|
|
70
|
+
t.boolean('isExternal');
|
|
71
|
+
t.text('message').notNullable();
|
|
72
|
+
t.integer('createdAt').notNullable();
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
if (!(await instance.schema.hasTable('page_tags'))) {
|
|
76
|
+
await instance.schema.createTable('page_tags', (t) => {
|
|
77
|
+
// Wappalyzer-derived technology detection. One row per
|
|
78
|
+
// (provider × externalId) tuple per page. `category` is the first
|
|
79
|
+
// element of `categories`; the full list lives in the JSON
|
|
80
|
+
// `categories` column. `sources` records where the provider was
|
|
81
|
+
// detected (script-src / inline / iframe-src / window-global / …).
|
|
82
|
+
t.increments('id');
|
|
83
|
+
t.integer('pageId')
|
|
84
|
+
.notNullable()
|
|
85
|
+
.unsigned()
|
|
86
|
+
.references('content_items.id')
|
|
87
|
+
.onDelete('CASCADE');
|
|
88
|
+
t.string('provider').notNullable();
|
|
89
|
+
t.string('category');
|
|
90
|
+
t.string('externalId');
|
|
91
|
+
t.string('version');
|
|
92
|
+
t.integer('confidence');
|
|
93
|
+
t.json('categories');
|
|
94
|
+
t.json('sources');
|
|
95
|
+
t.index('pageId');
|
|
96
|
+
t.index('provider');
|
|
97
|
+
t.index('externalId');
|
|
98
|
+
});
|
|
99
|
+
// Compound indexes for the "find duplicate IDs across pages" and
|
|
100
|
+
// "list pages using provider X" hot paths. Knex's schema builder
|
|
101
|
+
// can't express compound indexes inline in a way that round-trips
|
|
102
|
+
// through libsql consistently, so raw SQL is used.
|
|
103
|
+
await instance.raw('CREATE INDEX page_tags_provider_extId ON page_tags(provider, externalId)');
|
|
104
|
+
await instance.raw('CREATE INDEX page_tags_provider_pageId ON page_tags(provider, pageId)');
|
|
105
|
+
}
|
|
106
|
+
if (!(await instance.schema.hasTable('page_jsonld'))) {
|
|
107
|
+
await instance.schema.createTable('page_jsonld', (t) => {
|
|
108
|
+
// JSON-LD and SpeculationRules entries captured from
|
|
109
|
+
// `<script type="application/ld+json">` and
|
|
110
|
+
// `<script type="speculationrules">`. `kind` discriminates; `type`
|
|
111
|
+
// is the top-level `@type` extracted by classify-jsonld-type for
|
|
112
|
+
// indexable filtering. `raw` is stored uncompressed; SQLite
|
|
113
|
+
// overflow pages handle multi-KB JSON bodies transparently.
|
|
114
|
+
t.increments('id');
|
|
115
|
+
t.integer('pageId')
|
|
116
|
+
.notNullable()
|
|
117
|
+
.unsigned()
|
|
118
|
+
.references('content_items.id')
|
|
119
|
+
.onDelete('CASCADE');
|
|
120
|
+
t.string('kind').notNullable();
|
|
121
|
+
t.string('type');
|
|
122
|
+
t.text('raw').notNullable();
|
|
123
|
+
t.json('parsed');
|
|
124
|
+
t.text('parseError');
|
|
125
|
+
t.index('pageId');
|
|
126
|
+
t.index('type');
|
|
127
|
+
});
|
|
128
|
+
// Compound `(type, pageId)` accelerates streaming
|
|
129
|
+
// `list_pages_by_jsonld_type` JOINs.
|
|
130
|
+
await instance.raw('CREATE INDEX page_jsonld_type_pageId ON page_jsonld(type, pageId)');
|
|
131
|
+
}
|
|
132
|
+
if (!(await instance.schema.hasTable('inventory_runs'))) {
|
|
133
|
+
await instance.schema.createTable('inventory_runs', (t) => {
|
|
134
|
+
// One row per successful `--inventory <list>` invocation. The
|
|
135
|
+
// archive's audit log of "when did we apply which deploy list
|
|
136
|
+
// at what scale". `.bak` is removed on success so this table
|
|
137
|
+
// is the only durable provenance record. Column semantics live
|
|
138
|
+
// on the `InventoryRunMeta` interface in `archive/types.ts`.
|
|
139
|
+
t.increments('id');
|
|
140
|
+
t.string('ran_at').notNullable();
|
|
141
|
+
t.string('list_label').nullable();
|
|
142
|
+
t.string('source_file_sha256', 64).nullable();
|
|
143
|
+
t.integer('total_lines').nullable();
|
|
144
|
+
t.integer('new_pages').nullable();
|
|
145
|
+
t.integer('new_resources').nullable();
|
|
146
|
+
t.integer('scope_skipped').nullable();
|
|
147
|
+
t.text('notes').nullable();
|
|
148
|
+
t.index('ran_at');
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
if (!(await instance.schema.hasTable('analysis_text_refs'))) {
|
|
152
|
+
await instance.raw(`
|
|
153
|
+
CREATE TABLE analysis_text_refs (
|
|
154
|
+
id integer primary key,
|
|
155
|
+
text text not null,
|
|
156
|
+
sha256 text not null,
|
|
157
|
+
unique(sha256, text)
|
|
158
|
+
)
|
|
159
|
+
`);
|
|
160
|
+
}
|
|
161
|
+
if (!(await instance.schema.hasTable('analysis_violations'))) {
|
|
162
|
+
await instance.raw(`
|
|
163
|
+
CREATE TABLE analysis_violations (
|
|
164
|
+
id integer primary key,
|
|
165
|
+
page_id integer not null references content_items(id),
|
|
166
|
+
validator text not null,
|
|
167
|
+
severity text not null,
|
|
168
|
+
rule text not null,
|
|
169
|
+
message_text_id integer not null references analysis_text_refs(id),
|
|
170
|
+
code_text_id integer references analysis_text_refs(id),
|
|
171
|
+
page_url_sort_key text not null,
|
|
172
|
+
message_sort_key text not null,
|
|
173
|
+
code_sort_key text not null,
|
|
174
|
+
line integer,
|
|
175
|
+
col integer
|
|
176
|
+
)
|
|
177
|
+
`);
|
|
178
|
+
await instance.raw('CREATE INDEX av_url_order ON analysis_violations(page_url_sort_key, id)');
|
|
179
|
+
await instance.raw('CREATE INDEX av_filter_url ON analysis_violations(validator, severity, rule, page_url_sort_key, id)');
|
|
180
|
+
await instance.raw('CREATE INDEX av_validator_url ON analysis_violations(validator, page_url_sort_key, id)');
|
|
181
|
+
await instance.raw('CREATE INDEX av_severity_url ON analysis_violations(severity, page_url_sort_key, id)');
|
|
182
|
+
await instance.raw('CREATE INDEX av_rule_url ON analysis_violations(rule, page_url_sort_key, id)');
|
|
183
|
+
await instance.raw('CREATE INDEX av_message_order ON analysis_violations(message_sort_key, id)');
|
|
184
|
+
await instance.raw('CREATE INDEX av_code_order ON analysis_violations(code_sort_key, id)');
|
|
185
|
+
await instance.raw('CREATE INDEX av_page ON analysis_violations(page_id, id)');
|
|
186
|
+
}
|
|
187
|
+
// Content-addressable HTML blob storage. Knex's schema builder doesn't
|
|
188
|
+
// expose a WITHOUT ROWID toggle, so the BLOB tables are created via raw
|
|
189
|
+
// SQL. WITHOUT ROWID keeps the rows packed inside the b-tree leaves
|
|
190
|
+
// (no hidden rowid + secondary index pair), which matters for the blob
|
|
191
|
+
// table where a 32-byte hash PK + multi-KB body is the dominant row
|
|
192
|
+
// shape.
|
|
193
|
+
if (!(await instance.schema.hasTable('page_html_blobs'))) {
|
|
194
|
+
await instance.raw(`
|
|
195
|
+
CREATE TABLE page_html_blobs (
|
|
196
|
+
hash BLOB PRIMARY KEY,
|
|
197
|
+
body BLOB NOT NULL,
|
|
198
|
+
codec TEXT NOT NULL CHECK(codec IN ('zstd', 'none')),
|
|
199
|
+
size_raw INTEGER NOT NULL,
|
|
200
|
+
size_stored INTEGER NOT NULL
|
|
201
|
+
) WITHOUT ROWID
|
|
202
|
+
`);
|
|
203
|
+
}
|
|
204
|
+
if (!(await instance.schema.hasTable('page_html_ref'))) {
|
|
205
|
+
await instance.raw(`
|
|
206
|
+
CREATE TABLE page_html_ref (
|
|
207
|
+
page_id INTEGER PRIMARY KEY REFERENCES content_items(id) ON DELETE CASCADE,
|
|
208
|
+
hash BLOB NOT NULL REFERENCES page_html_blobs(hash)
|
|
209
|
+
) WITHOUT ROWID
|
|
210
|
+
`);
|
|
211
|
+
await instance.raw('CREATE INDEX idx_page_html_ref_hash ON page_html_ref(hash)');
|
|
212
|
+
}
|
|
213
|
+
}
|