@nitpicker/crawler 0.12.0 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (349) hide show
  1. package/README.md +6 -4
  2. package/lib/archive/archive-accessor.d.ts +2 -2
  3. package/lib/archive/archive-accessor.js +2 -2
  4. package/lib/archive/archive-lock.d.ts +7 -0
  5. package/lib/archive/archive-lock.js +7 -0
  6. package/lib/archive/archive.d.ts +63 -16
  7. package/lib/archive/archive.js +56 -17
  8. package/lib/archive/create-adjunct-tables.d.ts +43 -0
  9. package/lib/archive/create-adjunct-tables.js +213 -0
  10. package/lib/archive/create-entity-tables.d.ts +173 -0
  11. package/lib/archive/create-entity-tables.js +318 -0
  12. package/lib/archive/create-progress-reporter.d.ts +30 -0
  13. package/lib/archive/create-progress-reporter.js +38 -0
  14. package/lib/archive/create-ref-tables.d.ts +35 -0
  15. package/lib/archive/create-ref-tables.js +188 -0
  16. package/lib/archive/database.d.ts +92 -345
  17. package/lib/archive/database.js +168 -1942
  18. package/lib/archive/db-ops/_shared/clear-write-ref-caches.d.ts +27 -0
  19. package/lib/archive/db-ops/_shared/clear-write-ref-caches.js +34 -0
  20. package/lib/archive/db-ops/_shared/create-write-ref-caches.d.ts +17 -0
  21. package/lib/archive/db-ops/_shared/create-write-ref-caches.js +26 -0
  22. package/lib/archive/db-ops/_shared/decode-json-ref.d.ts +17 -0
  23. package/lib/archive/db-ops/_shared/decode-json-ref.js +31 -0
  24. package/lib/archive/db-ops/_shared/load-response-headers-by-set-ids.d.ts +20 -0
  25. package/lib/archive/db-ops/_shared/load-response-headers-by-set-ids.js +53 -0
  26. package/lib/archive/db-ops/_shared/resolve-content-item-id.d.ts +61 -0
  27. package/lib/archive/db-ops/_shared/resolve-content-item-id.js +111 -0
  28. package/lib/archive/db-ops/_shared/resolve-url-or-blob.d.ts +23 -0
  29. package/lib/archive/db-ops/_shared/resolve-url-or-blob.js +29 -0
  30. package/lib/archive/db-ops/_shared/retry-setting.d.ts +16 -0
  31. package/lib/archive/db-ops/_shared/retry-setting.js +18 -0
  32. package/lib/archive/db-ops/_shared/safe-parse-json.d.ts +11 -0
  33. package/lib/archive/db-ops/_shared/safe-parse-json.js +18 -0
  34. package/lib/archive/db-ops/_shared/types.d.ts +53 -0
  35. package/lib/archive/db-ops/_shared/types.js +1 -0
  36. package/lib/archive/db-ops/_shared/upsert-blob-ref.d.ts +25 -0
  37. package/lib/archive/db-ops/_shared/upsert-blob-ref.js +48 -0
  38. package/lib/archive/db-ops/_shared/upsert-content-type-ref.d.ts +30 -0
  39. package/lib/archive/db-ops/_shared/upsert-content-type-ref.js +45 -0
  40. package/lib/archive/db-ops/_shared/upsert-json-ref.d.ts +22 -0
  41. package/lib/archive/db-ops/_shared/upsert-json-ref.js +41 -0
  42. package/lib/archive/db-ops/_shared/upsert-response-headers.d.ts +35 -0
  43. package/lib/archive/db-ops/_shared/upsert-response-headers.js +49 -0
  44. package/lib/archive/db-ops/_shared/upsert-url-ref.d.ts +39 -0
  45. package/lib/archive/db-ops/_shared/upsert-url-ref.js +62 -0
  46. package/lib/archive/db-ops/analysis/replace-analysis-violations.d.ts +28 -0
  47. package/lib/archive/db-ops/analysis/replace-analysis-violations.js +152 -0
  48. package/lib/archive/db-ops/anchors/get-anchors-on-page.d.ts +10 -0
  49. package/lib/archive/db-ops/anchors/get-anchors-on-page.js +21 -0
  50. package/lib/archive/db-ops/config/get-base-url.d.ts +8 -0
  51. package/lib/archive/db-ops/config/get-base-url.js +14 -0
  52. package/lib/archive/db-ops/config/get-config.d.ts +10 -0
  53. package/lib/archive/db-ops/config/get-config.js +27 -0
  54. package/lib/archive/db-ops/config/get-name.d.ts +8 -0
  55. package/lib/archive/db-ops/config/get-name.js +14 -0
  56. package/lib/archive/db-ops/config/info-column-allowlist.d.ts +7 -0
  57. package/lib/archive/db-ops/config/info-column-allowlist.js +26 -0
  58. package/lib/archive/db-ops/config/info-json-columns.d.ts +5 -0
  59. package/lib/archive/db-ops/config/info-json-columns.js +10 -0
  60. package/lib/archive/db-ops/config/set-config.d.ts +12 -0
  61. package/lib/archive/db-ops/config/set-config.js +21 -0
  62. package/lib/archive/db-ops/config/update-config.d.ts +17 -0
  63. package/lib/archive/db-ops/config/update-config.js +36 -0
  64. package/lib/archive/db-ops/errors/insert-crawl-error.d.ts +15 -0
  65. package/lib/archive/db-ops/errors/insert-crawl-error.js +21 -0
  66. package/lib/archive/db-ops/errors/insert-page-error.d.ts +21 -0
  67. package/lib/archive/db-ops/errors/insert-page-error.js +28 -0
  68. package/lib/archive/db-ops/errors/list-dns-burned-host-candidates.d.ts +22 -0
  69. package/lib/archive/db-ops/errors/list-dns-burned-host-candidates.js +141 -0
  70. package/lib/archive/db-ops/html/get-html-of-page-by-id.d.ts +18 -0
  71. package/lib/archive/db-ops/html/get-html-of-page-by-id.js +29 -0
  72. package/lib/archive/db-ops/inventory/record-inventory-run.d.ts +21 -0
  73. package/lib/archive/db-ops/inventory/record-inventory-run.js +38 -0
  74. package/lib/archive/db-ops/lifecycle/checkpoint.d.ts +8 -0
  75. package/lib/archive/db-ops/lifecycle/checkpoint.js +9 -0
  76. package/lib/archive/db-ops/lifecycle/destroy.d.ts +6 -0
  77. package/lib/archive/db-ops/lifecycle/destroy.js +7 -0
  78. package/lib/archive/db-ops/lifecycle/init.d.ts +22 -0
  79. package/lib/archive/db-ops/lifecycle/init.js +42 -0
  80. package/lib/archive/db-ops/meta/get-jsonld-of-page.d.ts +13 -0
  81. package/lib/archive/db-ops/meta/get-jsonld-of-page.js +27 -0
  82. package/lib/archive/db-ops/meta/get-tags-of-page.d.ts +12 -0
  83. package/lib/archive/db-ops/meta/get-tags-of-page.js +28 -0
  84. package/lib/archive/db-ops/pages/order/set-url-order.d.ts +8 -0
  85. package/lib/archive/db-ops/pages/order/set-url-order.js +32 -0
  86. package/lib/archive/db-ops/pages/read/build-page-query.d.ts +18 -0
  87. package/lib/archive/db-ops/pages/read/build-page-query.js +40 -0
  88. package/lib/archive/db-ops/pages/read/get-crawling-state.d.ts +70 -0
  89. package/lib/archive/db-ops/pages/read/get-crawling-state.js +98 -0
  90. package/lib/archive/db-ops/pages/read/get-existing-page-urls.d.ts +15 -0
  91. package/lib/archive/db-ops/pages/read/get-existing-page-urls.js +30 -0
  92. package/lib/archive/db-ops/pages/read/get-page-count.d.ts +12 -0
  93. package/lib/archive/db-ops/pages/read/get-page-count.js +21 -0
  94. package/lib/archive/db-ops/pages/read/get-page-source-by-url.d.ts +24 -0
  95. package/lib/archive/db-ops/pages/read/get-page-source-by-url.js +28 -0
  96. package/lib/archive/db-ops/pages/read/get-pages-with-rels.d.ts +38 -0
  97. package/lib/archive/db-ops/pages/read/get-pages-with-rels.js +107 -0
  98. package/lib/archive/db-ops/pages/read/get-pages.d.ts +11 -0
  99. package/lib/archive/db-ops/pages/read/get-pages.js +51 -0
  100. package/lib/archive/db-ops/pages/read/get-scraped-html-page-count.d.ts +18 -0
  101. package/lib/archive/db-ops/pages/read/get-scraped-html-page-count.js +25 -0
  102. package/lib/archive/db-ops/pages/read/reconstruct-page-rows.d.ts +31 -0
  103. package/lib/archive/db-ops/pages/read/reconstruct-page-rows.js +32 -0
  104. package/lib/archive/db-ops/pages/reset/repromote-external-pages.d.ts +24 -0
  105. package/lib/archive/db-ops/pages/reset/repromote-external-pages.js +93 -0
  106. package/lib/archive/db-ops/pages/reset/reset-failed-pages.d.ts +47 -0
  107. package/lib/archive/db-ops/pages/reset/reset-failed-pages.js +124 -0
  108. package/lib/archive/db-ops/pages/write/insert-inventory-seeds.d.ts +37 -0
  109. package/lib/archive/db-ops/pages/write/insert-inventory-seeds.js +72 -0
  110. package/lib/archive/db-ops/pages/write/insert-jsonld.d.ts +17 -0
  111. package/lib/archive/db-ops/pages/write/insert-jsonld.js +49 -0
  112. package/lib/archive/db-ops/pages/write/insert-page.d.ts +36 -0
  113. package/lib/archive/db-ops/pages/write/insert-page.js +208 -0
  114. package/lib/archive/db-ops/pages/write/insert-tags.d.ts +16 -0
  115. package/lib/archive/db-ops/pages/write/insert-tags.js +34 -0
  116. package/lib/archive/db-ops/pages/write/link-redirect-sources.d.ts +36 -0
  117. package/lib/archive/db-ops/pages/write/link-redirect-sources.js +93 -0
  118. package/lib/archive/db-ops/pages/write/record-redirect.d.ts +35 -0
  119. package/lib/archive/db-ops/pages/write/record-redirect.js +100 -0
  120. package/lib/archive/db-ops/pages/write/set-skipped-page.d.ts +13 -0
  121. package/lib/archive/db-ops/pages/write/set-skipped-page.js +22 -0
  122. package/lib/archive/db-ops/pages/write/update-page.d.ts +29 -0
  123. package/lib/archive/db-ops/pages/write/update-page.js +334 -0
  124. package/lib/archive/db-ops/pages/write/write-page-html-blob.d.ts +19 -0
  125. package/lib/archive/db-ops/pages/write/write-page-html-blob.js +41 -0
  126. package/lib/archive/db-ops/referrers/get-redirects-for-pages.d.ts +9 -0
  127. package/lib/archive/db-ops/referrers/get-redirects-for-pages.js +15 -0
  128. package/lib/archive/db-ops/referrers/get-referrers-of-page.d.ts +17 -0
  129. package/lib/archive/db-ops/referrers/get-referrers-of-page.js +32 -0
  130. package/lib/archive/db-ops/referrers/get-referrers-of-resource.d.ts +8 -0
  131. package/lib/archive/db-ops/referrers/get-referrers-of-resource.js +15 -0
  132. package/lib/archive/db-ops/resources/build-resource-query.d.ts +25 -0
  133. package/lib/archive/db-ops/resources/build-resource-query.js +29 -0
  134. package/lib/archive/db-ops/resources/get-existing-resource-urls.d.ts +9 -0
  135. package/lib/archive/db-ops/resources/get-existing-resource-urls.js +24 -0
  136. package/lib/archive/db-ops/resources/get-resource-by-url.d.ts +13 -0
  137. package/lib/archive/db-ops/resources/get-resource-by-url.js +22 -0
  138. package/lib/archive/db-ops/resources/get-resource-url-list.d.ts +9 -0
  139. package/lib/archive/db-ops/resources/get-resource-url-list.js +13 -0
  140. package/lib/archive/db-ops/resources/get-resources.d.ts +8 -0
  141. package/lib/archive/db-ops/resources/get-resources.js +11 -0
  142. package/lib/archive/db-ops/resources/insert-inventory-resources.d.ts +24 -0
  143. package/lib/archive/db-ops/resources/insert-inventory-resources.js +64 -0
  144. package/lib/archive/db-ops/resources/insert-resource-referrers.d.ts +15 -0
  145. package/lib/archive/db-ops/resources/insert-resource-referrers.js +54 -0
  146. package/lib/archive/db-ops/resources/insert-resource.d.ts +34 -0
  147. package/lib/archive/db-ops/resources/insert-resource.js +73 -0
  148. package/lib/archive/db-ops/resources/reconstruct-resource-rows.d.ts +26 -0
  149. package/lib/archive/db-ops/resources/reconstruct-resource-rows.js +30 -0
  150. package/lib/archive/decode-html-blob.d.ts +18 -0
  151. package/lib/archive/decode-html-blob.js +31 -0
  152. package/lib/archive/derive-lineage-from-parent.d.ts +1 -1
  153. package/lib/archive/derive-lineage-from-parent.js +1 -1
  154. package/lib/archive/drop-legacy-tables.d.ts +45 -0
  155. package/lib/archive/drop-legacy-tables.js +56 -0
  156. package/lib/archive/filesystem/rename.js +1 -1
  157. package/lib/archive/get-failed-page-messages.d.ts +5 -4
  158. package/lib/archive/get-failed-page-messages.js +5 -4
  159. package/lib/archive/init-schema.d.ts +35 -39
  160. package/lib/archive/init-schema.js +99 -460
  161. package/lib/archive/limited-page-ids.d.ts +2 -1
  162. package/lib/archive/limited-page-ids.js +5 -4
  163. package/lib/archive/meta/assert-compatible-version.d.ts +24 -3
  164. package/lib/archive/meta/assert-compatible-version.js +24 -3
  165. package/lib/archive/meta/types.d.ts +87 -1
  166. package/lib/archive/meta/types.js +34 -2
  167. package/lib/archive/migrate-entity-tables.d.ts +45 -0
  168. package/lib/archive/migrate-entity-tables.js +56 -0
  169. package/lib/archive/migrate-ref-tables.d.ts +25 -0
  170. package/lib/archive/migrate-ref-tables.js +38 -0
  171. package/lib/archive/page-meta-column-maps.d.ts +32 -0
  172. package/lib/archive/page-meta-column-maps.js +43 -0
  173. package/lib/archive/page.d.ts +6 -6
  174. package/lib/archive/page.js +5 -5
  175. package/lib/archive/peek-archive-lock.d.ts +2 -2
  176. package/lib/archive/peek-archive-lock.js +2 -2
  177. package/lib/archive/populate-entity-tables/collapse-anchor-rows.d.ts +41 -0
  178. package/lib/archive/populate-entity-tables/collapse-anchor-rows.js +87 -0
  179. package/lib/archive/populate-entity-tables/derive-dom-path.d.ts +35 -0
  180. package/lib/archive/populate-entity-tables/derive-dom-path.js +72 -0
  181. package/lib/archive/populate-entity-tables/is-blob-ref-value.d.ts +16 -0
  182. package/lib/archive/populate-entity-tables/is-blob-ref-value.js +19 -0
  183. package/lib/archive/populate-entity-tables/match-images-to-dom-paths.d.ts +66 -0
  184. package/lib/archive/populate-entity-tables/match-images-to-dom-paths.js +96 -0
  185. package/lib/archive/populate-entity-tables/populate-anchor-edges.d.ts +33 -0
  186. package/lib/archive/populate-entity-tables/populate-anchor-edges.js +153 -0
  187. package/lib/archive/populate-entity-tables/populate-content-items.d.ts +40 -0
  188. package/lib/archive/populate-entity-tables/populate-content-items.js +141 -0
  189. package/lib/archive/populate-entity-tables/populate-entities.d.ts +81 -0
  190. package/lib/archive/populate-entity-tables/populate-entities.js +111 -0
  191. package/lib/archive/populate-entity-tables/populate-image-items.d.ts +91 -0
  192. package/lib/archive/populate-entity-tables/populate-image-items.js +223 -0
  193. package/lib/archive/populate-entity-tables/populate-page-meta.d.ts +33 -0
  194. package/lib/archive/populate-entity-tables/populate-page-meta.js +267 -0
  195. package/lib/archive/populate-entity-tables/populate-resource-items.d.ts +22 -0
  196. package/lib/archive/populate-entity-tables/populate-resource-items.js +114 -0
  197. package/lib/archive/populate-entity-tables/populate-resource-ref-edges.d.ts +31 -0
  198. package/lib/archive/populate-entity-tables/populate-resource-ref-edges.js +33 -0
  199. package/lib/archive/populate-entity-tables/resolve-blob-refs.d.ts +31 -0
  200. package/lib/archive/populate-entity-tables/resolve-blob-refs.js +100 -0
  201. package/lib/archive/populate-entity-tables/resolve-content-type-refs.d.ts +22 -0
  202. package/lib/archive/populate-entity-tables/resolve-content-type-refs.js +27 -0
  203. package/lib/archive/populate-entity-tables/resolve-header-sets.d.ts +49 -0
  204. package/lib/archive/populate-entity-tables/resolve-header-sets.js +122 -0
  205. package/lib/archive/populate-entity-tables/resolve-json-refs.d.ts +25 -0
  206. package/lib/archive/populate-entity-tables/resolve-json-refs.js +67 -0
  207. package/lib/archive/populate-entity-tables/resolve-text-refs.d.ts +30 -0
  208. package/lib/archive/populate-entity-tables/resolve-text-refs.js +61 -0
  209. package/lib/archive/populate-entity-tables/resolve-url-or-blob-from-maps.d.ts +21 -0
  210. package/lib/archive/populate-entity-tables/resolve-url-or-blob-from-maps.js +27 -0
  211. package/lib/archive/populate-entity-tables/resolve-url-refs.d.ts +33 -0
  212. package/lib/archive/populate-entity-tables/resolve-url-refs.js +60 -0
  213. package/lib/archive/populate-entity-tables/test-utils/count-rows.d.ts +17 -0
  214. package/lib/archive/populate-entity-tables/test-utils/count-rows.js +20 -0
  215. package/lib/archive/populate-entity-tables/test-utils/seed-content-items.d.ts +25 -0
  216. package/lib/archive/populate-entity-tables/test-utils/seed-content-items.js +42 -0
  217. package/lib/archive/populate-entity-tables/test-utils/setup-entities-db.d.ts +23 -0
  218. package/lib/archive/populate-entity-tables/test-utils/setup-entities-db.js +178 -0
  219. package/lib/archive/populate-entity-tables/types.d.ts +157 -0
  220. package/lib/archive/populate-entity-tables/types.js +12 -0
  221. package/lib/archive/populate-entity-tables/upsert-text-refs.d.ts +38 -0
  222. package/lib/archive/populate-entity-tables/upsert-text-refs.js +78 -0
  223. package/lib/archive/populate-ref-tables/classify-content-type.d.ts +16 -0
  224. package/lib/archive/populate-ref-tables/classify-content-type.js +52 -0
  225. package/lib/archive/populate-ref-tables/compute-content-hash.d.ts +22 -0
  226. package/lib/archive/populate-ref-tables/compute-content-hash.js +26 -0
  227. package/lib/archive/populate-ref-tables/compute-header-flags.d.ts +16 -0
  228. package/lib/archive/populate-ref-tables/compute-header-flags.js +70 -0
  229. package/lib/archive/populate-ref-tables/content-type-rules.d.ts +38 -0
  230. package/lib/archive/populate-ref-tables/content-type-rules.js +133 -0
  231. package/lib/archive/populate-ref-tables/create-header-table-caches.d.ts +25 -0
  232. package/lib/archive/populate-ref-tables/create-header-table-caches.js +49 -0
  233. package/lib/archive/populate-ref-tables/data-uri-url-refs-limit.d.ts +15 -0
  234. package/lib/archive/populate-ref-tables/data-uri-url-refs-limit.js +15 -0
  235. package/lib/archive/populate-ref-tables/decode-data-uri.d.ts +21 -0
  236. package/lib/archive/populate-ref-tables/decode-data-uri.js +126 -0
  237. package/lib/archive/populate-ref-tables/decompose-header-set.d.ts +29 -0
  238. package/lib/archive/populate-ref-tables/decompose-header-set.js +157 -0
  239. package/lib/archive/populate-ref-tables/decompose-url.d.ts +25 -0
  240. package/lib/archive/populate-ref-tables/decompose-url.js +70 -0
  241. package/lib/archive/populate-ref-tables/header-stability.d.ts +19 -0
  242. package/lib/archive/populate-ref-tables/header-stability.js +22 -0
  243. package/lib/archive/populate-ref-tables/header-value-cache-key.d.ts +17 -0
  244. package/lib/archive/populate-ref-tables/header-value-cache-key.js +19 -0
  245. package/lib/archive/populate-ref-tables/normalize-mime.d.ts +24 -0
  246. package/lib/archive/populate-ref-tables/normalize-mime.js +36 -0
  247. package/lib/archive/populate-ref-tables/populate-blob-refs.d.ts +38 -0
  248. package/lib/archive/populate-ref-tables/populate-blob-refs.js +134 -0
  249. package/lib/archive/populate-ref-tables/populate-content-type-refs.d.ts +27 -0
  250. package/lib/archive/populate-ref-tables/populate-content-type-refs.js +70 -0
  251. package/lib/archive/populate-ref-tables/populate-header-tables.d.ts +35 -0
  252. package/lib/archive/populate-ref-tables/populate-header-tables.js +80 -0
  253. package/lib/archive/populate-ref-tables/populate-json-refs.d.ts +29 -0
  254. package/lib/archive/populate-ref-tables/populate-json-refs.js +101 -0
  255. package/lib/archive/populate-ref-tables/populate-refs.d.ts +51 -0
  256. package/lib/archive/populate-ref-tables/populate-refs.js +62 -0
  257. package/lib/archive/populate-ref-tables/populate-text-refs.d.ts +32 -0
  258. package/lib/archive/populate-ref-tables/populate-text-refs.js +133 -0
  259. package/lib/archive/populate-ref-tables/populate-url-refs.d.ts +28 -0
  260. package/lib/archive/populate-ref-tables/populate-url-refs.js +148 -0
  261. package/lib/archive/populate-ref-tables/test-utils/count-rows.d.ts +15 -0
  262. package/lib/archive/populate-ref-tables/test-utils/count-rows.js +17 -0
  263. package/lib/archive/populate-ref-tables/types.d.ts +197 -0
  264. package/lib/archive/populate-ref-tables/types.js +7 -0
  265. package/lib/archive/populate-ref-tables/upsert-one-header-set.d.ts +34 -0
  266. package/lib/archive/populate-ref-tables/upsert-one-header-set.js +208 -0
  267. package/lib/archive/populate-ref-tables/volatile-header-names.d.ts +20 -0
  268. package/lib/archive/populate-ref-tables/volatile-header-names.js +33 -0
  269. package/lib/archive/redirect-table.d.ts +4 -2
  270. package/lib/archive/redirect-table.js +15 -10
  271. package/lib/archive/resolve-redirect-chain.d.ts +3 -3
  272. package/lib/archive/resolve-redirect-chain.js +2 -2
  273. package/lib/archive/resource.d.ts +1 -1
  274. package/lib/archive/retarget-legacy-fk-tables.d.ts +47 -0
  275. package/lib/archive/retarget-legacy-fk-tables.js +107 -0
  276. package/lib/archive/test-utils/fk-parent-tables.d.ts +15 -0
  277. package/lib/archive/test-utils/fk-parent-tables.js +19 -0
  278. package/lib/archive/test-utils/seed-content-item.d.ts +35 -0
  279. package/lib/archive/test-utils/seed-content-item.js +42 -0
  280. package/lib/archive/test-utils/setup-legacy-fk-db.d.ts +33 -0
  281. package/lib/archive/test-utils/setup-legacy-fk-db.js +270 -0
  282. package/lib/archive/types.d.ts +127 -24
  283. package/lib/archive/verify-migration/capture-rejection.d.ts +24 -0
  284. package/lib/archive/verify-migration/capture-rejection.js +31 -0
  285. package/lib/archive/verify-migration/check-anchor-edges-count.d.ts +34 -0
  286. package/lib/archive/verify-migration/check-anchor-edges-count.js +72 -0
  287. package/lib/archive/verify-migration/check-anchor-edges-sum.d.ts +13 -0
  288. package/lib/archive/verify-migration/check-anchor-edges-sum.js +27 -0
  289. package/lib/archive/verify-migration/check-content-items-count.d.ts +16 -0
  290. package/lib/archive/verify-migration/check-content-items-count.js +30 -0
  291. package/lib/archive/verify-migration/check-content-type-preservation.d.ts +22 -0
  292. package/lib/archive/verify-migration/check-content-type-preservation.js +40 -0
  293. package/lib/archive/verify-migration/check-foreign-key-integrity.d.ts +31 -0
  294. package/lib/archive/verify-migration/check-foreign-key-integrity.js +47 -0
  295. package/lib/archive/verify-migration/check-image-items-count.d.ts +12 -0
  296. package/lib/archive/verify-migration/check-image-items-count.js +26 -0
  297. package/lib/archive/verify-migration/check-page-meta-count.d.ts +15 -0
  298. package/lib/archive/verify-migration/check-page-meta-count.js +31 -0
  299. package/lib/archive/verify-migration/check-reader-parity.d.ts +23 -0
  300. package/lib/archive/verify-migration/check-reader-parity.js +211 -0
  301. package/lib/archive/verify-migration/check-resource-items-count.d.ts +17 -0
  302. package/lib/archive/verify-migration/check-resource-items-count.js +33 -0
  303. package/lib/archive/verify-migration/check-url-round-trip.d.ts +43 -0
  304. package/lib/archive/verify-migration/check-url-round-trip.js +112 -0
  305. package/lib/archive/verify-migration/types.d.ts +70 -0
  306. package/lib/archive/verify-migration/types.js +63 -0
  307. package/lib/archive/verify-migration/verify-migration.d.ts +41 -0
  308. package/lib/archive/verify-migration/verify-migration.js +120 -0
  309. package/lib/crawler/build-redirect-event.d.ts +1 -1
  310. package/lib/crawler/build-redirect-event.js +1 -1
  311. package/lib/crawler/capture-image-dom-paths.d.ts +33 -0
  312. package/lib/crawler/capture-image-dom-paths.js +39 -0
  313. package/lib/crawler/clear-dns-burned-host-cache.d.ts +1 -1
  314. package/lib/crawler/clear-dns-burned-host-cache.js +1 -1
  315. package/lib/crawler/collect-image-dom-paths.d.ts +23 -0
  316. package/lib/crawler/collect-image-dom-paths.js +64 -0
  317. package/lib/crawler/crawler.d.ts +19 -0
  318. package/lib/crawler/crawler.js +40 -26
  319. package/lib/crawler/dns-burned-host-cache.d.ts +3 -3
  320. package/lib/crawler/dns-burned-host-cache.js +3 -3
  321. package/lib/crawler/dns-burned-host-short-circuit-counter.d.ts +2 -2
  322. package/lib/crawler/dns-burned-host-short-circuit-counter.js +2 -2
  323. package/lib/crawler/inject-scope-auth.d.ts +1 -1
  324. package/lib/crawler/inject-scope-auth.js +1 -1
  325. package/lib/crawler/normalize-content-type.d.ts +1 -1
  326. package/lib/crawler/normalize-content-type.js +1 -1
  327. package/lib/crawler/types.d.ts +3 -3
  328. package/lib/crawler-orchestrator.d.ts +9 -0
  329. package/lib/crawler-orchestrator.js +44 -28
  330. package/lib/crawler.d.ts +12 -0
  331. package/lib/crawler.js +21 -0
  332. package/lib/permanent-error-kinds.d.ts +1 -1
  333. package/lib/permanent-error-kinds.js +1 -1
  334. package/lib/types.d.ts +1 -1
  335. package/lib/utils/compute-file-sha256.d.ts +5 -4
  336. package/lib/utils/compute-file-sha256.js +5 -4
  337. package/lib/utils/error/emit-error-with-retry.d.ts +1 -1
  338. package/lib/utils/error/emit-error-with-retry.js +1 -1
  339. package/package.json +10 -10
  340. package/lib/archive/migrate-crawl-errors.d.ts +0 -20
  341. package/lib/archive/migrate-crawl-errors.js +0 -38
  342. package/lib/archive/migrate-html-blob-tables.d.ts +0 -24
  343. package/lib/archive/migrate-html-blob-tables.js +0 -53
  344. package/lib/archive/migrate-inventory-runs.d.ts +0 -29
  345. package/lib/archive/migrate-inventory-runs.js +0 -52
  346. package/lib/archive/migrate-page-errors.d.ts +0 -16
  347. package/lib/archive/migrate-page-errors.js +0 -35
  348. package/lib/archive/migrate-pages-resources-source.d.ts +0 -16
  349. package/lib/archive/migrate-pages-resources-source.js +0 -46
package/README.md CHANGED
@@ -1,12 +1,14 @@
1
1
  # @nitpicker/crawler
2
2
 
3
- ヘッドレスブラウザによる Web クローラーエンジン。
3
+ ヘッドレスブラウザでWebサイトをクロールし、`.nitpicker` アーカイブを生成・更新する内部パッケージです。
4
4
 
5
- ## 概要
5
+ 通常は [@nitpicker/cli](../cli/README.md) の `crawl` コマンドから利用します。
6
6
 
7
- Puppeteer を使用して Web サイトをクロールし、各ページのメタデータ・リンク構造・ネットワークリソース・レンダリング後 HTML スナップショットを SQLite ベースのアーカイブ(`.nitpicker`)に保存します。
7
+ ## 関連リンク
8
8
 
9
- このパッケージは [Nitpicker](../../README.md) モノレポの内部パッケージです。単体での利用は想定していません。
9
+ - [Nitpicker README](../../../README.md)
10
+ - [CLI crawl docs](../cli/docs/crawl.md)
11
+ - [ARCHITECTURE.md](../../../ARCHITECTURE.md)
10
12
 
11
13
  ## ライセンス
12
14
 
@@ -39,7 +39,7 @@ export declare class ArchiveAccessor extends EventEmitter<DatabaseEvent> {
39
39
  * tmpDir), where touching the filesystem would race with — or destroy —
40
40
  * the live crawler's working state.
41
41
  *
42
- * Subclasses that own the archive's lifecycle (notably {@link Archive})
42
+ * Subclasses that own the archive's lifecycle (notably `Archive`)
43
43
  * override this to add write/cleanup steps.
44
44
  *
45
45
  * **Idempotent and concurrent-safe**: the first invocation captures the
@@ -171,7 +171,7 @@ export declare class ArchiveAccessor extends EventEmitter<DatabaseEvent> {
171
171
  * Retrieves a flat list of all resource URLs stored in the archive.
172
172
  * @returns An array of resource URL strings.
173
173
  */
174
- getResourceUrlList(): Promise<any[]>;
174
+ getResourceUrlList(): Promise<string[]>;
175
175
  /**
176
176
  * Retrieves the Wappalyzer tag entries for the given page, parsed back
177
177
  * from the `page_tags` table.
@@ -30,7 +30,7 @@ export class ArchiveAccessor extends EventEmitter {
30
30
  * Promise tracking an in-progress (or completed) close. `null` means the
31
31
  * accessor is open and idle; a settled promise means we are closed (the
32
32
  * accessor stays "closed" even if `db.destroy()` rejected, because there
33
- * is nothing safe to retry — see {@link close}).
33
+ * is nothing safe to retry — see {@link ArchiveAccessor.close}).
34
34
  */
35
35
  #closeOnce = null;
36
36
  /** The SQLite database instance for querying archived data. */
@@ -91,7 +91,7 @@ export class ArchiveAccessor extends EventEmitter {
91
91
  * tmpDir), where touching the filesystem would race with — or destroy —
92
92
  * the live crawler's working state.
93
93
  *
94
- * Subclasses that own the archive's lifecycle (notably {@link Archive})
94
+ * Subclasses that own the archive's lifecycle (notably `Archive`)
95
95
  * override this to add write/cleanup steps.
96
96
  *
97
97
  * **Idempotent and concurrent-safe**: the first invocation captures the
@@ -34,5 +34,12 @@ export declare class ArchiveLockError extends Error {
34
34
  * @param tmpDir - Absolute path to the archive's temporary working directory.
35
35
  * @returns A release function to be called when the work is done.
36
36
  * @throws {ArchiveLockError} When the lock cannot be acquired even after a stale-lock retry.
37
+ * @example
38
+ * const releaseLock = await acquireArchiveLock(tmpDir);
39
+ * try {
40
+ * // ... exclusive work against tmpDir ...
41
+ * } finally {
42
+ * await releaseLock();
43
+ * }
37
44
  */
38
45
  export declare function acquireArchiveLock(tmpDir: string): Promise<() => Promise<void>>;
@@ -42,6 +42,13 @@ export class ArchiveLockError extends Error {
42
42
  * @param tmpDir - Absolute path to the archive's temporary working directory.
43
43
  * @returns A release function to be called when the work is done.
44
44
  * @throws {ArchiveLockError} When the lock cannot be acquired even after a stale-lock retry.
45
+ * @example
46
+ * const releaseLock = await acquireArchiveLock(tmpDir);
47
+ * try {
48
+ * // ... exclusive work against tmpDir ...
49
+ * } finally {
50
+ * await releaseLock();
51
+ * }
45
52
  */
46
53
  export async function acquireArchiveLock(tmpDir) {
47
54
  const lockPath = `${tmpDir}.lock`;
@@ -14,6 +14,15 @@ import { ArchiveAccessor } from './archive-accessor.js';
14
14
  * Use the static factory methods ({@link Archive.create}, {@link Archive.open},
15
15
  * {@link Archive.resume}, {@link Archive.connect}) to obtain instances.
16
16
  * The constructor is private.
17
+ * @example
18
+ * const archive = await Archive.create({ filePath: '/path/to/site.nitpicker' });
19
+ * try {
20
+ * await archive.setConfig(config);
21
+ * const pageId = await archive.setPage(pageData);
22
+ * } finally {
23
+ * // Writes the `.nitpicker` tar (if absent), removes tmpDir, releases the lock.
24
+ * await archive.close();
25
+ * }
17
26
  */
18
27
  export default class Archive extends ArchiveAccessor {
19
28
  #private;
@@ -112,15 +121,14 @@ export default class Archive extends ArchiveAccessor {
112
121
  * Retrieves the base URL of the crawl session from the archive database.
113
122
  * @returns The base URL string.
114
123
  */
115
- getUrl(): Promise<any>;
124
+ getUrl(): Promise<string>;
116
125
  /**
117
126
  * Pre-insert inventory non-HTML URLs as `source='inventory-seed'`
118
127
  * placeholders in the `resources` table — the non-HTML counterpart of
119
- * {@link Archive.insertInventorySeeds}. Replaces the previous per-URL
120
- * `setResources` loop in `CrawlerOrchestrator.inventory` so the
121
- * ingestion phase commits all non-HTML rows in one chunked round-trip
122
- * per 500 (a 50k-URL inventory list dropped from minutes-inside-`.bak`
123
- * to seconds).
128
+ * {@link Archive.insertInventorySeeds}. Rows are committed in chunked
129
+ * bulk inserts (500 per round-trip) rather than per-URL awaits — a
130
+ * per-URL loop would keep a 50k-URL inventory list inside the `.bak`
131
+ * window for minutes instead of seconds.
124
132
  *
125
133
  * Thin facade over {@link Database.insertInventoryResources}.
126
134
  * `ExURL.href` is the storage key for `resources.url` (matches what
@@ -137,7 +145,7 @@ export default class Archive extends ArchiveAccessor {
137
145
  * Ctrl+C-tolerance rationale and the `getCrawlingState` interaction.
138
146
  *
139
147
  * `ExURL` inputs are normalised to `withoutHashAndAuth` here so the storage
140
- * key matches what `#getIdByUrl` writes for crawled rows, keeping the
148
+ * key matches what `resolveContentItemId` writes for crawled rows, keeping the
141
149
  * crawled-wins downgrade and the existing-URL filter (`getExistingPageUrls`)
142
150
  * lookups consistent.
143
151
  * @param urls - HTML seed URLs to pre-insert. No-op when empty.
@@ -180,6 +188,24 @@ export default class Archive extends ArchiveAccessor {
180
188
  * are mutually exclusive (the first one called wins).
181
189
  */
182
190
  releaseHandle(): Promise<void>;
191
+ /**
192
+ * Replaces the archive's analysis violations with a fresh SQL-backed set.
193
+ *
194
+ * Thin facade over {@link Database.replaceAnalysisViolations}; kept on
195
+ * `Archive` so the analyze pipeline can persist violations without
196
+ * reaching into the low-level database class directly.
197
+ * @param violations - Flat analyze violations.
198
+ */
199
+ replaceAnalysisViolations(violations: readonly {
200
+ validator: string;
201
+ severity: string;
202
+ rule: string;
203
+ code?: string | null;
204
+ message: string;
205
+ url: string;
206
+ line?: number | null;
207
+ col?: number | null;
208
+ }[]): Promise<void>;
183
209
  /**
184
210
  * Promote previously-external pages that now fall under the (possibly extended)
185
211
  * scope back to a pending state so that the crawler re-scrapes them as fully
@@ -287,23 +313,44 @@ export default class Archive extends ArchiveAccessor {
287
313
  /** The prefix used for temporary working directories during archive operations. */
288
314
  static TMP_DIR_PREFIX: string;
289
315
  /**
290
- * Opens a read-only connection to an existing archive's database.
316
+ * Opens a connection to an existing archive's database, defaulting to
317
+ * read-only.
291
318
  *
292
- * Returns an {@link ArchiveAccessor} that provides query methods
293
- * without the ability to modify or write the archive. The DB is opened
294
- * in **read-only mode**: no schema migrations run, and the connection
319
+ * Returns an {@link ArchiveAccessor} that provides query methods. In the
320
+ * default read-only mode, no schema migrations run and the connection
295
321
  * refuses to resurrect a missing parent directory or db file (so a
296
322
  * TOCTOU window between source classification and this call cannot
297
- * silently produce an empty phantom tmpDir).
323
+ * silently produce an empty phantom tmpDir); the returned accessor is
324
+ * also marked read-only so consumer-facing helpers (e.g.
325
+ * {@link ArchiveAccessor.getHtmlOfPage}) avoid any filesystem mutation
326
+ * on the user's tmpDir.
298
327
  *
299
- * The returned accessor is also marked read-only so consumer-facing
300
- * helpers (e.g. {@link ArchiveAccessor.getHtmlOfPage}) avoid any
301
- * filesystem mutation on the user's tmpDir.
328
+ * `options.readOnly: false` is a narrow escape hatch for opening a
329
+ * second, writable connection to a `tmpDir` that {@link Archive.openCached}
330
+ * already extracted (and migrated) into an OS-temp cache directory —
331
+ * never the caller's live/interrupted crawl tmpDir, which must stay
332
+ * read-only. A read-only open (`Archive.openCached`/`ArchiveManager.open`)
333
+ * must never take this path itself — blocking or writing during what
334
+ * must be a read-only open is forbidden (issue #177). This escape
335
+ * hatch has no current production caller; any future
336
+ * one is responsible for its own cross-process coordination (see
337
+ * `acquireArchiveLock`) — this method does not acquire any lock itself.
302
338
  * @param tmpDir - The path to the temporary directory containing the database.
303
339
  * @param namespace - An optional namespace for scoping data access within the archive.
340
+ * @param options - Connection options.
341
+ * @param options.readOnly - Defaults to `true`. Pass `false` to obtain a
342
+ * writable accessor against an already-extracted cache directory.
304
343
  * @returns An ArchiveAccessor instance for querying the archive data.
344
+ * @example
345
+ * // Default (read-only) — safe for stub mode and cache reads:
346
+ * const accessor = await Archive.connect(tmpDir);
347
+ * @example
348
+ * // Writable escape hatch — only against a tar-cache extraction:
349
+ * const writable = await Archive.connect(cacheDir, null, { readOnly: false });
305
350
  */
306
- static connect(tmpDir: string, namespace?: string | null): Promise<ArchiveAccessor>;
351
+ static connect(tmpDir: string, namespace?: string | null, options?: {
352
+ readOnly?: boolean;
353
+ }): Promise<ArchiveAccessor>;
307
354
  /**
308
355
  * Open a `.nitpicker` archive through the read-only tar cache.
309
356
  *
@@ -27,6 +27,15 @@ import { untar } from './filesystem/untar.js';
27
27
  * Use the static factory methods ({@link Archive.create}, {@link Archive.open},
28
28
  * {@link Archive.resume}, {@link Archive.connect}) to obtain instances.
29
29
  * The constructor is private.
30
+ * @example
31
+ * const archive = await Archive.create({ filePath: '/path/to/site.nitpicker' });
32
+ * try {
33
+ * await archive.setConfig(config);
34
+ * const pageId = await archive.setPage(pageData);
35
+ * } finally {
36
+ * // Writes the `.nitpicker` tar (if absent), removes tmpDir, releases the lock.
37
+ * await archive.close();
38
+ * }
30
39
  */
31
40
  export default class Archive extends ArchiveAccessor {
32
41
  /**
@@ -180,11 +189,10 @@ export default class Archive extends ArchiveAccessor {
180
189
  /**
181
190
  * Pre-insert inventory non-HTML URLs as `source='inventory-seed'`
182
191
  * placeholders in the `resources` table — the non-HTML counterpart of
183
- * {@link Archive.insertInventorySeeds}. Replaces the previous per-URL
184
- * `setResources` loop in `CrawlerOrchestrator.inventory` so the
185
- * ingestion phase commits all non-HTML rows in one chunked round-trip
186
- * per 500 (a 50k-URL inventory list dropped from minutes-inside-`.bak`
187
- * to seconds).
192
+ * {@link Archive.insertInventorySeeds}. Rows are committed in chunked
193
+ * bulk inserts (500 per round-trip) rather than per-URL awaits — a
194
+ * per-URL loop would keep a 50k-URL inventory list inside the `.bak`
195
+ * window for minutes instead of seconds.
188
196
  *
189
197
  * Thin facade over {@link Database.insertInventoryResources}.
190
198
  * `ExURL.href` is the storage key for `resources.url` (matches what
@@ -207,7 +215,7 @@ export default class Archive extends ArchiveAccessor {
207
215
  * Ctrl+C-tolerance rationale and the `getCrawlingState` interaction.
208
216
  *
209
217
  * `ExURL` inputs are normalised to `withoutHashAndAuth` here so the storage
210
- * key matches what `#getIdByUrl` writes for crawled rows, keeping the
218
+ * key matches what `resolveContentItemId` writes for crawled rows, keeping the
211
219
  * crawled-wins downgrade and the existing-URL filter (`getExistingPageUrls`)
212
220
  * lookups consistent.
213
221
  * @param urls - HTML seed URLs to pre-insert. No-op when empty.
@@ -267,6 +275,17 @@ export default class Archive extends ArchiveAccessor {
267
275
  this.#closeOnce = this.#runReleaseHandle();
268
276
  return this.#closeOnce;
269
277
  }
278
+ /**
279
+ * Replaces the archive's analysis violations with a fresh SQL-backed set.
280
+ *
281
+ * Thin facade over {@link Database.replaceAnalysisViolations}; kept on
282
+ * `Archive` so the analyze pipeline can persist violations without
283
+ * reaching into the low-level database class directly.
284
+ * @param violations - Flat analyze violations.
285
+ */
286
+ async replaceAnalysisViolations(violations) {
287
+ await this.#db.replaceAnalysisViolations(violations);
288
+ }
270
289
  /**
271
290
  * Promote previously-external pages that now fall under the (possibly extended)
272
291
  * scope back to a pending state so that the crawler re-scrapes them as fully
@@ -452,25 +471,45 @@ export default class Archive extends ArchiveAccessor {
452
471
  /** The prefix used for temporary working directories during archive operations. */
453
472
  static TMP_DIR_PREFIX = '._nitpicker-';
454
473
  /**
455
- * Opens a read-only connection to an existing archive's database.
474
+ * Opens a connection to an existing archive's database, defaulting to
475
+ * read-only.
456
476
  *
457
- * Returns an {@link ArchiveAccessor} that provides query methods
458
- * without the ability to modify or write the archive. The DB is opened
459
- * in **read-only mode**: no schema migrations run, and the connection
477
+ * Returns an {@link ArchiveAccessor} that provides query methods. In the
478
+ * default read-only mode, no schema migrations run and the connection
460
479
  * refuses to resurrect a missing parent directory or db file (so a
461
480
  * TOCTOU window between source classification and this call cannot
462
- * silently produce an empty phantom tmpDir).
481
+ * silently produce an empty phantom tmpDir); the returned accessor is
482
+ * also marked read-only so consumer-facing helpers (e.g.
483
+ * {@link ArchiveAccessor.getHtmlOfPage}) avoid any filesystem mutation
484
+ * on the user's tmpDir.
463
485
  *
464
- * The returned accessor is also marked read-only so consumer-facing
465
- * helpers (e.g. {@link ArchiveAccessor.getHtmlOfPage}) avoid any
466
- * filesystem mutation on the user's tmpDir.
486
+ * `options.readOnly: false` is a narrow escape hatch for opening a
487
+ * second, writable connection to a `tmpDir` that {@link Archive.openCached}
488
+ * already extracted (and migrated) into an OS-temp cache directory —
489
+ * never the caller's live/interrupted crawl tmpDir, which must stay
490
+ * read-only. A read-only open (`Archive.openCached`/`ArchiveManager.open`)
491
+ * must never take this path itself — blocking or writing during what
492
+ * must be a read-only open is forbidden (issue #177). This escape
493
+ * hatch has no current production caller; any future
494
+ * one is responsible for its own cross-process coordination (see
495
+ * `acquireArchiveLock`) — this method does not acquire any lock itself.
467
496
  * @param tmpDir - The path to the temporary directory containing the database.
468
497
  * @param namespace - An optional namespace for scoping data access within the archive.
498
+ * @param options - Connection options.
499
+ * @param options.readOnly - Defaults to `true`. Pass `false` to obtain a
500
+ * writable accessor against an already-extracted cache directory.
469
501
  * @returns An ArchiveAccessor instance for querying the archive data.
502
+ * @example
503
+ * // Default (read-only) — safe for stub mode and cache reads:
504
+ * const accessor = await Archive.connect(tmpDir);
505
+ * @example
506
+ * // Writable escape hatch — only against a tar-cache extraction:
507
+ * const writable = await Archive.connect(cacheDir, null, { readOnly: false });
470
508
  */
471
- static async connect(tmpDir, namespace = null) {
472
- const db = await Archive.#connectDB(tmpDir, { readOnly: true });
473
- const archive = new ArchiveAccessor(tmpDir, db, namespace, { readOnly: true });
509
+ static async connect(tmpDir, namespace = null, options = {}) {
510
+ const readOnly = options.readOnly ?? true;
511
+ const db = await Archive.#connectDB(tmpDir, { readOnly });
512
+ const archive = new ArchiveAccessor(tmpDir, db, namespace, { readOnly });
474
513
  return archive;
475
514
  }
476
515
  /**
@@ -0,0 +1,43 @@
1
+ import type { Knex } from 'knex';
2
+ /**
3
+ * Creates the adjunct tables that hang off `content_items` (plus the two
4
+ * standalone log tables), guarded per-table so partially-provisioned
5
+ * archives converge to the full set:
6
+ *
7
+ * - `page_errors` — partial scrape failures, FK → `content_items(id)`
8
+ * - `crawl_errors` — crawler-level error channel (no FK; the URL may be
9
+ * an external link that failed DNS, or null for a process-level error)
10
+ * - `page_tags` — Wappalyzer detections, FK → `content_items(id)`
11
+ * - `page_jsonld` — JSON-LD / SpeculationRules, FK → `content_items(id)`
12
+ * - `inventory_runs` — `--inventory` audit log (no FK; append-only)
13
+ * - `analysis_text_refs` + `analysis_violations` — analyze-phase findings,
14
+ * FK → `content_items(id)`
15
+ * - `page_html_blobs` + `page_html_ref` — content-addressable HTML
16
+ * snapshots, FK → `content_items(id)`
17
+ *
18
+ * The DDL is shared between fresh-archive provisioning ({@link initSchema}
19
+ * calls this right after `createEntityTables`) and the migration script
20
+ * (`scripts/migrate-to-0.13.mjs` calls it before retargeting FK
21
+ * declarations and dropping the legacy tables). Keeping the schema in one
22
+ * function guarantees both origin points produce identical tables — the
23
+ * pre-0.13 era kept per-table copies of this DDL in separate lazy-migration
24
+ * modules, and those copies drifted: they declared `REFERENCES pages(id)`
25
+ * while `initSchema` had moved on to `content_items(id)`, leaving migrated
26
+ * archives with stale FK targets that only `scripts/migrate-to-0.13.mjs`'s
27
+ * rename-copy-drop pass can now repair.
28
+ *
29
+ * Unlike `createRefTables` / `createEntityTables` (whose callers guard with
30
+ * a single sentinel table), each table here is guarded individually because
31
+ * the migration-script caller sees archives where any subset may already
32
+ * exist (e.g. `page_tags` from the 0.10 migration but no `inventory_runs`).
33
+ * Index creation stays inside each guard: an existing table keeps whatever
34
+ * indexes its creation path declared.
35
+ * @param instance - The Knex query builder instance connected to the database.
36
+ * @example
37
+ * // Must run after createEntityTables — the page-scoped tables FK into
38
+ * // content_items(id).
39
+ * await createRefTables(db);
40
+ * await createEntityTables(db);
41
+ * await createAdjunctTables(db);
42
+ */
43
+ export declare function createAdjunctTables(instance: Knex): Promise<void>;
@@ -0,0 +1,213 @@
1
+ /**
2
+ * Creates the adjunct tables that hang off `content_items` (plus the two
3
+ * standalone log tables), guarded per-table so partially-provisioned
4
+ * archives converge to the full set:
5
+ *
6
+ * - `page_errors` — partial scrape failures, FK → `content_items(id)`
7
+ * - `crawl_errors` — crawler-level error channel (no FK; the URL may be
8
+ * an external link that failed DNS, or null for a process-level error)
9
+ * - `page_tags` — Wappalyzer detections, FK → `content_items(id)`
10
+ * - `page_jsonld` — JSON-LD / SpeculationRules, FK → `content_items(id)`
11
+ * - `inventory_runs` — `--inventory` audit log (no FK; append-only)
12
+ * - `analysis_text_refs` + `analysis_violations` — analyze-phase findings,
13
+ * FK → `content_items(id)`
14
+ * - `page_html_blobs` + `page_html_ref` — content-addressable HTML
15
+ * snapshots, FK → `content_items(id)`
16
+ *
17
+ * The DDL is shared between fresh-archive provisioning ({@link initSchema}
18
+ * calls this right after `createEntityTables`) and the migration script
19
+ * (`scripts/migrate-to-0.13.mjs` calls it before retargeting FK
20
+ * declarations and dropping the legacy tables). Keeping the schema in one
21
+ * function guarantees both origin points produce identical tables — the
22
+ * pre-0.13 era kept per-table copies of this DDL in separate lazy-migration
23
+ * modules, and those copies drifted: they declared `REFERENCES pages(id)`
24
+ * while `initSchema` had moved on to `content_items(id)`, leaving migrated
25
+ * archives with stale FK targets that only `scripts/migrate-to-0.13.mjs`'s
26
+ * rename-copy-drop pass can now repair.
27
+ *
28
+ * Unlike `createRefTables` / `createEntityTables` (whose callers guard with
29
+ * a single sentinel table), each table here is guarded individually because
30
+ * the migration-script caller sees archives where any subset may already
31
+ * exist (e.g. `page_tags` from the 0.10 migration but no `inventory_runs`).
32
+ * Index creation stays inside each guard: an existing table keeps whatever
33
+ * indexes its creation path declared.
34
+ * @param instance - The Knex query builder instance connected to the database.
35
+ * @example
36
+ * // Must run after createEntityTables — the page-scoped tables FK into
37
+ * // content_items(id).
38
+ * await createRefTables(db);
39
+ * await createEntityTables(db);
40
+ * await createAdjunctTables(db);
41
+ */
42
+ export async function createAdjunctTables(instance) {
43
+ if (!(await instance.schema.hasTable('page_errors'))) {
44
+ await instance.schema.createTable('page_errors', (t) => {
45
+ // Records partial scrape failures (e.g. a viewport switch that
46
+ // detaches the frame and trips beholder's @retryable into the
47
+ // `retryExhausted` phase). A page can have zero or more rows here
48
+ // in addition to its normal `content_items` entry — the page
49
+ // itself is considered successfully scraped, but image capture or
50
+ // another secondary step failed for at least one device preset.
51
+ t.increments('id');
52
+ t.integer('pageId').notNullable().unsigned().references('content_items.id');
53
+ t.string('phase').notNullable();
54
+ t.text('message').notNullable();
55
+ t.integer('createdAt').notNullable();
56
+ t.index('pageId');
57
+ });
58
+ }
59
+ if (!(await instance.schema.hasTable('crawl_errors'))) {
60
+ await instance.schema.createTable('crawl_errors', (t) => {
61
+ // Structured form of the crawler-level `error` channel that otherwise
62
+ // only lands in `error.log`. Unlike `page_errors` these are not tied to
63
+ // a scraped page (the URL may be an external link that failed DNS, or
64
+ // null for a process-level error), so there is no `pageId` FK and `url`
65
+ // is nullable. The cause is NOT stored — it is classified on read from
66
+ // `message` so older archives (which only have `error.log`) classify the
67
+ // same way.
68
+ t.increments('id');
69
+ t.string('url', 8190).nullable();
70
+ t.boolean('isExternal');
71
+ t.text('message').notNullable();
72
+ t.integer('createdAt').notNullable();
73
+ });
74
+ }
75
+ if (!(await instance.schema.hasTable('page_tags'))) {
76
+ await instance.schema.createTable('page_tags', (t) => {
77
+ // Wappalyzer-derived technology detection. One row per
78
+ // (provider × externalId) tuple per page. `category` is the first
79
+ // element of `categories`; the full list lives in the JSON
80
+ // `categories` column. `sources` records where the provider was
81
+ // detected (script-src / inline / iframe-src / window-global / …).
82
+ t.increments('id');
83
+ t.integer('pageId')
84
+ .notNullable()
85
+ .unsigned()
86
+ .references('content_items.id')
87
+ .onDelete('CASCADE');
88
+ t.string('provider').notNullable();
89
+ t.string('category');
90
+ t.string('externalId');
91
+ t.string('version');
92
+ t.integer('confidence');
93
+ t.json('categories');
94
+ t.json('sources');
95
+ t.index('pageId');
96
+ t.index('provider');
97
+ t.index('externalId');
98
+ });
99
+ // Compound indexes for the "find duplicate IDs across pages" and
100
+ // "list pages using provider X" hot paths. Knex's schema builder
101
+ // can't express compound indexes inline in a way that round-trips
102
+ // through libsql consistently, so raw SQL is used.
103
+ await instance.raw('CREATE INDEX page_tags_provider_extId ON page_tags(provider, externalId)');
104
+ await instance.raw('CREATE INDEX page_tags_provider_pageId ON page_tags(provider, pageId)');
105
+ }
106
+ if (!(await instance.schema.hasTable('page_jsonld'))) {
107
+ await instance.schema.createTable('page_jsonld', (t) => {
108
+ // JSON-LD and SpeculationRules entries captured from
109
+ // `<script type="application/ld+json">` and
110
+ // `<script type="speculationrules">`. `kind` discriminates; `type`
111
+ // is the top-level `@type` extracted by classify-jsonld-type for
112
+ // indexable filtering. `raw` is stored uncompressed; SQLite
113
+ // overflow pages handle multi-KB JSON bodies transparently.
114
+ t.increments('id');
115
+ t.integer('pageId')
116
+ .notNullable()
117
+ .unsigned()
118
+ .references('content_items.id')
119
+ .onDelete('CASCADE');
120
+ t.string('kind').notNullable();
121
+ t.string('type');
122
+ t.text('raw').notNullable();
123
+ t.json('parsed');
124
+ t.text('parseError');
125
+ t.index('pageId');
126
+ t.index('type');
127
+ });
128
+ // Compound `(type, pageId)` accelerates streaming
129
+ // `list_pages_by_jsonld_type` JOINs.
130
+ await instance.raw('CREATE INDEX page_jsonld_type_pageId ON page_jsonld(type, pageId)');
131
+ }
132
+ if (!(await instance.schema.hasTable('inventory_runs'))) {
133
+ await instance.schema.createTable('inventory_runs', (t) => {
134
+ // One row per successful `--inventory <list>` invocation. The
135
+ // archive's audit log of "when did we apply which deploy list
136
+ // at what scale". `.bak` is removed on success so this table
137
+ // is the only durable provenance record. Column semantics live
138
+ // on the `InventoryRunMeta` interface in `archive/types.ts`.
139
+ t.increments('id');
140
+ t.string('ran_at').notNullable();
141
+ t.string('list_label').nullable();
142
+ t.string('source_file_sha256', 64).nullable();
143
+ t.integer('total_lines').nullable();
144
+ t.integer('new_pages').nullable();
145
+ t.integer('new_resources').nullable();
146
+ t.integer('scope_skipped').nullable();
147
+ t.text('notes').nullable();
148
+ t.index('ran_at');
149
+ });
150
+ }
151
+ if (!(await instance.schema.hasTable('analysis_text_refs'))) {
152
+ await instance.raw(`
153
+ CREATE TABLE analysis_text_refs (
154
+ id integer primary key,
155
+ text text not null,
156
+ sha256 text not null,
157
+ unique(sha256, text)
158
+ )
159
+ `);
160
+ }
161
+ if (!(await instance.schema.hasTable('analysis_violations'))) {
162
+ await instance.raw(`
163
+ CREATE TABLE analysis_violations (
164
+ id integer primary key,
165
+ page_id integer not null references content_items(id),
166
+ validator text not null,
167
+ severity text not null,
168
+ rule text not null,
169
+ message_text_id integer not null references analysis_text_refs(id),
170
+ code_text_id integer references analysis_text_refs(id),
171
+ page_url_sort_key text not null,
172
+ message_sort_key text not null,
173
+ code_sort_key text not null,
174
+ line integer,
175
+ col integer
176
+ )
177
+ `);
178
+ await instance.raw('CREATE INDEX av_url_order ON analysis_violations(page_url_sort_key, id)');
179
+ await instance.raw('CREATE INDEX av_filter_url ON analysis_violations(validator, severity, rule, page_url_sort_key, id)');
180
+ await instance.raw('CREATE INDEX av_validator_url ON analysis_violations(validator, page_url_sort_key, id)');
181
+ await instance.raw('CREATE INDEX av_severity_url ON analysis_violations(severity, page_url_sort_key, id)');
182
+ await instance.raw('CREATE INDEX av_rule_url ON analysis_violations(rule, page_url_sort_key, id)');
183
+ await instance.raw('CREATE INDEX av_message_order ON analysis_violations(message_sort_key, id)');
184
+ await instance.raw('CREATE INDEX av_code_order ON analysis_violations(code_sort_key, id)');
185
+ await instance.raw('CREATE INDEX av_page ON analysis_violations(page_id, id)');
186
+ }
187
+ // Content-addressable HTML blob storage. Knex's schema builder doesn't
188
+ // expose a WITHOUT ROWID toggle, so the BLOB tables are created via raw
189
+ // SQL. WITHOUT ROWID keeps the rows packed inside the b-tree leaves
190
+ // (no hidden rowid + secondary index pair), which matters for the blob
191
+ // table where a 32-byte hash PK + multi-KB body is the dominant row
192
+ // shape.
193
+ if (!(await instance.schema.hasTable('page_html_blobs'))) {
194
+ await instance.raw(`
195
+ CREATE TABLE page_html_blobs (
196
+ hash BLOB PRIMARY KEY,
197
+ body BLOB NOT NULL,
198
+ codec TEXT NOT NULL CHECK(codec IN ('zstd', 'none')),
199
+ size_raw INTEGER NOT NULL,
200
+ size_stored INTEGER NOT NULL
201
+ ) WITHOUT ROWID
202
+ `);
203
+ }
204
+ if (!(await instance.schema.hasTable('page_html_ref'))) {
205
+ await instance.raw(`
206
+ CREATE TABLE page_html_ref (
207
+ page_id INTEGER PRIMARY KEY REFERENCES content_items(id) ON DELETE CASCADE,
208
+ hash BLOB NOT NULL REFERENCES page_html_blobs(hash)
209
+ ) WITHOUT ROWID
210
+ `);
211
+ await instance.raw('CREATE INDEX idx_page_html_ref_hash ON page_html_ref(hash)');
212
+ }
213
+ }