mcp-scraper 0.88.2 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/CHANGELOG.md +36 -15
  2. package/README.md +6 -5
  3. package/THIRD_PARTY_NOTICES.html +203 -0
  4. package/dist/analytics-repository-2BMT5JNE.js +1 -0
  5. package/dist/bin/api-server.js +2 -41
  6. package/dist/bin/mcp-scraper-cli.js +39 -756
  7. package/dist/bin/mcp-scraper-core.js +1 -60
  8. package/dist/bin/mcp-scraper-install.js +2 -25
  9. package/dist/bin/mcp-stdio-server.js +1 -19
  10. package/dist/bin/paa-harvest.js +1 -41
  11. package/dist/chunk-3GP5CYZX.js +1 -0
  12. package/dist/chunk-4AI7DOS7.js +59 -0
  13. package/dist/chunk-4FROKQJN.js +1 -0
  14. package/dist/chunk-7YGVI5J4.js +21710 -0
  15. package/dist/chunk-CCYWSJNG.js +16 -0
  16. package/dist/chunk-CFI6CXIV.js +182 -0
  17. package/dist/chunk-E2WRWV3A.js +1 -0
  18. package/dist/chunk-GMWKPIYX.js +1172 -0
  19. package/dist/chunk-HDPYG3XV.js +102 -0
  20. package/dist/chunk-HE45FFBU.js +1 -0
  21. package/dist/chunk-HUV2WTRW.js +1 -0
  22. package/dist/chunk-KJQXUZ4Y.js +4 -0
  23. package/dist/chunk-L4CGLFPU.js +4 -0
  24. package/dist/chunk-M22MM4N4.js +84 -0
  25. package/dist/chunk-MASR22K4.js +73 -0
  26. package/dist/chunk-MZN4U5BL.js +1 -0
  27. package/dist/chunk-PUHFVA7P.js +1280 -0
  28. package/dist/chunk-QPWPR5XG.js +10 -0
  29. package/dist/chunk-TMB56NCA.js +1 -0
  30. package/dist/chunk-TXENITMS.js +20 -0
  31. package/dist/chunk-W2BVJ7S2.js +13 -0
  32. package/dist/chunk-WO3N5FH2.js +5 -0
  33. package/dist/chunk-WSCGYRWA.js +2595 -0
  34. package/dist/chunk-X54CQLK2.js +1 -0
  35. package/dist/chunk-XLWNEVUZ.js +27 -0
  36. package/dist/chunk-XPZVJIZ2.js +100 -0
  37. package/dist/chunk-YQZGZBB4.js +1 -0
  38. package/dist/chunk-Z2QGQJS2.js +1 -0
  39. package/dist/db-F2MX63GI.js +1 -0
  40. package/dist/extract-bundle-SNUIHM3J.js +26 -0
  41. package/dist/gmail-service-BZ3H75XC.js +1 -0
  42. package/dist/index.cjs +21750 -6045
  43. package/dist/index.d.cts +14 -14
  44. package/dist/index.d.ts +14 -14
  45. package/dist/index.js +18 -315
  46. package/dist/lead-list-enrichment-repository-S2H3U7T7.js +1 -0
  47. package/dist/location-data-repository-OTWHWMV6.js +1 -0
  48. package/dist/server-RFR2A5UJ.js +7303 -0
  49. package/dist/site-extract-repository-SE776XDC.js +1 -0
  50. package/dist/worker-XUDSM3AL.js +1 -0
  51. package/package.json +17 -124
  52. package/dist/analytics-repository-GGJJCVVP.js +0 -194
  53. package/dist/chunk-4QMUF6XM.js +0 -1013
  54. package/dist/chunk-6HAV7LCE.js +0 -265
  55. package/dist/chunk-ABF2CGOZ.js +0 -113
  56. package/dist/chunk-C5Z4OFKW.js +0 -404
  57. package/dist/chunk-DNM65UCK.js +0 -299
  58. package/dist/chunk-EQGTEHLZ.js +0 -592
  59. package/dist/chunk-F5GQJWZU.js +0 -732
  60. package/dist/chunk-GGZEC22A.js +0 -215
  61. package/dist/chunk-GXBZXWXB.js +0 -184
  62. package/dist/chunk-IHXAXYIS.js +0 -843
  63. package/dist/chunk-K3Z5AQYE.js +0 -683
  64. package/dist/chunk-K45K75OF.js +0 -6
  65. package/dist/chunk-LFW2FRPJ.js +0 -224
  66. package/dist/chunk-MZDNZQWT.js +0 -2078
  67. package/dist/chunk-NVUKO5NN.js +0 -256
  68. package/dist/chunk-OM7HVEJ3.js +0 -26
  69. package/dist/chunk-OPQIGAFB.js +0 -286
  70. package/dist/chunk-OZJMVCDK.js +0 -16
  71. package/dist/chunk-P7FWOMU7.js +0 -505
  72. package/dist/chunk-PGJQDMC2.js +0 -383
  73. package/dist/chunk-PKZS6SHW.js +0 -33139
  74. package/dist/chunk-RJ7JVYKU.js +0 -68
  75. package/dist/chunk-S24LFPL7.js +0 -5262
  76. package/dist/chunk-T3MZISOF.js +0 -240
  77. package/dist/chunk-UZPTGUDV.js +0 -1915
  78. package/dist/chunk-X623GTBV.js +0 -8290
  79. package/dist/chunk-YXNDOQXN.js +0 -4018
  80. package/dist/db-Z34LPZNR.js +0 -284
  81. package/dist/extract-bundle-565SBZCR.js +0 -1003
  82. package/dist/gmail-service-E6ALS7JG.js +0 -25
  83. package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
  84. package/dist/location-data-repository-WPRG62GE.js +0 -34
  85. package/dist/server-SQZ3A7SY.js +0 -86606
  86. package/dist/site-extract-repository-VYFZASPU.js +0 -69
  87. package/dist/worker-LDCAULWL.js +0 -146
@@ -1,843 +0,0 @@
1
- import {
2
- createPrivateArtifact,
3
- createPrivateArtifactFromStream,
4
- privateArtifactOwnerId,
5
- readPrivateArtifactBuffer,
6
- renewPrivateArtifactDownload
7
- } from "./chunk-DNM65UCK.js";
8
- import {
9
- LedgerOperation
10
- } from "./chunk-4QMUF6XM.js";
11
- import {
12
- CREDIT_LOT_TTL,
13
- getDb,
14
- parsePublicErrorEnvelope,
15
- sanitizeVendorName,
16
- serializePublicErrorEnvelope,
17
- settleDebitMcIdempotent
18
- } from "./chunk-YXNDOQXN.js";
19
-
20
- // src/api/site-extract-content-store.ts
21
- import { createHash } from "crypto";
22
- import { gunzipSync, gzipSync } from "zlib";
23
-
24
- // src/api/site-extract-artifacts.ts
25
- var SITE_EXTRACT_ARTIFACT_PREFIX = "site-extracts/";
26
- var SITE_EXTRACT_ARTIFACT_TTL_MS = 7 * 24 * 60 * 60 * 1e3;
27
- var SITE_EXTRACT_DOWNLOAD_TTL_MS = 15 * 60 * 1e3;
28
- function siteExtractArtifactToken() {
29
- return process.env.SITE_EXTRACT_ARTIFACT_READ_WRITE_TOKEN?.trim() || process.env.PRIVATE_ARTIFACT_READ_WRITE_TOKEN?.trim() || process.env.CONNECTED_DATA_READ_WRITE_TOKEN?.trim() || process.env.CONNECTED_DATA_BLOB_READ_WRITE_TOKEN?.trim() || null;
30
- }
31
- function hostedByEnvironment() {
32
- return process.env.VERCEL === "1" || process.env.NODE_ENV === "production";
33
- }
34
- function policy() {
35
- return {
36
- prefix: SITE_EXTRACT_ARTIFACT_PREFIX,
37
- artifactTtlMs: SITE_EXTRACT_ARTIFACT_TTL_MS,
38
- downloadTtlMs: SITE_EXTRACT_DOWNLOAD_TTL_MS,
39
- token: siteExtractArtifactToken()
40
- };
41
- }
42
- function siteExtractArtifactOwnerId(artifactId) {
43
- return privateArtifactOwnerId(artifactId, SITE_EXTRACT_ARTIFACT_PREFIX);
44
- }
45
- async function createSiteExtractBundleArtifactStream(args) {
46
- const pointer = await createPrivateArtifactFromStream({
47
- policy: policy(),
48
- ownerId: args.ownerId,
49
- artifactKey: `${args.jobId}.zip`,
50
- createdAt: args.createdAt,
51
- filename: `${args.jobId}-site-export.zip`,
52
- contentType: "application/zip",
53
- content: args.content
54
- });
55
- return {
56
- key: pointer.artifactId,
57
- url: pointer.downloadUrl ?? "",
58
- bytes: pointer.bytes,
59
- contentType: pointer.contentType,
60
- filename: pointer.filename,
61
- sha256: pointer.sha256,
62
- expiresAt: pointer.expiresAt,
63
- downloadUrlExpiresAt: pointer.downloadUrlExpiresAt,
64
- kind: "bundle"
65
- };
66
- }
67
- async function createSiteExtractImageArtifact(args) {
68
- const pointer = await createPrivateArtifact({
69
- policy: policy(),
70
- ownerId: args.ownerId,
71
- scopeSegments: [args.jobId, "images"],
72
- artifactKey: `${args.imageId}.bin`,
73
- createdAt: /* @__PURE__ */ new Date(),
74
- filename: args.filename,
75
- contentType: args.contentType,
76
- content: args.content
77
- });
78
- return {
79
- key: pointer.artifactId,
80
- url: pointer.downloadUrl ?? "",
81
- bytes: pointer.bytes,
82
- contentType: pointer.contentType,
83
- filename: pointer.filename,
84
- sha256: pointer.sha256,
85
- expiresAt: pointer.expiresAt,
86
- downloadUrlExpiresAt: pointer.downloadUrlExpiresAt,
87
- kind: "image",
88
- imageId: args.imageId,
89
- sourceUrl: args.sourceUrl,
90
- sourcePage: args.sourcePage
91
- };
92
- }
93
- async function createSiteExtractContentChunkArtifact(args) {
94
- return createPrivateArtifact({
95
- policy: policy(),
96
- ownerId: args.ownerId,
97
- scopeSegments: [args.jobId, "content"],
98
- artifactKey: `${args.chunkKey}.json.gz`,
99
- createdAt: args.createdAt,
100
- filename: `${args.chunkKey}.json.gz`,
101
- contentType: "application/gzip",
102
- content: args.content
103
- });
104
- }
105
- async function renewSiteExtractArtifactDownload(args) {
106
- return renewPrivateArtifactDownload({
107
- policy: policy(),
108
- artifactId: args.artifactId,
109
- ownerId: args.ownerId
110
- });
111
- }
112
- async function readSiteExtractArtifactBuffer(artifactId) {
113
- return readPrivateArtifactBuffer({
114
- policy: policy(),
115
- artifactId,
116
- maxBytes: 50 * 1024 * 1024
117
- });
118
- }
119
- async function readOwnedSiteExtractArtifactBuffer(args) {
120
- if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
121
- return readSiteExtractArtifactBuffer(args.artifactId);
122
- }
123
- async function readOwnedSiteExtractContentChunk(args) {
124
- if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
125
- return readPrivateArtifactBuffer({
126
- policy: policy(),
127
- artifactId: args.artifactId,
128
- maxBytes: 25 * 1024 * 1024
129
- });
130
- }
131
- async function readOwnedSiteExtractImageArtifact(args) {
132
- if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
133
- return readPrivateArtifactBuffer({
134
- policy: policy(),
135
- artifactId: args.artifactId,
136
- maxBytes: 10 * 1024 * 1024
137
- });
138
- }
139
- async function cleanupExpiredSiteExtractArtifacts(args = {}) {
140
- const now = args.now ?? /* @__PURE__ */ new Date();
141
- const token = siteExtractArtifactToken();
142
- if (!token) return { deleted: 0, store: hostedByEnvironment() ? "none" : "local" };
143
- if (!args.force && !(now.getUTCHours() === 3 && now.getUTCMinutes() === 21)) {
144
- return { deleted: 0, store: "private-vercel-blob", skipped: true };
145
- }
146
- const cutoff = now.getTime() - SITE_EXTRACT_ARTIFACT_TTL_MS;
147
- const { list, del } = await import("@vercel/blob");
148
- let cursor;
149
- let deleted = 0;
150
- for (let page = 0; page < 20; page += 1) {
151
- const result = await list({ prefix: SITE_EXTRACT_ARTIFACT_PREFIX, token, limit: 1e3, cursor });
152
- const expired = result.blobs.filter((blob) => new Date(blob.uploadedAt).getTime() <= cutoff);
153
- if (expired.length > 0) {
154
- await del(expired.map((blob) => blob.pathname), { token });
155
- deleted += expired.length;
156
- }
157
- if (!result.hasMore || !result.cursor) break;
158
- cursor = result.cursor;
159
- }
160
- return { deleted, store: "private-vercel-blob" };
161
- }
162
-
163
- // src/api/site-extract-content-store.ts
164
- var MAX_PAGES_PER_CHUNK = 10;
165
- var MAX_UNCOMPRESSED_CHUNK_BYTES = 20 * 1024 * 1024;
166
- function sha256(value) {
167
- return createHash("sha256").update(value).digest("hex");
168
- }
169
- function pageContent(page) {
170
- if (!page.acquiredHtml || page.extractionStatus === "failed") return null;
171
- const pageId = page.pageId ?? sha256(page.archivedUrl ?? page.url);
172
- return {
173
- pageId,
174
- url: page.archivedUrl ?? page.url,
175
- html: page.acquiredHtml,
176
- mainHtml: page.mainHtml ?? "",
177
- markdown: page.bodyMarkdown,
178
- htmlSha256: page.htmlSha256 ?? sha256(page.acquiredHtml),
179
- markdownSha256: page.markdownSha256 ?? sha256(page.bodyMarkdown)
180
- };
181
- }
182
- function serializedChunk(pages) {
183
- return Buffer.from(JSON.stringify({ version: "site-extract-content.v1", pages }));
184
- }
185
- async function persistSiteExtractPageContent(input) {
186
- const contentPages = input.pages.flatMap((page) => {
187
- const content = pageContent(page);
188
- return content ? [content] : [];
189
- });
190
- const refs = /* @__PURE__ */ new Map();
191
- let chunk = [];
192
- let chunkNumber = 0;
193
- const flush = async () => {
194
- if (!chunk.length) return;
195
- const uncompressed = serializedChunk(chunk);
196
- if (uncompressed.length > MAX_UNCOMPRESSED_CHUNK_BYTES) {
197
- throw new Error(`site export content chunk exceeds ${MAX_UNCOMPRESSED_CHUNK_BYTES} bytes`);
198
- }
199
- const identity = sha256(chunk.map((page) => page.pageId).join("\0")).slice(0, 16);
200
- const pointer = await createSiteExtractContentChunkArtifact({
201
- ownerId: input.ownerId,
202
- jobId: input.jobId,
203
- createdAt: input.createdAt,
204
- chunkKey: `content-${String(chunkNumber).padStart(4, "0")}-${identity}`,
205
- content: gzipSync(uncompressed, { level: 6 })
206
- });
207
- for (const page of chunk) {
208
- refs.set(page.url, {
209
- artifactId: pointer.artifactId,
210
- artifactSha256: pointer.sha256,
211
- pageId: page.pageId,
212
- uncompressedBytes: uncompressed.length
213
- });
214
- }
215
- chunkNumber++;
216
- chunk = [];
217
- };
218
- for (const page of contentPages) {
219
- const candidate = [...chunk, page];
220
- if (chunk.length > 0 && (candidate.length > MAX_PAGES_PER_CHUNK || serializedChunk(candidate).length > MAX_UNCOMPRESSED_CHUNK_BYTES)) {
221
- await flush();
222
- }
223
- chunk.push(page);
224
- if (serializedChunk(chunk).length > MAX_UNCOMPRESSED_CHUNK_BYTES) {
225
- throw new Error(`page ${page.pageId} exceeds the durable content chunk limit`);
226
- }
227
- }
228
- await flush();
229
- return refs;
230
- }
231
- function createSiteExtractContentReader(ownerId) {
232
- const cache = /* @__PURE__ */ new Map();
233
- return async (ref) => {
234
- let chunk = cache.get(ref.artifactId);
235
- if (!chunk) {
236
- const compressed = await readOwnedSiteExtractContentChunk({ artifactId: ref.artifactId, ownerId });
237
- if (!compressed) throw new Error("site export content chunk is missing or unauthorized");
238
- if (sha256(compressed) !== ref.artifactSha256) throw new Error("site export content chunk checksum mismatch");
239
- const decoded = gunzipSync(compressed);
240
- if (decoded.length > MAX_UNCOMPRESSED_CHUNK_BYTES) throw new Error("site export content chunk exceeds its read limit");
241
- chunk = JSON.parse(decoded.toString("utf8"));
242
- if (chunk.version !== "site-extract-content.v1" || !Array.isArray(chunk.pages)) {
243
- throw new Error("site export content chunk has an unsupported contract");
244
- }
245
- cache.set(ref.artifactId, chunk);
246
- while (cache.size > 2) cache.delete(cache.keys().next().value);
247
- }
248
- const page = chunk.pages.find((candidate) => candidate.pageId === ref.pageId);
249
- if (!page) throw new Error("site export page is missing from its content chunk");
250
- if (sha256(page.html) !== page.htmlSha256 || sha256(page.markdown) !== page.markdownSha256) {
251
- throw new Error("site export page checksum mismatch");
252
- }
253
- return page;
254
- };
255
- }
256
-
257
- // src/api/site-extract-repository.ts
258
- var XRAY_ENTITLED_EXTRACT_BILLING_CLASS = "xray_entitlement";
259
- function extractJobLimitInfo(job) {
260
- const effectiveMaxPages = Math.max(1, Number(job.options.effectiveMaxPages ?? job.options.maxPages ?? 1));
261
- const requestedMaxPages = Math.max(effectiveMaxPages, Number(job.options.requestedMaxPages ?? effectiveMaxPages));
262
- const creditLimited = requestedMaxPages > effectiveMaxPages;
263
- return {
264
- requestedMaxPages,
265
- effectiveMaxPages,
266
- creditLimited,
267
- // If discovery exhausted below the funded cap, the smaller hold did not
268
- // actually truncate this crawl. Hitting the cap is conservatively partial.
269
- creditTruncated: creditLimited && job.totalUrls >= effectiveMaxPages
270
- };
271
- }
272
- function terminalExtractJobStatus(progress, creditTruncated = false) {
273
- if (progress.successfulUrls === 0) return "failed";
274
- if (progress.failedUrls > 0 || progress.remainingUrls > 0 || creditTruncated) return "partial";
275
- return "complete";
276
- }
277
- function rowToJob(r) {
278
- const totalUrls = Number(r.total_urls ?? 0);
279
- const legacyProgress = r.attempted_urls == null;
280
- const attemptedUrls = Number(legacyProgress ? r.done_urls ?? 0 : r.attempted_urls);
281
- const successfulUrls = Number(legacyProgress ? r.done_urls ?? 0 : r.successful_urls ?? 0);
282
- const failedUrls = Number(r.failed_urls ?? Math.max(0, attemptedUrls - successfulUrls));
283
- return {
284
- id: String(r.id),
285
- userId: r.user_id != null ? Number(r.user_id) : null,
286
- idempotencyKey: r.idempotency_key != null ? String(r.idempotency_key) : null,
287
- requestFingerprint: r.request_fingerprint != null ? String(r.request_fingerprint) : null,
288
- status: r.status != null ? String(r.status) : "pending",
289
- startUrl: String(r.start_url ?? ""),
290
- options: r.options ? JSON.parse(String(r.options)) : {},
291
- totalUrls,
292
- doneUrls: attemptedUrls,
293
- attemptedUrls,
294
- successfulUrls,
295
- failedUrls,
296
- remainingUrls: Math.max(0, totalUrls - attemptedUrls),
297
- artifacts: r.artifacts ? JSON.parse(String(r.artifacts)) : null,
298
- error: r.error != null ? String(r.error) : null,
299
- publicError: parsePublicErrorEnvelope(r.public_error_json),
300
- billedMc: r.billed_mc != null ? Number(r.billed_mc) : null,
301
- createdAt: String(r.created_at ?? ""),
302
- updatedAt: String(r.updated_at ?? "")
303
- };
304
- }
305
- async function createExtractJob(jobId, userId, startUrl, options) {
306
- const db = getDb();
307
- await db.execute({
308
- sql: `INSERT INTO site_extract_jobs (id, user_id, status, start_url, options, created_at, updated_at)
309
- VALUES (?, ?, 'pending', ?, ?, datetime('now'), datetime('now'))`,
310
- args: [jobId, userId, startUrl, JSON.stringify(options)]
311
- });
312
- }
313
- async function getExtractJobByIdempotencyKey(userId, idempotencyKey) {
314
- const db = getDb();
315
- const res = await db.execute({
316
- sql: `SELECT * FROM site_extract_jobs WHERE user_id = ? AND idempotency_key = ? LIMIT 1`,
317
- args: [userId, idempotencyKey]
318
- });
319
- return res.rows[0] ? rowToJob(res.rows[0]) : null;
320
- }
321
- async function createOrGetExtractJob(input) {
322
- const db = getDb();
323
- const inserted = await db.execute({
324
- sql: `INSERT OR IGNORE INTO site_extract_jobs
325
- (id, user_id, status, start_url, options, idempotency_key, request_fingerprint, created_at, updated_at)
326
- VALUES (?, ?, 'pending', ?, ?, ?, ?, datetime('now'), datetime('now'))`,
327
- args: [
328
- input.jobId,
329
- input.userId,
330
- input.startUrl,
331
- JSON.stringify(input.options),
332
- input.idempotencyKey,
333
- input.requestFingerprint
334
- ]
335
- });
336
- const job = await getExtractJobByIdempotencyKey(input.userId, input.idempotencyKey);
337
- if (!job) throw new Error("idempotent extract job was not persisted");
338
- return {
339
- job,
340
- created: inserted.rowsAffected === 1,
341
- conflict: job.requestFingerprint !== input.requestFingerprint
342
- };
343
- }
344
- async function getExtractJob(jobId) {
345
- const db = getDb();
346
- const res = await db.execute({ sql: `SELECT * FROM site_extract_jobs WHERE id = ?`, args: [jobId] });
347
- return res.rows[0] ? rowToJob(res.rows[0]) : null;
348
- }
349
- async function listExtractJobs(userId) {
350
- const db = getDb();
351
- const res = await db.execute({ sql: `SELECT * FROM site_extract_jobs WHERE user_id = ? ORDER BY created_at DESC LIMIT 50`, args: [userId] });
352
- return res.rows.map((r) => rowToJob(r));
353
- }
354
- async function listUnsettledExtractJobs(limit = 25) {
355
- const db = getDb();
356
- const res = await db.execute({
357
- sql: `SELECT * FROM site_extract_jobs
358
- WHERE billed_mc IS NULL AND status IN ('complete', 'partial', 'failed')
359
- AND (next_settlement_at IS NULL OR next_settlement_at <= datetime('now'))
360
- ORDER BY updated_at ASC LIMIT ?`,
361
- args: [Math.max(1, Math.min(100, Math.round(limit)))]
362
- });
363
- return res.rows.map((row) => rowToJob(row));
364
- }
365
- async function listFundedPendingExtractJobs(limit = 25, staleMinutes = 2) {
366
- const db = getDb();
367
- const safeStaleMinutes = Math.max(1, Math.min(60, Math.round(staleMinutes)));
368
- const res = await db.execute({
369
- sql: `SELECT j.* FROM site_extract_jobs j
370
- LEFT JOIN billing_debits d
371
- ON d.idempotency_key = json_extract(j.options, '$.debitKey')
372
- AND d.user_id = j.user_id
373
- AND d.status = 'applied'
374
- WHERE j.status = 'pending'
375
- AND j.billed_mc IS NULL
376
- AND (
377
- d.idempotency_key IS NOT NULL
378
- OR (
379
- json_extract(j.options, '$.billingClass') = ?
380
- AND json_extract(j.options, '$.heldMc') = 0
381
- AND json_extract(j.options, '$.debitKey') IS NULL
382
- AND json_extract(j.options, '$.xraySetupReceipt.billingClass') = ?
383
- AND json_extract(j.options, '$.xraySetupReceipt.consumesMcpScraperCredits') = 0
384
- )
385
- )
386
- AND (j.next_dispatch_at IS NULL OR j.next_dispatch_at <= datetime('now'))
387
- AND j.updated_at <= datetime('now', ?)
388
- ORDER BY j.updated_at ASC LIMIT ?`,
389
- args: [
390
- XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
391
- XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
392
- `-${safeStaleMinutes} minutes`,
393
- Math.max(1, Math.min(100, Math.round(limit)))
394
- ]
395
- });
396
- return res.rows.map((row) => rowToJob(row));
397
- }
398
- async function markExtractJobDispatchAttempt(jobId) {
399
- const db = getDb();
400
- await db.execute({
401
- sql: `UPDATE site_extract_jobs
402
- SET dispatch_attempts = dispatch_attempts + 1,
403
- updated_at = datetime('now'),
404
- next_dispatch_at = datetime('now', '+5 minutes'),
405
- dispatch_error = NULL,
406
- status = CASE
407
- WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours') THEN 'failed'
408
- ELSE status
409
- END,
410
- error = CASE
411
- WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours')
412
- THEN 'Background crawl was acknowledged repeatedly but never started; the hold will be refunded.'
413
- ELSE error
414
- END
415
- WHERE id = ? AND status = 'pending'`,
416
- args: [jobId]
417
- });
418
- return getExtractJob(jobId);
419
- }
420
- async function recordExtractJobDispatchFailure(jobId, error) {
421
- const db = getDb();
422
- const safeError = sanitizeVendorName(error).slice(0, 1e3);
423
- await db.execute({
424
- sql: `UPDATE site_extract_jobs SET
425
- dispatch_attempts = dispatch_attempts + 1,
426
- dispatch_error = ?,
427
- next_dispatch_at = datetime('now', '+5 minutes'),
428
- updated_at = datetime('now'),
429
- status = CASE
430
- WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours') THEN 'failed'
431
- ELSE status
432
- END,
433
- error = CASE
434
- WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours')
435
- THEN 'Background crawl could not be dispatched after repeated attempts; the hold will be refunded.'
436
- ELSE error
437
- END
438
- WHERE id = ? AND status = 'pending'`,
439
- args: [safeError, jobId]
440
- });
441
- return getExtractJob(jobId);
442
- }
443
- async function recordExtractSettlementFailure(jobId, error) {
444
- const db = getDb();
445
- await db.execute({
446
- sql: `UPDATE site_extract_jobs SET
447
- settlement_attempts = settlement_attempts + 1,
448
- settlement_error = ?,
449
- next_settlement_at = datetime('now', '+15 minutes'),
450
- updated_at = datetime('now')
451
- WHERE id = ? AND billed_mc IS NULL`,
452
- args: [sanitizeVendorName(error).slice(0, 1e3), jobId]
453
- });
454
- }
455
- async function listStaleRunningExtractJobs(limit = 25, staleMinutes = 60) {
456
- const db = getDb();
457
- const safeStaleMinutes = Math.max(15, Math.min(24 * 60, Math.round(staleMinutes)));
458
- const res = await db.execute({
459
- sql: `SELECT j.* FROM site_extract_jobs j
460
- LEFT JOIN billing_debits d
461
- ON d.idempotency_key = json_extract(j.options, '$.debitKey')
462
- AND d.user_id = j.user_id
463
- AND d.status = 'applied'
464
- WHERE j.status = 'running'
465
- AND j.billed_mc IS NULL
466
- AND (
467
- json_extract(j.options, '$.debitKey') IS NULL
468
- OR d.idempotency_key IS NOT NULL
469
- )
470
- AND j.updated_at <= datetime('now', ?)
471
- ORDER BY j.updated_at ASC LIMIT ?`,
472
- args: [`-${safeStaleMinutes} minutes`, Math.max(1, Math.min(100, Math.round(limit)))]
473
- });
474
- return res.rows.map((row) => rowToJob(row));
475
- }
476
- async function failStaleRunningExtractJob(jobId, staleMinutes = 60) {
477
- const db = getDb();
478
- const safeStaleMinutes = Math.max(15, Math.min(24 * 60, Math.round(staleMinutes)));
479
- const result = await db.execute({
480
- sql: `UPDATE site_extract_jobs
481
- SET status = 'failed',
482
- error = 'Background crawl stopped without a heartbeat; completed pages were preserved and settlement is pending.',
483
- public_error_json = ?,
484
- updated_at = datetime('now')
485
- WHERE id = ?
486
- AND status = 'running'
487
- AND billed_mc IS NULL
488
- AND updated_at <= datetime('now', ?)`,
489
- args: [
490
- serializePublicErrorEnvelope({
491
- error_code: "extraction_failed",
492
- error_type: "extraction",
493
- message: "The background crawl stopped without a heartbeat. Completed pages were preserved; retry after settlement completes.",
494
- retryable: true,
495
- charge_status: "refund_pending"
496
- }),
497
- jobId,
498
- `-${safeStaleMinutes} minutes`
499
- ]
500
- });
501
- return result.rowsAffected === 1;
502
- }
503
- async function claimFailedExtractJobForRefinalize(jobId) {
504
- const db = getDb();
505
- const claimed = await db.execute({
506
- sql: `UPDATE site_extract_jobs
507
- SET status = 'running', updated_at = datetime('now')
508
- WHERE id = ? AND status = 'failed'`,
509
- args: [jobId]
510
- });
511
- if (claimed.rowsAffected !== 1) return null;
512
- return getExtractJob(jobId);
513
- }
514
- async function abandonExtractSettlement(jobId, error) {
515
- const db = getDb();
516
- await db.execute({
517
- sql: `UPDATE site_extract_jobs SET billed_mc = 0, settlement_error = ?, updated_at = datetime('now')
518
- WHERE id = ? AND billed_mc IS NULL`,
519
- args: [sanitizeVendorName(error).slice(0, 1e3), jobId]
520
- });
521
- }
522
- async function setExtractJobTotal(jobId, totalUrls) {
523
- const db = getDb();
524
- await db.execute({
525
- sql: `UPDATE site_extract_jobs
526
- SET status = 'running', total_urls = MAX(total_urls, ?), updated_at = datetime('now')
527
- WHERE id = ? AND status IN ('pending', 'running')`,
528
- args: [totalUrls, jobId]
529
- });
530
- }
531
- async function saveExtractPages(jobId, pages) {
532
- if (pages.length === 0) return;
533
- const db = getDb();
534
- const job = await getExtractJob(jobId);
535
- const contentRefs = job?.userId != null ? await persistSiteExtractPageContent({
536
- ownerId: String(job.userId),
537
- jobId,
538
- createdAt: job.createdAt,
539
- pages
540
- }) : /* @__PURE__ */ new Map();
541
- await db.batch([
542
- ...pages.map((p) => {
543
- const {
544
- discoveryLinks: _ephemeralDiscoveryLinks,
545
- acquiredHtml: _durableHtml,
546
- mainHtml: _durableMainHtml,
547
- ...storedPage
548
- } = p;
549
- storedPage.contentRef = contentRefs.get(p.archivedUrl ?? p.url) ?? p.contentRef;
550
- return {
551
- sql: `INSERT INTO site_extract_pages (job_id, url, page)
552
- SELECT ?, ?, ?
553
- WHERE EXISTS (
554
- SELECT 1 FROM site_extract_jobs
555
- WHERE id = ? AND status IN ('pending', 'running')
556
- )
557
- ON CONFLICT(job_id, url) DO UPDATE SET page = excluded.page
558
- WHERE NOT CASE
559
- WHEN json_valid(site_extract_pages.page) = 0 THEN 0
560
- WHEN json_extract(site_extract_pages.page, '$.extractionStatus') IS NOT NULL
561
- THEN json_extract(site_extract_pages.page, '$.extractionStatus') = 'successful'
562
- ELSE COALESCE(
563
- json_extract(site_extract_pages.page, '$.status') >= 200
564
- AND json_extract(site_extract_pages.page, '$.status') < 300
565
- AND json_extract(site_extract_pages.page, '$.wordCount') > 0,
566
- 0
567
- )
568
- END
569
- OR CASE
570
- WHEN json_valid(excluded.page) = 0 THEN 0
571
- WHEN json_extract(excluded.page, '$.extractionStatus') IS NOT NULL
572
- THEN json_extract(excluded.page, '$.extractionStatus') = 'successful'
573
- ELSE COALESCE(
574
- json_extract(excluded.page, '$.status') >= 200
575
- AND json_extract(excluded.page, '$.status') < 300
576
- AND json_extract(excluded.page, '$.wordCount') > 0,
577
- 0
578
- )
579
- END`,
580
- args: [jobId, p.archivedUrl ?? p.url, JSON.stringify(storedPage), jobId]
581
- };
582
- }),
583
- {
584
- sql: `UPDATE site_extract_jobs SET
585
- done_urls = (SELECT COUNT(*) FROM site_extract_pages WHERE job_id = ?),
586
- attempted_urls = (SELECT COUNT(*) FROM site_extract_pages WHERE job_id = ?),
587
- successful_urls = (
588
- SELECT COUNT(*) FROM site_extract_pages
589
- WHERE job_id = ? AND CASE
590
- WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
591
- THEN json_extract(page, '$.extractionStatus') = 'successful'
592
- ELSE json_extract(page, '$.status') >= 200
593
- AND json_extract(page, '$.status') < 300
594
- AND json_extract(page, '$.wordCount') > 0
595
- END
596
- ),
597
- failed_urls = (
598
- SELECT COUNT(*) FROM site_extract_pages
599
- WHERE job_id = ? AND NOT CASE
600
- WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
601
- THEN json_extract(page, '$.extractionStatus') = 'successful'
602
- ELSE COALESCE(
603
- json_extract(page, '$.status') >= 200
604
- AND json_extract(page, '$.status') < 300
605
- AND json_extract(page, '$.wordCount') > 0,
606
- 0
607
- )
608
- END
609
- ),
610
- updated_at = datetime('now')
611
- WHERE id = ? AND status IN ('pending', 'running')`,
612
- args: [jobId, jobId, jobId, jobId, jobId]
613
- }
614
- ], "write");
615
- }
616
- async function listExtractPages(jobId) {
617
- const db = getDb();
618
- const result = await db.execute({
619
- sql: `SELECT page FROM site_extract_pages WHERE job_id = ? ORDER BY rowid`,
620
- args: [jobId]
621
- });
622
- return result.rows.map((row) => JSON.parse(String(row.page)));
623
- }
624
- async function getExtractedPages(jobId) {
625
- const db = getDb();
626
- const res = await db.execute({ sql: `SELECT page FROM site_extract_pages WHERE job_id = ?`, args: [jobId] });
627
- return res.rows.map((r) => JSON.parse(String(r.page)));
628
- }
629
- async function getExtractedImageLinks(jobId, limit = 181) {
630
- const db = getDb();
631
- const safeLimit = Math.max(1, Math.min(5001, Math.round(limit)));
632
- const res = await db.execute({
633
- sql: `SELECT DISTINCT CAST(images.value AS TEXT) AS url
634
- FROM site_extract_pages p,
635
- json_each(CASE WHEN json_valid(p.page) THEN p.page ELSE '{}' END, '$.imageLinks') images
636
- WHERE p.job_id = ?
637
- AND images.type = 'text'
638
- AND length(CAST(images.value AS TEXT)) BETWEEN 1 AND 4096
639
- LIMIT ?`,
640
- args: [jobId, safeLimit]
641
- });
642
- return res.rows.map((row) => String(row.url));
643
- }
644
- async function countSuccessfulPages(jobId) {
645
- const db = getDb();
646
- const res = await db.execute({
647
- sql: `SELECT COUNT(*) AS n FROM site_extract_pages
648
- WHERE job_id = ? AND CASE
649
- WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
650
- THEN json_extract(page, '$.extractionStatus') = 'successful'
651
- ELSE COALESCE(
652
- json_extract(page, '$.status') >= 200
653
- AND json_extract(page, '$.status') < 300
654
- AND json_extract(page, '$.wordCount') > 0,
655
- 0
656
- )
657
- END`,
658
- args: [jobId]
659
- });
660
- return Number(res.rows[0]?.n ?? 0);
661
- }
662
- async function getExtractedUrls(jobId) {
663
- const db = getDb();
664
- const res = await db.execute({ sql: `SELECT url FROM site_extract_pages WHERE job_id = ?`, args: [jobId] });
665
- return new Set(res.rows.map((r) => String(r.url)));
666
- }
667
- async function finishExtractJob(jobId, artifacts, status, error = null, allowFailedRecovery = false, publicError = null) {
668
- const db = getDb();
669
- const result = await db.execute({
670
- sql: `UPDATE site_extract_jobs
671
- SET status = ?, artifacts = ?, error = ?, public_error_json = ?, updated_at = datetime('now')
672
- WHERE id = ? AND (status IN ('pending', 'running') OR (? = 1 AND status = 'failed'))`,
673
- args: [
674
- status,
675
- JSON.stringify(artifacts ?? []),
676
- error ? sanitizeVendorName(error).slice(0, 2e3) : null,
677
- serializePublicErrorEnvelope(publicError),
678
- jobId,
679
- allowFailedRecovery ? 1 : 0
680
- ]
681
- });
682
- return result.rowsAffected === 1;
683
- }
684
- async function completeExtractJob(jobId, artifacts) {
685
- await finishExtractJob(jobId, artifacts, "complete");
686
- }
687
- async function failExtractJob(jobId, error, publicError) {
688
- const db = getDb();
689
- await db.execute({
690
- sql: `UPDATE site_extract_jobs
691
- SET status = 'failed', error = ?, public_error_json = ?, updated_at = datetime('now')
692
- WHERE id = ? AND status IN ('pending', 'running')`,
693
- args: [sanitizeVendorName(error).slice(0, 2e3), serializePublicErrorEnvelope(publicError), jobId]
694
- });
695
- }
696
- async function failUnfundedExtractJob(jobId, error, publicError) {
697
- const db = getDb();
698
- await db.execute({
699
- sql: `UPDATE site_extract_jobs
700
- SET status = 'failed', billed_mc = 0, error = ?, public_error_json = ?, updated_at = datetime('now')
701
- WHERE id = ? AND status = 'pending' AND billed_mc IS NULL`,
702
- args: [sanitizeVendorName(error).slice(0, 2e3), serializePublicErrorEnvelope(publicError), jobId]
703
- });
704
- }
705
- async function setExtractJobPublicError(jobId, publicError) {
706
- const job = await getExtractJob(jobId);
707
- const normalized = job?.options.billingClass === XRAY_ENTITLED_EXTRACT_BILLING_CLASS && publicError ? { ...publicError, charge_status: "not_charged" } : publicError;
708
- await getDb().execute({
709
- sql: `UPDATE site_extract_jobs SET public_error_json = ?, updated_at = datetime('now')
710
- WHERE id = ? AND status IN ('complete', 'partial', 'failed')`,
711
- args: [serializePublicErrorEnvelope(normalized), jobId]
712
- });
713
- }
714
- async function settleExtractJob(jobId, userId, refundMc, netChargeMc, reference) {
715
- const db = getDb();
716
- const job = await getExtractJob(jobId);
717
- if (!job || job.billedMc != null) return;
718
- if (job.options.billingClass === XRAY_ENTITLED_EXTRACT_BILLING_CLASS) {
719
- if (!await finalizeXRayEntitledExtractJobCharge(jobId)) {
720
- throw new Error("xray_extract_zero_charge_finalization_rejected");
721
- }
722
- return;
723
- }
724
- const debitKey = typeof job.options.debitKey === "string" ? job.options.debitKey : null;
725
- if (debitKey) {
726
- const settlement = await settleDebitMcIdempotent(
727
- userId,
728
- debitKey,
729
- Math.max(0, netChargeMc),
730
- LedgerOperation.EXTRACT_SITE_REFUND,
731
- reference,
732
- "extract_site"
733
- );
734
- await db.execute({
735
- sql: `UPDATE site_extract_jobs SET billed_mc = ?, updated_at = datetime('now') WHERE id = ? AND billed_mc IS NULL`,
736
- args: [settlement.final_amount_mc, jobId]
737
- });
738
- return;
739
- }
740
- const safeRefundMc = Math.max(0, refundMc);
741
- const safeNetChargeMc = Math.max(0, netChargeMc);
742
- await db.batch([
743
- {
744
- sql: `UPDATE site_extract_jobs SET billed_mc = -1 WHERE id = ? AND billed_mc IS NULL`,
745
- args: [jobId]
746
- },
747
- {
748
- sql: `INSERT INTO credit_lots (user_id, amount_mc, remaining_mc, source, expires_at)
749
- SELECT ?, ?, ?, ?, datetime(j.created_at, ?)
750
- FROM site_extract_jobs j
751
- WHERE j.id = ?
752
- AND j.billed_mc = -1
753
- AND ? > 0
754
- AND changes() = 1
755
- AND datetime(j.created_at, ?) > datetime('now')`,
756
- args: [
757
- userId,
758
- safeRefundMc,
759
- safeRefundMc,
760
- LedgerOperation.EXTRACT_SITE_REFUND,
761
- CREDIT_LOT_TTL,
762
- jobId,
763
- safeRefundMc,
764
- CREDIT_LOT_TTL
765
- ]
766
- },
767
- {
768
- sql: `UPDATE users SET balance_mc = balance_mc + ?
769
- WHERE id = ? AND ? > 0 AND changes() = 1`,
770
- args: [safeRefundMc, userId, safeRefundMc]
771
- },
772
- {
773
- sql: `INSERT INTO ledger (user_id, amount_mc, operation, description)
774
- SELECT ?, ?, ?, ? WHERE ? > 0 AND changes() = 1`,
775
- args: [userId, safeRefundMc, LedgerOperation.EXTRACT_SITE_REFUND, reference, safeRefundMc]
776
- },
777
- {
778
- sql: `UPDATE site_extract_jobs SET billed_mc = ?, updated_at = datetime('now')
779
- WHERE id = ? AND billed_mc = -1`,
780
- args: [safeNetChargeMc, jobId]
781
- }
782
- ], "write");
783
- }
784
- async function finalizeXRayEntitledExtractJobCharge(jobId) {
785
- const db = getDb();
786
- const result = await db.execute({
787
- sql: `UPDATE site_extract_jobs
788
- SET billed_mc = 0, updated_at = datetime('now')
789
- WHERE id = ?
790
- AND billed_mc IS NULL
791
- AND json_extract(options, '$.billingClass') = ?
792
- AND json_extract(options, '$.heldMc') = 0
793
- AND json_extract(options, '$.debitKey') IS NULL
794
- AND json_extract(options, '$.xraySetupReceipt.billingClass') = ?
795
- AND json_extract(options, '$.xraySetupReceipt.consumesMcpScraperCredits') = 0
796
- AND json_extract(options, '$.xraySetupReceipt.heldMc') = 0
797
- AND json_extract(options, '$.xraySetupReceipt.billedMc') = 0`,
798
- args: [jobId, XRAY_ENTITLED_EXTRACT_BILLING_CLASS, XRAY_ENTITLED_EXTRACT_BILLING_CLASS]
799
- });
800
- return result.rowsAffected === 1;
801
- }
802
-
803
- export {
804
- SITE_EXTRACT_ARTIFACT_PREFIX,
805
- createSiteExtractBundleArtifactStream,
806
- createSiteExtractImageArtifact,
807
- renewSiteExtractArtifactDownload,
808
- readOwnedSiteExtractArtifactBuffer,
809
- readOwnedSiteExtractImageArtifact,
810
- cleanupExpiredSiteExtractArtifacts,
811
- createSiteExtractContentReader,
812
- XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
813
- extractJobLimitInfo,
814
- terminalExtractJobStatus,
815
- createExtractJob,
816
- getExtractJobByIdempotencyKey,
817
- createOrGetExtractJob,
818
- getExtractJob,
819
- listExtractJobs,
820
- listUnsettledExtractJobs,
821
- listFundedPendingExtractJobs,
822
- markExtractJobDispatchAttempt,
823
- recordExtractJobDispatchFailure,
824
- recordExtractSettlementFailure,
825
- listStaleRunningExtractJobs,
826
- failStaleRunningExtractJob,
827
- claimFailedExtractJobForRefinalize,
828
- abandonExtractSettlement,
829
- setExtractJobTotal,
830
- saveExtractPages,
831
- listExtractPages,
832
- getExtractedPages,
833
- getExtractedImageLinks,
834
- countSuccessfulPages,
835
- getExtractedUrls,
836
- finishExtractJob,
837
- completeExtractJob,
838
- failExtractJob,
839
- failUnfundedExtractJob,
840
- setExtractJobPublicError,
841
- settleExtractJob,
842
- finalizeXRayEntitledExtractJobCharge
843
- };