mcp-scraper 0.88.2 → 0.89.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/CHANGELOG.md +41 -2
  2. package/README.md +6 -3
  3. package/dist/analytics-repository-J25XR5E7.js +1 -0
  4. package/dist/bin/api-server.js +2 -41
  5. package/dist/bin/mcp-scraper-cli.js +39 -756
  6. package/dist/bin/mcp-scraper-core.js +1 -60
  7. package/dist/bin/mcp-scraper-install.js +2 -25
  8. package/dist/bin/mcp-stdio-server.js +1 -19
  9. package/dist/bin/paa-harvest.js +1 -41
  10. package/dist/chunk-3GP5CYZX.js +1 -0
  11. package/dist/chunk-4FROKQJN.js +1 -0
  12. package/dist/chunk-57AHYL2O.js +20 -0
  13. package/dist/chunk-5H3MCTTP.js +1 -0
  14. package/dist/chunk-63T4W2LX.js +1271 -0
  15. package/dist/chunk-645LQBE7.js +2595 -0
  16. package/dist/chunk-6QO4F2KC.js +21710 -0
  17. package/dist/chunk-CCYWSJNG.js +16 -0
  18. package/dist/chunk-D6IQJP64.js +84 -0
  19. package/dist/chunk-DGBLGFNJ.js +182 -0
  20. package/dist/chunk-FFPD2DOF.js +59 -0
  21. package/dist/chunk-FXEWHK54.js +1 -0
  22. package/dist/chunk-HE45FFBU.js +1 -0
  23. package/dist/chunk-HUV2WTRW.js +1 -0
  24. package/dist/chunk-IJ2AO7AO.js +102 -0
  25. package/dist/chunk-KJQXUZ4Y.js +4 -0
  26. package/dist/chunk-MEM4D2QN.js +1280 -0
  27. package/dist/chunk-MZN4U5BL.js +1 -0
  28. package/dist/chunk-PRWZNHQB.js +1 -0
  29. package/dist/chunk-QPWPR5XG.js +10 -0
  30. package/dist/chunk-RYNHNEF3.js +73 -0
  31. package/dist/chunk-TMB56NCA.js +1 -0
  32. package/dist/chunk-W2BVJ7S2.js +13 -0
  33. package/dist/chunk-WO3N5FH2.js +5 -0
  34. package/dist/chunk-XLWNEVUZ.js +27 -0
  35. package/dist/chunk-YQZGZBB4.js +1 -0
  36. package/dist/chunk-YUPLOOTB.js +4 -0
  37. package/dist/chunk-ZHORX44A.js +100 -0
  38. package/dist/db-JJWJLAVQ.js +1 -0
  39. package/dist/extract-bundle-TTSEGWYF.js +26 -0
  40. package/dist/gmail-service-34PIJRLK.js +1 -0
  41. package/dist/index.cjs +21838 -6034
  42. package/dist/index.js +18 -315
  43. package/dist/lead-list-enrichment-repository-VJBBMBCG.js +1 -0
  44. package/dist/location-data-repository-W3LPT6G3.js +1 -0
  45. package/dist/server-UEY5YQL6.js +7329 -0
  46. package/dist/site-extract-repository-7W5ZUNPR.js +1 -0
  47. package/dist/worker-ZOTOKG56.js +1 -0
  48. package/package.json +11 -4
  49. package/dist/analytics-repository-GGJJCVVP.js +0 -194
  50. package/dist/chunk-4QMUF6XM.js +0 -1013
  51. package/dist/chunk-6HAV7LCE.js +0 -265
  52. package/dist/chunk-ABF2CGOZ.js +0 -113
  53. package/dist/chunk-C5Z4OFKW.js +0 -404
  54. package/dist/chunk-DNM65UCK.js +0 -299
  55. package/dist/chunk-EQGTEHLZ.js +0 -592
  56. package/dist/chunk-F5GQJWZU.js +0 -732
  57. package/dist/chunk-GGZEC22A.js +0 -215
  58. package/dist/chunk-GXBZXWXB.js +0 -184
  59. package/dist/chunk-IHXAXYIS.js +0 -843
  60. package/dist/chunk-K3Z5AQYE.js +0 -683
  61. package/dist/chunk-K45K75OF.js +0 -6
  62. package/dist/chunk-LFW2FRPJ.js +0 -224
  63. package/dist/chunk-MZDNZQWT.js +0 -2078
  64. package/dist/chunk-NVUKO5NN.js +0 -256
  65. package/dist/chunk-OM7HVEJ3.js +0 -26
  66. package/dist/chunk-OPQIGAFB.js +0 -286
  67. package/dist/chunk-OZJMVCDK.js +0 -16
  68. package/dist/chunk-P7FWOMU7.js +0 -505
  69. package/dist/chunk-PGJQDMC2.js +0 -383
  70. package/dist/chunk-PKZS6SHW.js +0 -33139
  71. package/dist/chunk-RJ7JVYKU.js +0 -68
  72. package/dist/chunk-S24LFPL7.js +0 -5262
  73. package/dist/chunk-T3MZISOF.js +0 -240
  74. package/dist/chunk-UZPTGUDV.js +0 -1915
  75. package/dist/chunk-X623GTBV.js +0 -8290
  76. package/dist/chunk-YXNDOQXN.js +0 -4018
  77. package/dist/db-Z34LPZNR.js +0 -284
  78. package/dist/extract-bundle-565SBZCR.js +0 -1003
  79. package/dist/gmail-service-E6ALS7JG.js +0 -25
  80. package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
  81. package/dist/location-data-repository-WPRG62GE.js +0 -34
  82. package/dist/server-SQZ3A7SY.js +0 -86606
  83. package/dist/site-extract-repository-VYFZASPU.js +0 -69
  84. package/dist/worker-LDCAULWL.js +0 -146
@@ -1,843 +0,0 @@
1
- import {
2
- createPrivateArtifact,
3
- createPrivateArtifactFromStream,
4
- privateArtifactOwnerId,
5
- readPrivateArtifactBuffer,
6
- renewPrivateArtifactDownload
7
- } from "./chunk-DNM65UCK.js";
8
- import {
9
- LedgerOperation
10
- } from "./chunk-4QMUF6XM.js";
11
- import {
12
- CREDIT_LOT_TTL,
13
- getDb,
14
- parsePublicErrorEnvelope,
15
- sanitizeVendorName,
16
- serializePublicErrorEnvelope,
17
- settleDebitMcIdempotent
18
- } from "./chunk-YXNDOQXN.js";
19
-
20
- // src/api/site-extract-content-store.ts
21
- import { createHash } from "crypto";
22
- import { gunzipSync, gzipSync } from "zlib";
23
-
24
- // src/api/site-extract-artifacts.ts
25
- var SITE_EXTRACT_ARTIFACT_PREFIX = "site-extracts/";
26
- var SITE_EXTRACT_ARTIFACT_TTL_MS = 7 * 24 * 60 * 60 * 1e3;
27
- var SITE_EXTRACT_DOWNLOAD_TTL_MS = 15 * 60 * 1e3;
28
- function siteExtractArtifactToken() {
29
- return process.env.SITE_EXTRACT_ARTIFACT_READ_WRITE_TOKEN?.trim() || process.env.PRIVATE_ARTIFACT_READ_WRITE_TOKEN?.trim() || process.env.CONNECTED_DATA_READ_WRITE_TOKEN?.trim() || process.env.CONNECTED_DATA_BLOB_READ_WRITE_TOKEN?.trim() || null;
30
- }
31
- function hostedByEnvironment() {
32
- return process.env.VERCEL === "1" || process.env.NODE_ENV === "production";
33
- }
34
- function policy() {
35
- return {
36
- prefix: SITE_EXTRACT_ARTIFACT_PREFIX,
37
- artifactTtlMs: SITE_EXTRACT_ARTIFACT_TTL_MS,
38
- downloadTtlMs: SITE_EXTRACT_DOWNLOAD_TTL_MS,
39
- token: siteExtractArtifactToken()
40
- };
41
- }
42
- function siteExtractArtifactOwnerId(artifactId) {
43
- return privateArtifactOwnerId(artifactId, SITE_EXTRACT_ARTIFACT_PREFIX);
44
- }
45
- async function createSiteExtractBundleArtifactStream(args) {
46
- const pointer = await createPrivateArtifactFromStream({
47
- policy: policy(),
48
- ownerId: args.ownerId,
49
- artifactKey: `${args.jobId}.zip`,
50
- createdAt: args.createdAt,
51
- filename: `${args.jobId}-site-export.zip`,
52
- contentType: "application/zip",
53
- content: args.content
54
- });
55
- return {
56
- key: pointer.artifactId,
57
- url: pointer.downloadUrl ?? "",
58
- bytes: pointer.bytes,
59
- contentType: pointer.contentType,
60
- filename: pointer.filename,
61
- sha256: pointer.sha256,
62
- expiresAt: pointer.expiresAt,
63
- downloadUrlExpiresAt: pointer.downloadUrlExpiresAt,
64
- kind: "bundle"
65
- };
66
- }
67
- async function createSiteExtractImageArtifact(args) {
68
- const pointer = await createPrivateArtifact({
69
- policy: policy(),
70
- ownerId: args.ownerId,
71
- scopeSegments: [args.jobId, "images"],
72
- artifactKey: `${args.imageId}.bin`,
73
- createdAt: /* @__PURE__ */ new Date(),
74
- filename: args.filename,
75
- contentType: args.contentType,
76
- content: args.content
77
- });
78
- return {
79
- key: pointer.artifactId,
80
- url: pointer.downloadUrl ?? "",
81
- bytes: pointer.bytes,
82
- contentType: pointer.contentType,
83
- filename: pointer.filename,
84
- sha256: pointer.sha256,
85
- expiresAt: pointer.expiresAt,
86
- downloadUrlExpiresAt: pointer.downloadUrlExpiresAt,
87
- kind: "image",
88
- imageId: args.imageId,
89
- sourceUrl: args.sourceUrl,
90
- sourcePage: args.sourcePage
91
- };
92
- }
93
- async function createSiteExtractContentChunkArtifact(args) {
94
- return createPrivateArtifact({
95
- policy: policy(),
96
- ownerId: args.ownerId,
97
- scopeSegments: [args.jobId, "content"],
98
- artifactKey: `${args.chunkKey}.json.gz`,
99
- createdAt: args.createdAt,
100
- filename: `${args.chunkKey}.json.gz`,
101
- contentType: "application/gzip",
102
- content: args.content
103
- });
104
- }
105
- async function renewSiteExtractArtifactDownload(args) {
106
- return renewPrivateArtifactDownload({
107
- policy: policy(),
108
- artifactId: args.artifactId,
109
- ownerId: args.ownerId
110
- });
111
- }
112
- async function readSiteExtractArtifactBuffer(artifactId) {
113
- return readPrivateArtifactBuffer({
114
- policy: policy(),
115
- artifactId,
116
- maxBytes: 50 * 1024 * 1024
117
- });
118
- }
119
- async function readOwnedSiteExtractArtifactBuffer(args) {
120
- if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
121
- return readSiteExtractArtifactBuffer(args.artifactId);
122
- }
123
- async function readOwnedSiteExtractContentChunk(args) {
124
- if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
125
- return readPrivateArtifactBuffer({
126
- policy: policy(),
127
- artifactId: args.artifactId,
128
- maxBytes: 25 * 1024 * 1024
129
- });
130
- }
131
- async function readOwnedSiteExtractImageArtifact(args) {
132
- if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
133
- return readPrivateArtifactBuffer({
134
- policy: policy(),
135
- artifactId: args.artifactId,
136
- maxBytes: 10 * 1024 * 1024
137
- });
138
- }
139
- async function cleanupExpiredSiteExtractArtifacts(args = {}) {
140
- const now = args.now ?? /* @__PURE__ */ new Date();
141
- const token = siteExtractArtifactToken();
142
- if (!token) return { deleted: 0, store: hostedByEnvironment() ? "none" : "local" };
143
- if (!args.force && !(now.getUTCHours() === 3 && now.getUTCMinutes() === 21)) {
144
- return { deleted: 0, store: "private-vercel-blob", skipped: true };
145
- }
146
- const cutoff = now.getTime() - SITE_EXTRACT_ARTIFACT_TTL_MS;
147
- const { list, del } = await import("@vercel/blob");
148
- let cursor;
149
- let deleted = 0;
150
- for (let page = 0; page < 20; page += 1) {
151
- const result = await list({ prefix: SITE_EXTRACT_ARTIFACT_PREFIX, token, limit: 1e3, cursor });
152
- const expired = result.blobs.filter((blob) => new Date(blob.uploadedAt).getTime() <= cutoff);
153
- if (expired.length > 0) {
154
- await del(expired.map((blob) => blob.pathname), { token });
155
- deleted += expired.length;
156
- }
157
- if (!result.hasMore || !result.cursor) break;
158
- cursor = result.cursor;
159
- }
160
- return { deleted, store: "private-vercel-blob" };
161
- }
162
-
163
- // src/api/site-extract-content-store.ts
164
- var MAX_PAGES_PER_CHUNK = 10;
165
- var MAX_UNCOMPRESSED_CHUNK_BYTES = 20 * 1024 * 1024;
166
- function sha256(value) {
167
- return createHash("sha256").update(value).digest("hex");
168
- }
169
- function pageContent(page) {
170
- if (!page.acquiredHtml || page.extractionStatus === "failed") return null;
171
- const pageId = page.pageId ?? sha256(page.archivedUrl ?? page.url);
172
- return {
173
- pageId,
174
- url: page.archivedUrl ?? page.url,
175
- html: page.acquiredHtml,
176
- mainHtml: page.mainHtml ?? "",
177
- markdown: page.bodyMarkdown,
178
- htmlSha256: page.htmlSha256 ?? sha256(page.acquiredHtml),
179
- markdownSha256: page.markdownSha256 ?? sha256(page.bodyMarkdown)
180
- };
181
- }
182
- function serializedChunk(pages) {
183
- return Buffer.from(JSON.stringify({ version: "site-extract-content.v1", pages }));
184
- }
185
- async function persistSiteExtractPageContent(input) {
186
- const contentPages = input.pages.flatMap((page) => {
187
- const content = pageContent(page);
188
- return content ? [content] : [];
189
- });
190
- const refs = /* @__PURE__ */ new Map();
191
- let chunk = [];
192
- let chunkNumber = 0;
193
- const flush = async () => {
194
- if (!chunk.length) return;
195
- const uncompressed = serializedChunk(chunk);
196
- if (uncompressed.length > MAX_UNCOMPRESSED_CHUNK_BYTES) {
197
- throw new Error(`site export content chunk exceeds ${MAX_UNCOMPRESSED_CHUNK_BYTES} bytes`);
198
- }
199
- const identity = sha256(chunk.map((page) => page.pageId).join("\0")).slice(0, 16);
200
- const pointer = await createSiteExtractContentChunkArtifact({
201
- ownerId: input.ownerId,
202
- jobId: input.jobId,
203
- createdAt: input.createdAt,
204
- chunkKey: `content-${String(chunkNumber).padStart(4, "0")}-${identity}`,
205
- content: gzipSync(uncompressed, { level: 6 })
206
- });
207
- for (const page of chunk) {
208
- refs.set(page.url, {
209
- artifactId: pointer.artifactId,
210
- artifactSha256: pointer.sha256,
211
- pageId: page.pageId,
212
- uncompressedBytes: uncompressed.length
213
- });
214
- }
215
- chunkNumber++;
216
- chunk = [];
217
- };
218
- for (const page of contentPages) {
219
- const candidate = [...chunk, page];
220
- if (chunk.length > 0 && (candidate.length > MAX_PAGES_PER_CHUNK || serializedChunk(candidate).length > MAX_UNCOMPRESSED_CHUNK_BYTES)) {
221
- await flush();
222
- }
223
- chunk.push(page);
224
- if (serializedChunk(chunk).length > MAX_UNCOMPRESSED_CHUNK_BYTES) {
225
- throw new Error(`page ${page.pageId} exceeds the durable content chunk limit`);
226
- }
227
- }
228
- await flush();
229
- return refs;
230
- }
231
- function createSiteExtractContentReader(ownerId) {
232
- const cache = /* @__PURE__ */ new Map();
233
- return async (ref) => {
234
- let chunk = cache.get(ref.artifactId);
235
- if (!chunk) {
236
- const compressed = await readOwnedSiteExtractContentChunk({ artifactId: ref.artifactId, ownerId });
237
- if (!compressed) throw new Error("site export content chunk is missing or unauthorized");
238
- if (sha256(compressed) !== ref.artifactSha256) throw new Error("site export content chunk checksum mismatch");
239
- const decoded = gunzipSync(compressed);
240
- if (decoded.length > MAX_UNCOMPRESSED_CHUNK_BYTES) throw new Error("site export content chunk exceeds its read limit");
241
- chunk = JSON.parse(decoded.toString("utf8"));
242
- if (chunk.version !== "site-extract-content.v1" || !Array.isArray(chunk.pages)) {
243
- throw new Error("site export content chunk has an unsupported contract");
244
- }
245
- cache.set(ref.artifactId, chunk);
246
- while (cache.size > 2) cache.delete(cache.keys().next().value);
247
- }
248
- const page = chunk.pages.find((candidate) => candidate.pageId === ref.pageId);
249
- if (!page) throw new Error("site export page is missing from its content chunk");
250
- if (sha256(page.html) !== page.htmlSha256 || sha256(page.markdown) !== page.markdownSha256) {
251
- throw new Error("site export page checksum mismatch");
252
- }
253
- return page;
254
- };
255
- }
256
-
257
- // src/api/site-extract-repository.ts
258
- var XRAY_ENTITLED_EXTRACT_BILLING_CLASS = "xray_entitlement";
259
- function extractJobLimitInfo(job) {
260
- const effectiveMaxPages = Math.max(1, Number(job.options.effectiveMaxPages ?? job.options.maxPages ?? 1));
261
- const requestedMaxPages = Math.max(effectiveMaxPages, Number(job.options.requestedMaxPages ?? effectiveMaxPages));
262
- const creditLimited = requestedMaxPages > effectiveMaxPages;
263
- return {
264
- requestedMaxPages,
265
- effectiveMaxPages,
266
- creditLimited,
267
- // If discovery exhausted below the funded cap, the smaller hold did not
268
- // actually truncate this crawl. Hitting the cap is conservatively partial.
269
- creditTruncated: creditLimited && job.totalUrls >= effectiveMaxPages
270
- };
271
- }
272
- function terminalExtractJobStatus(progress, creditTruncated = false) {
273
- if (progress.successfulUrls === 0) return "failed";
274
- if (progress.failedUrls > 0 || progress.remainingUrls > 0 || creditTruncated) return "partial";
275
- return "complete";
276
- }
277
- function rowToJob(r) {
278
- const totalUrls = Number(r.total_urls ?? 0);
279
- const legacyProgress = r.attempted_urls == null;
280
- const attemptedUrls = Number(legacyProgress ? r.done_urls ?? 0 : r.attempted_urls);
281
- const successfulUrls = Number(legacyProgress ? r.done_urls ?? 0 : r.successful_urls ?? 0);
282
- const failedUrls = Number(r.failed_urls ?? Math.max(0, attemptedUrls - successfulUrls));
283
- return {
284
- id: String(r.id),
285
- userId: r.user_id != null ? Number(r.user_id) : null,
286
- idempotencyKey: r.idempotency_key != null ? String(r.idempotency_key) : null,
287
- requestFingerprint: r.request_fingerprint != null ? String(r.request_fingerprint) : null,
288
- status: r.status != null ? String(r.status) : "pending",
289
- startUrl: String(r.start_url ?? ""),
290
- options: r.options ? JSON.parse(String(r.options)) : {},
291
- totalUrls,
292
- doneUrls: attemptedUrls,
293
- attemptedUrls,
294
- successfulUrls,
295
- failedUrls,
296
- remainingUrls: Math.max(0, totalUrls - attemptedUrls),
297
- artifacts: r.artifacts ? JSON.parse(String(r.artifacts)) : null,
298
- error: r.error != null ? String(r.error) : null,
299
- publicError: parsePublicErrorEnvelope(r.public_error_json),
300
- billedMc: r.billed_mc != null ? Number(r.billed_mc) : null,
301
- createdAt: String(r.created_at ?? ""),
302
- updatedAt: String(r.updated_at ?? "")
303
- };
304
- }
305
- async function createExtractJob(jobId, userId, startUrl, options) {
306
- const db = getDb();
307
- await db.execute({
308
- sql: `INSERT INTO site_extract_jobs (id, user_id, status, start_url, options, created_at, updated_at)
309
- VALUES (?, ?, 'pending', ?, ?, datetime('now'), datetime('now'))`,
310
- args: [jobId, userId, startUrl, JSON.stringify(options)]
311
- });
312
- }
313
- async function getExtractJobByIdempotencyKey(userId, idempotencyKey) {
314
- const db = getDb();
315
- const res = await db.execute({
316
- sql: `SELECT * FROM site_extract_jobs WHERE user_id = ? AND idempotency_key = ? LIMIT 1`,
317
- args: [userId, idempotencyKey]
318
- });
319
- return res.rows[0] ? rowToJob(res.rows[0]) : null;
320
- }
321
- async function createOrGetExtractJob(input) {
322
- const db = getDb();
323
- const inserted = await db.execute({
324
- sql: `INSERT OR IGNORE INTO site_extract_jobs
325
- (id, user_id, status, start_url, options, idempotency_key, request_fingerprint, created_at, updated_at)
326
- VALUES (?, ?, 'pending', ?, ?, ?, ?, datetime('now'), datetime('now'))`,
327
- args: [
328
- input.jobId,
329
- input.userId,
330
- input.startUrl,
331
- JSON.stringify(input.options),
332
- input.idempotencyKey,
333
- input.requestFingerprint
334
- ]
335
- });
336
- const job = await getExtractJobByIdempotencyKey(input.userId, input.idempotencyKey);
337
- if (!job) throw new Error("idempotent extract job was not persisted");
338
- return {
339
- job,
340
- created: inserted.rowsAffected === 1,
341
- conflict: job.requestFingerprint !== input.requestFingerprint
342
- };
343
- }
344
- async function getExtractJob(jobId) {
345
- const db = getDb();
346
- const res = await db.execute({ sql: `SELECT * FROM site_extract_jobs WHERE id = ?`, args: [jobId] });
347
- return res.rows[0] ? rowToJob(res.rows[0]) : null;
348
- }
349
- async function listExtractJobs(userId) {
350
- const db = getDb();
351
- const res = await db.execute({ sql: `SELECT * FROM site_extract_jobs WHERE user_id = ? ORDER BY created_at DESC LIMIT 50`, args: [userId] });
352
- return res.rows.map((r) => rowToJob(r));
353
- }
354
- async function listUnsettledExtractJobs(limit = 25) {
355
- const db = getDb();
356
- const res = await db.execute({
357
- sql: `SELECT * FROM site_extract_jobs
358
- WHERE billed_mc IS NULL AND status IN ('complete', 'partial', 'failed')
359
- AND (next_settlement_at IS NULL OR next_settlement_at <= datetime('now'))
360
- ORDER BY updated_at ASC LIMIT ?`,
361
- args: [Math.max(1, Math.min(100, Math.round(limit)))]
362
- });
363
- return res.rows.map((row) => rowToJob(row));
364
- }
365
- async function listFundedPendingExtractJobs(limit = 25, staleMinutes = 2) {
366
- const db = getDb();
367
- const safeStaleMinutes = Math.max(1, Math.min(60, Math.round(staleMinutes)));
368
- const res = await db.execute({
369
- sql: `SELECT j.* FROM site_extract_jobs j
370
- LEFT JOIN billing_debits d
371
- ON d.idempotency_key = json_extract(j.options, '$.debitKey')
372
- AND d.user_id = j.user_id
373
- AND d.status = 'applied'
374
- WHERE j.status = 'pending'
375
- AND j.billed_mc IS NULL
376
- AND (
377
- d.idempotency_key IS NOT NULL
378
- OR (
379
- json_extract(j.options, '$.billingClass') = ?
380
- AND json_extract(j.options, '$.heldMc') = 0
381
- AND json_extract(j.options, '$.debitKey') IS NULL
382
- AND json_extract(j.options, '$.xraySetupReceipt.billingClass') = ?
383
- AND json_extract(j.options, '$.xraySetupReceipt.consumesMcpScraperCredits') = 0
384
- )
385
- )
386
- AND (j.next_dispatch_at IS NULL OR j.next_dispatch_at <= datetime('now'))
387
- AND j.updated_at <= datetime('now', ?)
388
- ORDER BY j.updated_at ASC LIMIT ?`,
389
- args: [
390
- XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
391
- XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
392
- `-${safeStaleMinutes} minutes`,
393
- Math.max(1, Math.min(100, Math.round(limit)))
394
- ]
395
- });
396
- return res.rows.map((row) => rowToJob(row));
397
- }
398
- async function markExtractJobDispatchAttempt(jobId) {
399
- const db = getDb();
400
- await db.execute({
401
- sql: `UPDATE site_extract_jobs
402
- SET dispatch_attempts = dispatch_attempts + 1,
403
- updated_at = datetime('now'),
404
- next_dispatch_at = datetime('now', '+5 minutes'),
405
- dispatch_error = NULL,
406
- status = CASE
407
- WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours') THEN 'failed'
408
- ELSE status
409
- END,
410
- error = CASE
411
- WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours')
412
- THEN 'Background crawl was acknowledged repeatedly but never started; the hold will be refunded.'
413
- ELSE error
414
- END
415
- WHERE id = ? AND status = 'pending'`,
416
- args: [jobId]
417
- });
418
- return getExtractJob(jobId);
419
- }
420
- async function recordExtractJobDispatchFailure(jobId, error) {
421
- const db = getDb();
422
- const safeError = sanitizeVendorName(error).slice(0, 1e3);
423
- await db.execute({
424
- sql: `UPDATE site_extract_jobs SET
425
- dispatch_attempts = dispatch_attempts + 1,
426
- dispatch_error = ?,
427
- next_dispatch_at = datetime('now', '+5 minutes'),
428
- updated_at = datetime('now'),
429
- status = CASE
430
- WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours') THEN 'failed'
431
- ELSE status
432
- END,
433
- error = CASE
434
- WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours')
435
- THEN 'Background crawl could not be dispatched after repeated attempts; the hold will be refunded.'
436
- ELSE error
437
- END
438
- WHERE id = ? AND status = 'pending'`,
439
- args: [safeError, jobId]
440
- });
441
- return getExtractJob(jobId);
442
- }
443
- async function recordExtractSettlementFailure(jobId, error) {
444
- const db = getDb();
445
- await db.execute({
446
- sql: `UPDATE site_extract_jobs SET
447
- settlement_attempts = settlement_attempts + 1,
448
- settlement_error = ?,
449
- next_settlement_at = datetime('now', '+15 minutes'),
450
- updated_at = datetime('now')
451
- WHERE id = ? AND billed_mc IS NULL`,
452
- args: [sanitizeVendorName(error).slice(0, 1e3), jobId]
453
- });
454
- }
455
- async function listStaleRunningExtractJobs(limit = 25, staleMinutes = 60) {
456
- const db = getDb();
457
- const safeStaleMinutes = Math.max(15, Math.min(24 * 60, Math.round(staleMinutes)));
458
- const res = await db.execute({
459
- sql: `SELECT j.* FROM site_extract_jobs j
460
- LEFT JOIN billing_debits d
461
- ON d.idempotency_key = json_extract(j.options, '$.debitKey')
462
- AND d.user_id = j.user_id
463
- AND d.status = 'applied'
464
- WHERE j.status = 'running'
465
- AND j.billed_mc IS NULL
466
- AND (
467
- json_extract(j.options, '$.debitKey') IS NULL
468
- OR d.idempotency_key IS NOT NULL
469
- )
470
- AND j.updated_at <= datetime('now', ?)
471
- ORDER BY j.updated_at ASC LIMIT ?`,
472
- args: [`-${safeStaleMinutes} minutes`, Math.max(1, Math.min(100, Math.round(limit)))]
473
- });
474
- return res.rows.map((row) => rowToJob(row));
475
- }
476
- async function failStaleRunningExtractJob(jobId, staleMinutes = 60) {
477
- const db = getDb();
478
- const safeStaleMinutes = Math.max(15, Math.min(24 * 60, Math.round(staleMinutes)));
479
- const result = await db.execute({
480
- sql: `UPDATE site_extract_jobs
481
- SET status = 'failed',
482
- error = 'Background crawl stopped without a heartbeat; completed pages were preserved and settlement is pending.',
483
- public_error_json = ?,
484
- updated_at = datetime('now')
485
- WHERE id = ?
486
- AND status = 'running'
487
- AND billed_mc IS NULL
488
- AND updated_at <= datetime('now', ?)`,
489
- args: [
490
- serializePublicErrorEnvelope({
491
- error_code: "extraction_failed",
492
- error_type: "extraction",
493
- message: "The background crawl stopped without a heartbeat. Completed pages were preserved; retry after settlement completes.",
494
- retryable: true,
495
- charge_status: "refund_pending"
496
- }),
497
- jobId,
498
- `-${safeStaleMinutes} minutes`
499
- ]
500
- });
501
- return result.rowsAffected === 1;
502
- }
503
- async function claimFailedExtractJobForRefinalize(jobId) {
504
- const db = getDb();
505
- const claimed = await db.execute({
506
- sql: `UPDATE site_extract_jobs
507
- SET status = 'running', updated_at = datetime('now')
508
- WHERE id = ? AND status = 'failed'`,
509
- args: [jobId]
510
- });
511
- if (claimed.rowsAffected !== 1) return null;
512
- return getExtractJob(jobId);
513
- }
514
- async function abandonExtractSettlement(jobId, error) {
515
- const db = getDb();
516
- await db.execute({
517
- sql: `UPDATE site_extract_jobs SET billed_mc = 0, settlement_error = ?, updated_at = datetime('now')
518
- WHERE id = ? AND billed_mc IS NULL`,
519
- args: [sanitizeVendorName(error).slice(0, 1e3), jobId]
520
- });
521
- }
522
- async function setExtractJobTotal(jobId, totalUrls) {
523
- const db = getDb();
524
- await db.execute({
525
- sql: `UPDATE site_extract_jobs
526
- SET status = 'running', total_urls = MAX(total_urls, ?), updated_at = datetime('now')
527
- WHERE id = ? AND status IN ('pending', 'running')`,
528
- args: [totalUrls, jobId]
529
- });
530
- }
531
- async function saveExtractPages(jobId, pages) {
532
- if (pages.length === 0) return;
533
- const db = getDb();
534
- const job = await getExtractJob(jobId);
535
- const contentRefs = job?.userId != null ? await persistSiteExtractPageContent({
536
- ownerId: String(job.userId),
537
- jobId,
538
- createdAt: job.createdAt,
539
- pages
540
- }) : /* @__PURE__ */ new Map();
541
- await db.batch([
542
- ...pages.map((p) => {
543
- const {
544
- discoveryLinks: _ephemeralDiscoveryLinks,
545
- acquiredHtml: _durableHtml,
546
- mainHtml: _durableMainHtml,
547
- ...storedPage
548
- } = p;
549
- storedPage.contentRef = contentRefs.get(p.archivedUrl ?? p.url) ?? p.contentRef;
550
- return {
551
- sql: `INSERT INTO site_extract_pages (job_id, url, page)
552
- SELECT ?, ?, ?
553
- WHERE EXISTS (
554
- SELECT 1 FROM site_extract_jobs
555
- WHERE id = ? AND status IN ('pending', 'running')
556
- )
557
- ON CONFLICT(job_id, url) DO UPDATE SET page = excluded.page
558
- WHERE NOT CASE
559
- WHEN json_valid(site_extract_pages.page) = 0 THEN 0
560
- WHEN json_extract(site_extract_pages.page, '$.extractionStatus') IS NOT NULL
561
- THEN json_extract(site_extract_pages.page, '$.extractionStatus') = 'successful'
562
- ELSE COALESCE(
563
- json_extract(site_extract_pages.page, '$.status') >= 200
564
- AND json_extract(site_extract_pages.page, '$.status') < 300
565
- AND json_extract(site_extract_pages.page, '$.wordCount') > 0,
566
- 0
567
- )
568
- END
569
- OR CASE
570
- WHEN json_valid(excluded.page) = 0 THEN 0
571
- WHEN json_extract(excluded.page, '$.extractionStatus') IS NOT NULL
572
- THEN json_extract(excluded.page, '$.extractionStatus') = 'successful'
573
- ELSE COALESCE(
574
- json_extract(excluded.page, '$.status') >= 200
575
- AND json_extract(excluded.page, '$.status') < 300
576
- AND json_extract(excluded.page, '$.wordCount') > 0,
577
- 0
578
- )
579
- END`,
580
- args: [jobId, p.archivedUrl ?? p.url, JSON.stringify(storedPage), jobId]
581
- };
582
- }),
583
- {
584
- sql: `UPDATE site_extract_jobs SET
585
- done_urls = (SELECT COUNT(*) FROM site_extract_pages WHERE job_id = ?),
586
- attempted_urls = (SELECT COUNT(*) FROM site_extract_pages WHERE job_id = ?),
587
- successful_urls = (
588
- SELECT COUNT(*) FROM site_extract_pages
589
- WHERE job_id = ? AND CASE
590
- WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
591
- THEN json_extract(page, '$.extractionStatus') = 'successful'
592
- ELSE json_extract(page, '$.status') >= 200
593
- AND json_extract(page, '$.status') < 300
594
- AND json_extract(page, '$.wordCount') > 0
595
- END
596
- ),
597
- failed_urls = (
598
- SELECT COUNT(*) FROM site_extract_pages
599
- WHERE job_id = ? AND NOT CASE
600
- WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
601
- THEN json_extract(page, '$.extractionStatus') = 'successful'
602
- ELSE COALESCE(
603
- json_extract(page, '$.status') >= 200
604
- AND json_extract(page, '$.status') < 300
605
- AND json_extract(page, '$.wordCount') > 0,
606
- 0
607
- )
608
- END
609
- ),
610
- updated_at = datetime('now')
611
- WHERE id = ? AND status IN ('pending', 'running')`,
612
- args: [jobId, jobId, jobId, jobId, jobId]
613
- }
614
- ], "write");
615
- }
616
- async function listExtractPages(jobId) {
617
- const db = getDb();
618
- const result = await db.execute({
619
- sql: `SELECT page FROM site_extract_pages WHERE job_id = ? ORDER BY rowid`,
620
- args: [jobId]
621
- });
622
- return result.rows.map((row) => JSON.parse(String(row.page)));
623
- }
624
- async function getExtractedPages(jobId) {
625
- const db = getDb();
626
- const res = await db.execute({ sql: `SELECT page FROM site_extract_pages WHERE job_id = ?`, args: [jobId] });
627
- return res.rows.map((r) => JSON.parse(String(r.page)));
628
- }
629
- async function getExtractedImageLinks(jobId, limit = 181) {
630
- const db = getDb();
631
- const safeLimit = Math.max(1, Math.min(5001, Math.round(limit)));
632
- const res = await db.execute({
633
- sql: `SELECT DISTINCT CAST(images.value AS TEXT) AS url
634
- FROM site_extract_pages p,
635
- json_each(CASE WHEN json_valid(p.page) THEN p.page ELSE '{}' END, '$.imageLinks') images
636
- WHERE p.job_id = ?
637
- AND images.type = 'text'
638
- AND length(CAST(images.value AS TEXT)) BETWEEN 1 AND 4096
639
- LIMIT ?`,
640
- args: [jobId, safeLimit]
641
- });
642
- return res.rows.map((row) => String(row.url));
643
- }
644
- async function countSuccessfulPages(jobId) {
645
- const db = getDb();
646
- const res = await db.execute({
647
- sql: `SELECT COUNT(*) AS n FROM site_extract_pages
648
- WHERE job_id = ? AND CASE
649
- WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
650
- THEN json_extract(page, '$.extractionStatus') = 'successful'
651
- ELSE COALESCE(
652
- json_extract(page, '$.status') >= 200
653
- AND json_extract(page, '$.status') < 300
654
- AND json_extract(page, '$.wordCount') > 0,
655
- 0
656
- )
657
- END`,
658
- args: [jobId]
659
- });
660
- return Number(res.rows[0]?.n ?? 0);
661
- }
662
- async function getExtractedUrls(jobId) {
663
- const db = getDb();
664
- const res = await db.execute({ sql: `SELECT url FROM site_extract_pages WHERE job_id = ?`, args: [jobId] });
665
- return new Set(res.rows.map((r) => String(r.url)));
666
- }
667
- async function finishExtractJob(jobId, artifacts, status, error = null, allowFailedRecovery = false, publicError = null) {
668
- const db = getDb();
669
- const result = await db.execute({
670
- sql: `UPDATE site_extract_jobs
671
- SET status = ?, artifacts = ?, error = ?, public_error_json = ?, updated_at = datetime('now')
672
- WHERE id = ? AND (status IN ('pending', 'running') OR (? = 1 AND status = 'failed'))`,
673
- args: [
674
- status,
675
- JSON.stringify(artifacts ?? []),
676
- error ? sanitizeVendorName(error).slice(0, 2e3) : null,
677
- serializePublicErrorEnvelope(publicError),
678
- jobId,
679
- allowFailedRecovery ? 1 : 0
680
- ]
681
- });
682
- return result.rowsAffected === 1;
683
- }
684
- async function completeExtractJob(jobId, artifacts) {
685
- await finishExtractJob(jobId, artifacts, "complete");
686
- }
687
- async function failExtractJob(jobId, error, publicError) {
688
- const db = getDb();
689
- await db.execute({
690
- sql: `UPDATE site_extract_jobs
691
- SET status = 'failed', error = ?, public_error_json = ?, updated_at = datetime('now')
692
- WHERE id = ? AND status IN ('pending', 'running')`,
693
- args: [sanitizeVendorName(error).slice(0, 2e3), serializePublicErrorEnvelope(publicError), jobId]
694
- });
695
- }
696
- async function failUnfundedExtractJob(jobId, error, publicError) {
697
- const db = getDb();
698
- await db.execute({
699
- sql: `UPDATE site_extract_jobs
700
- SET status = 'failed', billed_mc = 0, error = ?, public_error_json = ?, updated_at = datetime('now')
701
- WHERE id = ? AND status = 'pending' AND billed_mc IS NULL`,
702
- args: [sanitizeVendorName(error).slice(0, 2e3), serializePublicErrorEnvelope(publicError), jobId]
703
- });
704
- }
705
- async function setExtractJobPublicError(jobId, publicError) {
706
- const job = await getExtractJob(jobId);
707
- const normalized = job?.options.billingClass === XRAY_ENTITLED_EXTRACT_BILLING_CLASS && publicError ? { ...publicError, charge_status: "not_charged" } : publicError;
708
- await getDb().execute({
709
- sql: `UPDATE site_extract_jobs SET public_error_json = ?, updated_at = datetime('now')
710
- WHERE id = ? AND status IN ('complete', 'partial', 'failed')`,
711
- args: [serializePublicErrorEnvelope(normalized), jobId]
712
- });
713
- }
714
- async function settleExtractJob(jobId, userId, refundMc, netChargeMc, reference) {
715
- const db = getDb();
716
- const job = await getExtractJob(jobId);
717
- if (!job || job.billedMc != null) return;
718
- if (job.options.billingClass === XRAY_ENTITLED_EXTRACT_BILLING_CLASS) {
719
- if (!await finalizeXRayEntitledExtractJobCharge(jobId)) {
720
- throw new Error("xray_extract_zero_charge_finalization_rejected");
721
- }
722
- return;
723
- }
724
- const debitKey = typeof job.options.debitKey === "string" ? job.options.debitKey : null;
725
- if (debitKey) {
726
- const settlement = await settleDebitMcIdempotent(
727
- userId,
728
- debitKey,
729
- Math.max(0, netChargeMc),
730
- LedgerOperation.EXTRACT_SITE_REFUND,
731
- reference,
732
- "extract_site"
733
- );
734
- await db.execute({
735
- sql: `UPDATE site_extract_jobs SET billed_mc = ?, updated_at = datetime('now') WHERE id = ? AND billed_mc IS NULL`,
736
- args: [settlement.final_amount_mc, jobId]
737
- });
738
- return;
739
- }
740
- const safeRefundMc = Math.max(0, refundMc);
741
- const safeNetChargeMc = Math.max(0, netChargeMc);
742
- await db.batch([
743
- {
744
- sql: `UPDATE site_extract_jobs SET billed_mc = -1 WHERE id = ? AND billed_mc IS NULL`,
745
- args: [jobId]
746
- },
747
- {
748
- sql: `INSERT INTO credit_lots (user_id, amount_mc, remaining_mc, source, expires_at)
749
- SELECT ?, ?, ?, ?, datetime(j.created_at, ?)
750
- FROM site_extract_jobs j
751
- WHERE j.id = ?
752
- AND j.billed_mc = -1
753
- AND ? > 0
754
- AND changes() = 1
755
- AND datetime(j.created_at, ?) > datetime('now')`,
756
- args: [
757
- userId,
758
- safeRefundMc,
759
- safeRefundMc,
760
- LedgerOperation.EXTRACT_SITE_REFUND,
761
- CREDIT_LOT_TTL,
762
- jobId,
763
- safeRefundMc,
764
- CREDIT_LOT_TTL
765
- ]
766
- },
767
- {
768
- sql: `UPDATE users SET balance_mc = balance_mc + ?
769
- WHERE id = ? AND ? > 0 AND changes() = 1`,
770
- args: [safeRefundMc, userId, safeRefundMc]
771
- },
772
- {
773
- sql: `INSERT INTO ledger (user_id, amount_mc, operation, description)
774
- SELECT ?, ?, ?, ? WHERE ? > 0 AND changes() = 1`,
775
- args: [userId, safeRefundMc, LedgerOperation.EXTRACT_SITE_REFUND, reference, safeRefundMc]
776
- },
777
- {
778
- sql: `UPDATE site_extract_jobs SET billed_mc = ?, updated_at = datetime('now')
779
- WHERE id = ? AND billed_mc = -1`,
780
- args: [safeNetChargeMc, jobId]
781
- }
782
- ], "write");
783
- }
784
- async function finalizeXRayEntitledExtractJobCharge(jobId) {
785
- const db = getDb();
786
- const result = await db.execute({
787
- sql: `UPDATE site_extract_jobs
788
- SET billed_mc = 0, updated_at = datetime('now')
789
- WHERE id = ?
790
- AND billed_mc IS NULL
791
- AND json_extract(options, '$.billingClass') = ?
792
- AND json_extract(options, '$.heldMc') = 0
793
- AND json_extract(options, '$.debitKey') IS NULL
794
- AND json_extract(options, '$.xraySetupReceipt.billingClass') = ?
795
- AND json_extract(options, '$.xraySetupReceipt.consumesMcpScraperCredits') = 0
796
- AND json_extract(options, '$.xraySetupReceipt.heldMc') = 0
797
- AND json_extract(options, '$.xraySetupReceipt.billedMc') = 0`,
798
- args: [jobId, XRAY_ENTITLED_EXTRACT_BILLING_CLASS, XRAY_ENTITLED_EXTRACT_BILLING_CLASS]
799
- });
800
- return result.rowsAffected === 1;
801
- }
802
-
803
- export {
804
- SITE_EXTRACT_ARTIFACT_PREFIX,
805
- createSiteExtractBundleArtifactStream,
806
- createSiteExtractImageArtifact,
807
- renewSiteExtractArtifactDownload,
808
- readOwnedSiteExtractArtifactBuffer,
809
- readOwnedSiteExtractImageArtifact,
810
- cleanupExpiredSiteExtractArtifacts,
811
- createSiteExtractContentReader,
812
- XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
813
- extractJobLimitInfo,
814
- terminalExtractJobStatus,
815
- createExtractJob,
816
- getExtractJobByIdempotencyKey,
817
- createOrGetExtractJob,
818
- getExtractJob,
819
- listExtractJobs,
820
- listUnsettledExtractJobs,
821
- listFundedPendingExtractJobs,
822
- markExtractJobDispatchAttempt,
823
- recordExtractJobDispatchFailure,
824
- recordExtractSettlementFailure,
825
- listStaleRunningExtractJobs,
826
- failStaleRunningExtractJob,
827
- claimFailedExtractJobForRefinalize,
828
- abandonExtractSettlement,
829
- setExtractJobTotal,
830
- saveExtractPages,
831
- listExtractPages,
832
- getExtractedPages,
833
- getExtractedImageLinks,
834
- countSuccessfulPages,
835
- getExtractedUrls,
836
- finishExtractJob,
837
- completeExtractJob,
838
- failExtractJob,
839
- failUnfundedExtractJob,
840
- setExtractJobPublicError,
841
- settleExtractJob,
842
- finalizeXRayEntitledExtractJobCharge
843
- };