mcp-scraper 0.88.2 → 0.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -15
- package/README.md +6 -5
- package/THIRD_PARTY_NOTICES.html +203 -0
- package/dist/analytics-repository-2BMT5JNE.js +1 -0
- package/dist/bin/api-server.js +2 -41
- package/dist/bin/mcp-scraper-cli.js +39 -756
- package/dist/bin/mcp-scraper-core.js +1 -60
- package/dist/bin/mcp-scraper-install.js +2 -25
- package/dist/bin/mcp-stdio-server.js +1 -19
- package/dist/bin/paa-harvest.js +1 -41
- package/dist/chunk-3GP5CYZX.js +1 -0
- package/dist/chunk-4AI7DOS7.js +59 -0
- package/dist/chunk-4FROKQJN.js +1 -0
- package/dist/chunk-7YGVI5J4.js +21710 -0
- package/dist/chunk-CCYWSJNG.js +16 -0
- package/dist/chunk-CFI6CXIV.js +182 -0
- package/dist/chunk-E2WRWV3A.js +1 -0
- package/dist/chunk-GMWKPIYX.js +1172 -0
- package/dist/chunk-HDPYG3XV.js +102 -0
- package/dist/chunk-HE45FFBU.js +1 -0
- package/dist/chunk-HUV2WTRW.js +1 -0
- package/dist/chunk-KJQXUZ4Y.js +4 -0
- package/dist/chunk-L4CGLFPU.js +4 -0
- package/dist/chunk-M22MM4N4.js +84 -0
- package/dist/chunk-MASR22K4.js +73 -0
- package/dist/chunk-MZN4U5BL.js +1 -0
- package/dist/chunk-PUHFVA7P.js +1280 -0
- package/dist/chunk-QPWPR5XG.js +10 -0
- package/dist/chunk-TMB56NCA.js +1 -0
- package/dist/chunk-TXENITMS.js +20 -0
- package/dist/chunk-W2BVJ7S2.js +13 -0
- package/dist/chunk-WO3N5FH2.js +5 -0
- package/dist/chunk-WSCGYRWA.js +2595 -0
- package/dist/chunk-X54CQLK2.js +1 -0
- package/dist/chunk-XLWNEVUZ.js +27 -0
- package/dist/chunk-XPZVJIZ2.js +100 -0
- package/dist/chunk-YQZGZBB4.js +1 -0
- package/dist/chunk-Z2QGQJS2.js +1 -0
- package/dist/db-F2MX63GI.js +1 -0
- package/dist/extract-bundle-SNUIHM3J.js +26 -0
- package/dist/gmail-service-BZ3H75XC.js +1 -0
- package/dist/index.cjs +21750 -6045
- package/dist/index.d.cts +14 -14
- package/dist/index.d.ts +14 -14
- package/dist/index.js +18 -315
- package/dist/lead-list-enrichment-repository-S2H3U7T7.js +1 -0
- package/dist/location-data-repository-OTWHWMV6.js +1 -0
- package/dist/server-RFR2A5UJ.js +7303 -0
- package/dist/site-extract-repository-SE776XDC.js +1 -0
- package/dist/worker-XUDSM3AL.js +1 -0
- package/package.json +17 -124
- package/dist/analytics-repository-GGJJCVVP.js +0 -194
- package/dist/chunk-4QMUF6XM.js +0 -1013
- package/dist/chunk-6HAV7LCE.js +0 -265
- package/dist/chunk-ABF2CGOZ.js +0 -113
- package/dist/chunk-C5Z4OFKW.js +0 -404
- package/dist/chunk-DNM65UCK.js +0 -299
- package/dist/chunk-EQGTEHLZ.js +0 -592
- package/dist/chunk-F5GQJWZU.js +0 -732
- package/dist/chunk-GGZEC22A.js +0 -215
- package/dist/chunk-GXBZXWXB.js +0 -184
- package/dist/chunk-IHXAXYIS.js +0 -843
- package/dist/chunk-K3Z5AQYE.js +0 -683
- package/dist/chunk-K45K75OF.js +0 -6
- package/dist/chunk-LFW2FRPJ.js +0 -224
- package/dist/chunk-MZDNZQWT.js +0 -2078
- package/dist/chunk-NVUKO5NN.js +0 -256
- package/dist/chunk-OM7HVEJ3.js +0 -26
- package/dist/chunk-OPQIGAFB.js +0 -286
- package/dist/chunk-OZJMVCDK.js +0 -16
- package/dist/chunk-P7FWOMU7.js +0 -505
- package/dist/chunk-PGJQDMC2.js +0 -383
- package/dist/chunk-PKZS6SHW.js +0 -33139
- package/dist/chunk-RJ7JVYKU.js +0 -68
- package/dist/chunk-S24LFPL7.js +0 -5262
- package/dist/chunk-T3MZISOF.js +0 -240
- package/dist/chunk-UZPTGUDV.js +0 -1915
- package/dist/chunk-X623GTBV.js +0 -8290
- package/dist/chunk-YXNDOQXN.js +0 -4018
- package/dist/db-Z34LPZNR.js +0 -284
- package/dist/extract-bundle-565SBZCR.js +0 -1003
- package/dist/gmail-service-E6ALS7JG.js +0 -25
- package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
- package/dist/location-data-repository-WPRG62GE.js +0 -34
- package/dist/server-SQZ3A7SY.js +0 -86606
- package/dist/site-extract-repository-VYFZASPU.js +0 -69
- package/dist/worker-LDCAULWL.js +0 -146
package/dist/chunk-IHXAXYIS.js
DELETED
|
@@ -1,843 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
createPrivateArtifact,
|
|
3
|
-
createPrivateArtifactFromStream,
|
|
4
|
-
privateArtifactOwnerId,
|
|
5
|
-
readPrivateArtifactBuffer,
|
|
6
|
-
renewPrivateArtifactDownload
|
|
7
|
-
} from "./chunk-DNM65UCK.js";
|
|
8
|
-
import {
|
|
9
|
-
LedgerOperation
|
|
10
|
-
} from "./chunk-4QMUF6XM.js";
|
|
11
|
-
import {
|
|
12
|
-
CREDIT_LOT_TTL,
|
|
13
|
-
getDb,
|
|
14
|
-
parsePublicErrorEnvelope,
|
|
15
|
-
sanitizeVendorName,
|
|
16
|
-
serializePublicErrorEnvelope,
|
|
17
|
-
settleDebitMcIdempotent
|
|
18
|
-
} from "./chunk-YXNDOQXN.js";
|
|
19
|
-
|
|
20
|
-
// src/api/site-extract-content-store.ts
|
|
21
|
-
import { createHash } from "crypto";
|
|
22
|
-
import { gunzipSync, gzipSync } from "zlib";
|
|
23
|
-
|
|
24
|
-
// src/api/site-extract-artifacts.ts
|
|
25
|
-
var SITE_EXTRACT_ARTIFACT_PREFIX = "site-extracts/";
|
|
26
|
-
var SITE_EXTRACT_ARTIFACT_TTL_MS = 7 * 24 * 60 * 60 * 1e3;
|
|
27
|
-
var SITE_EXTRACT_DOWNLOAD_TTL_MS = 15 * 60 * 1e3;
|
|
28
|
-
function siteExtractArtifactToken() {
|
|
29
|
-
return process.env.SITE_EXTRACT_ARTIFACT_READ_WRITE_TOKEN?.trim() || process.env.PRIVATE_ARTIFACT_READ_WRITE_TOKEN?.trim() || process.env.CONNECTED_DATA_READ_WRITE_TOKEN?.trim() || process.env.CONNECTED_DATA_BLOB_READ_WRITE_TOKEN?.trim() || null;
|
|
30
|
-
}
|
|
31
|
-
function hostedByEnvironment() {
|
|
32
|
-
return process.env.VERCEL === "1" || process.env.NODE_ENV === "production";
|
|
33
|
-
}
|
|
34
|
-
function policy() {
|
|
35
|
-
return {
|
|
36
|
-
prefix: SITE_EXTRACT_ARTIFACT_PREFIX,
|
|
37
|
-
artifactTtlMs: SITE_EXTRACT_ARTIFACT_TTL_MS,
|
|
38
|
-
downloadTtlMs: SITE_EXTRACT_DOWNLOAD_TTL_MS,
|
|
39
|
-
token: siteExtractArtifactToken()
|
|
40
|
-
};
|
|
41
|
-
}
|
|
42
|
-
function siteExtractArtifactOwnerId(artifactId) {
|
|
43
|
-
return privateArtifactOwnerId(artifactId, SITE_EXTRACT_ARTIFACT_PREFIX);
|
|
44
|
-
}
|
|
45
|
-
async function createSiteExtractBundleArtifactStream(args) {
|
|
46
|
-
const pointer = await createPrivateArtifactFromStream({
|
|
47
|
-
policy: policy(),
|
|
48
|
-
ownerId: args.ownerId,
|
|
49
|
-
artifactKey: `${args.jobId}.zip`,
|
|
50
|
-
createdAt: args.createdAt,
|
|
51
|
-
filename: `${args.jobId}-site-export.zip`,
|
|
52
|
-
contentType: "application/zip",
|
|
53
|
-
content: args.content
|
|
54
|
-
});
|
|
55
|
-
return {
|
|
56
|
-
key: pointer.artifactId,
|
|
57
|
-
url: pointer.downloadUrl ?? "",
|
|
58
|
-
bytes: pointer.bytes,
|
|
59
|
-
contentType: pointer.contentType,
|
|
60
|
-
filename: pointer.filename,
|
|
61
|
-
sha256: pointer.sha256,
|
|
62
|
-
expiresAt: pointer.expiresAt,
|
|
63
|
-
downloadUrlExpiresAt: pointer.downloadUrlExpiresAt,
|
|
64
|
-
kind: "bundle"
|
|
65
|
-
};
|
|
66
|
-
}
|
|
67
|
-
async function createSiteExtractImageArtifact(args) {
|
|
68
|
-
const pointer = await createPrivateArtifact({
|
|
69
|
-
policy: policy(),
|
|
70
|
-
ownerId: args.ownerId,
|
|
71
|
-
scopeSegments: [args.jobId, "images"],
|
|
72
|
-
artifactKey: `${args.imageId}.bin`,
|
|
73
|
-
createdAt: /* @__PURE__ */ new Date(),
|
|
74
|
-
filename: args.filename,
|
|
75
|
-
contentType: args.contentType,
|
|
76
|
-
content: args.content
|
|
77
|
-
});
|
|
78
|
-
return {
|
|
79
|
-
key: pointer.artifactId,
|
|
80
|
-
url: pointer.downloadUrl ?? "",
|
|
81
|
-
bytes: pointer.bytes,
|
|
82
|
-
contentType: pointer.contentType,
|
|
83
|
-
filename: pointer.filename,
|
|
84
|
-
sha256: pointer.sha256,
|
|
85
|
-
expiresAt: pointer.expiresAt,
|
|
86
|
-
downloadUrlExpiresAt: pointer.downloadUrlExpiresAt,
|
|
87
|
-
kind: "image",
|
|
88
|
-
imageId: args.imageId,
|
|
89
|
-
sourceUrl: args.sourceUrl,
|
|
90
|
-
sourcePage: args.sourcePage
|
|
91
|
-
};
|
|
92
|
-
}
|
|
93
|
-
async function createSiteExtractContentChunkArtifact(args) {
|
|
94
|
-
return createPrivateArtifact({
|
|
95
|
-
policy: policy(),
|
|
96
|
-
ownerId: args.ownerId,
|
|
97
|
-
scopeSegments: [args.jobId, "content"],
|
|
98
|
-
artifactKey: `${args.chunkKey}.json.gz`,
|
|
99
|
-
createdAt: args.createdAt,
|
|
100
|
-
filename: `${args.chunkKey}.json.gz`,
|
|
101
|
-
contentType: "application/gzip",
|
|
102
|
-
content: args.content
|
|
103
|
-
});
|
|
104
|
-
}
|
|
105
|
-
async function renewSiteExtractArtifactDownload(args) {
|
|
106
|
-
return renewPrivateArtifactDownload({
|
|
107
|
-
policy: policy(),
|
|
108
|
-
artifactId: args.artifactId,
|
|
109
|
-
ownerId: args.ownerId
|
|
110
|
-
});
|
|
111
|
-
}
|
|
112
|
-
async function readSiteExtractArtifactBuffer(artifactId) {
|
|
113
|
-
return readPrivateArtifactBuffer({
|
|
114
|
-
policy: policy(),
|
|
115
|
-
artifactId,
|
|
116
|
-
maxBytes: 50 * 1024 * 1024
|
|
117
|
-
});
|
|
118
|
-
}
|
|
119
|
-
async function readOwnedSiteExtractArtifactBuffer(args) {
|
|
120
|
-
if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
|
|
121
|
-
return readSiteExtractArtifactBuffer(args.artifactId);
|
|
122
|
-
}
|
|
123
|
-
async function readOwnedSiteExtractContentChunk(args) {
|
|
124
|
-
if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
|
|
125
|
-
return readPrivateArtifactBuffer({
|
|
126
|
-
policy: policy(),
|
|
127
|
-
artifactId: args.artifactId,
|
|
128
|
-
maxBytes: 25 * 1024 * 1024
|
|
129
|
-
});
|
|
130
|
-
}
|
|
131
|
-
async function readOwnedSiteExtractImageArtifact(args) {
|
|
132
|
-
if (siteExtractArtifactOwnerId(args.artifactId) !== args.ownerId) return null;
|
|
133
|
-
return readPrivateArtifactBuffer({
|
|
134
|
-
policy: policy(),
|
|
135
|
-
artifactId: args.artifactId,
|
|
136
|
-
maxBytes: 10 * 1024 * 1024
|
|
137
|
-
});
|
|
138
|
-
}
|
|
139
|
-
async function cleanupExpiredSiteExtractArtifacts(args = {}) {
|
|
140
|
-
const now = args.now ?? /* @__PURE__ */ new Date();
|
|
141
|
-
const token = siteExtractArtifactToken();
|
|
142
|
-
if (!token) return { deleted: 0, store: hostedByEnvironment() ? "none" : "local" };
|
|
143
|
-
if (!args.force && !(now.getUTCHours() === 3 && now.getUTCMinutes() === 21)) {
|
|
144
|
-
return { deleted: 0, store: "private-vercel-blob", skipped: true };
|
|
145
|
-
}
|
|
146
|
-
const cutoff = now.getTime() - SITE_EXTRACT_ARTIFACT_TTL_MS;
|
|
147
|
-
const { list, del } = await import("@vercel/blob");
|
|
148
|
-
let cursor;
|
|
149
|
-
let deleted = 0;
|
|
150
|
-
for (let page = 0; page < 20; page += 1) {
|
|
151
|
-
const result = await list({ prefix: SITE_EXTRACT_ARTIFACT_PREFIX, token, limit: 1e3, cursor });
|
|
152
|
-
const expired = result.blobs.filter((blob) => new Date(blob.uploadedAt).getTime() <= cutoff);
|
|
153
|
-
if (expired.length > 0) {
|
|
154
|
-
await del(expired.map((blob) => blob.pathname), { token });
|
|
155
|
-
deleted += expired.length;
|
|
156
|
-
}
|
|
157
|
-
if (!result.hasMore || !result.cursor) break;
|
|
158
|
-
cursor = result.cursor;
|
|
159
|
-
}
|
|
160
|
-
return { deleted, store: "private-vercel-blob" };
|
|
161
|
-
}
|
|
162
|
-
|
|
163
|
-
// src/api/site-extract-content-store.ts
|
|
164
|
-
var MAX_PAGES_PER_CHUNK = 10;
|
|
165
|
-
var MAX_UNCOMPRESSED_CHUNK_BYTES = 20 * 1024 * 1024;
|
|
166
|
-
function sha256(value) {
|
|
167
|
-
return createHash("sha256").update(value).digest("hex");
|
|
168
|
-
}
|
|
169
|
-
function pageContent(page) {
|
|
170
|
-
if (!page.acquiredHtml || page.extractionStatus === "failed") return null;
|
|
171
|
-
const pageId = page.pageId ?? sha256(page.archivedUrl ?? page.url);
|
|
172
|
-
return {
|
|
173
|
-
pageId,
|
|
174
|
-
url: page.archivedUrl ?? page.url,
|
|
175
|
-
html: page.acquiredHtml,
|
|
176
|
-
mainHtml: page.mainHtml ?? "",
|
|
177
|
-
markdown: page.bodyMarkdown,
|
|
178
|
-
htmlSha256: page.htmlSha256 ?? sha256(page.acquiredHtml),
|
|
179
|
-
markdownSha256: page.markdownSha256 ?? sha256(page.bodyMarkdown)
|
|
180
|
-
};
|
|
181
|
-
}
|
|
182
|
-
function serializedChunk(pages) {
|
|
183
|
-
return Buffer.from(JSON.stringify({ version: "site-extract-content.v1", pages }));
|
|
184
|
-
}
|
|
185
|
-
async function persistSiteExtractPageContent(input) {
|
|
186
|
-
const contentPages = input.pages.flatMap((page) => {
|
|
187
|
-
const content = pageContent(page);
|
|
188
|
-
return content ? [content] : [];
|
|
189
|
-
});
|
|
190
|
-
const refs = /* @__PURE__ */ new Map();
|
|
191
|
-
let chunk = [];
|
|
192
|
-
let chunkNumber = 0;
|
|
193
|
-
const flush = async () => {
|
|
194
|
-
if (!chunk.length) return;
|
|
195
|
-
const uncompressed = serializedChunk(chunk);
|
|
196
|
-
if (uncompressed.length > MAX_UNCOMPRESSED_CHUNK_BYTES) {
|
|
197
|
-
throw new Error(`site export content chunk exceeds ${MAX_UNCOMPRESSED_CHUNK_BYTES} bytes`);
|
|
198
|
-
}
|
|
199
|
-
const identity = sha256(chunk.map((page) => page.pageId).join("\0")).slice(0, 16);
|
|
200
|
-
const pointer = await createSiteExtractContentChunkArtifact({
|
|
201
|
-
ownerId: input.ownerId,
|
|
202
|
-
jobId: input.jobId,
|
|
203
|
-
createdAt: input.createdAt,
|
|
204
|
-
chunkKey: `content-${String(chunkNumber).padStart(4, "0")}-${identity}`,
|
|
205
|
-
content: gzipSync(uncompressed, { level: 6 })
|
|
206
|
-
});
|
|
207
|
-
for (const page of chunk) {
|
|
208
|
-
refs.set(page.url, {
|
|
209
|
-
artifactId: pointer.artifactId,
|
|
210
|
-
artifactSha256: pointer.sha256,
|
|
211
|
-
pageId: page.pageId,
|
|
212
|
-
uncompressedBytes: uncompressed.length
|
|
213
|
-
});
|
|
214
|
-
}
|
|
215
|
-
chunkNumber++;
|
|
216
|
-
chunk = [];
|
|
217
|
-
};
|
|
218
|
-
for (const page of contentPages) {
|
|
219
|
-
const candidate = [...chunk, page];
|
|
220
|
-
if (chunk.length > 0 && (candidate.length > MAX_PAGES_PER_CHUNK || serializedChunk(candidate).length > MAX_UNCOMPRESSED_CHUNK_BYTES)) {
|
|
221
|
-
await flush();
|
|
222
|
-
}
|
|
223
|
-
chunk.push(page);
|
|
224
|
-
if (serializedChunk(chunk).length > MAX_UNCOMPRESSED_CHUNK_BYTES) {
|
|
225
|
-
throw new Error(`page ${page.pageId} exceeds the durable content chunk limit`);
|
|
226
|
-
}
|
|
227
|
-
}
|
|
228
|
-
await flush();
|
|
229
|
-
return refs;
|
|
230
|
-
}
|
|
231
|
-
function createSiteExtractContentReader(ownerId) {
|
|
232
|
-
const cache = /* @__PURE__ */ new Map();
|
|
233
|
-
return async (ref) => {
|
|
234
|
-
let chunk = cache.get(ref.artifactId);
|
|
235
|
-
if (!chunk) {
|
|
236
|
-
const compressed = await readOwnedSiteExtractContentChunk({ artifactId: ref.artifactId, ownerId });
|
|
237
|
-
if (!compressed) throw new Error("site export content chunk is missing or unauthorized");
|
|
238
|
-
if (sha256(compressed) !== ref.artifactSha256) throw new Error("site export content chunk checksum mismatch");
|
|
239
|
-
const decoded = gunzipSync(compressed);
|
|
240
|
-
if (decoded.length > MAX_UNCOMPRESSED_CHUNK_BYTES) throw new Error("site export content chunk exceeds its read limit");
|
|
241
|
-
chunk = JSON.parse(decoded.toString("utf8"));
|
|
242
|
-
if (chunk.version !== "site-extract-content.v1" || !Array.isArray(chunk.pages)) {
|
|
243
|
-
throw new Error("site export content chunk has an unsupported contract");
|
|
244
|
-
}
|
|
245
|
-
cache.set(ref.artifactId, chunk);
|
|
246
|
-
while (cache.size > 2) cache.delete(cache.keys().next().value);
|
|
247
|
-
}
|
|
248
|
-
const page = chunk.pages.find((candidate) => candidate.pageId === ref.pageId);
|
|
249
|
-
if (!page) throw new Error("site export page is missing from its content chunk");
|
|
250
|
-
if (sha256(page.html) !== page.htmlSha256 || sha256(page.markdown) !== page.markdownSha256) {
|
|
251
|
-
throw new Error("site export page checksum mismatch");
|
|
252
|
-
}
|
|
253
|
-
return page;
|
|
254
|
-
};
|
|
255
|
-
}
|
|
256
|
-
|
|
257
|
-
// src/api/site-extract-repository.ts
|
|
258
|
-
var XRAY_ENTITLED_EXTRACT_BILLING_CLASS = "xray_entitlement";
|
|
259
|
-
function extractJobLimitInfo(job) {
|
|
260
|
-
const effectiveMaxPages = Math.max(1, Number(job.options.effectiveMaxPages ?? job.options.maxPages ?? 1));
|
|
261
|
-
const requestedMaxPages = Math.max(effectiveMaxPages, Number(job.options.requestedMaxPages ?? effectiveMaxPages));
|
|
262
|
-
const creditLimited = requestedMaxPages > effectiveMaxPages;
|
|
263
|
-
return {
|
|
264
|
-
requestedMaxPages,
|
|
265
|
-
effectiveMaxPages,
|
|
266
|
-
creditLimited,
|
|
267
|
-
// If discovery exhausted below the funded cap, the smaller hold did not
|
|
268
|
-
// actually truncate this crawl. Hitting the cap is conservatively partial.
|
|
269
|
-
creditTruncated: creditLimited && job.totalUrls >= effectiveMaxPages
|
|
270
|
-
};
|
|
271
|
-
}
|
|
272
|
-
function terminalExtractJobStatus(progress, creditTruncated = false) {
|
|
273
|
-
if (progress.successfulUrls === 0) return "failed";
|
|
274
|
-
if (progress.failedUrls > 0 || progress.remainingUrls > 0 || creditTruncated) return "partial";
|
|
275
|
-
return "complete";
|
|
276
|
-
}
|
|
277
|
-
function rowToJob(r) {
|
|
278
|
-
const totalUrls = Number(r.total_urls ?? 0);
|
|
279
|
-
const legacyProgress = r.attempted_urls == null;
|
|
280
|
-
const attemptedUrls = Number(legacyProgress ? r.done_urls ?? 0 : r.attempted_urls);
|
|
281
|
-
const successfulUrls = Number(legacyProgress ? r.done_urls ?? 0 : r.successful_urls ?? 0);
|
|
282
|
-
const failedUrls = Number(r.failed_urls ?? Math.max(0, attemptedUrls - successfulUrls));
|
|
283
|
-
return {
|
|
284
|
-
id: String(r.id),
|
|
285
|
-
userId: r.user_id != null ? Number(r.user_id) : null,
|
|
286
|
-
idempotencyKey: r.idempotency_key != null ? String(r.idempotency_key) : null,
|
|
287
|
-
requestFingerprint: r.request_fingerprint != null ? String(r.request_fingerprint) : null,
|
|
288
|
-
status: r.status != null ? String(r.status) : "pending",
|
|
289
|
-
startUrl: String(r.start_url ?? ""),
|
|
290
|
-
options: r.options ? JSON.parse(String(r.options)) : {},
|
|
291
|
-
totalUrls,
|
|
292
|
-
doneUrls: attemptedUrls,
|
|
293
|
-
attemptedUrls,
|
|
294
|
-
successfulUrls,
|
|
295
|
-
failedUrls,
|
|
296
|
-
remainingUrls: Math.max(0, totalUrls - attemptedUrls),
|
|
297
|
-
artifacts: r.artifacts ? JSON.parse(String(r.artifacts)) : null,
|
|
298
|
-
error: r.error != null ? String(r.error) : null,
|
|
299
|
-
publicError: parsePublicErrorEnvelope(r.public_error_json),
|
|
300
|
-
billedMc: r.billed_mc != null ? Number(r.billed_mc) : null,
|
|
301
|
-
createdAt: String(r.created_at ?? ""),
|
|
302
|
-
updatedAt: String(r.updated_at ?? "")
|
|
303
|
-
};
|
|
304
|
-
}
|
|
305
|
-
async function createExtractJob(jobId, userId, startUrl, options) {
|
|
306
|
-
const db = getDb();
|
|
307
|
-
await db.execute({
|
|
308
|
-
sql: `INSERT INTO site_extract_jobs (id, user_id, status, start_url, options, created_at, updated_at)
|
|
309
|
-
VALUES (?, ?, 'pending', ?, ?, datetime('now'), datetime('now'))`,
|
|
310
|
-
args: [jobId, userId, startUrl, JSON.stringify(options)]
|
|
311
|
-
});
|
|
312
|
-
}
|
|
313
|
-
async function getExtractJobByIdempotencyKey(userId, idempotencyKey) {
|
|
314
|
-
const db = getDb();
|
|
315
|
-
const res = await db.execute({
|
|
316
|
-
sql: `SELECT * FROM site_extract_jobs WHERE user_id = ? AND idempotency_key = ? LIMIT 1`,
|
|
317
|
-
args: [userId, idempotencyKey]
|
|
318
|
-
});
|
|
319
|
-
return res.rows[0] ? rowToJob(res.rows[0]) : null;
|
|
320
|
-
}
|
|
321
|
-
async function createOrGetExtractJob(input) {
|
|
322
|
-
const db = getDb();
|
|
323
|
-
const inserted = await db.execute({
|
|
324
|
-
sql: `INSERT OR IGNORE INTO site_extract_jobs
|
|
325
|
-
(id, user_id, status, start_url, options, idempotency_key, request_fingerprint, created_at, updated_at)
|
|
326
|
-
VALUES (?, ?, 'pending', ?, ?, ?, ?, datetime('now'), datetime('now'))`,
|
|
327
|
-
args: [
|
|
328
|
-
input.jobId,
|
|
329
|
-
input.userId,
|
|
330
|
-
input.startUrl,
|
|
331
|
-
JSON.stringify(input.options),
|
|
332
|
-
input.idempotencyKey,
|
|
333
|
-
input.requestFingerprint
|
|
334
|
-
]
|
|
335
|
-
});
|
|
336
|
-
const job = await getExtractJobByIdempotencyKey(input.userId, input.idempotencyKey);
|
|
337
|
-
if (!job) throw new Error("idempotent extract job was not persisted");
|
|
338
|
-
return {
|
|
339
|
-
job,
|
|
340
|
-
created: inserted.rowsAffected === 1,
|
|
341
|
-
conflict: job.requestFingerprint !== input.requestFingerprint
|
|
342
|
-
};
|
|
343
|
-
}
|
|
344
|
-
async function getExtractJob(jobId) {
|
|
345
|
-
const db = getDb();
|
|
346
|
-
const res = await db.execute({ sql: `SELECT * FROM site_extract_jobs WHERE id = ?`, args: [jobId] });
|
|
347
|
-
return res.rows[0] ? rowToJob(res.rows[0]) : null;
|
|
348
|
-
}
|
|
349
|
-
async function listExtractJobs(userId) {
|
|
350
|
-
const db = getDb();
|
|
351
|
-
const res = await db.execute({ sql: `SELECT * FROM site_extract_jobs WHERE user_id = ? ORDER BY created_at DESC LIMIT 50`, args: [userId] });
|
|
352
|
-
return res.rows.map((r) => rowToJob(r));
|
|
353
|
-
}
|
|
354
|
-
async function listUnsettledExtractJobs(limit = 25) {
|
|
355
|
-
const db = getDb();
|
|
356
|
-
const res = await db.execute({
|
|
357
|
-
sql: `SELECT * FROM site_extract_jobs
|
|
358
|
-
WHERE billed_mc IS NULL AND status IN ('complete', 'partial', 'failed')
|
|
359
|
-
AND (next_settlement_at IS NULL OR next_settlement_at <= datetime('now'))
|
|
360
|
-
ORDER BY updated_at ASC LIMIT ?`,
|
|
361
|
-
args: [Math.max(1, Math.min(100, Math.round(limit)))]
|
|
362
|
-
});
|
|
363
|
-
return res.rows.map((row) => rowToJob(row));
|
|
364
|
-
}
|
|
365
|
-
async function listFundedPendingExtractJobs(limit = 25, staleMinutes = 2) {
|
|
366
|
-
const db = getDb();
|
|
367
|
-
const safeStaleMinutes = Math.max(1, Math.min(60, Math.round(staleMinutes)));
|
|
368
|
-
const res = await db.execute({
|
|
369
|
-
sql: `SELECT j.* FROM site_extract_jobs j
|
|
370
|
-
LEFT JOIN billing_debits d
|
|
371
|
-
ON d.idempotency_key = json_extract(j.options, '$.debitKey')
|
|
372
|
-
AND d.user_id = j.user_id
|
|
373
|
-
AND d.status = 'applied'
|
|
374
|
-
WHERE j.status = 'pending'
|
|
375
|
-
AND j.billed_mc IS NULL
|
|
376
|
-
AND (
|
|
377
|
-
d.idempotency_key IS NOT NULL
|
|
378
|
-
OR (
|
|
379
|
-
json_extract(j.options, '$.billingClass') = ?
|
|
380
|
-
AND json_extract(j.options, '$.heldMc') = 0
|
|
381
|
-
AND json_extract(j.options, '$.debitKey') IS NULL
|
|
382
|
-
AND json_extract(j.options, '$.xraySetupReceipt.billingClass') = ?
|
|
383
|
-
AND json_extract(j.options, '$.xraySetupReceipt.consumesMcpScraperCredits') = 0
|
|
384
|
-
)
|
|
385
|
-
)
|
|
386
|
-
AND (j.next_dispatch_at IS NULL OR j.next_dispatch_at <= datetime('now'))
|
|
387
|
-
AND j.updated_at <= datetime('now', ?)
|
|
388
|
-
ORDER BY j.updated_at ASC LIMIT ?`,
|
|
389
|
-
args: [
|
|
390
|
-
XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
|
|
391
|
-
XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
|
|
392
|
-
`-${safeStaleMinutes} minutes`,
|
|
393
|
-
Math.max(1, Math.min(100, Math.round(limit)))
|
|
394
|
-
]
|
|
395
|
-
});
|
|
396
|
-
return res.rows.map((row) => rowToJob(row));
|
|
397
|
-
}
|
|
398
|
-
async function markExtractJobDispatchAttempt(jobId) {
|
|
399
|
-
const db = getDb();
|
|
400
|
-
await db.execute({
|
|
401
|
-
sql: `UPDATE site_extract_jobs
|
|
402
|
-
SET dispatch_attempts = dispatch_attempts + 1,
|
|
403
|
-
updated_at = datetime('now'),
|
|
404
|
-
next_dispatch_at = datetime('now', '+5 minutes'),
|
|
405
|
-
dispatch_error = NULL,
|
|
406
|
-
status = CASE
|
|
407
|
-
WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours') THEN 'failed'
|
|
408
|
-
ELSE status
|
|
409
|
-
END,
|
|
410
|
-
error = CASE
|
|
411
|
-
WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours')
|
|
412
|
-
THEN 'Background crawl was acknowledged repeatedly but never started; the hold will be refunded.'
|
|
413
|
-
ELSE error
|
|
414
|
-
END
|
|
415
|
-
WHERE id = ? AND status = 'pending'`,
|
|
416
|
-
args: [jobId]
|
|
417
|
-
});
|
|
418
|
-
return getExtractJob(jobId);
|
|
419
|
-
}
|
|
420
|
-
async function recordExtractJobDispatchFailure(jobId, error) {
|
|
421
|
-
const db = getDb();
|
|
422
|
-
const safeError = sanitizeVendorName(error).slice(0, 1e3);
|
|
423
|
-
await db.execute({
|
|
424
|
-
sql: `UPDATE site_extract_jobs SET
|
|
425
|
-
dispatch_attempts = dispatch_attempts + 1,
|
|
426
|
-
dispatch_error = ?,
|
|
427
|
-
next_dispatch_at = datetime('now', '+5 minutes'),
|
|
428
|
-
updated_at = datetime('now'),
|
|
429
|
-
status = CASE
|
|
430
|
-
WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours') THEN 'failed'
|
|
431
|
-
ELSE status
|
|
432
|
-
END,
|
|
433
|
-
error = CASE
|
|
434
|
-
WHEN dispatch_attempts + 1 >= 10 OR created_at <= datetime('now', '-6 hours')
|
|
435
|
-
THEN 'Background crawl could not be dispatched after repeated attempts; the hold will be refunded.'
|
|
436
|
-
ELSE error
|
|
437
|
-
END
|
|
438
|
-
WHERE id = ? AND status = 'pending'`,
|
|
439
|
-
args: [safeError, jobId]
|
|
440
|
-
});
|
|
441
|
-
return getExtractJob(jobId);
|
|
442
|
-
}
|
|
443
|
-
async function recordExtractSettlementFailure(jobId, error) {
|
|
444
|
-
const db = getDb();
|
|
445
|
-
await db.execute({
|
|
446
|
-
sql: `UPDATE site_extract_jobs SET
|
|
447
|
-
settlement_attempts = settlement_attempts + 1,
|
|
448
|
-
settlement_error = ?,
|
|
449
|
-
next_settlement_at = datetime('now', '+15 minutes'),
|
|
450
|
-
updated_at = datetime('now')
|
|
451
|
-
WHERE id = ? AND billed_mc IS NULL`,
|
|
452
|
-
args: [sanitizeVendorName(error).slice(0, 1e3), jobId]
|
|
453
|
-
});
|
|
454
|
-
}
|
|
455
|
-
async function listStaleRunningExtractJobs(limit = 25, staleMinutes = 60) {
|
|
456
|
-
const db = getDb();
|
|
457
|
-
const safeStaleMinutes = Math.max(15, Math.min(24 * 60, Math.round(staleMinutes)));
|
|
458
|
-
const res = await db.execute({
|
|
459
|
-
sql: `SELECT j.* FROM site_extract_jobs j
|
|
460
|
-
LEFT JOIN billing_debits d
|
|
461
|
-
ON d.idempotency_key = json_extract(j.options, '$.debitKey')
|
|
462
|
-
AND d.user_id = j.user_id
|
|
463
|
-
AND d.status = 'applied'
|
|
464
|
-
WHERE j.status = 'running'
|
|
465
|
-
AND j.billed_mc IS NULL
|
|
466
|
-
AND (
|
|
467
|
-
json_extract(j.options, '$.debitKey') IS NULL
|
|
468
|
-
OR d.idempotency_key IS NOT NULL
|
|
469
|
-
)
|
|
470
|
-
AND j.updated_at <= datetime('now', ?)
|
|
471
|
-
ORDER BY j.updated_at ASC LIMIT ?`,
|
|
472
|
-
args: [`-${safeStaleMinutes} minutes`, Math.max(1, Math.min(100, Math.round(limit)))]
|
|
473
|
-
});
|
|
474
|
-
return res.rows.map((row) => rowToJob(row));
|
|
475
|
-
}
|
|
476
|
-
async function failStaleRunningExtractJob(jobId, staleMinutes = 60) {
|
|
477
|
-
const db = getDb();
|
|
478
|
-
const safeStaleMinutes = Math.max(15, Math.min(24 * 60, Math.round(staleMinutes)));
|
|
479
|
-
const result = await db.execute({
|
|
480
|
-
sql: `UPDATE site_extract_jobs
|
|
481
|
-
SET status = 'failed',
|
|
482
|
-
error = 'Background crawl stopped without a heartbeat; completed pages were preserved and settlement is pending.',
|
|
483
|
-
public_error_json = ?,
|
|
484
|
-
updated_at = datetime('now')
|
|
485
|
-
WHERE id = ?
|
|
486
|
-
AND status = 'running'
|
|
487
|
-
AND billed_mc IS NULL
|
|
488
|
-
AND updated_at <= datetime('now', ?)`,
|
|
489
|
-
args: [
|
|
490
|
-
serializePublicErrorEnvelope({
|
|
491
|
-
error_code: "extraction_failed",
|
|
492
|
-
error_type: "extraction",
|
|
493
|
-
message: "The background crawl stopped without a heartbeat. Completed pages were preserved; retry after settlement completes.",
|
|
494
|
-
retryable: true,
|
|
495
|
-
charge_status: "refund_pending"
|
|
496
|
-
}),
|
|
497
|
-
jobId,
|
|
498
|
-
`-${safeStaleMinutes} minutes`
|
|
499
|
-
]
|
|
500
|
-
});
|
|
501
|
-
return result.rowsAffected === 1;
|
|
502
|
-
}
|
|
503
|
-
async function claimFailedExtractJobForRefinalize(jobId) {
|
|
504
|
-
const db = getDb();
|
|
505
|
-
const claimed = await db.execute({
|
|
506
|
-
sql: `UPDATE site_extract_jobs
|
|
507
|
-
SET status = 'running', updated_at = datetime('now')
|
|
508
|
-
WHERE id = ? AND status = 'failed'`,
|
|
509
|
-
args: [jobId]
|
|
510
|
-
});
|
|
511
|
-
if (claimed.rowsAffected !== 1) return null;
|
|
512
|
-
return getExtractJob(jobId);
|
|
513
|
-
}
|
|
514
|
-
async function abandonExtractSettlement(jobId, error) {
|
|
515
|
-
const db = getDb();
|
|
516
|
-
await db.execute({
|
|
517
|
-
sql: `UPDATE site_extract_jobs SET billed_mc = 0, settlement_error = ?, updated_at = datetime('now')
|
|
518
|
-
WHERE id = ? AND billed_mc IS NULL`,
|
|
519
|
-
args: [sanitizeVendorName(error).slice(0, 1e3), jobId]
|
|
520
|
-
});
|
|
521
|
-
}
|
|
522
|
-
async function setExtractJobTotal(jobId, totalUrls) {
|
|
523
|
-
const db = getDb();
|
|
524
|
-
await db.execute({
|
|
525
|
-
sql: `UPDATE site_extract_jobs
|
|
526
|
-
SET status = 'running', total_urls = MAX(total_urls, ?), updated_at = datetime('now')
|
|
527
|
-
WHERE id = ? AND status IN ('pending', 'running')`,
|
|
528
|
-
args: [totalUrls, jobId]
|
|
529
|
-
});
|
|
530
|
-
}
|
|
531
|
-
async function saveExtractPages(jobId, pages) {
|
|
532
|
-
if (pages.length === 0) return;
|
|
533
|
-
const db = getDb();
|
|
534
|
-
const job = await getExtractJob(jobId);
|
|
535
|
-
const contentRefs = job?.userId != null ? await persistSiteExtractPageContent({
|
|
536
|
-
ownerId: String(job.userId),
|
|
537
|
-
jobId,
|
|
538
|
-
createdAt: job.createdAt,
|
|
539
|
-
pages
|
|
540
|
-
}) : /* @__PURE__ */ new Map();
|
|
541
|
-
await db.batch([
|
|
542
|
-
...pages.map((p) => {
|
|
543
|
-
const {
|
|
544
|
-
discoveryLinks: _ephemeralDiscoveryLinks,
|
|
545
|
-
acquiredHtml: _durableHtml,
|
|
546
|
-
mainHtml: _durableMainHtml,
|
|
547
|
-
...storedPage
|
|
548
|
-
} = p;
|
|
549
|
-
storedPage.contentRef = contentRefs.get(p.archivedUrl ?? p.url) ?? p.contentRef;
|
|
550
|
-
return {
|
|
551
|
-
sql: `INSERT INTO site_extract_pages (job_id, url, page)
|
|
552
|
-
SELECT ?, ?, ?
|
|
553
|
-
WHERE EXISTS (
|
|
554
|
-
SELECT 1 FROM site_extract_jobs
|
|
555
|
-
WHERE id = ? AND status IN ('pending', 'running')
|
|
556
|
-
)
|
|
557
|
-
ON CONFLICT(job_id, url) DO UPDATE SET page = excluded.page
|
|
558
|
-
WHERE NOT CASE
|
|
559
|
-
WHEN json_valid(site_extract_pages.page) = 0 THEN 0
|
|
560
|
-
WHEN json_extract(site_extract_pages.page, '$.extractionStatus') IS NOT NULL
|
|
561
|
-
THEN json_extract(site_extract_pages.page, '$.extractionStatus') = 'successful'
|
|
562
|
-
ELSE COALESCE(
|
|
563
|
-
json_extract(site_extract_pages.page, '$.status') >= 200
|
|
564
|
-
AND json_extract(site_extract_pages.page, '$.status') < 300
|
|
565
|
-
AND json_extract(site_extract_pages.page, '$.wordCount') > 0,
|
|
566
|
-
0
|
|
567
|
-
)
|
|
568
|
-
END
|
|
569
|
-
OR CASE
|
|
570
|
-
WHEN json_valid(excluded.page) = 0 THEN 0
|
|
571
|
-
WHEN json_extract(excluded.page, '$.extractionStatus') IS NOT NULL
|
|
572
|
-
THEN json_extract(excluded.page, '$.extractionStatus') = 'successful'
|
|
573
|
-
ELSE COALESCE(
|
|
574
|
-
json_extract(excluded.page, '$.status') >= 200
|
|
575
|
-
AND json_extract(excluded.page, '$.status') < 300
|
|
576
|
-
AND json_extract(excluded.page, '$.wordCount') > 0,
|
|
577
|
-
0
|
|
578
|
-
)
|
|
579
|
-
END`,
|
|
580
|
-
args: [jobId, p.archivedUrl ?? p.url, JSON.stringify(storedPage), jobId]
|
|
581
|
-
};
|
|
582
|
-
}),
|
|
583
|
-
{
|
|
584
|
-
sql: `UPDATE site_extract_jobs SET
|
|
585
|
-
done_urls = (SELECT COUNT(*) FROM site_extract_pages WHERE job_id = ?),
|
|
586
|
-
attempted_urls = (SELECT COUNT(*) FROM site_extract_pages WHERE job_id = ?),
|
|
587
|
-
successful_urls = (
|
|
588
|
-
SELECT COUNT(*) FROM site_extract_pages
|
|
589
|
-
WHERE job_id = ? AND CASE
|
|
590
|
-
WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
|
|
591
|
-
THEN json_extract(page, '$.extractionStatus') = 'successful'
|
|
592
|
-
ELSE json_extract(page, '$.status') >= 200
|
|
593
|
-
AND json_extract(page, '$.status') < 300
|
|
594
|
-
AND json_extract(page, '$.wordCount') > 0
|
|
595
|
-
END
|
|
596
|
-
),
|
|
597
|
-
failed_urls = (
|
|
598
|
-
SELECT COUNT(*) FROM site_extract_pages
|
|
599
|
-
WHERE job_id = ? AND NOT CASE
|
|
600
|
-
WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
|
|
601
|
-
THEN json_extract(page, '$.extractionStatus') = 'successful'
|
|
602
|
-
ELSE COALESCE(
|
|
603
|
-
json_extract(page, '$.status') >= 200
|
|
604
|
-
AND json_extract(page, '$.status') < 300
|
|
605
|
-
AND json_extract(page, '$.wordCount') > 0,
|
|
606
|
-
0
|
|
607
|
-
)
|
|
608
|
-
END
|
|
609
|
-
),
|
|
610
|
-
updated_at = datetime('now')
|
|
611
|
-
WHERE id = ? AND status IN ('pending', 'running')`,
|
|
612
|
-
args: [jobId, jobId, jobId, jobId, jobId]
|
|
613
|
-
}
|
|
614
|
-
], "write");
|
|
615
|
-
}
|
|
616
|
-
async function listExtractPages(jobId) {
|
|
617
|
-
const db = getDb();
|
|
618
|
-
const result = await db.execute({
|
|
619
|
-
sql: `SELECT page FROM site_extract_pages WHERE job_id = ? ORDER BY rowid`,
|
|
620
|
-
args: [jobId]
|
|
621
|
-
});
|
|
622
|
-
return result.rows.map((row) => JSON.parse(String(row.page)));
|
|
623
|
-
}
|
|
624
|
-
async function getExtractedPages(jobId) {
|
|
625
|
-
const db = getDb();
|
|
626
|
-
const res = await db.execute({ sql: `SELECT page FROM site_extract_pages WHERE job_id = ?`, args: [jobId] });
|
|
627
|
-
return res.rows.map((r) => JSON.parse(String(r.page)));
|
|
628
|
-
}
|
|
629
|
-
async function getExtractedImageLinks(jobId, limit = 181) {
|
|
630
|
-
const db = getDb();
|
|
631
|
-
const safeLimit = Math.max(1, Math.min(5001, Math.round(limit)));
|
|
632
|
-
const res = await db.execute({
|
|
633
|
-
sql: `SELECT DISTINCT CAST(images.value AS TEXT) AS url
|
|
634
|
-
FROM site_extract_pages p,
|
|
635
|
-
json_each(CASE WHEN json_valid(p.page) THEN p.page ELSE '{}' END, '$.imageLinks') images
|
|
636
|
-
WHERE p.job_id = ?
|
|
637
|
-
AND images.type = 'text'
|
|
638
|
-
AND length(CAST(images.value AS TEXT)) BETWEEN 1 AND 4096
|
|
639
|
-
LIMIT ?`,
|
|
640
|
-
args: [jobId, safeLimit]
|
|
641
|
-
});
|
|
642
|
-
return res.rows.map((row) => String(row.url));
|
|
643
|
-
}
|
|
644
|
-
async function countSuccessfulPages(jobId) {
|
|
645
|
-
const db = getDb();
|
|
646
|
-
const res = await db.execute({
|
|
647
|
-
sql: `SELECT COUNT(*) AS n FROM site_extract_pages
|
|
648
|
-
WHERE job_id = ? AND CASE
|
|
649
|
-
WHEN json_extract(page, '$.extractionStatus') IS NOT NULL
|
|
650
|
-
THEN json_extract(page, '$.extractionStatus') = 'successful'
|
|
651
|
-
ELSE COALESCE(
|
|
652
|
-
json_extract(page, '$.status') >= 200
|
|
653
|
-
AND json_extract(page, '$.status') < 300
|
|
654
|
-
AND json_extract(page, '$.wordCount') > 0,
|
|
655
|
-
0
|
|
656
|
-
)
|
|
657
|
-
END`,
|
|
658
|
-
args: [jobId]
|
|
659
|
-
});
|
|
660
|
-
return Number(res.rows[0]?.n ?? 0);
|
|
661
|
-
}
|
|
662
|
-
async function getExtractedUrls(jobId) {
|
|
663
|
-
const db = getDb();
|
|
664
|
-
const res = await db.execute({ sql: `SELECT url FROM site_extract_pages WHERE job_id = ?`, args: [jobId] });
|
|
665
|
-
return new Set(res.rows.map((r) => String(r.url)));
|
|
666
|
-
}
|
|
667
|
-
async function finishExtractJob(jobId, artifacts, status, error = null, allowFailedRecovery = false, publicError = null) {
|
|
668
|
-
const db = getDb();
|
|
669
|
-
const result = await db.execute({
|
|
670
|
-
sql: `UPDATE site_extract_jobs
|
|
671
|
-
SET status = ?, artifacts = ?, error = ?, public_error_json = ?, updated_at = datetime('now')
|
|
672
|
-
WHERE id = ? AND (status IN ('pending', 'running') OR (? = 1 AND status = 'failed'))`,
|
|
673
|
-
args: [
|
|
674
|
-
status,
|
|
675
|
-
JSON.stringify(artifacts ?? []),
|
|
676
|
-
error ? sanitizeVendorName(error).slice(0, 2e3) : null,
|
|
677
|
-
serializePublicErrorEnvelope(publicError),
|
|
678
|
-
jobId,
|
|
679
|
-
allowFailedRecovery ? 1 : 0
|
|
680
|
-
]
|
|
681
|
-
});
|
|
682
|
-
return result.rowsAffected === 1;
|
|
683
|
-
}
|
|
684
|
-
async function completeExtractJob(jobId, artifacts) {
|
|
685
|
-
await finishExtractJob(jobId, artifacts, "complete");
|
|
686
|
-
}
|
|
687
|
-
async function failExtractJob(jobId, error, publicError) {
|
|
688
|
-
const db = getDb();
|
|
689
|
-
await db.execute({
|
|
690
|
-
sql: `UPDATE site_extract_jobs
|
|
691
|
-
SET status = 'failed', error = ?, public_error_json = ?, updated_at = datetime('now')
|
|
692
|
-
WHERE id = ? AND status IN ('pending', 'running')`,
|
|
693
|
-
args: [sanitizeVendorName(error).slice(0, 2e3), serializePublicErrorEnvelope(publicError), jobId]
|
|
694
|
-
});
|
|
695
|
-
}
|
|
696
|
-
async function failUnfundedExtractJob(jobId, error, publicError) {
|
|
697
|
-
const db = getDb();
|
|
698
|
-
await db.execute({
|
|
699
|
-
sql: `UPDATE site_extract_jobs
|
|
700
|
-
SET status = 'failed', billed_mc = 0, error = ?, public_error_json = ?, updated_at = datetime('now')
|
|
701
|
-
WHERE id = ? AND status = 'pending' AND billed_mc IS NULL`,
|
|
702
|
-
args: [sanitizeVendorName(error).slice(0, 2e3), serializePublicErrorEnvelope(publicError), jobId]
|
|
703
|
-
});
|
|
704
|
-
}
|
|
705
|
-
async function setExtractJobPublicError(jobId, publicError) {
|
|
706
|
-
const job = await getExtractJob(jobId);
|
|
707
|
-
const normalized = job?.options.billingClass === XRAY_ENTITLED_EXTRACT_BILLING_CLASS && publicError ? { ...publicError, charge_status: "not_charged" } : publicError;
|
|
708
|
-
await getDb().execute({
|
|
709
|
-
sql: `UPDATE site_extract_jobs SET public_error_json = ?, updated_at = datetime('now')
|
|
710
|
-
WHERE id = ? AND status IN ('complete', 'partial', 'failed')`,
|
|
711
|
-
args: [serializePublicErrorEnvelope(normalized), jobId]
|
|
712
|
-
});
|
|
713
|
-
}
|
|
714
|
-
async function settleExtractJob(jobId, userId, refundMc, netChargeMc, reference) {
|
|
715
|
-
const db = getDb();
|
|
716
|
-
const job = await getExtractJob(jobId);
|
|
717
|
-
if (!job || job.billedMc != null) return;
|
|
718
|
-
if (job.options.billingClass === XRAY_ENTITLED_EXTRACT_BILLING_CLASS) {
|
|
719
|
-
if (!await finalizeXRayEntitledExtractJobCharge(jobId)) {
|
|
720
|
-
throw new Error("xray_extract_zero_charge_finalization_rejected");
|
|
721
|
-
}
|
|
722
|
-
return;
|
|
723
|
-
}
|
|
724
|
-
const debitKey = typeof job.options.debitKey === "string" ? job.options.debitKey : null;
|
|
725
|
-
if (debitKey) {
|
|
726
|
-
const settlement = await settleDebitMcIdempotent(
|
|
727
|
-
userId,
|
|
728
|
-
debitKey,
|
|
729
|
-
Math.max(0, netChargeMc),
|
|
730
|
-
LedgerOperation.EXTRACT_SITE_REFUND,
|
|
731
|
-
reference,
|
|
732
|
-
"extract_site"
|
|
733
|
-
);
|
|
734
|
-
await db.execute({
|
|
735
|
-
sql: `UPDATE site_extract_jobs SET billed_mc = ?, updated_at = datetime('now') WHERE id = ? AND billed_mc IS NULL`,
|
|
736
|
-
args: [settlement.final_amount_mc, jobId]
|
|
737
|
-
});
|
|
738
|
-
return;
|
|
739
|
-
}
|
|
740
|
-
const safeRefundMc = Math.max(0, refundMc);
|
|
741
|
-
const safeNetChargeMc = Math.max(0, netChargeMc);
|
|
742
|
-
await db.batch([
|
|
743
|
-
{
|
|
744
|
-
sql: `UPDATE site_extract_jobs SET billed_mc = -1 WHERE id = ? AND billed_mc IS NULL`,
|
|
745
|
-
args: [jobId]
|
|
746
|
-
},
|
|
747
|
-
{
|
|
748
|
-
sql: `INSERT INTO credit_lots (user_id, amount_mc, remaining_mc, source, expires_at)
|
|
749
|
-
SELECT ?, ?, ?, ?, datetime(j.created_at, ?)
|
|
750
|
-
FROM site_extract_jobs j
|
|
751
|
-
WHERE j.id = ?
|
|
752
|
-
AND j.billed_mc = -1
|
|
753
|
-
AND ? > 0
|
|
754
|
-
AND changes() = 1
|
|
755
|
-
AND datetime(j.created_at, ?) > datetime('now')`,
|
|
756
|
-
args: [
|
|
757
|
-
userId,
|
|
758
|
-
safeRefundMc,
|
|
759
|
-
safeRefundMc,
|
|
760
|
-
LedgerOperation.EXTRACT_SITE_REFUND,
|
|
761
|
-
CREDIT_LOT_TTL,
|
|
762
|
-
jobId,
|
|
763
|
-
safeRefundMc,
|
|
764
|
-
CREDIT_LOT_TTL
|
|
765
|
-
]
|
|
766
|
-
},
|
|
767
|
-
{
|
|
768
|
-
sql: `UPDATE users SET balance_mc = balance_mc + ?
|
|
769
|
-
WHERE id = ? AND ? > 0 AND changes() = 1`,
|
|
770
|
-
args: [safeRefundMc, userId, safeRefundMc]
|
|
771
|
-
},
|
|
772
|
-
{
|
|
773
|
-
sql: `INSERT INTO ledger (user_id, amount_mc, operation, description)
|
|
774
|
-
SELECT ?, ?, ?, ? WHERE ? > 0 AND changes() = 1`,
|
|
775
|
-
args: [userId, safeRefundMc, LedgerOperation.EXTRACT_SITE_REFUND, reference, safeRefundMc]
|
|
776
|
-
},
|
|
777
|
-
{
|
|
778
|
-
sql: `UPDATE site_extract_jobs SET billed_mc = ?, updated_at = datetime('now')
|
|
779
|
-
WHERE id = ? AND billed_mc = -1`,
|
|
780
|
-
args: [safeNetChargeMc, jobId]
|
|
781
|
-
}
|
|
782
|
-
], "write");
|
|
783
|
-
}
|
|
784
|
-
async function finalizeXRayEntitledExtractJobCharge(jobId) {
|
|
785
|
-
const db = getDb();
|
|
786
|
-
const result = await db.execute({
|
|
787
|
-
sql: `UPDATE site_extract_jobs
|
|
788
|
-
SET billed_mc = 0, updated_at = datetime('now')
|
|
789
|
-
WHERE id = ?
|
|
790
|
-
AND billed_mc IS NULL
|
|
791
|
-
AND json_extract(options, '$.billingClass') = ?
|
|
792
|
-
AND json_extract(options, '$.heldMc') = 0
|
|
793
|
-
AND json_extract(options, '$.debitKey') IS NULL
|
|
794
|
-
AND json_extract(options, '$.xraySetupReceipt.billingClass') = ?
|
|
795
|
-
AND json_extract(options, '$.xraySetupReceipt.consumesMcpScraperCredits') = 0
|
|
796
|
-
AND json_extract(options, '$.xraySetupReceipt.heldMc') = 0
|
|
797
|
-
AND json_extract(options, '$.xraySetupReceipt.billedMc') = 0`,
|
|
798
|
-
args: [jobId, XRAY_ENTITLED_EXTRACT_BILLING_CLASS, XRAY_ENTITLED_EXTRACT_BILLING_CLASS]
|
|
799
|
-
});
|
|
800
|
-
return result.rowsAffected === 1;
|
|
801
|
-
}
|
|
802
|
-
|
|
803
|
-
export {
|
|
804
|
-
SITE_EXTRACT_ARTIFACT_PREFIX,
|
|
805
|
-
createSiteExtractBundleArtifactStream,
|
|
806
|
-
createSiteExtractImageArtifact,
|
|
807
|
-
renewSiteExtractArtifactDownload,
|
|
808
|
-
readOwnedSiteExtractArtifactBuffer,
|
|
809
|
-
readOwnedSiteExtractImageArtifact,
|
|
810
|
-
cleanupExpiredSiteExtractArtifacts,
|
|
811
|
-
createSiteExtractContentReader,
|
|
812
|
-
XRAY_ENTITLED_EXTRACT_BILLING_CLASS,
|
|
813
|
-
extractJobLimitInfo,
|
|
814
|
-
terminalExtractJobStatus,
|
|
815
|
-
createExtractJob,
|
|
816
|
-
getExtractJobByIdempotencyKey,
|
|
817
|
-
createOrGetExtractJob,
|
|
818
|
-
getExtractJob,
|
|
819
|
-
listExtractJobs,
|
|
820
|
-
listUnsettledExtractJobs,
|
|
821
|
-
listFundedPendingExtractJobs,
|
|
822
|
-
markExtractJobDispatchAttempt,
|
|
823
|
-
recordExtractJobDispatchFailure,
|
|
824
|
-
recordExtractSettlementFailure,
|
|
825
|
-
listStaleRunningExtractJobs,
|
|
826
|
-
failStaleRunningExtractJob,
|
|
827
|
-
claimFailedExtractJobForRefinalize,
|
|
828
|
-
abandonExtractSettlement,
|
|
829
|
-
setExtractJobTotal,
|
|
830
|
-
saveExtractPages,
|
|
831
|
-
listExtractPages,
|
|
832
|
-
getExtractedPages,
|
|
833
|
-
getExtractedImageLinks,
|
|
834
|
-
countSuccessfulPages,
|
|
835
|
-
getExtractedUrls,
|
|
836
|
-
finishExtractJob,
|
|
837
|
-
completeExtractJob,
|
|
838
|
-
failExtractJob,
|
|
839
|
-
failUnfundedExtractJob,
|
|
840
|
-
setExtractJobPublicError,
|
|
841
|
-
settleExtractJob,
|
|
842
|
-
finalizeXRayEntitledExtractJobCharge
|
|
843
|
-
};
|