@semiont/jobs 0.5.34 → 0.5.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +130 -18
- package/dist/index.js +241 -108
- package/dist/index.js.map +1 -1
- package/dist/worker-main.js +281 -154
- package/dist/worker-main.js.map +1 -1
- package/package.json +8 -8
package/dist/index.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { promises, mkdtempSync, writeFileSync, readFileSync, rmSync } from 'fs';
|
|
2
2
|
import * as path from 'path';
|
|
3
3
|
import { join } from 'path';
|
|
4
|
-
import { replyChannelsFor, jobId, deriveViews, GENERATABLE_MEDIA_TYPES, estimateTokens,
|
|
4
|
+
import { replyChannelsFor, jobId, deriveViews, GENERATABLE_MEDIA_TYPES, estimateTokens, isObject, isString, reconcileSelector, getLocaleEnglishName, cutChunk, chunkText } from '@semiont/core';
|
|
5
5
|
import { withSpan, recordAnchorOutcome, recordDetectionCall } from '@semiont/observability';
|
|
6
6
|
import { StructuredReadError } from '@semiont/inference';
|
|
7
7
|
import { execFileSync } from 'child_process';
|
|
@@ -22,6 +22,19 @@ function willRetryAfter(metadata, failureClass) {
|
|
|
22
22
|
var REANNOUNCE_INTERVAL_MS = 3e4;
|
|
23
23
|
var STALE_RUNNING_MS = 30 * 6e4;
|
|
24
24
|
var PROGRESS_WRITE_MIN_INTERVAL_MS = 5e3;
|
|
25
|
+
function mergeUnitCursors(existing, incoming, completed) {
|
|
26
|
+
const done = new Set(completed);
|
|
27
|
+
const merged = {};
|
|
28
|
+
for (const [unit, cursor] of Object.entries({ ...existing })) {
|
|
29
|
+
if (!done.has(unit)) merged[unit] = cursor;
|
|
30
|
+
}
|
|
31
|
+
for (const [unit, cursor] of Object.entries({ ...incoming })) {
|
|
32
|
+
if (done.has(unit)) continue;
|
|
33
|
+
const held = merged[unit];
|
|
34
|
+
if (!held || cursor.next > held.next) merged[unit] = cursor;
|
|
35
|
+
}
|
|
36
|
+
return merged;
|
|
37
|
+
}
|
|
25
38
|
var RETENTION_HOURS = 24;
|
|
26
39
|
var CLEANUP_INTERVAL_MS = 36e5;
|
|
27
40
|
var FsJobQueue = class {
|
|
@@ -202,14 +215,20 @@ var FsJobQueue = class {
|
|
|
202
215
|
* re-announced); after that it lands in `failed` with the error.
|
|
203
216
|
* Returns null (and changes nothing) if the job isn't running.
|
|
204
217
|
*/
|
|
205
|
-
async failJob(jobId, error, completedUnits, failureClass) {
|
|
218
|
+
async failJob(jobId, error, completedUnits, failureClass, unitCursors) {
|
|
206
219
|
const job = await this.getJob(jobId);
|
|
207
220
|
if (!job || job.status !== "running") {
|
|
208
221
|
return null;
|
|
209
222
|
}
|
|
210
223
|
this.lastProgressWrite.delete(jobId);
|
|
211
224
|
const checkpoint = [.../* @__PURE__ */ new Set([...job.metadata.completedUnits ?? [], ...completedUnits ?? []])];
|
|
212
|
-
const
|
|
225
|
+
const cursors = mergeUnitCursors(job.metadata.unitCursors, unitCursors, checkpoint);
|
|
226
|
+
const { unitCursors: superseded, ...base } = job.metadata;
|
|
227
|
+
const metadata = {
|
|
228
|
+
...base,
|
|
229
|
+
...checkpoint.length > 0 ? { completedUnits: checkpoint } : {},
|
|
230
|
+
...Object.keys(cursors).length > 0 ? { unitCursors: cursors } : {}
|
|
231
|
+
};
|
|
213
232
|
if (willRetryAfter(job.metadata, failureClass)) {
|
|
214
233
|
const retried = {
|
|
215
234
|
status: "pending",
|
|
@@ -245,15 +264,24 @@ var FsJobQueue = class {
|
|
|
245
264
|
* non-running jobs. Written directly, like `recordProgress`, so it also
|
|
246
265
|
* refreshes the mtime heartbeat.
|
|
247
266
|
*/
|
|
248
|
-
async checkpointUnits(jobId, completedUnits) {
|
|
267
|
+
async checkpointUnits(jobId, completedUnits, unitCursors) {
|
|
249
268
|
const job = await this.getJob(jobId);
|
|
250
269
|
if (!job || job.status !== "running") {
|
|
251
270
|
return;
|
|
252
271
|
}
|
|
253
272
|
const merged = [.../* @__PURE__ */ new Set([...job.metadata.completedUnits ?? [], ...completedUnits])];
|
|
273
|
+
const cursors = mergeUnitCursors(job.metadata.unitCursors, unitCursors, merged);
|
|
274
|
+
const { unitCursors: superseded, ...metadata } = job.metadata;
|
|
254
275
|
const updated = {
|
|
255
276
|
...job,
|
|
256
|
-
metadata: {
|
|
277
|
+
metadata: {
|
|
278
|
+
...metadata,
|
|
279
|
+
completedUnits: merged,
|
|
280
|
+
// Absent rather than `{}`: an empty object would assert that units were
|
|
281
|
+
// tracked and none had progress, a different claim from a job that
|
|
282
|
+
// never reported a cursor at all.
|
|
283
|
+
...Object.keys(cursors).length > 0 ? { unitCursors: cursors } : {}
|
|
284
|
+
}
|
|
257
285
|
};
|
|
258
286
|
await promises.writeFile(this.getJobPath(jobId, "running"), JSON.stringify(updated, null, 2), "utf-8");
|
|
259
287
|
}
|
|
@@ -539,10 +567,33 @@ var DeterministicJobError = class extends Error {
|
|
|
539
567
|
name = "DeterministicJobError";
|
|
540
568
|
};
|
|
541
569
|
|
|
570
|
+
// src/workers/detection/chunk-size-controller.ts
|
|
571
|
+
var DEFAULT_CHUNK_SIZING_POLICY = {
|
|
572
|
+
growBelow: 0.5,
|
|
573
|
+
shrinkAbove: 0.8,
|
|
574
|
+
growFactor: 1.5,
|
|
575
|
+
shrinkFactor: 0.7
|
|
576
|
+
};
|
|
577
|
+
function clamp(value, lo, hi) {
|
|
578
|
+
return Math.max(lo, Math.min(hi, value));
|
|
579
|
+
}
|
|
580
|
+
function nextChunkSize(outcome, current, bounds, policy = DEFAULT_CHUNK_SIZING_POLICY) {
|
|
581
|
+
const grow = () => clamp(Math.floor(current * policy.growFactor), bounds.floor, bounds.ceiling);
|
|
582
|
+
const shrink = () => clamp(Math.floor(current * policy.shrinkFactor), bounds.floor, bounds.ceiling);
|
|
583
|
+
const hold = () => clamp(current, bounds.floor, bounds.ceiling);
|
|
584
|
+
if (outcome.truncated) return shrink();
|
|
585
|
+
if (outcome.outputTokens === void 0) return hold();
|
|
586
|
+
if (bounds.outputBudget <= 0) return hold();
|
|
587
|
+
const utilization = outcome.outputTokens / bounds.outputBudget;
|
|
588
|
+
if (utilization < policy.growBelow) return grow();
|
|
589
|
+
if (utilization > policy.shrinkAbove) return shrink();
|
|
590
|
+
return hold();
|
|
591
|
+
}
|
|
592
|
+
|
|
542
593
|
// src/workers/detection/detection-chunking.ts
|
|
543
|
-
function assertNotTruncated(response, label,
|
|
594
|
+
function assertNotTruncated(response, label, at, totalChars, outputBudget) {
|
|
544
595
|
if (response.stopReason === "max_tokens") {
|
|
545
|
-
throw new DeterministicJobError(`${label} response truncated (max_tokens)
|
|
596
|
+
throw new DeterministicJobError(`${label} response truncated (max_tokens) at character ${at} of ${totalChars} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
|
|
546
597
|
}
|
|
547
598
|
}
|
|
548
599
|
var SELECTOR_CONTEXT_CHARS = 64;
|
|
@@ -593,6 +644,7 @@ function deriveDetectionBudget(limits, scaffoldTokens, typesPerCall) {
|
|
|
593
644
|
outputBudget = durationSafeOutput;
|
|
594
645
|
}
|
|
595
646
|
}
|
|
647
|
+
const capacityInput = contextTokens - scaffoldTokens - outputBudget;
|
|
596
648
|
inputBudget = Math.min(inputBudget, Math.floor(outputBudget / (2 * typesPerCall)));
|
|
597
649
|
if (inputBudget <= OVERLAP_TOKENS) {
|
|
598
650
|
throw new Error(
|
|
@@ -601,15 +653,41 @@ function deriveDetectionBudget(limits, scaffoldTokens, typesPerCall) {
|
|
|
601
653
|
}
|
|
602
654
|
return {
|
|
603
655
|
chunking: { chunkSize: inputBudget, overlap: OVERLAP_TOKENS },
|
|
604
|
-
outputBudget
|
|
656
|
+
outputBudget,
|
|
657
|
+
bounds: {
|
|
658
|
+
// Two overlaps: at the floor a chunk is still half text it has not seen
|
|
659
|
+
// before, so the cursor keeps making progress rather than re-reading its
|
|
660
|
+
// own tail. Clamped under the opening size, because a cramped window can
|
|
661
|
+
// derive an opening below even that — and a floor above the opening would
|
|
662
|
+
// clamp every proposal upward, silently reversing the sizer.
|
|
663
|
+
floor: Math.min(inputBudget, 2 * OVERLAP_TOKENS),
|
|
664
|
+
// Never below the opening: a degenerate window can make the fit
|
|
665
|
+
// arithmetic smaller than the size already chosen, and a ceiling under
|
|
666
|
+
// the opening would clamp every proposal DOWN on the first call.
|
|
667
|
+
ceiling: Math.max(inputBudget, capacityInput),
|
|
668
|
+
outputBudget
|
|
669
|
+
}
|
|
605
670
|
};
|
|
606
671
|
}
|
|
672
|
+
async function runAdaptiveChunks(text, budget, onChunk, resume) {
|
|
673
|
+
let at = resume?.next ?? 0;
|
|
674
|
+
let size = resume ? nextChunkSize({ truncated: true }, resume.size, budget.bounds) : budget.chunking.chunkSize;
|
|
675
|
+
while (at < text.length) {
|
|
676
|
+
const { piece, next } = cutChunk(text, at, { chunkSize: size, overlap: budget.chunking.overlap });
|
|
677
|
+
const outcome = await onChunk({ piece, size, at, next, totalChars: text.length });
|
|
678
|
+
at = next;
|
|
679
|
+
size = nextChunkSize(outcome, size, budget.bounds);
|
|
680
|
+
}
|
|
681
|
+
}
|
|
607
682
|
var MAX_SUBDIVISION_DEPTH = 2;
|
|
608
683
|
function unknownUnreadable(error) {
|
|
609
684
|
return error instanceof StructuredReadError && error.stopReason === "unknown";
|
|
610
685
|
}
|
|
686
|
+
function unreadableDespiteFinishing(error) {
|
|
687
|
+
return error instanceof StructuredReadError && error.stopReason === "end_turn";
|
|
688
|
+
}
|
|
611
689
|
function subdividable(error) {
|
|
612
|
-
return error instanceof InferenceTimeoutError || truncation(error) || unknownUnreadable(error);
|
|
690
|
+
return error instanceof InferenceTimeoutError || truncation(error) || unknownUnreadable(error) || unreadableDespiteFinishing(error);
|
|
613
691
|
}
|
|
614
692
|
function truncation(error) {
|
|
615
693
|
return error instanceof DeterministicJobError || error instanceof StructuredReadError && error.stopReason === "max_tokens";
|
|
@@ -621,10 +699,19 @@ function outcomeOf(error) {
|
|
|
621
699
|
return "error";
|
|
622
700
|
}
|
|
623
701
|
async function callChunkSubdividing(label, chunk, chunking, call, logger, onUnderReport, onCounted) {
|
|
702
|
+
let outputTokens = 0;
|
|
703
|
+
let sizeShaped = false;
|
|
704
|
+
let callsMade = 0;
|
|
705
|
+
let callsMeasured = 0;
|
|
624
706
|
async function recorded(piece, depth, reroll) {
|
|
625
707
|
const start = performance.now();
|
|
626
708
|
try {
|
|
627
709
|
const result = await call(piece);
|
|
710
|
+
callsMade += 1;
|
|
711
|
+
if (result.usage) {
|
|
712
|
+
callsMeasured += 1;
|
|
713
|
+
outputTokens += result.usage.outputTokens;
|
|
714
|
+
}
|
|
628
715
|
recordDetectionCall({
|
|
629
716
|
label,
|
|
630
717
|
pieceChars: piece.length,
|
|
@@ -655,6 +742,7 @@ async function callChunkSubdividing(label, chunk, chunking, call, logger, onUnde
|
|
|
655
742
|
return (await recorded(piece, depth, false)).items;
|
|
656
743
|
} catch (error) {
|
|
657
744
|
if (!subdividable(error)) throw error;
|
|
745
|
+
sizeShaped = true;
|
|
658
746
|
const half = Math.floor(chunkSize / 2);
|
|
659
747
|
const pieces = chunkText(piece, { chunkSize: half, overlap: chunking.overlap });
|
|
660
748
|
const shrinks = pieces.length > 1 || pieces[0] !== piece;
|
|
@@ -690,7 +778,12 @@ async function callChunkSubdividing(label, chunk, chunking, call, logger, onUnde
|
|
|
690
778
|
return collected;
|
|
691
779
|
}
|
|
692
780
|
}
|
|
693
|
-
|
|
781
|
+
const items = await attempt(chunk, chunking.chunkSize, 0);
|
|
782
|
+
const measured = callsMade > 0 && callsMeasured === callsMade;
|
|
783
|
+
return {
|
|
784
|
+
items,
|
|
785
|
+
outcome: { truncated: sizeShaped, ...measured ? { outputTokens } : {} }
|
|
786
|
+
};
|
|
694
787
|
}
|
|
695
788
|
function languageName(tag) {
|
|
696
789
|
return getLocaleEnglishName(tag) || tag;
|
|
@@ -1191,33 +1284,37 @@ function logAnchorMethod(motivation, exact, anchorMethod) {
|
|
|
1191
1284
|
}
|
|
1192
1285
|
|
|
1193
1286
|
// src/workers/annotation-detection.ts
|
|
1194
|
-
async function detectInChunks(client, content, buildPrompt, motivation, elementSchema, parse, onActivity, onChunkResults) {
|
|
1287
|
+
async function detectInChunks(client, content, buildPrompt, motivation, elementSchema, parse, onActivity, resume, onChunkResults) {
|
|
1195
1288
|
const limits = await client.limits();
|
|
1196
1289
|
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
1197
|
-
const
|
|
1198
|
-
const
|
|
1290
|
+
const budget = deriveDetectionBudget(limits, scaffoldTokens, 1);
|
|
1291
|
+
const { outputBudget } = budget;
|
|
1199
1292
|
const collected = [];
|
|
1200
|
-
|
|
1201
|
-
const items = await callChunkSubdividing(
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
1206
|
-
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1293
|
+
await runAdaptiveChunks(content, budget, async ({ piece: chunk, size, at, next, totalChars }) => {
|
|
1294
|
+
const { items, outcome } = await callChunkSubdividing(
|
|
1295
|
+
motivation,
|
|
1296
|
+
chunk,
|
|
1297
|
+
{ chunkSize: size, overlap: budget.chunking.overlap },
|
|
1298
|
+
async (piece) => {
|
|
1299
|
+
const response = await boundedGenerateStructured(
|
|
1300
|
+
client,
|
|
1301
|
+
buildPrompt(piece),
|
|
1302
|
+
outputBudget,
|
|
1303
|
+
DETECTION_TEMPERATURE,
|
|
1304
|
+
elementSchema,
|
|
1305
|
+
// Still alive, same position (a long single call is otherwise silent).
|
|
1306
|
+
() => onActivity?.(at, totalChars)
|
|
1307
|
+
);
|
|
1308
|
+
assertNotTruncated(response, `${motivation} detection`, at, totalChars, outputBudget);
|
|
1309
|
+
return { items: response.items, ...response.usage ? { usage: response.usage } : {} };
|
|
1310
|
+
}
|
|
1311
|
+
);
|
|
1214
1312
|
const fromChunk = parse(items);
|
|
1215
1313
|
collected.push(...fromChunk);
|
|
1216
|
-
await onChunkResults?.(fromChunk);
|
|
1217
|
-
if (
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
}
|
|
1314
|
+
await onChunkResults?.(fromChunk, { next, size });
|
|
1315
|
+
if (next < totalChars) onActivity?.(next, totalChars);
|
|
1316
|
+
return outcome;
|
|
1317
|
+
}, resume);
|
|
1221
1318
|
return collected;
|
|
1222
1319
|
}
|
|
1223
1320
|
var AnnotationDetection = class {
|
|
@@ -1229,7 +1326,7 @@ var AnnotationDetection = class {
|
|
|
1229
1326
|
* (source-resource locale). See `types.ts` "Locale conventions" for the
|
|
1230
1327
|
* full discussion.
|
|
1231
1328
|
*/
|
|
1232
|
-
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onActivity, onChunkResults) {
|
|
1329
|
+
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onActivity, resume, onChunkResults) {
|
|
1233
1330
|
return detectInChunks(
|
|
1234
1331
|
client,
|
|
1235
1332
|
content,
|
|
@@ -1238,6 +1335,7 @@ var AnnotationDetection = class {
|
|
|
1238
1335
|
COMMENT_ELEMENT_SCHEMA,
|
|
1239
1336
|
(items) => MotivationParsers.parseComments(items, content),
|
|
1240
1337
|
onActivity,
|
|
1338
|
+
resume,
|
|
1241
1339
|
onChunkResults
|
|
1242
1340
|
);
|
|
1243
1341
|
}
|
|
@@ -1248,7 +1346,7 @@ var AnnotationDetection = class {
|
|
|
1248
1346
|
* applies, used in the prompt so the LLM analyzes non-English source
|
|
1249
1347
|
* correctly.
|
|
1250
1348
|
*/
|
|
1251
|
-
static async detectHighlights(content, client, instructions, density, sourceLanguage, onActivity, onChunkResults) {
|
|
1349
|
+
static async detectHighlights(content, client, instructions, density, sourceLanguage, onActivity, resume, onChunkResults) {
|
|
1252
1350
|
return detectInChunks(
|
|
1253
1351
|
client,
|
|
1254
1352
|
content,
|
|
@@ -1257,6 +1355,7 @@ var AnnotationDetection = class {
|
|
|
1257
1355
|
HIGHLIGHT_ELEMENT_SCHEMA,
|
|
1258
1356
|
(items) => MotivationParsers.parseHighlights(items, content),
|
|
1259
1357
|
onActivity,
|
|
1358
|
+
resume,
|
|
1260
1359
|
onChunkResults
|
|
1261
1360
|
);
|
|
1262
1361
|
}
|
|
@@ -1267,7 +1366,7 @@ var AnnotationDetection = class {
|
|
|
1267
1366
|
* (annotation body locale). `sourceLanguage` is the locale of the content
|
|
1268
1367
|
* being analyzed (source-resource locale).
|
|
1269
1368
|
*/
|
|
1270
|
-
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onActivity, onChunkResults) {
|
|
1369
|
+
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onActivity, resume, onChunkResults) {
|
|
1271
1370
|
return detectInChunks(
|
|
1272
1371
|
client,
|
|
1273
1372
|
content,
|
|
@@ -1276,6 +1375,7 @@ var AnnotationDetection = class {
|
|
|
1276
1375
|
ASSESSMENT_ELEMENT_SCHEMA,
|
|
1277
1376
|
(items) => MotivationParsers.parseAssessments(items, content),
|
|
1278
1377
|
onActivity,
|
|
1378
|
+
resume,
|
|
1279
1379
|
onChunkResults
|
|
1280
1380
|
);
|
|
1281
1381
|
}
|
|
@@ -1291,7 +1391,7 @@ var AnnotationDetection = class {
|
|
|
1291
1391
|
* identifiers, not LLM-generated text — so it's consumed at the body-stamp
|
|
1292
1392
|
* site, not here.
|
|
1293
1393
|
*/
|
|
1294
|
-
static async detectTags(content, client, schema, category, sourceLanguage, onActivity, onChunkResults) {
|
|
1394
|
+
static async detectTags(content, client, schema, category, sourceLanguage, onActivity, resume, onChunkResults) {
|
|
1295
1395
|
const categoryInfo = schema.tags.find((t) => t.name === category);
|
|
1296
1396
|
if (!categoryInfo) {
|
|
1297
1397
|
throw new Error(`Invalid category "${category}" for schema ${schema.id}`);
|
|
@@ -1313,7 +1413,8 @@ var AnnotationDetection = class {
|
|
|
1313
1413
|
TAG_ELEMENT_SCHEMA,
|
|
1314
1414
|
(items) => MotivationParsers.parseTags(items),
|
|
1315
1415
|
onActivity,
|
|
1316
|
-
|
|
1416
|
+
resume,
|
|
1417
|
+
onChunkResults ? async (raw, cursor) => onChunkResults(MotivationParsers.validateTagOffsets(raw, content, category), cursor) : void 0
|
|
1317
1418
|
);
|
|
1318
1419
|
return MotivationParsers.validateTagOffsets(parsedTags, content, category);
|
|
1319
1420
|
}
|
|
@@ -1365,7 +1466,7 @@ ${piece}
|
|
|
1365
1466
|
}
|
|
1366
1467
|
return counted;
|
|
1367
1468
|
}
|
|
1368
|
-
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger, sourceLanguage, onActivity, onUnderReport, onCounted, onChunkResults) {
|
|
1469
|
+
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger, sourceLanguage, onActivity, onUnderReport, onCounted, resume, onChunkResults) {
|
|
1369
1470
|
const entityTypesDescription = entityTypes.map((et) => {
|
|
1370
1471
|
if (typeof et === "string") {
|
|
1371
1472
|
return et;
|
|
@@ -1417,42 +1518,55 @@ Example output:
|
|
|
1417
1518
|
const limits = await client.limits();
|
|
1418
1519
|
const verifyYield = client.verifyDetectionYield;
|
|
1419
1520
|
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
1420
|
-
const
|
|
1421
|
-
const
|
|
1521
|
+
const budget = deriveDetectionBudget(limits, scaffoldTokens, entityTypes.length);
|
|
1522
|
+
const { chunking, outputBudget } = budget;
|
|
1422
1523
|
logger.debug("Sending entity extraction request", {
|
|
1423
1524
|
entityTypes: entityTypesDescription,
|
|
1424
|
-
|
|
1425
|
-
|
|
1525
|
+
chars: exact.length,
|
|
1526
|
+
// The size the run OPENS at, and how far measured yield may move it. The
|
|
1527
|
+
// chunk COUNT is deliberately absent: with sizing decided as the run goes,
|
|
1528
|
+
// there is no honest total until the cursor reaches the end.
|
|
1529
|
+
openingChunkSizeTokens: chunking.chunkSize,
|
|
1530
|
+
ceilingChunkSizeTokens: budget.bounds.ceiling,
|
|
1426
1531
|
outputBudget
|
|
1427
1532
|
});
|
|
1428
1533
|
const collected = [];
|
|
1429
|
-
|
|
1430
|
-
const items = await callChunkSubdividing(
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
|
|
1434
|
-
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
|
|
1440
|
-
|
|
1441
|
-
|
|
1442
|
-
|
|
1443
|
-
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1447
|
-
|
|
1448
|
-
|
|
1449
|
-
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
|
|
1534
|
+
await runAdaptiveChunks(exact, budget, async ({ piece: chunk, size, at, next, totalChars }) => {
|
|
1535
|
+
const { items, outcome } = await callChunkSubdividing(
|
|
1536
|
+
"reference",
|
|
1537
|
+
chunk,
|
|
1538
|
+
{ chunkSize: size, overlap: chunking.overlap },
|
|
1539
|
+
async (piece) => {
|
|
1540
|
+
const response = await boundedGenerateStructured(
|
|
1541
|
+
client,
|
|
1542
|
+
buildPrompt(piece),
|
|
1543
|
+
outputBudget,
|
|
1544
|
+
DETECTION_TEMPERATURE,
|
|
1545
|
+
ENTITY_ELEMENT_SCHEMA,
|
|
1546
|
+
// Still alive, same position: a long single call would otherwise emit
|
|
1547
|
+
// nothing at all between start and finish.
|
|
1548
|
+
() => onActivity?.(at, totalChars),
|
|
1549
|
+
logger
|
|
1550
|
+
);
|
|
1551
|
+
logger.debug("Got entity extraction response", {
|
|
1552
|
+
at,
|
|
1553
|
+
totalChars,
|
|
1554
|
+
chunkSizeTokens: size,
|
|
1555
|
+
pieceChars: piece.length,
|
|
1556
|
+
items: response.items.length
|
|
1557
|
+
});
|
|
1558
|
+
assertNotTruncated(response, "Entity extraction", at, totalChars, outputBudget);
|
|
1559
|
+
const counted = verifyYield ? await assertYieldNotCollapsed(client, piece, response.items, entityTypesDescription, logger) : void 0;
|
|
1560
|
+
return {
|
|
1561
|
+
items: response.items,
|
|
1562
|
+
...response.usage ? { usage: response.usage } : {},
|
|
1563
|
+
...counted !== void 0 ? { counted } : {}
|
|
1564
|
+
};
|
|
1565
|
+
},
|
|
1566
|
+
logger,
|
|
1567
|
+
onUnderReport,
|
|
1568
|
+
onCounted
|
|
1569
|
+
);
|
|
1456
1570
|
const fromChunk = [];
|
|
1457
1571
|
for (const e of items) {
|
|
1458
1572
|
if (isObject(e) && isString(e.exact) && isString(e.entityType)) {
|
|
@@ -1467,11 +1581,10 @@ Example output:
|
|
|
1467
1581
|
}
|
|
1468
1582
|
}
|
|
1469
1583
|
collected.push(...fromChunk);
|
|
1470
|
-
await onChunkResults?.(fromChunk);
|
|
1471
|
-
if (
|
|
1472
|
-
|
|
1473
|
-
|
|
1474
|
-
}
|
|
1584
|
+
await onChunkResults?.(fromChunk, { next, size });
|
|
1585
|
+
if (next < totalChars) onActivity?.(next, totalChars);
|
|
1586
|
+
return outcome;
|
|
1587
|
+
}, resume);
|
|
1475
1588
|
return collected;
|
|
1476
1589
|
}
|
|
1477
1590
|
function getLanguageName(locale) {
|
|
@@ -1836,13 +1949,14 @@ function makeSpanDeduper() {
|
|
|
1836
1949
|
return out;
|
|
1837
1950
|
};
|
|
1838
1951
|
}
|
|
1839
|
-
async function processHighlightJob(content, inferenceClient, params, buildAnnotation, onProgress, onChunkComplete) {
|
|
1952
|
+
async function processHighlightJob(content, inferenceClient, params, buildAnnotation, onProgress, onChunkComplete, resumeCursors) {
|
|
1840
1953
|
const echo = detectionEcho(params);
|
|
1841
1954
|
onProgress(10, { code: "loading" }, echo);
|
|
1842
1955
|
onProgress(30, { code: "analyzing" }, echo);
|
|
1843
1956
|
const dedupe = makeSpanDeduper();
|
|
1844
|
-
|
|
1845
|
-
let
|
|
1957
|
+
const prior = resumeCursors?.["highlighting"];
|
|
1958
|
+
let found = prior?.found ?? 0;
|
|
1959
|
+
let created = prior?.emitted ?? 0;
|
|
1846
1960
|
await AnnotationDetection.detectHighlights(
|
|
1847
1961
|
content,
|
|
1848
1962
|
inferenceClient,
|
|
@@ -1850,13 +1964,14 @@ async function processHighlightJob(content, inferenceClient, params, buildAnnota
|
|
|
1850
1964
|
params.density,
|
|
1851
1965
|
params.sourceLanguage,
|
|
1852
1966
|
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
1853
|
-
(
|
|
1854
|
-
|
|
1967
|
+
(consumedChars, totalChars) => onProgress(30 + Math.round(consumedChars / totalChars * 30), { code: "analyzing" }, echo),
|
|
1968
|
+
resumeCursors?.["highlighting"],
|
|
1969
|
+
async (matches, cursor) => {
|
|
1855
1970
|
found += matches.length;
|
|
1856
1971
|
const fresh = dedupe(matches.map((h) => buildAnnotation("highlighting", h)));
|
|
1857
1972
|
created += fresh.length;
|
|
1858
1973
|
onProgress(60, { code: "creating-annotations", count: created }, echo);
|
|
1859
|
-
await onChunkComplete(fresh);
|
|
1974
|
+
await onChunkComplete(fresh, { unit: "highlighting", cursor: { ...cursor, found, emitted: created } });
|
|
1860
1975
|
}
|
|
1861
1976
|
);
|
|
1862
1977
|
onProgress(100, { code: "complete-created", count: created, kind: "highlight" }, echo);
|
|
@@ -1871,14 +1986,15 @@ function detectionEcho(p) {
|
|
|
1871
1986
|
if (p.density !== void 0) requestParams.push({ label: "density", value: String(p.density) });
|
|
1872
1987
|
return requestParams.length > 0 ? { requestParams } : {};
|
|
1873
1988
|
}
|
|
1874
|
-
async function processCommentJob(content, inferenceClient, params, buildAnnotation, onProgress, onChunkComplete) {
|
|
1989
|
+
async function processCommentJob(content, inferenceClient, params, buildAnnotation, onProgress, onChunkComplete, resumeCursors) {
|
|
1875
1990
|
const echo = detectionEcho(params);
|
|
1876
1991
|
onProgress(10, { code: "loading" }, echo);
|
|
1877
1992
|
onProgress(30, { code: "analyzing" }, echo);
|
|
1878
1993
|
const bodyLanguage = params.language ?? "en";
|
|
1879
1994
|
const dedupe = makeSpanDeduper();
|
|
1880
|
-
|
|
1881
|
-
let
|
|
1995
|
+
const prior = resumeCursors?.["commenting"];
|
|
1996
|
+
let found = prior?.found ?? 0;
|
|
1997
|
+
let created = prior?.emitted ?? 0;
|
|
1882
1998
|
await AnnotationDetection.detectComments(
|
|
1883
1999
|
content,
|
|
1884
2000
|
inferenceClient,
|
|
@@ -1888,8 +2004,9 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
1888
2004
|
params.language,
|
|
1889
2005
|
params.sourceLanguage,
|
|
1890
2006
|
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
1891
|
-
(
|
|
1892
|
-
|
|
2007
|
+
(consumedChars, totalChars) => onProgress(30 + Math.round(consumedChars / totalChars * 30), { code: "analyzing" }, echo),
|
|
2008
|
+
resumeCursors?.["commenting"],
|
|
2009
|
+
async (comments, cursor) => {
|
|
1893
2010
|
found += comments.length;
|
|
1894
2011
|
const fresh = dedupe(comments.map(
|
|
1895
2012
|
(c) => (
|
|
@@ -1903,7 +2020,7 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
1903
2020
|
));
|
|
1904
2021
|
created += fresh.length;
|
|
1905
2022
|
onProgress(60, { code: "creating-annotations", count: created }, echo);
|
|
1906
|
-
await onChunkComplete(fresh);
|
|
2023
|
+
await onChunkComplete(fresh, { unit: "commenting", cursor: { ...cursor, found, emitted: created } });
|
|
1907
2024
|
}
|
|
1908
2025
|
);
|
|
1909
2026
|
onProgress(100, { code: "complete-created", count: created, kind: "comment" }, echo);
|
|
@@ -1911,14 +2028,15 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
1911
2028
|
result: { kind: "comment-annotation", commentsFound: found, commentsCreated: created }
|
|
1912
2029
|
};
|
|
1913
2030
|
}
|
|
1914
|
-
async function processAssessmentJob(content, inferenceClient, params, buildAnnotation, onProgress, onChunkComplete) {
|
|
2031
|
+
async function processAssessmentJob(content, inferenceClient, params, buildAnnotation, onProgress, onChunkComplete, resumeCursors) {
|
|
1915
2032
|
const echo = detectionEcho(params);
|
|
1916
2033
|
onProgress(10, { code: "loading" }, echo);
|
|
1917
2034
|
onProgress(30, { code: "analyzing" }, echo);
|
|
1918
2035
|
const bodyLanguage = params.language ?? "en";
|
|
1919
2036
|
const dedupe = makeSpanDeduper();
|
|
1920
|
-
|
|
1921
|
-
let
|
|
2037
|
+
const prior = resumeCursors?.["assessing"];
|
|
2038
|
+
let found = prior?.found ?? 0;
|
|
2039
|
+
let created = prior?.emitted ?? 0;
|
|
1922
2040
|
await AnnotationDetection.detectAssessments(
|
|
1923
2041
|
content,
|
|
1924
2042
|
inferenceClient,
|
|
@@ -1928,8 +2046,9 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
1928
2046
|
params.language,
|
|
1929
2047
|
params.sourceLanguage,
|
|
1930
2048
|
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
1931
|
-
(
|
|
1932
|
-
|
|
2049
|
+
(consumedChars, totalChars) => onProgress(30 + Math.round(consumedChars / totalChars * 30), { code: "analyzing" }, echo),
|
|
2050
|
+
resumeCursors?.["assessing"],
|
|
2051
|
+
async (assessments, cursor) => {
|
|
1933
2052
|
found += assessments.length;
|
|
1934
2053
|
const fresh = dedupe(assessments.map(
|
|
1935
2054
|
(a) => (
|
|
@@ -1950,7 +2069,7 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
1950
2069
|
));
|
|
1951
2070
|
created += fresh.length;
|
|
1952
2071
|
onProgress(60, { code: "creating-annotations", count: created }, echo);
|
|
1953
|
-
await onChunkComplete(fresh);
|
|
2072
|
+
await onChunkComplete(fresh, { unit: "assessing", cursor: { ...cursor, found, emitted: created } });
|
|
1954
2073
|
}
|
|
1955
2074
|
);
|
|
1956
2075
|
onProgress(100, { code: "complete-created", count: created, kind: "assessment" }, echo);
|
|
@@ -1958,12 +2077,12 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
1958
2077
|
result: { kind: "assessment-annotation", assessmentsFound: found, assessmentsCreated: created }
|
|
1959
2078
|
};
|
|
1960
2079
|
}
|
|
1961
|
-
async function processReferenceJob(content, inferenceClient, params, buildAnnotation, onProgress, logger, onUnitComplete, signal, onChunkComplete) {
|
|
2080
|
+
async function processReferenceJob(content, inferenceClient, params, buildAnnotation, onProgress, logger, onUnitComplete, signal, onChunkComplete, resumeCursors) {
|
|
1962
2081
|
const entityTypeNames = params.entityTypes.map(String);
|
|
1963
2082
|
const requestParams = [{ label: "entity-types", value: entityTypeNames.join(", ") }];
|
|
1964
2083
|
const completedItems = [];
|
|
1965
|
-
let totalFound = 0;
|
|
1966
|
-
let totalEmitted = 0;
|
|
2084
|
+
let totalFound = Object.values(resumeCursors ?? {}).reduce((n, c) => n + c.found, 0);
|
|
2085
|
+
let totalEmitted = Object.values(resumeCursors ?? {}).reduce((n, c) => n + c.emitted, 0);
|
|
1967
2086
|
let errors = 0;
|
|
1968
2087
|
let totalUnderReportedPieces = 0;
|
|
1969
2088
|
let totalExpected = 0;
|
|
@@ -1992,8 +2111,9 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
1992
2111
|
{ type: "TextualBody", value: entityTypeName, purpose: "tagging", format: "text/plain", language: bodyLanguage }
|
|
1993
2112
|
];
|
|
1994
2113
|
const dedupe = makeSpanDeduper();
|
|
1995
|
-
|
|
1996
|
-
let
|
|
2114
|
+
const priorUnit = resumeCursors?.[entityTypeName];
|
|
2115
|
+
let unitFound = priorUnit?.found ?? 0;
|
|
2116
|
+
let unitPersisted = priorUnit?.emitted ?? 0;
|
|
1997
2117
|
let underReported;
|
|
1998
2118
|
await extractEntities(
|
|
1999
2119
|
content,
|
|
@@ -2019,7 +2139,8 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
2019
2139
|
totalExpected += counted;
|
|
2020
2140
|
emitTypeProgress(entityTypeName);
|
|
2021
2141
|
},
|
|
2022
|
-
|
|
2142
|
+
resumeCursors?.[entityTypeName],
|
|
2143
|
+
async (chunkEntities, cursor) => {
|
|
2023
2144
|
const built = [];
|
|
2024
2145
|
for (const entity of chunkEntities) {
|
|
2025
2146
|
const reconciled = reconcileSelector(content, {
|
|
@@ -2039,9 +2160,14 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
2039
2160
|
built.push(buildAnnotation("linking", toMatch(reconciled), unresolvedBody));
|
|
2040
2161
|
}
|
|
2041
2162
|
const fresh = dedupe(built);
|
|
2042
|
-
|
|
2043
|
-
|
|
2044
|
-
|
|
2163
|
+
const nextFound = unitFound + chunkEntities.length;
|
|
2164
|
+
const nextEmitted = unitPersisted + fresh.length;
|
|
2165
|
+
await onChunkComplete?.(fresh, {
|
|
2166
|
+
unit: entityTypeName,
|
|
2167
|
+
cursor: { ...cursor, found: nextFound, emitted: nextEmitted }
|
|
2168
|
+
});
|
|
2169
|
+
unitFound = nextFound;
|
|
2170
|
+
unitPersisted = nextEmitted;
|
|
2045
2171
|
totalFound += chunkEntities.length;
|
|
2046
2172
|
totalEmitted += fresh.length;
|
|
2047
2173
|
emitTypeProgress(entityTypeName);
|
|
@@ -2073,13 +2199,13 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
2073
2199
|
}
|
|
2074
2200
|
};
|
|
2075
2201
|
}
|
|
2076
|
-
async function processTagJob(content, inferenceClient, params, buildAnnotation, onProgress, onChunkComplete) {
|
|
2202
|
+
async function processTagJob(content, inferenceClient, params, buildAnnotation, onProgress, onChunkComplete, resumeCursors) {
|
|
2077
2203
|
onProgress(10, { code: "loading" });
|
|
2078
2204
|
onProgress(30, { code: "analyzing-tags" });
|
|
2079
2205
|
const bodyLanguage = params.language ?? "en";
|
|
2080
2206
|
const dedupe = makeSpanDeduper();
|
|
2081
2207
|
let found = 0;
|
|
2082
|
-
let created = 0;
|
|
2208
|
+
let created = Object.values(resumeCursors ?? {}).reduce((n, c) => n + c.emitted, 0);
|
|
2083
2209
|
const byCategory = {};
|
|
2084
2210
|
const completedItems = [];
|
|
2085
2211
|
for (let c = 0; c < params.categories.length; c++) {
|
|
@@ -2095,7 +2221,9 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
2095
2221
|
{ code: "analyzing-tags" },
|
|
2096
2222
|
position()
|
|
2097
2223
|
);
|
|
2098
|
-
|
|
2224
|
+
const priorCategory = resumeCursors?.[category];
|
|
2225
|
+
let categoryFound = priorCategory?.found ?? 0;
|
|
2226
|
+
let categoryCreated = priorCategory?.emitted ?? 0;
|
|
2099
2227
|
await AnnotationDetection.detectTags(
|
|
2100
2228
|
content,
|
|
2101
2229
|
inferenceClient,
|
|
@@ -2104,12 +2232,13 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
2104
2232
|
params.sourceLanguage,
|
|
2105
2233
|
// Liveness (chunk boundaries + in-flight heartbeat): this category's
|
|
2106
2234
|
// slice of the 30–60 band.
|
|
2107
|
-
(
|
|
2108
|
-
30 + Math.round((c +
|
|
2235
|
+
(consumedChars, totalChars) => onProgress(
|
|
2236
|
+
30 + Math.round((c + consumedChars / totalChars) / params.categories.length * 30),
|
|
2109
2237
|
{ code: "analyzing-tags" },
|
|
2110
2238
|
position()
|
|
2111
2239
|
),
|
|
2112
|
-
|
|
2240
|
+
resumeCursors?.[category],
|
|
2241
|
+
async (matches, cursor) => {
|
|
2113
2242
|
categoryFound += matches.length;
|
|
2114
2243
|
const fresh = dedupe(matches.map((t) => {
|
|
2115
2244
|
const cat = t.category ?? "unknown";
|
|
@@ -2125,7 +2254,11 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
2125
2254
|
byCategory[cat] = (byCategory[cat] ?? 0) + 1;
|
|
2126
2255
|
}
|
|
2127
2256
|
onProgress(60, { code: "creating-tag-annotations", count: created });
|
|
2128
|
-
|
|
2257
|
+
categoryCreated += fresh.length;
|
|
2258
|
+
await onChunkComplete(fresh, {
|
|
2259
|
+
unit: category,
|
|
2260
|
+
cursor: { ...cursor, found: categoryFound, emitted: categoryCreated }
|
|
2261
|
+
});
|
|
2129
2262
|
}
|
|
2130
2263
|
);
|
|
2131
2264
|
found += categoryFound;
|