@semiont/jobs 0.5.26 → 0.5.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -12
- package/dist/index.d.ts +26 -72
- package/dist/index.js +129 -75
- package/dist/index.js.map +1 -1
- package/dist/worker-main.js +159 -92
- package/dist/worker-main.js.map +1 -1
- package/package.json +8 -8
package/dist/worker-main.js
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText, isObject, isString, deriveViews } from '@semiont/core';
|
|
1
|
+
import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, isGenerationJobParams, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText, isObject, isString, deriveViews } from '@semiont/core';
|
|
2
2
|
import { anchoredTextStoreOverTransport, deriveStorageUri, extractPdfTextLayer, EXTRACTORS, calculateChecksum, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
|
|
3
|
+
import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
|
|
3
4
|
import { execFileSync } from 'child_process';
|
|
4
5
|
import { existsSync, readFileSync, mkdtempSync, writeFileSync, rmSync } from 'fs';
|
|
5
6
|
import { homedir, hostname, tmpdir } from 'os';
|
|
6
7
|
import { join } from 'path';
|
|
7
8
|
import { generateAnnotationId } from '@semiont/event-sourcing';
|
|
8
|
-
import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
|
|
9
9
|
import { InMemorySessionStorage, setStoredSession, kbBackendUrl, SemiontClient, SemiontSession } from '@semiont/sdk';
|
|
10
10
|
import { HttpTransport, HttpContentTransport } from '@semiont/http-transport';
|
|
11
11
|
import { createInferenceClient } from '@semiont/inference';
|
|
@@ -9348,10 +9348,18 @@ function asJobParams(params) {
|
|
|
9348
9348
|
}
|
|
9349
9349
|
return params;
|
|
9350
9350
|
}
|
|
9351
|
-
|
|
9352
|
-
// src/workers/inference-call.ts
|
|
9353
9351
|
var INFERENCE_TIMEOUT_MS = 10 * 6e4;
|
|
9354
|
-
|
|
9352
|
+
var INFERENCE_HEARTBEAT_MS = 15e3;
|
|
9353
|
+
function spanned(client, kind, maxTokens, work) {
|
|
9354
|
+
return withSpan(`inference:${kind}`, work, {
|
|
9355
|
+
attrs: {
|
|
9356
|
+
"inference.provider": client.type,
|
|
9357
|
+
"inference.model": client.modelId,
|
|
9358
|
+
"inference.max_tokens": maxTokens
|
|
9359
|
+
}
|
|
9360
|
+
});
|
|
9361
|
+
}
|
|
9362
|
+
async function withTimeout(work, label, onHeartbeat) {
|
|
9355
9363
|
let timer;
|
|
9356
9364
|
const timedOut = new Promise((_, reject) => {
|
|
9357
9365
|
timer = setTimeout(() => {
|
|
@@ -9361,6 +9369,16 @@ async function withTimeout(work, label) {
|
|
|
9361
9369
|
}, INFERENCE_TIMEOUT_MS);
|
|
9362
9370
|
timer.unref?.();
|
|
9363
9371
|
});
|
|
9372
|
+
let heartbeat;
|
|
9373
|
+
if (onHeartbeat) {
|
|
9374
|
+
heartbeat = setInterval(() => {
|
|
9375
|
+
try {
|
|
9376
|
+
onHeartbeat();
|
|
9377
|
+
} catch {
|
|
9378
|
+
}
|
|
9379
|
+
}, INFERENCE_HEARTBEAT_MS);
|
|
9380
|
+
heartbeat.unref?.();
|
|
9381
|
+
}
|
|
9364
9382
|
try {
|
|
9365
9383
|
return await Promise.race([work, timedOut]);
|
|
9366
9384
|
} catch (err) {
|
|
@@ -9369,19 +9387,22 @@ async function withTimeout(work, label) {
|
|
|
9369
9387
|
throw err;
|
|
9370
9388
|
} finally {
|
|
9371
9389
|
clearTimeout(timer);
|
|
9390
|
+
if (heartbeat) clearInterval(heartbeat);
|
|
9372
9391
|
}
|
|
9373
9392
|
}
|
|
9374
|
-
function boundedGenerate(client, prompt, maxTokens, temperature) {
|
|
9375
|
-
return withTimeout(
|
|
9393
|
+
function boundedGenerate(client, prompt, maxTokens, temperature, onHeartbeat) {
|
|
9394
|
+
return spanned(client, "text", maxTokens, () => withTimeout(
|
|
9376
9395
|
client.generateText(prompt, maxTokens, temperature),
|
|
9377
|
-
`${client.type}:${client.modelId}
|
|
9378
|
-
|
|
9396
|
+
`${client.type}:${client.modelId}`,
|
|
9397
|
+
onHeartbeat
|
|
9398
|
+
));
|
|
9379
9399
|
}
|
|
9380
|
-
function boundedGenerateStructured(client, prompt, maxTokens, temperature, elementSchema) {
|
|
9381
|
-
return withTimeout(
|
|
9400
|
+
function boundedGenerateStructured(client, prompt, maxTokens, temperature, elementSchema, onHeartbeat) {
|
|
9401
|
+
return spanned(client, "structured", maxTokens, () => withTimeout(
|
|
9382
9402
|
client.generateStructured(prompt, maxTokens, temperature, elementSchema),
|
|
9383
|
-
`${client.type}:${client.modelId}
|
|
9384
|
-
|
|
9403
|
+
`${client.type}:${client.modelId}`,
|
|
9404
|
+
onHeartbeat
|
|
9405
|
+
));
|
|
9385
9406
|
}
|
|
9386
9407
|
|
|
9387
9408
|
// src/workers/detection/detection-chunking.ts
|
|
@@ -9912,7 +9933,7 @@ function assertNotTruncated(response, motivation, chunk, totalChunks, outputBudg
|
|
|
9912
9933
|
throw new Error(`${motivation} detection response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
|
|
9913
9934
|
}
|
|
9914
9935
|
}
|
|
9915
|
-
async function detectInChunks(client, content, buildPrompt, temperature, motivation, elementSchema, parse,
|
|
9936
|
+
async function detectInChunks(client, content, buildPrompt, temperature, motivation, elementSchema, parse, onActivity) {
|
|
9916
9937
|
const limits = await client.limits();
|
|
9917
9938
|
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
9918
9939
|
const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
|
|
@@ -9924,12 +9945,14 @@ async function detectInChunks(client, content, buildPrompt, temperature, motivat
|
|
|
9924
9945
|
buildPrompt(chunks[i]),
|
|
9925
9946
|
outputBudget,
|
|
9926
9947
|
temperature,
|
|
9927
|
-
elementSchema
|
|
9948
|
+
elementSchema,
|
|
9949
|
+
// Still alive, same position (a long single call is otherwise silent).
|
|
9950
|
+
() => onActivity?.(i, chunks.length)
|
|
9928
9951
|
);
|
|
9929
9952
|
assertNotTruncated(response, motivation, i + 1, chunks.length, outputBudget);
|
|
9930
9953
|
collected.push(...parse(response.items));
|
|
9931
9954
|
if (i < chunks.length - 1) {
|
|
9932
|
-
|
|
9955
|
+
onActivity?.(i + 1, chunks.length);
|
|
9933
9956
|
}
|
|
9934
9957
|
}
|
|
9935
9958
|
return collected;
|
|
@@ -9943,7 +9966,7 @@ var AnnotationDetection = class {
|
|
|
9943
9966
|
* (source-resource locale). See `types.ts` "Locale conventions" for the
|
|
9944
9967
|
* full discussion.
|
|
9945
9968
|
*/
|
|
9946
|
-
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage,
|
|
9969
|
+
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onActivity) {
|
|
9947
9970
|
return detectInChunks(
|
|
9948
9971
|
client,
|
|
9949
9972
|
content,
|
|
@@ -9952,7 +9975,7 @@ var AnnotationDetection = class {
|
|
|
9952
9975
|
"comment",
|
|
9953
9976
|
COMMENT_ELEMENT_SCHEMA,
|
|
9954
9977
|
(items) => MotivationParsers.parseComments(items, content),
|
|
9955
|
-
|
|
9978
|
+
onActivity
|
|
9956
9979
|
);
|
|
9957
9980
|
}
|
|
9958
9981
|
/**
|
|
@@ -9962,7 +9985,7 @@ var AnnotationDetection = class {
|
|
|
9962
9985
|
* applies, used in the prompt so the LLM analyzes non-English source
|
|
9963
9986
|
* correctly.
|
|
9964
9987
|
*/
|
|
9965
|
-
static async detectHighlights(content, client, instructions, density, sourceLanguage,
|
|
9988
|
+
static async detectHighlights(content, client, instructions, density, sourceLanguage, onActivity) {
|
|
9966
9989
|
return detectInChunks(
|
|
9967
9990
|
client,
|
|
9968
9991
|
content,
|
|
@@ -9971,7 +9994,7 @@ var AnnotationDetection = class {
|
|
|
9971
9994
|
"highlight",
|
|
9972
9995
|
HIGHLIGHT_ELEMENT_SCHEMA,
|
|
9973
9996
|
(items) => MotivationParsers.parseHighlights(items, content),
|
|
9974
|
-
|
|
9997
|
+
onActivity
|
|
9975
9998
|
);
|
|
9976
9999
|
}
|
|
9977
10000
|
/**
|
|
@@ -9981,7 +10004,7 @@ var AnnotationDetection = class {
|
|
|
9981
10004
|
* (annotation body locale). `sourceLanguage` is the locale of the content
|
|
9982
10005
|
* being analyzed (source-resource locale).
|
|
9983
10006
|
*/
|
|
9984
|
-
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage,
|
|
10007
|
+
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onActivity) {
|
|
9985
10008
|
return detectInChunks(
|
|
9986
10009
|
client,
|
|
9987
10010
|
content,
|
|
@@ -9990,7 +10013,7 @@ var AnnotationDetection = class {
|
|
|
9990
10013
|
"assessment",
|
|
9991
10014
|
ASSESSMENT_ELEMENT_SCHEMA,
|
|
9992
10015
|
(items) => MotivationParsers.parseAssessments(items, content),
|
|
9993
|
-
|
|
10016
|
+
onActivity
|
|
9994
10017
|
);
|
|
9995
10018
|
}
|
|
9996
10019
|
/**
|
|
@@ -10005,7 +10028,7 @@ var AnnotationDetection = class {
|
|
|
10005
10028
|
* identifiers, not LLM-generated text — so it's consumed at the body-stamp
|
|
10006
10029
|
* site, not here.
|
|
10007
10030
|
*/
|
|
10008
|
-
static async detectTags(content, client, schema, category, sourceLanguage,
|
|
10031
|
+
static async detectTags(content, client, schema, category, sourceLanguage, onActivity) {
|
|
10009
10032
|
const categoryInfo = schema.tags.find((t) => t.name === category);
|
|
10010
10033
|
if (!categoryInfo) {
|
|
10011
10034
|
throw new Error(`Invalid category "${category}" for schema ${schema.id}`);
|
|
@@ -10027,7 +10050,7 @@ var AnnotationDetection = class {
|
|
|
10027
10050
|
"tag",
|
|
10028
10051
|
TAG_ELEMENT_SCHEMA,
|
|
10029
10052
|
(items) => MotivationParsers.parseTags(items),
|
|
10030
|
-
|
|
10053
|
+
onActivity
|
|
10031
10054
|
);
|
|
10032
10055
|
return MotivationParsers.validateTagOffsets(parsedTags, content, category);
|
|
10033
10056
|
}
|
|
@@ -10043,7 +10066,7 @@ var ENTITY_ELEMENT_SCHEMA = {
|
|
|
10043
10066
|
required: ["exact", "entityType"],
|
|
10044
10067
|
additionalProperties: false
|
|
10045
10068
|
};
|
|
10046
|
-
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage,
|
|
10069
|
+
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage, onActivity) {
|
|
10047
10070
|
const entityTypesDescription = entityTypes.map((et) => {
|
|
10048
10071
|
if (typeof et === "string") {
|
|
10049
10072
|
return et;
|
|
@@ -10110,7 +10133,10 @@ Example output:
|
|
|
10110
10133
|
outputBudget,
|
|
10111
10134
|
0.3,
|
|
10112
10135
|
// Lower temperature for more consistent extraction
|
|
10113
|
-
ENTITY_ELEMENT_SCHEMA
|
|
10136
|
+
ENTITY_ELEMENT_SCHEMA,
|
|
10137
|
+
// Still alive, same position: a long single call would otherwise emit
|
|
10138
|
+
// nothing at all between start and finish.
|
|
10139
|
+
() => onActivity?.(i, chunks.length)
|
|
10114
10140
|
);
|
|
10115
10141
|
logger2.debug("Got entity extraction response", {
|
|
10116
10142
|
chunk: i + 1,
|
|
@@ -10135,7 +10161,7 @@ Example output:
|
|
|
10135
10161
|
}
|
|
10136
10162
|
}
|
|
10137
10163
|
if (i < chunks.length - 1) {
|
|
10138
|
-
|
|
10164
|
+
onActivity?.(i + 1, chunks.length);
|
|
10139
10165
|
}
|
|
10140
10166
|
}
|
|
10141
10167
|
return collected;
|
|
@@ -10571,30 +10597,39 @@ function buildPdfAnnotation(anchored, resourceId, userId, generator, motivation,
|
|
|
10571
10597
|
};
|
|
10572
10598
|
}
|
|
10573
10599
|
async function processHighlightJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
10574
|
-
|
|
10575
|
-
onProgress(
|
|
10600
|
+
const echo = detectionEcho(params);
|
|
10601
|
+
onProgress(10, { code: "loading" }, echo);
|
|
10602
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
10576
10603
|
const highlights = await AnnotationDetection.detectHighlights(
|
|
10577
10604
|
content,
|
|
10578
10605
|
inferenceClient,
|
|
10579
10606
|
params.instructions,
|
|
10580
10607
|
params.density,
|
|
10581
10608
|
params.sourceLanguage,
|
|
10582
|
-
//
|
|
10583
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
10609
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
10610
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
10584
10611
|
);
|
|
10585
|
-
onProgress(60,
|
|
10612
|
+
onProgress(60, { code: "creating-annotations", count: highlights.length }, echo);
|
|
10586
10613
|
const annotations = dedupeAnnotations(highlights.map(
|
|
10587
10614
|
(h) => buildAnnotation("highlighting", h)
|
|
10588
10615
|
));
|
|
10589
|
-
onProgress(100,
|
|
10616
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "highlight" }, echo);
|
|
10590
10617
|
return {
|
|
10591
10618
|
annotations,
|
|
10592
10619
|
result: { highlightsFound: highlights.length, highlightsCreated: annotations.length }
|
|
10593
10620
|
};
|
|
10594
10621
|
}
|
|
10622
|
+
function detectionEcho(p) {
|
|
10623
|
+
const requestParams = [];
|
|
10624
|
+
if (p.instructions?.trim()) requestParams.push({ label: "instructions", value: p.instructions.trim() });
|
|
10625
|
+
if (p.tone?.trim()) requestParams.push({ label: "tone", value: p.tone.trim() });
|
|
10626
|
+
if (p.density !== void 0) requestParams.push({ label: "density", value: String(p.density) });
|
|
10627
|
+
return requestParams.length > 0 ? { requestParams } : {};
|
|
10628
|
+
}
|
|
10595
10629
|
async function processCommentJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
10596
|
-
|
|
10597
|
-
onProgress(
|
|
10630
|
+
const echo = detectionEcho(params);
|
|
10631
|
+
onProgress(10, { code: "loading" }, echo);
|
|
10632
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
10598
10633
|
const comments = await AnnotationDetection.detectComments(
|
|
10599
10634
|
content,
|
|
10600
10635
|
inferenceClient,
|
|
@@ -10603,10 +10638,10 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
10603
10638
|
params.density,
|
|
10604
10639
|
params.language,
|
|
10605
10640
|
params.sourceLanguage,
|
|
10606
|
-
//
|
|
10607
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
10641
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
10642
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
10608
10643
|
);
|
|
10609
|
-
onProgress(60,
|
|
10644
|
+
onProgress(60, { code: "creating-annotations", count: comments.length }, echo);
|
|
10610
10645
|
const bodyLanguage = params.language ?? "en";
|
|
10611
10646
|
const annotations = dedupeAnnotations(comments.map(
|
|
10612
10647
|
(c) => (
|
|
@@ -10618,15 +10653,16 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
10618
10653
|
])
|
|
10619
10654
|
)
|
|
10620
10655
|
));
|
|
10621
|
-
onProgress(100,
|
|
10656
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "comment" }, echo);
|
|
10622
10657
|
return {
|
|
10623
10658
|
annotations,
|
|
10624
10659
|
result: { commentsFound: comments.length, commentsCreated: annotations.length }
|
|
10625
10660
|
};
|
|
10626
10661
|
}
|
|
10627
10662
|
async function processAssessmentJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
10628
|
-
|
|
10629
|
-
onProgress(
|
|
10663
|
+
const echo = detectionEcho(params);
|
|
10664
|
+
onProgress(10, { code: "loading" }, echo);
|
|
10665
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
10630
10666
|
const assessments = await AnnotationDetection.detectAssessments(
|
|
10631
10667
|
content,
|
|
10632
10668
|
inferenceClient,
|
|
@@ -10635,10 +10671,10 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
10635
10671
|
params.density,
|
|
10636
10672
|
params.language,
|
|
10637
10673
|
params.sourceLanguage,
|
|
10638
|
-
//
|
|
10639
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
10674
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
10675
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
10640
10676
|
);
|
|
10641
|
-
onProgress(60,
|
|
10677
|
+
onProgress(60, { code: "creating-annotations", count: assessments.length }, echo);
|
|
10642
10678
|
const bodyLanguage = params.language ?? "en";
|
|
10643
10679
|
const annotations = dedupeAnnotations(assessments.map(
|
|
10644
10680
|
(a) => (
|
|
@@ -10657,7 +10693,7 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
10657
10693
|
})
|
|
10658
10694
|
)
|
|
10659
10695
|
));
|
|
10660
|
-
onProgress(100,
|
|
10696
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "assessment" }, echo);
|
|
10661
10697
|
return {
|
|
10662
10698
|
annotations,
|
|
10663
10699
|
result: { assessmentsFound: assessments.length, assessmentsCreated: annotations.length }
|
|
@@ -10665,25 +10701,27 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
10665
10701
|
}
|
|
10666
10702
|
async function processReferenceJob(content, inferenceClient, params, buildAnnotation, onProgress, logger2) {
|
|
10667
10703
|
const entityTypeNames = params.entityTypes.map(String);
|
|
10668
|
-
const requestParams = [{ label: "
|
|
10669
|
-
const
|
|
10704
|
+
const requestParams = [{ label: "entity-types", value: entityTypeNames.join(", ") }];
|
|
10705
|
+
const completedItems = [];
|
|
10670
10706
|
let totalFound = 0;
|
|
10671
10707
|
let totalEmitted = 0;
|
|
10672
10708
|
let errors = 0;
|
|
10673
10709
|
const allAnnotations = [];
|
|
10674
|
-
onProgress(10,
|
|
10710
|
+
onProgress(10, { code: "loading" }, { requestParams });
|
|
10675
10711
|
const bodyLanguage = params.language ?? "en";
|
|
10676
10712
|
for (let i = 0; i < entityTypeNames.length; i++) {
|
|
10677
10713
|
const entityTypeName = entityTypeNames[i];
|
|
10678
10714
|
if (!entityTypeName) continue;
|
|
10679
10715
|
const pct = 20 + Math.round(i / entityTypeNames.length * 60);
|
|
10680
|
-
onProgress(pct,
|
|
10681
|
-
|
|
10682
|
-
|
|
10683
|
-
|
|
10716
|
+
onProgress(pct, { code: "detecting-entities", entityType: entityTypeName }, {
|
|
10717
|
+
// One vocabulary for "what is in flight" (CLEAN-PROGRESS D2): the entity
|
|
10718
|
+
// type is KB data, `kind` is the code the client localizes around it.
|
|
10719
|
+
current: { kind: "entity-type", value: entityTypeName },
|
|
10720
|
+
processed: i,
|
|
10721
|
+
total: entityTypeNames.length,
|
|
10684
10722
|
entitiesFound: totalFound,
|
|
10685
10723
|
entitiesEmitted: totalEmitted,
|
|
10686
|
-
|
|
10724
|
+
completedItems: [...completedItems],
|
|
10687
10725
|
requestParams
|
|
10688
10726
|
});
|
|
10689
10727
|
const extractedEntities = await extractEntities(
|
|
@@ -10693,25 +10731,28 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
10693
10731
|
params.includeDescriptiveReferences ?? false,
|
|
10694
10732
|
logger2,
|
|
10695
10733
|
params.sourceLanguage,
|
|
10696
|
-
//
|
|
10697
|
-
//
|
|
10698
|
-
//
|
|
10699
|
-
//
|
|
10734
|
+
// Liveness: fires at chunk boundaries AND every ~15 s while a single
|
|
10735
|
+
// inference call is in flight (DETECTION-HEARTBEAT). Progress feeds the
|
|
10736
|
+
// stall watchdog, the janitor, AND the client's inter-emission timeout,
|
|
10737
|
+
// so a long single-chunk call must not be silent. Percentage
|
|
10738
|
+
// interpolates within this entity type's band of the 20–80 range; a
|
|
10739
|
+
// heartbeat repeats the current position rather than inventing an
|
|
10740
|
+
// advance.
|
|
10700
10741
|
(completed, total) => {
|
|
10701
10742
|
const interpolated = 20 + Math.round((i + completed / total) / entityTypeNames.length * 60);
|
|
10702
|
-
onProgress(interpolated,
|
|
10703
|
-
|
|
10704
|
-
|
|
10705
|
-
|
|
10743
|
+
onProgress(interpolated, { code: "detecting-entities", entityType: entityTypeName }, {
|
|
10744
|
+
current: { kind: "entity-type", value: entityTypeName },
|
|
10745
|
+
processed: i,
|
|
10746
|
+
total: entityTypeNames.length,
|
|
10706
10747
|
entitiesFound: totalFound,
|
|
10707
10748
|
entitiesEmitted: totalEmitted,
|
|
10708
|
-
|
|
10749
|
+
completedItems: [...completedItems],
|
|
10709
10750
|
requestParams
|
|
10710
10751
|
});
|
|
10711
10752
|
}
|
|
10712
10753
|
);
|
|
10713
10754
|
totalFound += extractedEntities.length;
|
|
10714
|
-
|
|
10755
|
+
completedItems.push({ value: entityTypeName, foundCount: extractedEntities.length });
|
|
10715
10756
|
const unresolvedBody = [
|
|
10716
10757
|
{ type: "TextualBody", value: entityTypeName, purpose: "tagging", format: "text/plain", language: bodyLanguage }
|
|
10717
10758
|
];
|
|
@@ -10742,36 +10783,49 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
10742
10783
|
}
|
|
10743
10784
|
}
|
|
10744
10785
|
const annotations = dedupeAnnotations(allAnnotations);
|
|
10745
|
-
onProgress(100,
|
|
10786
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "reference" }, { requestParams });
|
|
10746
10787
|
return {
|
|
10747
10788
|
annotations,
|
|
10748
10789
|
result: { totalFound, totalEmitted: annotations.length, errors }
|
|
10749
10790
|
};
|
|
10750
10791
|
}
|
|
10751
10792
|
async function processTagJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
10752
|
-
onProgress(10,
|
|
10753
|
-
onProgress(30,
|
|
10793
|
+
onProgress(10, { code: "loading" });
|
|
10794
|
+
onProgress(30, { code: "analyzing-tags" });
|
|
10754
10795
|
const allTags = [];
|
|
10796
|
+
const completedItems = [];
|
|
10755
10797
|
for (let c = 0; c < params.categories.length; c++) {
|
|
10756
10798
|
const category = params.categories[c];
|
|
10799
|
+
const position = () => ({
|
|
10800
|
+
current: { kind: "category", value: category },
|
|
10801
|
+
processed: c,
|
|
10802
|
+
total: params.categories.length,
|
|
10803
|
+
completedItems: [...completedItems]
|
|
10804
|
+
});
|
|
10805
|
+
onProgress(
|
|
10806
|
+
30 + Math.round(c / params.categories.length * 30),
|
|
10807
|
+
{ code: "analyzing-tags" },
|
|
10808
|
+
position()
|
|
10809
|
+
);
|
|
10757
10810
|
const categoryTags = await AnnotationDetection.detectTags(
|
|
10758
10811
|
content,
|
|
10759
10812
|
inferenceClient,
|
|
10760
10813
|
params.schema,
|
|
10761
10814
|
category,
|
|
10762
10815
|
params.sourceLanguage,
|
|
10763
|
-
//
|
|
10764
|
-
//
|
|
10816
|
+
// Liveness (chunk boundaries + in-flight heartbeat): this category's
|
|
10817
|
+
// slice of the 30–60 band.
|
|
10765
10818
|
(completed, total) => onProgress(
|
|
10766
10819
|
30 + Math.round((c + completed / total) / params.categories.length * 30),
|
|
10767
|
-
|
|
10768
|
-
|
|
10820
|
+
{ code: "analyzing-tags" },
|
|
10821
|
+
position()
|
|
10769
10822
|
)
|
|
10770
10823
|
);
|
|
10824
|
+
completedItems.push({ value: category, foundCount: categoryTags.length });
|
|
10771
10825
|
allTags.push(...categoryTags);
|
|
10772
10826
|
}
|
|
10773
10827
|
const tags = allTags;
|
|
10774
|
-
onProgress(60,
|
|
10828
|
+
onProgress(60, { code: "creating-tag-annotations", count: tags.length });
|
|
10775
10829
|
const bodyLanguage = params.language ?? "en";
|
|
10776
10830
|
const annotations = dedupeAnnotations(tags.map((t) => {
|
|
10777
10831
|
const category = t.category ?? "unknown";
|
|
@@ -10786,7 +10840,7 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
10786
10840
|
const category = Array.isArray(body) && typeof body[0]?.value === "string" ? body[0].value : "unknown";
|
|
10787
10841
|
byCategory[category] = (byCategory[category] ?? 0) + 1;
|
|
10788
10842
|
}
|
|
10789
|
-
onProgress(100,
|
|
10843
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "tag" });
|
|
10790
10844
|
return {
|
|
10791
10845
|
annotations,
|
|
10792
10846
|
result: { tagsFound: tags.length, tagsCreated: annotations.length, byCategory }
|
|
@@ -10809,7 +10863,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10809
10863
|
const title = params.title ?? "Untitled";
|
|
10810
10864
|
const entityTypes = (params.entityTypes ?? []).map(String);
|
|
10811
10865
|
if (outputMediaType === "application/pdf") {
|
|
10812
|
-
onProgress(5,
|
|
10866
|
+
onProgress(5, { code: "generating-resource" });
|
|
10813
10867
|
const validIds = params.cite === true ? collectContextResourceIds(params.context) : null;
|
|
10814
10868
|
let generated2 = await generateResourceFromTopic(
|
|
10815
10869
|
title,
|
|
@@ -10874,7 +10928,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10874
10928
|
);
|
|
10875
10929
|
}
|
|
10876
10930
|
assertWithinOutputBudget(compiled.pdf.byteLength);
|
|
10877
|
-
onProgress(95,
|
|
10931
|
+
onProgress(95, { code: "creating-resource" });
|
|
10878
10932
|
return {
|
|
10879
10933
|
content: compiled.pdf,
|
|
10880
10934
|
title: generated2.title ?? title,
|
|
@@ -10886,7 +10940,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10886
10940
|
}
|
|
10887
10941
|
};
|
|
10888
10942
|
}
|
|
10889
|
-
onProgress(5,
|
|
10943
|
+
onProgress(5, { code: "generating-resource" });
|
|
10890
10944
|
const generated = await generateResourceFromTopic(
|
|
10891
10945
|
title,
|
|
10892
10946
|
entityTypes,
|
|
@@ -10910,7 +10964,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10910
10964
|
content = resolved.content;
|
|
10911
10965
|
citations = resolved.citations;
|
|
10912
10966
|
}
|
|
10913
|
-
onProgress(95,
|
|
10967
|
+
onProgress(95, { code: "creating-resource" });
|
|
10914
10968
|
const artifact = new TextEncoder().encode(content);
|
|
10915
10969
|
assertWithinOutputBudget(artifact.byteLength);
|
|
10916
10970
|
return {
|
|
@@ -10950,13 +11004,18 @@ async function prepareDetection(mediaType, session, resourceId, userId, generato
|
|
|
10950
11004
|
buildAnnotation: (motivation, match, body) => buildTextAnnotation(extracted.text, resourceId, userId, generator, motivation, match, body)
|
|
10951
11005
|
};
|
|
10952
11006
|
}
|
|
10953
|
-
|
|
10954
|
-
|
|
10955
|
-
|
|
10956
|
-
|
|
10957
|
-
|
|
10958
|
-
|
|
10959
|
-
}
|
|
11007
|
+
function referenceIdOf(job) {
|
|
11008
|
+
if (job.type === "generation") {
|
|
11009
|
+
const context = job.params.context;
|
|
11010
|
+
const focus = context?.focus;
|
|
11011
|
+
if (focus?.kind === "annotation" && typeof focus.annotation?.id === "string") {
|
|
11012
|
+
return focus.annotation.id;
|
|
11013
|
+
}
|
|
11014
|
+
return void 0;
|
|
11015
|
+
}
|
|
11016
|
+
const ref = job.params.referenceId;
|
|
11017
|
+
return typeof ref === "string" ? ref : void 0;
|
|
11018
|
+
}
|
|
10960
11019
|
async function emitEvent(session, channel, payload) {
|
|
10961
11020
|
await session.client.transport.emit(channel, payload);
|
|
10962
11021
|
}
|
|
@@ -10973,7 +11032,7 @@ function startWorkerProcess(config) {
|
|
|
10973
11032
|
handleJob(adapter, config, job).catch((error) => {
|
|
10974
11033
|
const message = error instanceof Error ? error.message : String(error);
|
|
10975
11034
|
logger2.error("Job failed", { jobId: job.jobId, error: message, stack: error instanceof Error ? error.stack : void 0 });
|
|
10976
|
-
const failAnnotationId = job
|
|
11035
|
+
const failAnnotationId = referenceIdOf(job);
|
|
10977
11036
|
if (isJobType(job.type)) {
|
|
10978
11037
|
emitEvent(session, "job:fail", {
|
|
10979
11038
|
resourceId: job.resourceId,
|
|
@@ -11022,7 +11081,7 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11022
11081
|
}
|
|
11023
11082
|
const jobType = job.type;
|
|
11024
11083
|
const resourceId$1 = resourceId(job.resourceId);
|
|
11025
|
-
const annotationId = job
|
|
11084
|
+
const annotationId = referenceIdOf(job);
|
|
11026
11085
|
const lifecycleBase = {
|
|
11027
11086
|
resourceId: resourceId$1,
|
|
11028
11087
|
jobId,
|
|
@@ -11038,7 +11097,11 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11038
11097
|
if (jobType !== "generation") {
|
|
11039
11098
|
const descriptor = await session.client.browse.resource(resourceId$1).fresh();
|
|
11040
11099
|
const mediaType = getPrimaryMediaType(descriptor);
|
|
11041
|
-
const source = await
|
|
11100
|
+
const source = await withSpan(
|
|
11101
|
+
"detection:prepare",
|
|
11102
|
+
() => prepareDetection(mediaType ?? "", session, resourceId$1, userId, generator, config.anchoredTextStore),
|
|
11103
|
+
{ attrs: { "resource.id": resourceId$1, "media.type": mediaType ?? "unknown" } }
|
|
11104
|
+
);
|
|
11042
11105
|
if ("declined" in source) {
|
|
11043
11106
|
if (source.declined === "no-extractor") {
|
|
11044
11107
|
throw new Error(`Cannot run ${jobType} on resource ${resourceId$1}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
|
|
@@ -11047,8 +11110,7 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11047
11110
|
...lifecycleBase,
|
|
11048
11111
|
result: {
|
|
11049
11112
|
declined: true,
|
|
11050
|
-
reason: source.declined
|
|
11051
|
-
message: DECLINE_MESSAGES[source.declined]
|
|
11113
|
+
reason: source.declined
|
|
11052
11114
|
}
|
|
11053
11115
|
});
|
|
11054
11116
|
adapter.completeJob();
|
|
@@ -11056,13 +11118,12 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11056
11118
|
}
|
|
11057
11119
|
ready = source;
|
|
11058
11120
|
}
|
|
11059
|
-
const onProgress = (percentage, message,
|
|
11121
|
+
const onProgress = (percentage, message, extra) => {
|
|
11060
11122
|
adapter.touchActivity();
|
|
11061
11123
|
emitEvent(session, "job:report-progress", {
|
|
11062
11124
|
...lifecycleBase,
|
|
11063
11125
|
percentage,
|
|
11064
11126
|
progress: {
|
|
11065
|
-
stage,
|
|
11066
11127
|
percentage,
|
|
11067
11128
|
message,
|
|
11068
11129
|
...annotationId ? { annotationId } : {},
|
|
@@ -11153,6 +11214,11 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11153
11214
|
});
|
|
11154
11215
|
adapter.completeJob();
|
|
11155
11216
|
} else if (jobType === "generation") {
|
|
11217
|
+
if (!isGenerationJobParams(job.params)) {
|
|
11218
|
+
throw new Error(
|
|
11219
|
+
`generation job ${job.jobId}: params do not satisfy GenerationJobParams (title, storageUri, and context are required)`
|
|
11220
|
+
);
|
|
11221
|
+
}
|
|
11156
11222
|
const genResult = await processGenerationJob(
|
|
11157
11223
|
inferenceClient,
|
|
11158
11224
|
job.params,
|
|
@@ -11160,6 +11226,7 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11160
11226
|
config.logger
|
|
11161
11227
|
);
|
|
11162
11228
|
const genParams = job.params;
|
|
11229
|
+
const genReferenceId = referenceIdOf(job);
|
|
11163
11230
|
const storageUri = deriveStorageUri(genResult.title, genResult.format);
|
|
11164
11231
|
const { resourceId: newResourceId } = await session.client.yield.resource({
|
|
11165
11232
|
name: genResult.title,
|
|
@@ -11167,13 +11234,13 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11167
11234
|
format: genResult.format,
|
|
11168
11235
|
storageUri,
|
|
11169
11236
|
sourceResourceId: resourceId$1,
|
|
11170
|
-
...
|
|
11237
|
+
...genReferenceId ? { sourceAnnotationId: genReferenceId } : {},
|
|
11171
11238
|
...genParams.prompt ? { generationPrompt: genParams.prompt } : {},
|
|
11172
11239
|
...genParams.language ? { language: genParams.language } : {},
|
|
11173
11240
|
...genParams.entityTypes && genParams.entityTypes.length > 0 ? { entityTypes: genParams.entityTypes } : {},
|
|
11174
11241
|
generator
|
|
11175
11242
|
});
|
|
11176
|
-
if (!
|
|
11243
|
+
if (!genReferenceId) {
|
|
11177
11244
|
const { annotation: provenanceRef } = assembleAnnotation(
|
|
11178
11245
|
{
|
|
11179
11246
|
motivation: "linking",
|