@semiont/jobs 0.5.25 → 0.5.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -12
- package/dist/index.d.ts +27 -72
- package/dist/index.js +211 -147
- package/dist/index.js.map +1 -1
- package/dist/worker-main.js +246 -169
- package/dist/worker-main.js.map +1 -1
- package/package.json +8 -8
package/dist/worker-main.js
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText,
|
|
1
|
+
import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, isGenerationJobParams, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText, isObject, isString, deriveViews } from '@semiont/core';
|
|
2
2
|
import { anchoredTextStoreOverTransport, deriveStorageUri, extractPdfTextLayer, EXTRACTORS, calculateChecksum, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
|
|
3
|
+
import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
|
|
3
4
|
import { execFileSync } from 'child_process';
|
|
4
5
|
import { existsSync, readFileSync, mkdtempSync, writeFileSync, rmSync } from 'fs';
|
|
5
6
|
import { homedir, hostname, tmpdir } from 'os';
|
|
6
7
|
import { join } from 'path';
|
|
7
8
|
import { generateAnnotationId } from '@semiont/event-sourcing';
|
|
8
|
-
import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
|
|
9
9
|
import { InMemorySessionStorage, setStoredSession, kbBackendUrl, SemiontClient, SemiontSession } from '@semiont/sdk';
|
|
10
10
|
import { HttpTransport, HttpContentTransport } from '@semiont/http-transport';
|
|
11
11
|
import { createInferenceClient } from '@semiont/inference';
|
|
@@ -3766,9 +3766,9 @@ var require_mapOneOrManyArgs = __commonJS({
|
|
|
3766
3766
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3767
3767
|
exports.mapOneOrManyArgs = void 0;
|
|
3768
3768
|
var map_1 = require_map();
|
|
3769
|
-
var
|
|
3769
|
+
var isArray = Array.isArray;
|
|
3770
3770
|
function callOrApply(fn, args) {
|
|
3771
|
-
return
|
|
3771
|
+
return isArray(args) ? fn.apply(void 0, __spreadArray([], __read(args))) : fn(args);
|
|
3772
3772
|
}
|
|
3773
3773
|
function mapOneOrManyArgs(fn) {
|
|
3774
3774
|
return map_1.map(function(args) {
|
|
@@ -3913,14 +3913,14 @@ var require_argsArgArrayOrObject = __commonJS({
|
|
|
3913
3913
|
"../../node_modules/rxjs/dist/cjs/internal/util/argsArgArrayOrObject.js"(exports) {
|
|
3914
3914
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3915
3915
|
exports.argsArgArrayOrObject = void 0;
|
|
3916
|
-
var
|
|
3916
|
+
var isArray = Array.isArray;
|
|
3917
3917
|
var getPrototypeOf = Object.getPrototypeOf;
|
|
3918
3918
|
var objectProto = Object.prototype;
|
|
3919
3919
|
var getKeys = Object.keys;
|
|
3920
3920
|
function argsArgArrayOrObject(args) {
|
|
3921
3921
|
if (args.length === 1) {
|
|
3922
3922
|
var first_1 = args[0];
|
|
3923
|
-
if (
|
|
3923
|
+
if (isArray(first_1)) {
|
|
3924
3924
|
return { args: first_1, keys: null };
|
|
3925
3925
|
}
|
|
3926
3926
|
if (isPOJO(first_1)) {
|
|
@@ -4668,9 +4668,9 @@ var require_argsOrArgArray = __commonJS({
|
|
|
4668
4668
|
"../../node_modules/rxjs/dist/cjs/internal/util/argsOrArgArray.js"(exports) {
|
|
4669
4669
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
4670
4670
|
exports.argsOrArgArray = void 0;
|
|
4671
|
-
var
|
|
4671
|
+
var isArray = Array.isArray;
|
|
4672
4672
|
function argsOrArgArray(args) {
|
|
4673
|
-
return args.length === 1 &&
|
|
4673
|
+
return args.length === 1 && isArray(args[0]) ? args[0] : args;
|
|
4674
4674
|
}
|
|
4675
4675
|
exports.argsOrArgArray = argsOrArgArray;
|
|
4676
4676
|
}
|
|
@@ -9348,10 +9348,18 @@ function asJobParams(params) {
|
|
|
9348
9348
|
}
|
|
9349
9349
|
return params;
|
|
9350
9350
|
}
|
|
9351
|
-
|
|
9352
|
-
// src/workers/inference-call.ts
|
|
9353
9351
|
var INFERENCE_TIMEOUT_MS = 10 * 6e4;
|
|
9354
|
-
|
|
9352
|
+
var INFERENCE_HEARTBEAT_MS = 15e3;
|
|
9353
|
+
function spanned(client, kind, maxTokens, work) {
|
|
9354
|
+
return withSpan(`inference:${kind}`, work, {
|
|
9355
|
+
attrs: {
|
|
9356
|
+
"inference.provider": client.type,
|
|
9357
|
+
"inference.model": client.modelId,
|
|
9358
|
+
"inference.max_tokens": maxTokens
|
|
9359
|
+
}
|
|
9360
|
+
});
|
|
9361
|
+
}
|
|
9362
|
+
async function withTimeout(work, label, onHeartbeat) {
|
|
9355
9363
|
let timer;
|
|
9356
9364
|
const timedOut = new Promise((_, reject) => {
|
|
9357
9365
|
timer = setTimeout(() => {
|
|
@@ -9361,6 +9369,16 @@ async function withTimeout(work, label) {
|
|
|
9361
9369
|
}, INFERENCE_TIMEOUT_MS);
|
|
9362
9370
|
timer.unref?.();
|
|
9363
9371
|
});
|
|
9372
|
+
let heartbeat;
|
|
9373
|
+
if (onHeartbeat) {
|
|
9374
|
+
heartbeat = setInterval(() => {
|
|
9375
|
+
try {
|
|
9376
|
+
onHeartbeat();
|
|
9377
|
+
} catch {
|
|
9378
|
+
}
|
|
9379
|
+
}, INFERENCE_HEARTBEAT_MS);
|
|
9380
|
+
heartbeat.unref?.();
|
|
9381
|
+
}
|
|
9364
9382
|
try {
|
|
9365
9383
|
return await Promise.race([work, timedOut]);
|
|
9366
9384
|
} catch (err) {
|
|
@@ -9369,19 +9387,22 @@ async function withTimeout(work, label) {
|
|
|
9369
9387
|
throw err;
|
|
9370
9388
|
} finally {
|
|
9371
9389
|
clearTimeout(timer);
|
|
9390
|
+
if (heartbeat) clearInterval(heartbeat);
|
|
9372
9391
|
}
|
|
9373
9392
|
}
|
|
9374
|
-
function boundedGenerate(client, prompt, maxTokens, temperature,
|
|
9375
|
-
return withTimeout(
|
|
9376
|
-
client.generateText(prompt, maxTokens, temperature
|
|
9377
|
-
`${client.type}:${client.modelId}
|
|
9378
|
-
|
|
9393
|
+
function boundedGenerate(client, prompt, maxTokens, temperature, onHeartbeat) {
|
|
9394
|
+
return spanned(client, "text", maxTokens, () => withTimeout(
|
|
9395
|
+
client.generateText(prompt, maxTokens, temperature),
|
|
9396
|
+
`${client.type}:${client.modelId}`,
|
|
9397
|
+
onHeartbeat
|
|
9398
|
+
));
|
|
9379
9399
|
}
|
|
9380
|
-
function
|
|
9381
|
-
return withTimeout(
|
|
9382
|
-
client.
|
|
9383
|
-
`${client.type}:${client.modelId}
|
|
9384
|
-
|
|
9400
|
+
function boundedGenerateStructured(client, prompt, maxTokens, temperature, elementSchema, onHeartbeat) {
|
|
9401
|
+
return spanned(client, "structured", maxTokens, () => withTimeout(
|
|
9402
|
+
client.generateStructured(prompt, maxTokens, temperature, elementSchema),
|
|
9403
|
+
`${client.type}:${client.modelId}`,
|
|
9404
|
+
onHeartbeat
|
|
9405
|
+
));
|
|
9385
9406
|
}
|
|
9386
9407
|
|
|
9387
9408
|
// src/workers/detection/detection-chunking.ts
|
|
@@ -9713,32 +9734,57 @@ Example format:
|
|
|
9713
9734
|
return prompt;
|
|
9714
9735
|
}
|
|
9715
9736
|
};
|
|
9716
|
-
|
|
9717
|
-
|
|
9718
|
-
|
|
9719
|
-
|
|
9720
|
-
|
|
9721
|
-
|
|
9722
|
-
|
|
9723
|
-
|
|
9724
|
-
|
|
9725
|
-
|
|
9726
|
-
|
|
9727
|
-
|
|
9728
|
-
|
|
9729
|
-
|
|
9730
|
-
}
|
|
9737
|
+
var COMMENT_ELEMENT_SCHEMA = {
|
|
9738
|
+
type: "object",
|
|
9739
|
+
properties: {
|
|
9740
|
+
exact: { type: "string" },
|
|
9741
|
+
prefix: { type: "string" },
|
|
9742
|
+
suffix: { type: "string" },
|
|
9743
|
+
comment: { type: "string" }
|
|
9744
|
+
},
|
|
9745
|
+
required: ["exact", "comment"],
|
|
9746
|
+
additionalProperties: false
|
|
9747
|
+
};
|
|
9748
|
+
var HIGHLIGHT_ELEMENT_SCHEMA = {
|
|
9749
|
+
type: "object",
|
|
9750
|
+
properties: {
|
|
9751
|
+
exact: { type: "string" },
|
|
9752
|
+
prefix: { type: "string" },
|
|
9753
|
+
suffix: { type: "string" }
|
|
9754
|
+
},
|
|
9755
|
+
required: ["exact"],
|
|
9756
|
+
additionalProperties: false
|
|
9757
|
+
};
|
|
9758
|
+
var ASSESSMENT_ELEMENT_SCHEMA = {
|
|
9759
|
+
type: "object",
|
|
9760
|
+
properties: {
|
|
9761
|
+
exact: { type: "string" },
|
|
9762
|
+
prefix: { type: "string" },
|
|
9763
|
+
suffix: { type: "string" },
|
|
9764
|
+
assessment: { type: "string" }
|
|
9765
|
+
},
|
|
9766
|
+
required: ["exact", "assessment"],
|
|
9767
|
+
additionalProperties: false
|
|
9768
|
+
};
|
|
9769
|
+
var TAG_ELEMENT_SCHEMA = {
|
|
9770
|
+
type: "object",
|
|
9771
|
+
properties: {
|
|
9772
|
+
exact: { type: "string" },
|
|
9773
|
+
prefix: { type: "string" },
|
|
9774
|
+
suffix: { type: "string" }
|
|
9775
|
+
},
|
|
9776
|
+
required: ["exact"],
|
|
9777
|
+
additionalProperties: false
|
|
9778
|
+
};
|
|
9731
9779
|
var MotivationParsers = class {
|
|
9732
9780
|
/**
|
|
9733
|
-
*
|
|
9781
|
+
* Validate and reconcile structured comment elements.
|
|
9734
9782
|
*
|
|
9735
|
-
* @param
|
|
9783
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
9736
9784
|
* @param content - Original content to validate offsets against
|
|
9737
9785
|
* @returns Array of validated comment matches
|
|
9738
|
-
* @throws if the response is not a parseable JSON array
|
|
9739
9786
|
*/
|
|
9740
|
-
static parseComments(
|
|
9741
|
-
const parsed = parseJsonArray(response, "comment");
|
|
9787
|
+
static parseComments(parsed, content) {
|
|
9742
9788
|
const valid = parsed.filter(
|
|
9743
9789
|
(c) => isObject(c) && isString(c.exact) && isString(c.comment) && c.comment.trim().length > 0
|
|
9744
9790
|
);
|
|
@@ -9767,15 +9813,13 @@ var MotivationParsers = class {
|
|
|
9767
9813
|
return validatedComments;
|
|
9768
9814
|
}
|
|
9769
9815
|
/**
|
|
9770
|
-
*
|
|
9816
|
+
* Validate and reconcile structured highlight elements.
|
|
9771
9817
|
*
|
|
9772
|
-
* @param
|
|
9818
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
9773
9819
|
* @param content - Original content to validate offsets against
|
|
9774
9820
|
* @returns Array of validated highlight matches
|
|
9775
|
-
* @throws if the response is not a parseable JSON array
|
|
9776
9821
|
*/
|
|
9777
|
-
static parseHighlights(
|
|
9778
|
-
const parsed = parseJsonArray(response, "highlight");
|
|
9822
|
+
static parseHighlights(parsed, content) {
|
|
9779
9823
|
const highlights = parsed.filter(
|
|
9780
9824
|
(h) => isObject(h) && isString(h.exact)
|
|
9781
9825
|
);
|
|
@@ -9802,15 +9846,13 @@ var MotivationParsers = class {
|
|
|
9802
9846
|
return validatedHighlights;
|
|
9803
9847
|
}
|
|
9804
9848
|
/**
|
|
9805
|
-
*
|
|
9849
|
+
* Validate and reconcile structured assessment elements.
|
|
9806
9850
|
*
|
|
9807
|
-
* @param
|
|
9851
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
9808
9852
|
* @param content - Original content to validate offsets against
|
|
9809
9853
|
* @returns Array of validated assessment matches
|
|
9810
|
-
* @throws if the response is not a parseable JSON array
|
|
9811
9854
|
*/
|
|
9812
|
-
static parseAssessments(
|
|
9813
|
-
const parsed = parseJsonArray(response, "assessment");
|
|
9855
|
+
static parseAssessments(parsed, content) {
|
|
9814
9856
|
const assessments = parsed.filter(
|
|
9815
9857
|
(a) => isObject(a) && isString(a.exact) && isString(a.assessment)
|
|
9816
9858
|
);
|
|
@@ -9838,14 +9880,13 @@ var MotivationParsers = class {
|
|
|
9838
9880
|
return validatedAssessments;
|
|
9839
9881
|
}
|
|
9840
9882
|
/**
|
|
9841
|
-
*
|
|
9883
|
+
* Validate structured tag elements into raw, pre-reconciliation tag inputs.
|
|
9842
9884
|
* Reconciliation happens in `validateTagOffsets`, which adds `start`/`end`
|
|
9843
9885
|
* by anchoring `exact` against the source content.
|
|
9844
9886
|
*
|
|
9845
|
-
* @
|
|
9887
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
9846
9888
|
*/
|
|
9847
|
-
static parseTags(
|
|
9848
|
-
const parsed = parseJsonArray(response, "tag");
|
|
9889
|
+
static parseTags(parsed) {
|
|
9849
9890
|
const valid = parsed.filter(
|
|
9850
9891
|
(t) => isObject(t) && isString(t.exact) && t.exact.trim().length > 0
|
|
9851
9892
|
);
|
|
@@ -9892,24 +9933,26 @@ function assertNotTruncated(response, motivation, chunk, totalChunks, outputBudg
|
|
|
9892
9933
|
throw new Error(`${motivation} detection response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
|
|
9893
9934
|
}
|
|
9894
9935
|
}
|
|
9895
|
-
async function detectInChunks(client, content, buildPrompt, temperature, motivation, parse,
|
|
9936
|
+
async function detectInChunks(client, content, buildPrompt, temperature, motivation, elementSchema, parse, onActivity) {
|
|
9896
9937
|
const limits = await client.limits();
|
|
9897
9938
|
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
9898
9939
|
const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
|
|
9899
9940
|
const chunks = chunkText(content, chunking);
|
|
9900
9941
|
const collected = [];
|
|
9901
9942
|
for (let i = 0; i < chunks.length; i++) {
|
|
9902
|
-
const response = await
|
|
9943
|
+
const response = await boundedGenerateStructured(
|
|
9903
9944
|
client,
|
|
9904
9945
|
buildPrompt(chunks[i]),
|
|
9905
9946
|
outputBudget,
|
|
9906
9947
|
temperature,
|
|
9907
|
-
|
|
9948
|
+
elementSchema,
|
|
9949
|
+
// Still alive, same position (a long single call is otherwise silent).
|
|
9950
|
+
() => onActivity?.(i, chunks.length)
|
|
9908
9951
|
);
|
|
9909
9952
|
assertNotTruncated(response, motivation, i + 1, chunks.length, outputBudget);
|
|
9910
|
-
collected.push(...parse(response.
|
|
9953
|
+
collected.push(...parse(response.items));
|
|
9911
9954
|
if (i < chunks.length - 1) {
|
|
9912
|
-
|
|
9955
|
+
onActivity?.(i + 1, chunks.length);
|
|
9913
9956
|
}
|
|
9914
9957
|
}
|
|
9915
9958
|
return collected;
|
|
@@ -9923,15 +9966,16 @@ var AnnotationDetection = class {
|
|
|
9923
9966
|
* (source-resource locale). See `types.ts` "Locale conventions" for the
|
|
9924
9967
|
* full discussion.
|
|
9925
9968
|
*/
|
|
9926
|
-
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage,
|
|
9969
|
+
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onActivity) {
|
|
9927
9970
|
return detectInChunks(
|
|
9928
9971
|
client,
|
|
9929
9972
|
content,
|
|
9930
9973
|
(chunk) => MotivationPrompts.buildCommentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
|
|
9931
9974
|
0.4,
|
|
9932
9975
|
"comment",
|
|
9933
|
-
|
|
9934
|
-
|
|
9976
|
+
COMMENT_ELEMENT_SCHEMA,
|
|
9977
|
+
(items) => MotivationParsers.parseComments(items, content),
|
|
9978
|
+
onActivity
|
|
9935
9979
|
);
|
|
9936
9980
|
}
|
|
9937
9981
|
/**
|
|
@@ -9941,15 +9985,16 @@ var AnnotationDetection = class {
|
|
|
9941
9985
|
* applies, used in the prompt so the LLM analyzes non-English source
|
|
9942
9986
|
* correctly.
|
|
9943
9987
|
*/
|
|
9944
|
-
static async detectHighlights(content, client, instructions, density, sourceLanguage,
|
|
9988
|
+
static async detectHighlights(content, client, instructions, density, sourceLanguage, onActivity) {
|
|
9945
9989
|
return detectInChunks(
|
|
9946
9990
|
client,
|
|
9947
9991
|
content,
|
|
9948
9992
|
(chunk) => MotivationPrompts.buildHighlightPrompt(chunk, instructions, density, sourceLanguage),
|
|
9949
9993
|
0.3,
|
|
9950
9994
|
"highlight",
|
|
9951
|
-
|
|
9952
|
-
|
|
9995
|
+
HIGHLIGHT_ELEMENT_SCHEMA,
|
|
9996
|
+
(items) => MotivationParsers.parseHighlights(items, content),
|
|
9997
|
+
onActivity
|
|
9953
9998
|
);
|
|
9954
9999
|
}
|
|
9955
10000
|
/**
|
|
@@ -9959,15 +10004,16 @@ var AnnotationDetection = class {
|
|
|
9959
10004
|
* (annotation body locale). `sourceLanguage` is the locale of the content
|
|
9960
10005
|
* being analyzed (source-resource locale).
|
|
9961
10006
|
*/
|
|
9962
|
-
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage,
|
|
10007
|
+
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onActivity) {
|
|
9963
10008
|
return detectInChunks(
|
|
9964
10009
|
client,
|
|
9965
10010
|
content,
|
|
9966
10011
|
(chunk) => MotivationPrompts.buildAssessmentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
|
|
9967
10012
|
0.3,
|
|
9968
10013
|
"assessment",
|
|
9969
|
-
|
|
9970
|
-
|
|
10014
|
+
ASSESSMENT_ELEMENT_SCHEMA,
|
|
10015
|
+
(items) => MotivationParsers.parseAssessments(items, content),
|
|
10016
|
+
onActivity
|
|
9971
10017
|
);
|
|
9972
10018
|
}
|
|
9973
10019
|
/**
|
|
@@ -9982,7 +10028,7 @@ var AnnotationDetection = class {
|
|
|
9982
10028
|
* identifiers, not LLM-generated text — so it's consumed at the body-stamp
|
|
9983
10029
|
* site, not here.
|
|
9984
10030
|
*/
|
|
9985
|
-
static async detectTags(content, client, schema, category, sourceLanguage,
|
|
10031
|
+
static async detectTags(content, client, schema, category, sourceLanguage, onActivity) {
|
|
9986
10032
|
const categoryInfo = schema.tags.find((t) => t.name === category);
|
|
9987
10033
|
if (!categoryInfo) {
|
|
9988
10034
|
throw new Error(`Invalid category "${category}" for schema ${schema.id}`);
|
|
@@ -10002,13 +10048,25 @@ var AnnotationDetection = class {
|
|
|
10002
10048
|
),
|
|
10003
10049
|
0.2,
|
|
10004
10050
|
"tag",
|
|
10005
|
-
|
|
10006
|
-
|
|
10051
|
+
TAG_ELEMENT_SCHEMA,
|
|
10052
|
+
(items) => MotivationParsers.parseTags(items),
|
|
10053
|
+
onActivity
|
|
10007
10054
|
);
|
|
10008
10055
|
return MotivationParsers.validateTagOffsets(parsedTags, content, category);
|
|
10009
10056
|
}
|
|
10010
10057
|
};
|
|
10011
|
-
|
|
10058
|
+
var ENTITY_ELEMENT_SCHEMA = {
|
|
10059
|
+
type: "object",
|
|
10060
|
+
properties: {
|
|
10061
|
+
exact: { type: "string" },
|
|
10062
|
+
entityType: { type: "string" },
|
|
10063
|
+
prefix: { type: "string" },
|
|
10064
|
+
suffix: { type: "string" }
|
|
10065
|
+
},
|
|
10066
|
+
required: ["exact", "entityType"],
|
|
10067
|
+
additionalProperties: false
|
|
10068
|
+
};
|
|
10069
|
+
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage, onActivity) {
|
|
10012
10070
|
const entityTypesDescription = entityTypes.map((et) => {
|
|
10013
10071
|
if (typeof et === "string") {
|
|
10014
10072
|
return et;
|
|
@@ -10069,50 +10127,28 @@ Example output:
|
|
|
10069
10127
|
});
|
|
10070
10128
|
const collected = [];
|
|
10071
10129
|
for (let i = 0; i < chunks.length; i++) {
|
|
10072
|
-
const response = await
|
|
10130
|
+
const response = await boundedGenerateStructured(
|
|
10073
10131
|
client,
|
|
10074
10132
|
buildPrompt(chunks[i]),
|
|
10075
10133
|
outputBudget,
|
|
10076
10134
|
0.3,
|
|
10077
10135
|
// Lower temperature for more consistent extraction
|
|
10078
|
-
|
|
10079
|
-
//
|
|
10080
|
-
//
|
|
10081
|
-
|
|
10082
|
-
// governs *what* the JSON contains; `format: 'json'` governs that
|
|
10083
|
-
// it's syntactically valid.
|
|
10084
|
-
{ format: "json" }
|
|
10136
|
+
ENTITY_ELEMENT_SCHEMA,
|
|
10137
|
+
// Still alive, same position: a long single call would otherwise emit
|
|
10138
|
+
// nothing at all between start and finish.
|
|
10139
|
+
() => onActivity?.(i, chunks.length)
|
|
10085
10140
|
);
|
|
10086
10141
|
logger2.debug("Got entity extraction response", {
|
|
10087
10142
|
chunk: i + 1,
|
|
10088
10143
|
chunks: chunks.length,
|
|
10089
|
-
|
|
10144
|
+
items: response.items.length
|
|
10090
10145
|
});
|
|
10091
10146
|
if (response.stopReason === "max_tokens") {
|
|
10092
10147
|
const errorMsg = `Entity extraction response truncated (max_tokens) on chunk ${i + 1}/${chunks.length} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than dropping annotations.`;
|
|
10093
|
-
logger2.error(errorMsg, {
|
|
10148
|
+
logger2.error(errorMsg, { items: response.items.length });
|
|
10094
10149
|
throw new Error(errorMsg);
|
|
10095
10150
|
}
|
|
10096
|
-
|
|
10097
|
-
try {
|
|
10098
|
-
entities = JSON.parse(response.text.trim());
|
|
10099
|
-
} catch (error) {
|
|
10100
|
-
logger2.error("Failed to parse entity extraction response", {
|
|
10101
|
-
error: error instanceof Error ? error.message : String(error),
|
|
10102
|
-
response: response.text.slice(0, 500)
|
|
10103
|
-
});
|
|
10104
|
-
throw new Error("Failed to parse entity extraction response", {
|
|
10105
|
-
cause: error instanceof Error ? error : new Error(String(error))
|
|
10106
|
-
});
|
|
10107
|
-
}
|
|
10108
|
-
if (!isArray(entities)) {
|
|
10109
|
-
logger2.error("Failed to parse entity extraction response: expected a JSON array", {
|
|
10110
|
-
response: response.text.slice(0, 500)
|
|
10111
|
-
});
|
|
10112
|
-
throw new Error("Failed to parse entity extraction response: expected a JSON array");
|
|
10113
|
-
}
|
|
10114
|
-
logger2.debug("Parsed entities from AI response", { chunk: i + 1, count: entities.length });
|
|
10115
|
-
for (const e of entities) {
|
|
10151
|
+
for (const e of response.items) {
|
|
10116
10152
|
if (isObject(e) && isString(e.exact) && isString(e.entityType)) {
|
|
10117
10153
|
collected.push({
|
|
10118
10154
|
exact: e.exact,
|
|
@@ -10125,7 +10161,7 @@ Example output:
|
|
|
10125
10161
|
}
|
|
10126
10162
|
}
|
|
10127
10163
|
if (i < chunks.length - 1) {
|
|
10128
|
-
|
|
10164
|
+
onActivity?.(i + 1, chunks.length);
|
|
10129
10165
|
}
|
|
10130
10166
|
}
|
|
10131
10167
|
return collected;
|
|
@@ -10561,30 +10597,39 @@ function buildPdfAnnotation(anchored, resourceId, userId, generator, motivation,
|
|
|
10561
10597
|
};
|
|
10562
10598
|
}
|
|
10563
10599
|
async function processHighlightJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
10564
|
-
|
|
10565
|
-
onProgress(
|
|
10600
|
+
const echo = detectionEcho(params);
|
|
10601
|
+
onProgress(10, { code: "loading" }, echo);
|
|
10602
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
10566
10603
|
const highlights = await AnnotationDetection.detectHighlights(
|
|
10567
10604
|
content,
|
|
10568
10605
|
inferenceClient,
|
|
10569
10606
|
params.instructions,
|
|
10570
10607
|
params.density,
|
|
10571
10608
|
params.sourceLanguage,
|
|
10572
|
-
//
|
|
10573
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
10609
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
10610
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
10574
10611
|
);
|
|
10575
|
-
onProgress(60,
|
|
10612
|
+
onProgress(60, { code: "creating-annotations", count: highlights.length }, echo);
|
|
10576
10613
|
const annotations = dedupeAnnotations(highlights.map(
|
|
10577
10614
|
(h) => buildAnnotation("highlighting", h)
|
|
10578
10615
|
));
|
|
10579
|
-
onProgress(100,
|
|
10616
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "highlight" }, echo);
|
|
10580
10617
|
return {
|
|
10581
10618
|
annotations,
|
|
10582
10619
|
result: { highlightsFound: highlights.length, highlightsCreated: annotations.length }
|
|
10583
10620
|
};
|
|
10584
10621
|
}
|
|
10622
|
+
function detectionEcho(p) {
|
|
10623
|
+
const requestParams = [];
|
|
10624
|
+
if (p.instructions?.trim()) requestParams.push({ label: "instructions", value: p.instructions.trim() });
|
|
10625
|
+
if (p.tone?.trim()) requestParams.push({ label: "tone", value: p.tone.trim() });
|
|
10626
|
+
if (p.density !== void 0) requestParams.push({ label: "density", value: String(p.density) });
|
|
10627
|
+
return requestParams.length > 0 ? { requestParams } : {};
|
|
10628
|
+
}
|
|
10585
10629
|
async function processCommentJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
10586
|
-
|
|
10587
|
-
onProgress(
|
|
10630
|
+
const echo = detectionEcho(params);
|
|
10631
|
+
onProgress(10, { code: "loading" }, echo);
|
|
10632
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
10588
10633
|
const comments = await AnnotationDetection.detectComments(
|
|
10589
10634
|
content,
|
|
10590
10635
|
inferenceClient,
|
|
@@ -10593,10 +10638,10 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
10593
10638
|
params.density,
|
|
10594
10639
|
params.language,
|
|
10595
10640
|
params.sourceLanguage,
|
|
10596
|
-
//
|
|
10597
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
10641
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
10642
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
10598
10643
|
);
|
|
10599
|
-
onProgress(60,
|
|
10644
|
+
onProgress(60, { code: "creating-annotations", count: comments.length }, echo);
|
|
10600
10645
|
const bodyLanguage = params.language ?? "en";
|
|
10601
10646
|
const annotations = dedupeAnnotations(comments.map(
|
|
10602
10647
|
(c) => (
|
|
@@ -10608,15 +10653,16 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
10608
10653
|
])
|
|
10609
10654
|
)
|
|
10610
10655
|
));
|
|
10611
|
-
onProgress(100,
|
|
10656
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "comment" }, echo);
|
|
10612
10657
|
return {
|
|
10613
10658
|
annotations,
|
|
10614
10659
|
result: { commentsFound: comments.length, commentsCreated: annotations.length }
|
|
10615
10660
|
};
|
|
10616
10661
|
}
|
|
10617
10662
|
async function processAssessmentJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
10618
|
-
|
|
10619
|
-
onProgress(
|
|
10663
|
+
const echo = detectionEcho(params);
|
|
10664
|
+
onProgress(10, { code: "loading" }, echo);
|
|
10665
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
10620
10666
|
const assessments = await AnnotationDetection.detectAssessments(
|
|
10621
10667
|
content,
|
|
10622
10668
|
inferenceClient,
|
|
@@ -10625,10 +10671,10 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
10625
10671
|
params.density,
|
|
10626
10672
|
params.language,
|
|
10627
10673
|
params.sourceLanguage,
|
|
10628
|
-
//
|
|
10629
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
10674
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
10675
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
10630
10676
|
);
|
|
10631
|
-
onProgress(60,
|
|
10677
|
+
onProgress(60, { code: "creating-annotations", count: assessments.length }, echo);
|
|
10632
10678
|
const bodyLanguage = params.language ?? "en";
|
|
10633
10679
|
const annotations = dedupeAnnotations(assessments.map(
|
|
10634
10680
|
(a) => (
|
|
@@ -10647,7 +10693,7 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
10647
10693
|
})
|
|
10648
10694
|
)
|
|
10649
10695
|
));
|
|
10650
|
-
onProgress(100,
|
|
10696
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "assessment" }, echo);
|
|
10651
10697
|
return {
|
|
10652
10698
|
annotations,
|
|
10653
10699
|
result: { assessmentsFound: assessments.length, assessmentsCreated: annotations.length }
|
|
@@ -10655,25 +10701,27 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
10655
10701
|
}
|
|
10656
10702
|
async function processReferenceJob(content, inferenceClient, params, buildAnnotation, onProgress, logger2) {
|
|
10657
10703
|
const entityTypeNames = params.entityTypes.map(String);
|
|
10658
|
-
const requestParams = [{ label: "
|
|
10659
|
-
const
|
|
10704
|
+
const requestParams = [{ label: "entity-types", value: entityTypeNames.join(", ") }];
|
|
10705
|
+
const completedItems = [];
|
|
10660
10706
|
let totalFound = 0;
|
|
10661
10707
|
let totalEmitted = 0;
|
|
10662
10708
|
let errors = 0;
|
|
10663
10709
|
const allAnnotations = [];
|
|
10664
|
-
onProgress(10,
|
|
10710
|
+
onProgress(10, { code: "loading" }, { requestParams });
|
|
10665
10711
|
const bodyLanguage = params.language ?? "en";
|
|
10666
10712
|
for (let i = 0; i < entityTypeNames.length; i++) {
|
|
10667
10713
|
const entityTypeName = entityTypeNames[i];
|
|
10668
10714
|
if (!entityTypeName) continue;
|
|
10669
10715
|
const pct = 20 + Math.round(i / entityTypeNames.length * 60);
|
|
10670
|
-
onProgress(pct,
|
|
10671
|
-
|
|
10672
|
-
|
|
10673
|
-
|
|
10716
|
+
onProgress(pct, { code: "detecting-entities", entityType: entityTypeName }, {
|
|
10717
|
+
// One vocabulary for "what is in flight" (CLEAN-PROGRESS D2): the entity
|
|
10718
|
+
// type is KB data, `kind` is the code the client localizes around it.
|
|
10719
|
+
current: { kind: "entity-type", value: entityTypeName },
|
|
10720
|
+
processed: i,
|
|
10721
|
+
total: entityTypeNames.length,
|
|
10674
10722
|
entitiesFound: totalFound,
|
|
10675
10723
|
entitiesEmitted: totalEmitted,
|
|
10676
|
-
|
|
10724
|
+
completedItems: [...completedItems],
|
|
10677
10725
|
requestParams
|
|
10678
10726
|
});
|
|
10679
10727
|
const extractedEntities = await extractEntities(
|
|
@@ -10683,25 +10731,28 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
10683
10731
|
params.includeDescriptiveReferences ?? false,
|
|
10684
10732
|
logger2,
|
|
10685
10733
|
params.sourceLanguage,
|
|
10686
|
-
//
|
|
10687
|
-
//
|
|
10688
|
-
//
|
|
10689
|
-
//
|
|
10734
|
+
// Liveness: fires at chunk boundaries AND every ~15 s while a single
|
|
10735
|
+
// inference call is in flight (DETECTION-HEARTBEAT). Progress feeds the
|
|
10736
|
+
// stall watchdog, the janitor, AND the client's inter-emission timeout,
|
|
10737
|
+
// so a long single-chunk call must not be silent. Percentage
|
|
10738
|
+
// interpolates within this entity type's band of the 20–80 range; a
|
|
10739
|
+
// heartbeat repeats the current position rather than inventing an
|
|
10740
|
+
// advance.
|
|
10690
10741
|
(completed, total) => {
|
|
10691
10742
|
const interpolated = 20 + Math.round((i + completed / total) / entityTypeNames.length * 60);
|
|
10692
|
-
onProgress(interpolated,
|
|
10693
|
-
|
|
10694
|
-
|
|
10695
|
-
|
|
10743
|
+
onProgress(interpolated, { code: "detecting-entities", entityType: entityTypeName }, {
|
|
10744
|
+
current: { kind: "entity-type", value: entityTypeName },
|
|
10745
|
+
processed: i,
|
|
10746
|
+
total: entityTypeNames.length,
|
|
10696
10747
|
entitiesFound: totalFound,
|
|
10697
10748
|
entitiesEmitted: totalEmitted,
|
|
10698
|
-
|
|
10749
|
+
completedItems: [...completedItems],
|
|
10699
10750
|
requestParams
|
|
10700
10751
|
});
|
|
10701
10752
|
}
|
|
10702
10753
|
);
|
|
10703
10754
|
totalFound += extractedEntities.length;
|
|
10704
|
-
|
|
10755
|
+
completedItems.push({ value: entityTypeName, foundCount: extractedEntities.length });
|
|
10705
10756
|
const unresolvedBody = [
|
|
10706
10757
|
{ type: "TextualBody", value: entityTypeName, purpose: "tagging", format: "text/plain", language: bodyLanguage }
|
|
10707
10758
|
];
|
|
@@ -10732,36 +10783,49 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
10732
10783
|
}
|
|
10733
10784
|
}
|
|
10734
10785
|
const annotations = dedupeAnnotations(allAnnotations);
|
|
10735
|
-
onProgress(100,
|
|
10786
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "reference" }, { requestParams });
|
|
10736
10787
|
return {
|
|
10737
10788
|
annotations,
|
|
10738
10789
|
result: { totalFound, totalEmitted: annotations.length, errors }
|
|
10739
10790
|
};
|
|
10740
10791
|
}
|
|
10741
10792
|
async function processTagJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
10742
|
-
onProgress(10,
|
|
10743
|
-
onProgress(30,
|
|
10793
|
+
onProgress(10, { code: "loading" });
|
|
10794
|
+
onProgress(30, { code: "analyzing-tags" });
|
|
10744
10795
|
const allTags = [];
|
|
10796
|
+
const completedItems = [];
|
|
10745
10797
|
for (let c = 0; c < params.categories.length; c++) {
|
|
10746
10798
|
const category = params.categories[c];
|
|
10799
|
+
const position = () => ({
|
|
10800
|
+
current: { kind: "category", value: category },
|
|
10801
|
+
processed: c,
|
|
10802
|
+
total: params.categories.length,
|
|
10803
|
+
completedItems: [...completedItems]
|
|
10804
|
+
});
|
|
10805
|
+
onProgress(
|
|
10806
|
+
30 + Math.round(c / params.categories.length * 30),
|
|
10807
|
+
{ code: "analyzing-tags" },
|
|
10808
|
+
position()
|
|
10809
|
+
);
|
|
10747
10810
|
const categoryTags = await AnnotationDetection.detectTags(
|
|
10748
10811
|
content,
|
|
10749
10812
|
inferenceClient,
|
|
10750
10813
|
params.schema,
|
|
10751
10814
|
category,
|
|
10752
10815
|
params.sourceLanguage,
|
|
10753
|
-
//
|
|
10754
|
-
//
|
|
10816
|
+
// Liveness (chunk boundaries + in-flight heartbeat): this category's
|
|
10817
|
+
// slice of the 30–60 band.
|
|
10755
10818
|
(completed, total) => onProgress(
|
|
10756
10819
|
30 + Math.round((c + completed / total) / params.categories.length * 30),
|
|
10757
|
-
|
|
10758
|
-
|
|
10820
|
+
{ code: "analyzing-tags" },
|
|
10821
|
+
position()
|
|
10759
10822
|
)
|
|
10760
10823
|
);
|
|
10824
|
+
completedItems.push({ value: category, foundCount: categoryTags.length });
|
|
10761
10825
|
allTags.push(...categoryTags);
|
|
10762
10826
|
}
|
|
10763
10827
|
const tags = allTags;
|
|
10764
|
-
onProgress(60,
|
|
10828
|
+
onProgress(60, { code: "creating-tag-annotations", count: tags.length });
|
|
10765
10829
|
const bodyLanguage = params.language ?? "en";
|
|
10766
10830
|
const annotations = dedupeAnnotations(tags.map((t) => {
|
|
10767
10831
|
const category = t.category ?? "unknown";
|
|
@@ -10776,7 +10840,7 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
10776
10840
|
const category = Array.isArray(body) && typeof body[0]?.value === "string" ? body[0].value : "unknown";
|
|
10777
10841
|
byCategory[category] = (byCategory[category] ?? 0) + 1;
|
|
10778
10842
|
}
|
|
10779
|
-
onProgress(100,
|
|
10843
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "tag" });
|
|
10780
10844
|
return {
|
|
10781
10845
|
annotations,
|
|
10782
10846
|
result: { tagsFound: tags.length, tagsCreated: annotations.length, byCategory }
|
|
@@ -10799,7 +10863,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10799
10863
|
const title = params.title ?? "Untitled";
|
|
10800
10864
|
const entityTypes = (params.entityTypes ?? []).map(String);
|
|
10801
10865
|
if (outputMediaType === "application/pdf") {
|
|
10802
|
-
onProgress(5,
|
|
10866
|
+
onProgress(5, { code: "generating-resource" });
|
|
10803
10867
|
const validIds = params.cite === true ? collectContextResourceIds(params.context) : null;
|
|
10804
10868
|
let generated2 = await generateResourceFromTopic(
|
|
10805
10869
|
title,
|
|
@@ -10864,7 +10928,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10864
10928
|
);
|
|
10865
10929
|
}
|
|
10866
10930
|
assertWithinOutputBudget(compiled.pdf.byteLength);
|
|
10867
|
-
onProgress(95,
|
|
10931
|
+
onProgress(95, { code: "creating-resource" });
|
|
10868
10932
|
return {
|
|
10869
10933
|
content: compiled.pdf,
|
|
10870
10934
|
title: generated2.title ?? title,
|
|
@@ -10876,7 +10940,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10876
10940
|
}
|
|
10877
10941
|
};
|
|
10878
10942
|
}
|
|
10879
|
-
onProgress(5,
|
|
10943
|
+
onProgress(5, { code: "generating-resource" });
|
|
10880
10944
|
const generated = await generateResourceFromTopic(
|
|
10881
10945
|
title,
|
|
10882
10946
|
entityTypes,
|
|
@@ -10900,7 +10964,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10900
10964
|
content = resolved.content;
|
|
10901
10965
|
citations = resolved.citations;
|
|
10902
10966
|
}
|
|
10903
|
-
onProgress(95,
|
|
10967
|
+
onProgress(95, { code: "creating-resource" });
|
|
10904
10968
|
const artifact = new TextEncoder().encode(content);
|
|
10905
10969
|
assertWithinOutputBudget(artifact.byteLength);
|
|
10906
10970
|
return {
|
|
@@ -10940,13 +11004,18 @@ async function prepareDetection(mediaType, session, resourceId, userId, generato
|
|
|
10940
11004
|
buildAnnotation: (motivation, match, body) => buildTextAnnotation(extracted.text, resourceId, userId, generator, motivation, match, body)
|
|
10941
11005
|
};
|
|
10942
11006
|
}
|
|
10943
|
-
|
|
10944
|
-
|
|
10945
|
-
|
|
10946
|
-
|
|
10947
|
-
|
|
10948
|
-
|
|
10949
|
-
}
|
|
11007
|
+
function referenceIdOf(job) {
|
|
11008
|
+
if (job.type === "generation") {
|
|
11009
|
+
const context = job.params.context;
|
|
11010
|
+
const focus = context?.focus;
|
|
11011
|
+
if (focus?.kind === "annotation" && typeof focus.annotation?.id === "string") {
|
|
11012
|
+
return focus.annotation.id;
|
|
11013
|
+
}
|
|
11014
|
+
return void 0;
|
|
11015
|
+
}
|
|
11016
|
+
const ref = job.params.referenceId;
|
|
11017
|
+
return typeof ref === "string" ? ref : void 0;
|
|
11018
|
+
}
|
|
10950
11019
|
async function emitEvent(session, channel, payload) {
|
|
10951
11020
|
await session.client.transport.emit(channel, payload);
|
|
10952
11021
|
}
|
|
@@ -10963,7 +11032,7 @@ function startWorkerProcess(config) {
|
|
|
10963
11032
|
handleJob(adapter, config, job).catch((error) => {
|
|
10964
11033
|
const message = error instanceof Error ? error.message : String(error);
|
|
10965
11034
|
logger2.error("Job failed", { jobId: job.jobId, error: message, stack: error instanceof Error ? error.stack : void 0 });
|
|
10966
|
-
const failAnnotationId = job
|
|
11035
|
+
const failAnnotationId = referenceIdOf(job);
|
|
10967
11036
|
if (isJobType(job.type)) {
|
|
10968
11037
|
emitEvent(session, "job:fail", {
|
|
10969
11038
|
resourceId: job.resourceId,
|
|
@@ -11012,7 +11081,7 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11012
11081
|
}
|
|
11013
11082
|
const jobType = job.type;
|
|
11014
11083
|
const resourceId$1 = resourceId(job.resourceId);
|
|
11015
|
-
const annotationId = job
|
|
11084
|
+
const annotationId = referenceIdOf(job);
|
|
11016
11085
|
const lifecycleBase = {
|
|
11017
11086
|
resourceId: resourceId$1,
|
|
11018
11087
|
jobId,
|
|
@@ -11028,7 +11097,11 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11028
11097
|
if (jobType !== "generation") {
|
|
11029
11098
|
const descriptor = await session.client.browse.resource(resourceId$1).fresh();
|
|
11030
11099
|
const mediaType = getPrimaryMediaType(descriptor);
|
|
11031
|
-
const source = await
|
|
11100
|
+
const source = await withSpan(
|
|
11101
|
+
"detection:prepare",
|
|
11102
|
+
() => prepareDetection(mediaType ?? "", session, resourceId$1, userId, generator, config.anchoredTextStore),
|
|
11103
|
+
{ attrs: { "resource.id": resourceId$1, "media.type": mediaType ?? "unknown" } }
|
|
11104
|
+
);
|
|
11032
11105
|
if ("declined" in source) {
|
|
11033
11106
|
if (source.declined === "no-extractor") {
|
|
11034
11107
|
throw new Error(`Cannot run ${jobType} on resource ${resourceId$1}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
|
|
@@ -11037,8 +11110,7 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11037
11110
|
...lifecycleBase,
|
|
11038
11111
|
result: {
|
|
11039
11112
|
declined: true,
|
|
11040
|
-
reason: source.declined
|
|
11041
|
-
message: DECLINE_MESSAGES[source.declined]
|
|
11113
|
+
reason: source.declined
|
|
11042
11114
|
}
|
|
11043
11115
|
});
|
|
11044
11116
|
adapter.completeJob();
|
|
@@ -11046,13 +11118,12 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11046
11118
|
}
|
|
11047
11119
|
ready = source;
|
|
11048
11120
|
}
|
|
11049
|
-
const onProgress = (percentage, message,
|
|
11121
|
+
const onProgress = (percentage, message, extra) => {
|
|
11050
11122
|
adapter.touchActivity();
|
|
11051
11123
|
emitEvent(session, "job:report-progress", {
|
|
11052
11124
|
...lifecycleBase,
|
|
11053
11125
|
percentage,
|
|
11054
11126
|
progress: {
|
|
11055
|
-
stage,
|
|
11056
11127
|
percentage,
|
|
11057
11128
|
message,
|
|
11058
11129
|
...annotationId ? { annotationId } : {},
|
|
@@ -11143,6 +11214,11 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11143
11214
|
});
|
|
11144
11215
|
adapter.completeJob();
|
|
11145
11216
|
} else if (jobType === "generation") {
|
|
11217
|
+
if (!isGenerationJobParams(job.params)) {
|
|
11218
|
+
throw new Error(
|
|
11219
|
+
`generation job ${job.jobId}: params do not satisfy GenerationJobParams (title, storageUri, and context are required)`
|
|
11220
|
+
);
|
|
11221
|
+
}
|
|
11146
11222
|
const genResult = await processGenerationJob(
|
|
11147
11223
|
inferenceClient,
|
|
11148
11224
|
job.params,
|
|
@@ -11150,6 +11226,7 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11150
11226
|
config.logger
|
|
11151
11227
|
);
|
|
11152
11228
|
const genParams = job.params;
|
|
11229
|
+
const genReferenceId = referenceIdOf(job);
|
|
11153
11230
|
const storageUri = deriveStorageUri(genResult.title, genResult.format);
|
|
11154
11231
|
const { resourceId: newResourceId } = await session.client.yield.resource({
|
|
11155
11232
|
name: genResult.title,
|
|
@@ -11157,13 +11234,13 @@ async function handleJobInner(adapter, config, job) {
|
|
|
11157
11234
|
format: genResult.format,
|
|
11158
11235
|
storageUri,
|
|
11159
11236
|
sourceResourceId: resourceId$1,
|
|
11160
|
-
...
|
|
11237
|
+
...genReferenceId ? { sourceAnnotationId: genReferenceId } : {},
|
|
11161
11238
|
...genParams.prompt ? { generationPrompt: genParams.prompt } : {},
|
|
11162
11239
|
...genParams.language ? { language: genParams.language } : {},
|
|
11163
11240
|
...genParams.entityTypes && genParams.entityTypes.length > 0 ? { entityTypes: genParams.entityTypes } : {},
|
|
11164
11241
|
generator
|
|
11165
11242
|
});
|
|
11166
|
-
if (!
|
|
11243
|
+
if (!genReferenceId) {
|
|
11167
11244
|
const { annotation: provenanceRef } = assembleAnnotation(
|
|
11168
11245
|
{
|
|
11169
11246
|
motivation: "linking",
|