@semiont/jobs 0.5.25 → 0.5.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -12
- package/dist/index.d.ts +27 -72
- package/dist/index.js +211 -147
- package/dist/index.js.map +1 -1
- package/dist/worker-main.js +246 -169
- package/dist/worker-main.js.map +1 -1
- package/package.json +8 -8
package/dist/index.js
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { promises, mkdtempSync, writeFileSync, readFileSync, rmSync } from 'fs';
|
|
2
2
|
import * as path from 'path';
|
|
3
3
|
import { join } from 'path';
|
|
4
|
-
import { jobId, deriveViews, reconcileSelector, GENERATABLE_MEDIA_TYPES, estimateTokens, chunkText, isObject, isString, getLocaleEnglishName
|
|
4
|
+
import { jobId, deriveViews, reconcileSelector, GENERATABLE_MEDIA_TYPES, estimateTokens, chunkText, isObject, isString, getLocaleEnglishName } from '@semiont/core';
|
|
5
|
+
import { withSpan } from '@semiont/observability';
|
|
5
6
|
import { execFileSync } from 'child_process';
|
|
6
7
|
import { tmpdir } from 'os';
|
|
7
8
|
import { withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
|
|
8
9
|
import '@semiont/event-sourcing';
|
|
9
|
-
import '@semiont/observability';
|
|
10
10
|
import '@semiont/sdk';
|
|
11
11
|
import '@semiont/http-transport';
|
|
12
12
|
|
|
@@ -426,10 +426,18 @@ function isFailedJob(job) {
|
|
|
426
426
|
function isCancelledJob(job) {
|
|
427
427
|
return job.status === "cancelled";
|
|
428
428
|
}
|
|
429
|
-
|
|
430
|
-
// src/workers/inference-call.ts
|
|
431
429
|
var INFERENCE_TIMEOUT_MS = 10 * 6e4;
|
|
432
|
-
|
|
430
|
+
var INFERENCE_HEARTBEAT_MS = 15e3;
|
|
431
|
+
function spanned(client, kind, maxTokens, work) {
|
|
432
|
+
return withSpan(`inference:${kind}`, work, {
|
|
433
|
+
attrs: {
|
|
434
|
+
"inference.provider": client.type,
|
|
435
|
+
"inference.model": client.modelId,
|
|
436
|
+
"inference.max_tokens": maxTokens
|
|
437
|
+
}
|
|
438
|
+
});
|
|
439
|
+
}
|
|
440
|
+
async function withTimeout(work, label, onHeartbeat) {
|
|
433
441
|
let timer;
|
|
434
442
|
const timedOut = new Promise((_, reject) => {
|
|
435
443
|
timer = setTimeout(() => {
|
|
@@ -439,6 +447,16 @@ async function withTimeout(work, label) {
|
|
|
439
447
|
}, INFERENCE_TIMEOUT_MS);
|
|
440
448
|
timer.unref?.();
|
|
441
449
|
});
|
|
450
|
+
let heartbeat;
|
|
451
|
+
if (onHeartbeat) {
|
|
452
|
+
heartbeat = setInterval(() => {
|
|
453
|
+
try {
|
|
454
|
+
onHeartbeat();
|
|
455
|
+
} catch {
|
|
456
|
+
}
|
|
457
|
+
}, INFERENCE_HEARTBEAT_MS);
|
|
458
|
+
heartbeat.unref?.();
|
|
459
|
+
}
|
|
442
460
|
try {
|
|
443
461
|
return await Promise.race([work, timedOut]);
|
|
444
462
|
} catch (err) {
|
|
@@ -447,19 +465,22 @@ async function withTimeout(work, label) {
|
|
|
447
465
|
throw err;
|
|
448
466
|
} finally {
|
|
449
467
|
clearTimeout(timer);
|
|
468
|
+
if (heartbeat) clearInterval(heartbeat);
|
|
450
469
|
}
|
|
451
470
|
}
|
|
452
|
-
function boundedGenerate(client, prompt, maxTokens, temperature,
|
|
453
|
-
return withTimeout(
|
|
454
|
-
client.generateText(prompt, maxTokens, temperature
|
|
455
|
-
`${client.type}:${client.modelId}
|
|
456
|
-
|
|
471
|
+
function boundedGenerate(client, prompt, maxTokens, temperature, onHeartbeat) {
|
|
472
|
+
return spanned(client, "text", maxTokens, () => withTimeout(
|
|
473
|
+
client.generateText(prompt, maxTokens, temperature),
|
|
474
|
+
`${client.type}:${client.modelId}`,
|
|
475
|
+
onHeartbeat
|
|
476
|
+
));
|
|
457
477
|
}
|
|
458
|
-
function
|
|
459
|
-
return withTimeout(
|
|
460
|
-
client.
|
|
461
|
-
`${client.type}:${client.modelId}
|
|
462
|
-
|
|
478
|
+
function boundedGenerateStructured(client, prompt, maxTokens, temperature, elementSchema, onHeartbeat) {
|
|
479
|
+
return spanned(client, "structured", maxTokens, () => withTimeout(
|
|
480
|
+
client.generateStructured(prompt, maxTokens, temperature, elementSchema),
|
|
481
|
+
`${client.type}:${client.modelId}`,
|
|
482
|
+
onHeartbeat
|
|
483
|
+
));
|
|
463
484
|
}
|
|
464
485
|
|
|
465
486
|
// src/workers/detection/detection-chunking.ts
|
|
@@ -791,32 +812,57 @@ Example format:
|
|
|
791
812
|
return prompt;
|
|
792
813
|
}
|
|
793
814
|
};
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
}
|
|
815
|
+
var COMMENT_ELEMENT_SCHEMA = {
|
|
816
|
+
type: "object",
|
|
817
|
+
properties: {
|
|
818
|
+
exact: { type: "string" },
|
|
819
|
+
prefix: { type: "string" },
|
|
820
|
+
suffix: { type: "string" },
|
|
821
|
+
comment: { type: "string" }
|
|
822
|
+
},
|
|
823
|
+
required: ["exact", "comment"],
|
|
824
|
+
additionalProperties: false
|
|
825
|
+
};
|
|
826
|
+
var HIGHLIGHT_ELEMENT_SCHEMA = {
|
|
827
|
+
type: "object",
|
|
828
|
+
properties: {
|
|
829
|
+
exact: { type: "string" },
|
|
830
|
+
prefix: { type: "string" },
|
|
831
|
+
suffix: { type: "string" }
|
|
832
|
+
},
|
|
833
|
+
required: ["exact"],
|
|
834
|
+
additionalProperties: false
|
|
835
|
+
};
|
|
836
|
+
var ASSESSMENT_ELEMENT_SCHEMA = {
|
|
837
|
+
type: "object",
|
|
838
|
+
properties: {
|
|
839
|
+
exact: { type: "string" },
|
|
840
|
+
prefix: { type: "string" },
|
|
841
|
+
suffix: { type: "string" },
|
|
842
|
+
assessment: { type: "string" }
|
|
843
|
+
},
|
|
844
|
+
required: ["exact", "assessment"],
|
|
845
|
+
additionalProperties: false
|
|
846
|
+
};
|
|
847
|
+
var TAG_ELEMENT_SCHEMA = {
|
|
848
|
+
type: "object",
|
|
849
|
+
properties: {
|
|
850
|
+
exact: { type: "string" },
|
|
851
|
+
prefix: { type: "string" },
|
|
852
|
+
suffix: { type: "string" }
|
|
853
|
+
},
|
|
854
|
+
required: ["exact"],
|
|
855
|
+
additionalProperties: false
|
|
856
|
+
};
|
|
809
857
|
var MotivationParsers = class {
|
|
810
858
|
/**
|
|
811
|
-
*
|
|
859
|
+
* Validate and reconcile structured comment elements.
|
|
812
860
|
*
|
|
813
|
-
* @param
|
|
861
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
814
862
|
* @param content - Original content to validate offsets against
|
|
815
863
|
* @returns Array of validated comment matches
|
|
816
|
-
* @throws if the response is not a parseable JSON array
|
|
817
864
|
*/
|
|
818
|
-
static parseComments(
|
|
819
|
-
const parsed = parseJsonArray(response, "comment");
|
|
865
|
+
static parseComments(parsed, content) {
|
|
820
866
|
const valid = parsed.filter(
|
|
821
867
|
(c) => isObject(c) && isString(c.exact) && isString(c.comment) && c.comment.trim().length > 0
|
|
822
868
|
);
|
|
@@ -845,15 +891,13 @@ var MotivationParsers = class {
|
|
|
845
891
|
return validatedComments;
|
|
846
892
|
}
|
|
847
893
|
/**
|
|
848
|
-
*
|
|
894
|
+
* Validate and reconcile structured highlight elements.
|
|
849
895
|
*
|
|
850
|
-
* @param
|
|
896
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
851
897
|
* @param content - Original content to validate offsets against
|
|
852
898
|
* @returns Array of validated highlight matches
|
|
853
|
-
* @throws if the response is not a parseable JSON array
|
|
854
899
|
*/
|
|
855
|
-
static parseHighlights(
|
|
856
|
-
const parsed = parseJsonArray(response, "highlight");
|
|
900
|
+
static parseHighlights(parsed, content) {
|
|
857
901
|
const highlights = parsed.filter(
|
|
858
902
|
(h) => isObject(h) && isString(h.exact)
|
|
859
903
|
);
|
|
@@ -880,15 +924,13 @@ var MotivationParsers = class {
|
|
|
880
924
|
return validatedHighlights;
|
|
881
925
|
}
|
|
882
926
|
/**
|
|
883
|
-
*
|
|
927
|
+
* Validate and reconcile structured assessment elements.
|
|
884
928
|
*
|
|
885
|
-
* @param
|
|
929
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
886
930
|
* @param content - Original content to validate offsets against
|
|
887
931
|
* @returns Array of validated assessment matches
|
|
888
|
-
* @throws if the response is not a parseable JSON array
|
|
889
932
|
*/
|
|
890
|
-
static parseAssessments(
|
|
891
|
-
const parsed = parseJsonArray(response, "assessment");
|
|
933
|
+
static parseAssessments(parsed, content) {
|
|
892
934
|
const assessments = parsed.filter(
|
|
893
935
|
(a) => isObject(a) && isString(a.exact) && isString(a.assessment)
|
|
894
936
|
);
|
|
@@ -916,14 +958,13 @@ var MotivationParsers = class {
|
|
|
916
958
|
return validatedAssessments;
|
|
917
959
|
}
|
|
918
960
|
/**
|
|
919
|
-
*
|
|
961
|
+
* Validate structured tag elements into raw, pre-reconciliation tag inputs.
|
|
920
962
|
* Reconciliation happens in `validateTagOffsets`, which adds `start`/`end`
|
|
921
963
|
* by anchoring `exact` against the source content.
|
|
922
964
|
*
|
|
923
|
-
* @
|
|
965
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
924
966
|
*/
|
|
925
|
-
static parseTags(
|
|
926
|
-
const parsed = parseJsonArray(response, "tag");
|
|
967
|
+
static parseTags(parsed) {
|
|
927
968
|
const valid = parsed.filter(
|
|
928
969
|
(t) => isObject(t) && isString(t.exact) && t.exact.trim().length > 0
|
|
929
970
|
);
|
|
@@ -970,24 +1011,26 @@ function assertNotTruncated(response, motivation, chunk, totalChunks, outputBudg
|
|
|
970
1011
|
throw new Error(`${motivation} detection response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
|
|
971
1012
|
}
|
|
972
1013
|
}
|
|
973
|
-
async function detectInChunks(client, content, buildPrompt, temperature, motivation, parse,
|
|
1014
|
+
async function detectInChunks(client, content, buildPrompt, temperature, motivation, elementSchema, parse, onActivity) {
|
|
974
1015
|
const limits = await client.limits();
|
|
975
1016
|
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
976
1017
|
const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
|
|
977
1018
|
const chunks = chunkText(content, chunking);
|
|
978
1019
|
const collected = [];
|
|
979
1020
|
for (let i = 0; i < chunks.length; i++) {
|
|
980
|
-
const response = await
|
|
1021
|
+
const response = await boundedGenerateStructured(
|
|
981
1022
|
client,
|
|
982
1023
|
buildPrompt(chunks[i]),
|
|
983
1024
|
outputBudget,
|
|
984
1025
|
temperature,
|
|
985
|
-
|
|
1026
|
+
elementSchema,
|
|
1027
|
+
// Still alive, same position (a long single call is otherwise silent).
|
|
1028
|
+
() => onActivity?.(i, chunks.length)
|
|
986
1029
|
);
|
|
987
1030
|
assertNotTruncated(response, motivation, i + 1, chunks.length, outputBudget);
|
|
988
|
-
collected.push(...parse(response.
|
|
1031
|
+
collected.push(...parse(response.items));
|
|
989
1032
|
if (i < chunks.length - 1) {
|
|
990
|
-
|
|
1033
|
+
onActivity?.(i + 1, chunks.length);
|
|
991
1034
|
}
|
|
992
1035
|
}
|
|
993
1036
|
return collected;
|
|
@@ -1001,15 +1044,16 @@ var AnnotationDetection = class {
|
|
|
1001
1044
|
* (source-resource locale). See `types.ts` "Locale conventions" for the
|
|
1002
1045
|
* full discussion.
|
|
1003
1046
|
*/
|
|
1004
|
-
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage,
|
|
1047
|
+
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onActivity) {
|
|
1005
1048
|
return detectInChunks(
|
|
1006
1049
|
client,
|
|
1007
1050
|
content,
|
|
1008
1051
|
(chunk) => MotivationPrompts.buildCommentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
|
|
1009
1052
|
0.4,
|
|
1010
1053
|
"comment",
|
|
1011
|
-
|
|
1012
|
-
|
|
1054
|
+
COMMENT_ELEMENT_SCHEMA,
|
|
1055
|
+
(items) => MotivationParsers.parseComments(items, content),
|
|
1056
|
+
onActivity
|
|
1013
1057
|
);
|
|
1014
1058
|
}
|
|
1015
1059
|
/**
|
|
@@ -1019,15 +1063,16 @@ var AnnotationDetection = class {
|
|
|
1019
1063
|
* applies, used in the prompt so the LLM analyzes non-English source
|
|
1020
1064
|
* correctly.
|
|
1021
1065
|
*/
|
|
1022
|
-
static async detectHighlights(content, client, instructions, density, sourceLanguage,
|
|
1066
|
+
static async detectHighlights(content, client, instructions, density, sourceLanguage, onActivity) {
|
|
1023
1067
|
return detectInChunks(
|
|
1024
1068
|
client,
|
|
1025
1069
|
content,
|
|
1026
1070
|
(chunk) => MotivationPrompts.buildHighlightPrompt(chunk, instructions, density, sourceLanguage),
|
|
1027
1071
|
0.3,
|
|
1028
1072
|
"highlight",
|
|
1029
|
-
|
|
1030
|
-
|
|
1073
|
+
HIGHLIGHT_ELEMENT_SCHEMA,
|
|
1074
|
+
(items) => MotivationParsers.parseHighlights(items, content),
|
|
1075
|
+
onActivity
|
|
1031
1076
|
);
|
|
1032
1077
|
}
|
|
1033
1078
|
/**
|
|
@@ -1037,15 +1082,16 @@ var AnnotationDetection = class {
|
|
|
1037
1082
|
* (annotation body locale). `sourceLanguage` is the locale of the content
|
|
1038
1083
|
* being analyzed (source-resource locale).
|
|
1039
1084
|
*/
|
|
1040
|
-
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage,
|
|
1085
|
+
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onActivity) {
|
|
1041
1086
|
return detectInChunks(
|
|
1042
1087
|
client,
|
|
1043
1088
|
content,
|
|
1044
1089
|
(chunk) => MotivationPrompts.buildAssessmentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
|
|
1045
1090
|
0.3,
|
|
1046
1091
|
"assessment",
|
|
1047
|
-
|
|
1048
|
-
|
|
1092
|
+
ASSESSMENT_ELEMENT_SCHEMA,
|
|
1093
|
+
(items) => MotivationParsers.parseAssessments(items, content),
|
|
1094
|
+
onActivity
|
|
1049
1095
|
);
|
|
1050
1096
|
}
|
|
1051
1097
|
/**
|
|
@@ -1060,7 +1106,7 @@ var AnnotationDetection = class {
|
|
|
1060
1106
|
* identifiers, not LLM-generated text — so it's consumed at the body-stamp
|
|
1061
1107
|
* site, not here.
|
|
1062
1108
|
*/
|
|
1063
|
-
static async detectTags(content, client, schema, category, sourceLanguage,
|
|
1109
|
+
static async detectTags(content, client, schema, category, sourceLanguage, onActivity) {
|
|
1064
1110
|
const categoryInfo = schema.tags.find((t) => t.name === category);
|
|
1065
1111
|
if (!categoryInfo) {
|
|
1066
1112
|
throw new Error(`Invalid category "${category}" for schema ${schema.id}`);
|
|
@@ -1080,13 +1126,25 @@ var AnnotationDetection = class {
|
|
|
1080
1126
|
),
|
|
1081
1127
|
0.2,
|
|
1082
1128
|
"tag",
|
|
1083
|
-
|
|
1084
|
-
|
|
1129
|
+
TAG_ELEMENT_SCHEMA,
|
|
1130
|
+
(items) => MotivationParsers.parseTags(items),
|
|
1131
|
+
onActivity
|
|
1085
1132
|
);
|
|
1086
1133
|
return MotivationParsers.validateTagOffsets(parsedTags, content, category);
|
|
1087
1134
|
}
|
|
1088
1135
|
};
|
|
1089
|
-
|
|
1136
|
+
var ENTITY_ELEMENT_SCHEMA = {
|
|
1137
|
+
type: "object",
|
|
1138
|
+
properties: {
|
|
1139
|
+
exact: { type: "string" },
|
|
1140
|
+
entityType: { type: "string" },
|
|
1141
|
+
prefix: { type: "string" },
|
|
1142
|
+
suffix: { type: "string" }
|
|
1143
|
+
},
|
|
1144
|
+
required: ["exact", "entityType"],
|
|
1145
|
+
additionalProperties: false
|
|
1146
|
+
};
|
|
1147
|
+
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger, sourceLanguage, onActivity) {
|
|
1090
1148
|
const entityTypesDescription = entityTypes.map((et) => {
|
|
1091
1149
|
if (typeof et === "string") {
|
|
1092
1150
|
return et;
|
|
@@ -1147,50 +1205,28 @@ Example output:
|
|
|
1147
1205
|
});
|
|
1148
1206
|
const collected = [];
|
|
1149
1207
|
for (let i = 0; i < chunks.length; i++) {
|
|
1150
|
-
const response = await
|
|
1208
|
+
const response = await boundedGenerateStructured(
|
|
1151
1209
|
client,
|
|
1152
1210
|
buildPrompt(chunks[i]),
|
|
1153
1211
|
outputBudget,
|
|
1154
1212
|
0.3,
|
|
1155
1213
|
// Lower temperature for more consistent extraction
|
|
1156
|
-
|
|
1157
|
-
//
|
|
1158
|
-
//
|
|
1159
|
-
|
|
1160
|
-
// governs *what* the JSON contains; `format: 'json'` governs that
|
|
1161
|
-
// it's syntactically valid.
|
|
1162
|
-
{ format: "json" }
|
|
1214
|
+
ENTITY_ELEMENT_SCHEMA,
|
|
1215
|
+
// Still alive, same position: a long single call would otherwise emit
|
|
1216
|
+
// nothing at all between start and finish.
|
|
1217
|
+
() => onActivity?.(i, chunks.length)
|
|
1163
1218
|
);
|
|
1164
1219
|
logger.debug("Got entity extraction response", {
|
|
1165
1220
|
chunk: i + 1,
|
|
1166
1221
|
chunks: chunks.length,
|
|
1167
|
-
|
|
1222
|
+
items: response.items.length
|
|
1168
1223
|
});
|
|
1169
1224
|
if (response.stopReason === "max_tokens") {
|
|
1170
1225
|
const errorMsg = `Entity extraction response truncated (max_tokens) on chunk ${i + 1}/${chunks.length} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than dropping annotations.`;
|
|
1171
|
-
logger.error(errorMsg, {
|
|
1226
|
+
logger.error(errorMsg, { items: response.items.length });
|
|
1172
1227
|
throw new Error(errorMsg);
|
|
1173
1228
|
}
|
|
1174
|
-
|
|
1175
|
-
try {
|
|
1176
|
-
entities = JSON.parse(response.text.trim());
|
|
1177
|
-
} catch (error) {
|
|
1178
|
-
logger.error("Failed to parse entity extraction response", {
|
|
1179
|
-
error: error instanceof Error ? error.message : String(error),
|
|
1180
|
-
response: response.text.slice(0, 500)
|
|
1181
|
-
});
|
|
1182
|
-
throw new Error("Failed to parse entity extraction response", {
|
|
1183
|
-
cause: error instanceof Error ? error : new Error(String(error))
|
|
1184
|
-
});
|
|
1185
|
-
}
|
|
1186
|
-
if (!isArray(entities)) {
|
|
1187
|
-
logger.error("Failed to parse entity extraction response: expected a JSON array", {
|
|
1188
|
-
response: response.text.slice(0, 500)
|
|
1189
|
-
});
|
|
1190
|
-
throw new Error("Failed to parse entity extraction response: expected a JSON array");
|
|
1191
|
-
}
|
|
1192
|
-
logger.debug("Parsed entities from AI response", { chunk: i + 1, count: entities.length });
|
|
1193
|
-
for (const e of entities) {
|
|
1229
|
+
for (const e of response.items) {
|
|
1194
1230
|
if (isObject(e) && isString(e.exact) && isString(e.entityType)) {
|
|
1195
1231
|
collected.push({
|
|
1196
1232
|
exact: e.exact,
|
|
@@ -1203,7 +1239,7 @@ Example output:
|
|
|
1203
1239
|
}
|
|
1204
1240
|
}
|
|
1205
1241
|
if (i < chunks.length - 1) {
|
|
1206
|
-
|
|
1242
|
+
onActivity?.(i + 1, chunks.length);
|
|
1207
1243
|
}
|
|
1208
1244
|
}
|
|
1209
1245
|
return collected;
|
|
@@ -1547,30 +1583,39 @@ function dedupeAnnotations(annotations) {
|
|
|
1547
1583
|
return out;
|
|
1548
1584
|
}
|
|
1549
1585
|
async function processHighlightJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
1550
|
-
|
|
1551
|
-
onProgress(
|
|
1586
|
+
const echo = detectionEcho(params);
|
|
1587
|
+
onProgress(10, { code: "loading" }, echo);
|
|
1588
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
1552
1589
|
const highlights = await AnnotationDetection.detectHighlights(
|
|
1553
1590
|
content,
|
|
1554
1591
|
inferenceClient,
|
|
1555
1592
|
params.instructions,
|
|
1556
1593
|
params.density,
|
|
1557
1594
|
params.sourceLanguage,
|
|
1558
|
-
//
|
|
1559
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
1595
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
1596
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
1560
1597
|
);
|
|
1561
|
-
onProgress(60,
|
|
1598
|
+
onProgress(60, { code: "creating-annotations", count: highlights.length }, echo);
|
|
1562
1599
|
const annotations = dedupeAnnotations(highlights.map(
|
|
1563
1600
|
(h) => buildAnnotation("highlighting", h)
|
|
1564
1601
|
));
|
|
1565
|
-
onProgress(100,
|
|
1602
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "highlight" }, echo);
|
|
1566
1603
|
return {
|
|
1567
1604
|
annotations,
|
|
1568
1605
|
result: { highlightsFound: highlights.length, highlightsCreated: annotations.length }
|
|
1569
1606
|
};
|
|
1570
1607
|
}
|
|
1608
|
+
function detectionEcho(p) {
|
|
1609
|
+
const requestParams = [];
|
|
1610
|
+
if (p.instructions?.trim()) requestParams.push({ label: "instructions", value: p.instructions.trim() });
|
|
1611
|
+
if (p.tone?.trim()) requestParams.push({ label: "tone", value: p.tone.trim() });
|
|
1612
|
+
if (p.density !== void 0) requestParams.push({ label: "density", value: String(p.density) });
|
|
1613
|
+
return requestParams.length > 0 ? { requestParams } : {};
|
|
1614
|
+
}
|
|
1571
1615
|
async function processCommentJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
1572
|
-
|
|
1573
|
-
onProgress(
|
|
1616
|
+
const echo = detectionEcho(params);
|
|
1617
|
+
onProgress(10, { code: "loading" }, echo);
|
|
1618
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
1574
1619
|
const comments = await AnnotationDetection.detectComments(
|
|
1575
1620
|
content,
|
|
1576
1621
|
inferenceClient,
|
|
@@ -1579,10 +1624,10 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
1579
1624
|
params.density,
|
|
1580
1625
|
params.language,
|
|
1581
1626
|
params.sourceLanguage,
|
|
1582
|
-
//
|
|
1583
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
1627
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
1628
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
1584
1629
|
);
|
|
1585
|
-
onProgress(60,
|
|
1630
|
+
onProgress(60, { code: "creating-annotations", count: comments.length }, echo);
|
|
1586
1631
|
const bodyLanguage = params.language ?? "en";
|
|
1587
1632
|
const annotations = dedupeAnnotations(comments.map(
|
|
1588
1633
|
(c) => (
|
|
@@ -1594,15 +1639,16 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
1594
1639
|
])
|
|
1595
1640
|
)
|
|
1596
1641
|
));
|
|
1597
|
-
onProgress(100,
|
|
1642
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "comment" }, echo);
|
|
1598
1643
|
return {
|
|
1599
1644
|
annotations,
|
|
1600
1645
|
result: { commentsFound: comments.length, commentsCreated: annotations.length }
|
|
1601
1646
|
};
|
|
1602
1647
|
}
|
|
1603
1648
|
async function processAssessmentJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
1604
|
-
|
|
1605
|
-
onProgress(
|
|
1649
|
+
const echo = detectionEcho(params);
|
|
1650
|
+
onProgress(10, { code: "loading" }, echo);
|
|
1651
|
+
onProgress(30, { code: "analyzing" }, echo);
|
|
1606
1652
|
const assessments = await AnnotationDetection.detectAssessments(
|
|
1607
1653
|
content,
|
|
1608
1654
|
inferenceClient,
|
|
@@ -1611,10 +1657,10 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
1611
1657
|
params.density,
|
|
1612
1658
|
params.language,
|
|
1613
1659
|
params.sourceLanguage,
|
|
1614
|
-
//
|
|
1615
|
-
(completed, total) => onProgress(30 + Math.round(completed / total * 30),
|
|
1660
|
+
// Liveness (chunk boundaries + in-flight heartbeat): 30–60 band.
|
|
1661
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), { code: "analyzing" }, echo)
|
|
1616
1662
|
);
|
|
1617
|
-
onProgress(60,
|
|
1663
|
+
onProgress(60, { code: "creating-annotations", count: assessments.length }, echo);
|
|
1618
1664
|
const bodyLanguage = params.language ?? "en";
|
|
1619
1665
|
const annotations = dedupeAnnotations(assessments.map(
|
|
1620
1666
|
(a) => (
|
|
@@ -1633,7 +1679,7 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
1633
1679
|
})
|
|
1634
1680
|
)
|
|
1635
1681
|
));
|
|
1636
|
-
onProgress(100,
|
|
1682
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "assessment" }, echo);
|
|
1637
1683
|
return {
|
|
1638
1684
|
annotations,
|
|
1639
1685
|
result: { assessmentsFound: assessments.length, assessmentsCreated: annotations.length }
|
|
@@ -1641,25 +1687,27 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
1641
1687
|
}
|
|
1642
1688
|
async function processReferenceJob(content, inferenceClient, params, buildAnnotation, onProgress, logger) {
|
|
1643
1689
|
const entityTypeNames = params.entityTypes.map(String);
|
|
1644
|
-
const requestParams = [{ label: "
|
|
1645
|
-
const
|
|
1690
|
+
const requestParams = [{ label: "entity-types", value: entityTypeNames.join(", ") }];
|
|
1691
|
+
const completedItems = [];
|
|
1646
1692
|
let totalFound = 0;
|
|
1647
1693
|
let totalEmitted = 0;
|
|
1648
1694
|
let errors = 0;
|
|
1649
1695
|
const allAnnotations = [];
|
|
1650
|
-
onProgress(10,
|
|
1696
|
+
onProgress(10, { code: "loading" }, { requestParams });
|
|
1651
1697
|
const bodyLanguage = params.language ?? "en";
|
|
1652
1698
|
for (let i = 0; i < entityTypeNames.length; i++) {
|
|
1653
1699
|
const entityTypeName = entityTypeNames[i];
|
|
1654
1700
|
if (!entityTypeName) continue;
|
|
1655
1701
|
const pct = 20 + Math.round(i / entityTypeNames.length * 60);
|
|
1656
|
-
onProgress(pct,
|
|
1657
|
-
|
|
1658
|
-
|
|
1659
|
-
|
|
1702
|
+
onProgress(pct, { code: "detecting-entities", entityType: entityTypeName }, {
|
|
1703
|
+
// One vocabulary for "what is in flight" (CLEAN-PROGRESS D2): the entity
|
|
1704
|
+
// type is KB data, `kind` is the code the client localizes around it.
|
|
1705
|
+
current: { kind: "entity-type", value: entityTypeName },
|
|
1706
|
+
processed: i,
|
|
1707
|
+
total: entityTypeNames.length,
|
|
1660
1708
|
entitiesFound: totalFound,
|
|
1661
1709
|
entitiesEmitted: totalEmitted,
|
|
1662
|
-
|
|
1710
|
+
completedItems: [...completedItems],
|
|
1663
1711
|
requestParams
|
|
1664
1712
|
});
|
|
1665
1713
|
const extractedEntities = await extractEntities(
|
|
@@ -1669,25 +1717,28 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
1669
1717
|
params.includeDescriptiveReferences ?? false,
|
|
1670
1718
|
logger,
|
|
1671
1719
|
params.sourceLanguage,
|
|
1672
|
-
//
|
|
1673
|
-
//
|
|
1674
|
-
//
|
|
1675
|
-
//
|
|
1720
|
+
// Liveness: fires at chunk boundaries AND every ~15 s while a single
|
|
1721
|
+
// inference call is in flight (DETECTION-HEARTBEAT). Progress feeds the
|
|
1722
|
+
// stall watchdog, the janitor, AND the client's inter-emission timeout,
|
|
1723
|
+
// so a long single-chunk call must not be silent. Percentage
|
|
1724
|
+
// interpolates within this entity type's band of the 20–80 range; a
|
|
1725
|
+
// heartbeat repeats the current position rather than inventing an
|
|
1726
|
+
// advance.
|
|
1676
1727
|
(completed, total) => {
|
|
1677
1728
|
const interpolated = 20 + Math.round((i + completed / total) / entityTypeNames.length * 60);
|
|
1678
|
-
onProgress(interpolated,
|
|
1679
|
-
|
|
1680
|
-
|
|
1681
|
-
|
|
1729
|
+
onProgress(interpolated, { code: "detecting-entities", entityType: entityTypeName }, {
|
|
1730
|
+
current: { kind: "entity-type", value: entityTypeName },
|
|
1731
|
+
processed: i,
|
|
1732
|
+
total: entityTypeNames.length,
|
|
1682
1733
|
entitiesFound: totalFound,
|
|
1683
1734
|
entitiesEmitted: totalEmitted,
|
|
1684
|
-
|
|
1735
|
+
completedItems: [...completedItems],
|
|
1685
1736
|
requestParams
|
|
1686
1737
|
});
|
|
1687
1738
|
}
|
|
1688
1739
|
);
|
|
1689
1740
|
totalFound += extractedEntities.length;
|
|
1690
|
-
|
|
1741
|
+
completedItems.push({ value: entityTypeName, foundCount: extractedEntities.length });
|
|
1691
1742
|
const unresolvedBody = [
|
|
1692
1743
|
{ type: "TextualBody", value: entityTypeName, purpose: "tagging", format: "text/plain", language: bodyLanguage }
|
|
1693
1744
|
];
|
|
@@ -1718,36 +1769,49 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
1718
1769
|
}
|
|
1719
1770
|
}
|
|
1720
1771
|
const annotations = dedupeAnnotations(allAnnotations);
|
|
1721
|
-
onProgress(100,
|
|
1772
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "reference" }, { requestParams });
|
|
1722
1773
|
return {
|
|
1723
1774
|
annotations,
|
|
1724
1775
|
result: { totalFound, totalEmitted: annotations.length, errors }
|
|
1725
1776
|
};
|
|
1726
1777
|
}
|
|
1727
1778
|
async function processTagJob(content, inferenceClient, params, buildAnnotation, onProgress) {
|
|
1728
|
-
onProgress(10,
|
|
1729
|
-
onProgress(30,
|
|
1779
|
+
onProgress(10, { code: "loading" });
|
|
1780
|
+
onProgress(30, { code: "analyzing-tags" });
|
|
1730
1781
|
const allTags = [];
|
|
1782
|
+
const completedItems = [];
|
|
1731
1783
|
for (let c = 0; c < params.categories.length; c++) {
|
|
1732
1784
|
const category = params.categories[c];
|
|
1785
|
+
const position = () => ({
|
|
1786
|
+
current: { kind: "category", value: category },
|
|
1787
|
+
processed: c,
|
|
1788
|
+
total: params.categories.length,
|
|
1789
|
+
completedItems: [...completedItems]
|
|
1790
|
+
});
|
|
1791
|
+
onProgress(
|
|
1792
|
+
30 + Math.round(c / params.categories.length * 30),
|
|
1793
|
+
{ code: "analyzing-tags" },
|
|
1794
|
+
position()
|
|
1795
|
+
);
|
|
1733
1796
|
const categoryTags = await AnnotationDetection.detectTags(
|
|
1734
1797
|
content,
|
|
1735
1798
|
inferenceClient,
|
|
1736
1799
|
params.schema,
|
|
1737
1800
|
category,
|
|
1738
1801
|
params.sourceLanguage,
|
|
1739
|
-
//
|
|
1740
|
-
//
|
|
1802
|
+
// Liveness (chunk boundaries + in-flight heartbeat): this category's
|
|
1803
|
+
// slice of the 30–60 band.
|
|
1741
1804
|
(completed, total) => onProgress(
|
|
1742
1805
|
30 + Math.round((c + completed / total) / params.categories.length * 30),
|
|
1743
|
-
|
|
1744
|
-
|
|
1806
|
+
{ code: "analyzing-tags" },
|
|
1807
|
+
position()
|
|
1745
1808
|
)
|
|
1746
1809
|
);
|
|
1810
|
+
completedItems.push({ value: category, foundCount: categoryTags.length });
|
|
1747
1811
|
allTags.push(...categoryTags);
|
|
1748
1812
|
}
|
|
1749
1813
|
const tags = allTags;
|
|
1750
|
-
onProgress(60,
|
|
1814
|
+
onProgress(60, { code: "creating-tag-annotations", count: tags.length });
|
|
1751
1815
|
const bodyLanguage = params.language ?? "en";
|
|
1752
1816
|
const annotations = dedupeAnnotations(tags.map((t) => {
|
|
1753
1817
|
const category = t.category ?? "unknown";
|
|
@@ -1762,7 +1826,7 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
1762
1826
|
const category = Array.isArray(body) && typeof body[0]?.value === "string" ? body[0].value : "unknown";
|
|
1763
1827
|
byCategory[category] = (byCategory[category] ?? 0) + 1;
|
|
1764
1828
|
}
|
|
1765
|
-
onProgress(100,
|
|
1829
|
+
onProgress(100, { code: "complete-created", count: annotations.length, kind: "tag" });
|
|
1766
1830
|
return {
|
|
1767
1831
|
annotations,
|
|
1768
1832
|
result: { tagsFound: tags.length, tagsCreated: annotations.length, byCategory }
|
|
@@ -1785,7 +1849,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger)
|
|
|
1785
1849
|
const title = params.title ?? "Untitled";
|
|
1786
1850
|
const entityTypes = (params.entityTypes ?? []).map(String);
|
|
1787
1851
|
if (outputMediaType === "application/pdf") {
|
|
1788
|
-
onProgress(5,
|
|
1852
|
+
onProgress(5, { code: "generating-resource" });
|
|
1789
1853
|
const validIds = params.cite === true ? collectContextResourceIds(params.context) : null;
|
|
1790
1854
|
let generated2 = await generateResourceFromTopic(
|
|
1791
1855
|
title,
|
|
@@ -1850,7 +1914,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger)
|
|
|
1850
1914
|
);
|
|
1851
1915
|
}
|
|
1852
1916
|
assertWithinOutputBudget(compiled.pdf.byteLength);
|
|
1853
|
-
onProgress(95,
|
|
1917
|
+
onProgress(95, { code: "creating-resource" });
|
|
1854
1918
|
return {
|
|
1855
1919
|
content: compiled.pdf,
|
|
1856
1920
|
title: generated2.title ?? title,
|
|
@@ -1862,7 +1926,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger)
|
|
|
1862
1926
|
}
|
|
1863
1927
|
};
|
|
1864
1928
|
}
|
|
1865
|
-
onProgress(5,
|
|
1929
|
+
onProgress(5, { code: "generating-resource" });
|
|
1866
1930
|
const generated = await generateResourceFromTopic(
|
|
1867
1931
|
title,
|
|
1868
1932
|
entityTypes,
|
|
@@ -1886,7 +1950,7 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger)
|
|
|
1886
1950
|
content = resolved.content;
|
|
1887
1951
|
citations = resolved.citations;
|
|
1888
1952
|
}
|
|
1889
|
-
onProgress(95,
|
|
1953
|
+
onProgress(95, { code: "creating-resource" });
|
|
1890
1954
|
const artifact = new TextEncoder().encode(content);
|
|
1891
1955
|
assertWithinOutputBudget(artifact.byteLength);
|
|
1892
1956
|
return {
|