@semiont/jobs 0.5.23 → 0.5.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +15 -20
- package/dist/index.js +364 -111
- package/dist/index.js.map +1 -1
- package/dist/worker-main.js +523 -196
- package/dist/worker-main.js.map +1 -1
- package/package.json +10 -9
package/dist/worker-main.js
CHANGED
|
@@ -1,14 +1,15 @@
|
|
|
1
|
-
import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest,
|
|
2
|
-
import { deriveStorageUri, extractPdfTextLayer,
|
|
1
|
+
import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText, isArray, isObject, isString, deriveViews } from '@semiont/core';
|
|
2
|
+
import { anchoredTextStoreOverTransport, deriveStorageUri, extractPdfTextLayer, EXTRACTORS, calculateChecksum, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
|
|
3
|
+
import { execFileSync } from 'child_process';
|
|
4
|
+
import { existsSync, readFileSync, mkdtempSync, writeFileSync, rmSync } from 'fs';
|
|
5
|
+
import { homedir, hostname, tmpdir } from 'os';
|
|
6
|
+
import { join } from 'path';
|
|
3
7
|
import { generateAnnotationId } from '@semiont/event-sourcing';
|
|
4
8
|
import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
|
|
5
|
-
import { homedir, hostname } from 'os';
|
|
6
9
|
import { InMemorySessionStorage, setStoredSession, kbBackendUrl, SemiontClient, SemiontSession } from '@semiont/sdk';
|
|
7
10
|
import { HttpTransport, HttpContentTransport } from '@semiont/http-transport';
|
|
8
11
|
import { createInferenceClient } from '@semiont/inference';
|
|
9
12
|
import { createServer } from 'http';
|
|
10
|
-
import { existsSync, readFileSync } from 'fs';
|
|
11
|
-
import { join } from 'path';
|
|
12
13
|
import { createProcessLogger } from '@semiont/observability/process-logger';
|
|
13
14
|
|
|
14
15
|
var __create = Object.create;
|
|
@@ -9329,6 +9330,25 @@ function createJobClaimAdapter(options) {
|
|
|
9329
9330
|
};
|
|
9330
9331
|
}
|
|
9331
9332
|
|
|
9333
|
+
// src/types.ts
|
|
9334
|
+
var JOB_TYPES = /* @__PURE__ */ new Set([
|
|
9335
|
+
"reference-annotation",
|
|
9336
|
+
"generation",
|
|
9337
|
+
"highlight-annotation",
|
|
9338
|
+
"assessment-annotation",
|
|
9339
|
+
"comment-annotation",
|
|
9340
|
+
"tag-annotation"
|
|
9341
|
+
]);
|
|
9342
|
+
function isJobType(value) {
|
|
9343
|
+
return JOB_TYPES.has(value);
|
|
9344
|
+
}
|
|
9345
|
+
function asJobParams(params) {
|
|
9346
|
+
if (typeof params.resourceId !== "string") {
|
|
9347
|
+
throw new Error("Job params are missing a resourceId");
|
|
9348
|
+
}
|
|
9349
|
+
return params;
|
|
9350
|
+
}
|
|
9351
|
+
|
|
9332
9352
|
// src/workers/inference-call.ts
|
|
9333
9353
|
var INFERENCE_TIMEOUT_MS = 10 * 6e4;
|
|
9334
9354
|
async function withTimeout(work, label) {
|
|
@@ -9363,6 +9383,39 @@ function boundedGenerateWithMetadata(client, prompt, maxTokens, temperature, opt
|
|
|
9363
9383
|
`${client.type}:${client.modelId}`
|
|
9364
9384
|
);
|
|
9365
9385
|
}
|
|
9386
|
+
|
|
9387
|
+
// src/workers/detection/detection-chunking.ts
|
|
9388
|
+
var SELECTOR_CONTEXT_CHARS = 64;
|
|
9389
|
+
var OVERLAP_CHARS = SELECTOR_CONTEXT_CHARS + // prefix
|
|
9390
|
+
SELECTOR_CONTEXT_CHARS + // suffix
|
|
9391
|
+
2 * SELECTOR_CONTEXT_CHARS;
|
|
9392
|
+
var OVERLAP_TOKENS = Math.ceil(OVERLAP_CHARS / 4);
|
|
9393
|
+
function deriveDetectionBudget(limits, scaffoldTokens) {
|
|
9394
|
+
const { contextTokens, maxOutputTokens } = limits;
|
|
9395
|
+
const available = contextTokens - scaffoldTokens;
|
|
9396
|
+
let inputBudget;
|
|
9397
|
+
let outputBudget;
|
|
9398
|
+
if (maxOutputTokens >= contextTokens) {
|
|
9399
|
+
inputBudget = Math.floor(available / 3);
|
|
9400
|
+
outputBudget = available - inputBudget;
|
|
9401
|
+
} else {
|
|
9402
|
+
outputBudget = maxOutputTokens;
|
|
9403
|
+
inputBudget = contextTokens - outputBudget - scaffoldTokens;
|
|
9404
|
+
if (inputBudget <= 0) {
|
|
9405
|
+
inputBudget = Math.floor(available / 3);
|
|
9406
|
+
outputBudget = available - inputBudget;
|
|
9407
|
+
}
|
|
9408
|
+
}
|
|
9409
|
+
if (inputBudget <= OVERLAP_TOKENS) {
|
|
9410
|
+
throw new Error(
|
|
9411
|
+
`Inference window too small for detection: context ${contextTokens} tokens minus scaffold ${scaffoldTokens} leaves an input budget of ${inputBudget} (need > ${OVERLAP_TOKENS}). Use a model with a larger context window or reduce the prompt scaffold.`
|
|
9412
|
+
);
|
|
9413
|
+
}
|
|
9414
|
+
return {
|
|
9415
|
+
chunking: { chunkSize: inputBudget, overlap: OVERLAP_TOKENS },
|
|
9416
|
+
outputBudget
|
|
9417
|
+
};
|
|
9418
|
+
}
|
|
9366
9419
|
function languageName(tag) {
|
|
9367
9420
|
return getLocaleEnglishName(tag) || tag;
|
|
9368
9421
|
}
|
|
@@ -9382,7 +9435,9 @@ var MotivationPrompts = class {
|
|
|
9382
9435
|
/**
|
|
9383
9436
|
* Build a prompt for detecting comment-worthy passages
|
|
9384
9437
|
*
|
|
9385
|
-
* @param content - The text content to analyze
|
|
9438
|
+
* @param content - The text content to analyze — a chunk sized by the
|
|
9439
|
+
* caller from derived provider limits; NEVER re-truncated here (a
|
|
9440
|
+
* builder-level clip is silent input loss — see #738)
|
|
9386
9441
|
* @param instructions - Optional user-provided instructions
|
|
9387
9442
|
* @param tone - Optional tone guidance (e.g., "academic", "conversational")
|
|
9388
9443
|
* @param density - Optional target number of comments per 2000 words
|
|
@@ -9403,7 +9458,7 @@ ${instructions}${toneGuidance}${densityGuidance}${sourceLang}${bodyLang}
|
|
|
9403
9458
|
|
|
9404
9459
|
Text to analyze:
|
|
9405
9460
|
---
|
|
9406
|
-
${content
|
|
9461
|
+
${content}
|
|
9407
9462
|
---
|
|
9408
9463
|
|
|
9409
9464
|
Return a JSON array of comments. Each comment must have:
|
|
@@ -9437,7 +9492,7 @@ Guidelines:
|
|
|
9437
9492
|
|
|
9438
9493
|
Text to analyze:
|
|
9439
9494
|
---
|
|
9440
|
-
${content
|
|
9495
|
+
${content}
|
|
9441
9496
|
---
|
|
9442
9497
|
|
|
9443
9498
|
Return a JSON array of comments. Each comment should have:
|
|
@@ -9458,7 +9513,9 @@ Example format:
|
|
|
9458
9513
|
/**
|
|
9459
9514
|
* Build a prompt for detecting highlight-worthy passages
|
|
9460
9515
|
*
|
|
9461
|
-
* @param content - The text content to analyze
|
|
9516
|
+
* @param content - The text content to analyze — a chunk sized by the
|
|
9517
|
+
* caller from derived provider limits; NEVER re-truncated here (a
|
|
9518
|
+
* builder-level clip is silent input loss — see #738)
|
|
9462
9519
|
* @param instructions - Optional user-provided instructions
|
|
9463
9520
|
* @param density - Optional target number of highlights per 2000 words
|
|
9464
9521
|
* @returns Formatted prompt string
|
|
@@ -9476,7 +9533,7 @@ ${instructions}${densityGuidance}${sourceLang}
|
|
|
9476
9533
|
|
|
9477
9534
|
Text to analyze:
|
|
9478
9535
|
---
|
|
9479
|
-
${content
|
|
9536
|
+
${content}
|
|
9480
9537
|
---
|
|
9481
9538
|
|
|
9482
9539
|
Return a JSON array of highlights. Each highlight must have:
|
|
@@ -9507,7 +9564,7 @@ Guidelines:
|
|
|
9507
9564
|
|
|
9508
9565
|
Text to analyze:
|
|
9509
9566
|
---
|
|
9510
|
-
${content
|
|
9567
|
+
${content}
|
|
9511
9568
|
---
|
|
9512
9569
|
|
|
9513
9570
|
Return a JSON array of highlights. Each highlight should have:
|
|
@@ -9527,7 +9584,9 @@ Example format:
|
|
|
9527
9584
|
/**
|
|
9528
9585
|
* Build a prompt for detecting assessment-worthy passages
|
|
9529
9586
|
*
|
|
9530
|
-
* @param content - The text content to analyze
|
|
9587
|
+
* @param content - The text content to analyze — a chunk sized by the
|
|
9588
|
+
* caller from derived provider limits; NEVER re-truncated here (a
|
|
9589
|
+
* builder-level clip is silent input loss — see #738)
|
|
9531
9590
|
* @param instructions - Optional user-provided instructions
|
|
9532
9591
|
* @param tone - Optional tone guidance (e.g., "critical", "supportive")
|
|
9533
9592
|
* @param density - Optional target number of assessments per 2000 words
|
|
@@ -9548,7 +9607,7 @@ ${instructions}${toneGuidance}${densityGuidance}${sourceLang}${bodyLang}
|
|
|
9548
9607
|
|
|
9549
9608
|
Text to analyze:
|
|
9550
9609
|
---
|
|
9551
|
-
${content
|
|
9610
|
+
${content}
|
|
9552
9611
|
---
|
|
9553
9612
|
|
|
9554
9613
|
Return a JSON array of assessments. Each assessment must have:
|
|
@@ -9582,7 +9641,7 @@ Guidelines:
|
|
|
9582
9641
|
|
|
9583
9642
|
Text to analyze:
|
|
9584
9643
|
---
|
|
9585
|
-
${content
|
|
9644
|
+
${content}
|
|
9586
9645
|
---
|
|
9587
9646
|
|
|
9588
9647
|
Return a JSON array of assessments. Each assessment should have:
|
|
@@ -9828,10 +9887,32 @@ function logAnchorMethod(motivation, exact, anchorMethod) {
|
|
|
9828
9887
|
}
|
|
9829
9888
|
|
|
9830
9889
|
// src/workers/annotation-detection.ts
|
|
9831
|
-
function assertNotTruncated(response, motivation) {
|
|
9890
|
+
function assertNotTruncated(response, motivation, chunk, totalChunks, outputBudget) {
|
|
9832
9891
|
if (response.stopReason === "max_tokens") {
|
|
9833
|
-
throw new Error(`${motivation} detection response truncated (max_tokens)
|
|
9892
|
+
throw new Error(`${motivation} detection response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
|
|
9893
|
+
}
|
|
9894
|
+
}
|
|
9895
|
+
async function detectInChunks(client, content, buildPrompt, temperature, motivation, parse, onChunk) {
|
|
9896
|
+
const limits = await client.limits();
|
|
9897
|
+
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
9898
|
+
const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
|
|
9899
|
+
const chunks = chunkText(content, chunking);
|
|
9900
|
+
const collected = [];
|
|
9901
|
+
for (let i = 0; i < chunks.length; i++) {
|
|
9902
|
+
const response = await boundedGenerateWithMetadata(
|
|
9903
|
+
client,
|
|
9904
|
+
buildPrompt(chunks[i]),
|
|
9905
|
+
outputBudget,
|
|
9906
|
+
temperature,
|
|
9907
|
+
{ format: "json" }
|
|
9908
|
+
);
|
|
9909
|
+
assertNotTruncated(response, motivation, i + 1, chunks.length, outputBudget);
|
|
9910
|
+
collected.push(...parse(response.text));
|
|
9911
|
+
if (i < chunks.length - 1) {
|
|
9912
|
+
onChunk?.(i + 1, chunks.length);
|
|
9913
|
+
}
|
|
9834
9914
|
}
|
|
9915
|
+
return collected;
|
|
9835
9916
|
}
|
|
9836
9917
|
var AnnotationDetection = class {
|
|
9837
9918
|
/**
|
|
@@ -9842,11 +9923,16 @@ var AnnotationDetection = class {
|
|
|
9842
9923
|
* (source-resource locale). See `types.ts` "Locale conventions" for the
|
|
9843
9924
|
* full discussion.
|
|
9844
9925
|
*/
|
|
9845
|
-
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage) {
|
|
9846
|
-
|
|
9847
|
-
|
|
9848
|
-
|
|
9849
|
-
|
|
9926
|
+
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onChunk) {
|
|
9927
|
+
return detectInChunks(
|
|
9928
|
+
client,
|
|
9929
|
+
content,
|
|
9930
|
+
(chunk) => MotivationPrompts.buildCommentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
|
|
9931
|
+
0.4,
|
|
9932
|
+
"comment",
|
|
9933
|
+
(text) => MotivationParsers.parseComments(text, content),
|
|
9934
|
+
onChunk
|
|
9935
|
+
);
|
|
9850
9936
|
}
|
|
9851
9937
|
/**
|
|
9852
9938
|
* Detect highlights in content.
|
|
@@ -9855,11 +9941,16 @@ var AnnotationDetection = class {
|
|
|
9855
9941
|
* applies, used in the prompt so the LLM analyzes non-English source
|
|
9856
9942
|
* correctly.
|
|
9857
9943
|
*/
|
|
9858
|
-
static async detectHighlights(content, client, instructions, density, sourceLanguage) {
|
|
9859
|
-
|
|
9860
|
-
|
|
9861
|
-
|
|
9862
|
-
|
|
9944
|
+
static async detectHighlights(content, client, instructions, density, sourceLanguage, onChunk) {
|
|
9945
|
+
return detectInChunks(
|
|
9946
|
+
client,
|
|
9947
|
+
content,
|
|
9948
|
+
(chunk) => MotivationPrompts.buildHighlightPrompt(chunk, instructions, density, sourceLanguage),
|
|
9949
|
+
0.3,
|
|
9950
|
+
"highlight",
|
|
9951
|
+
(text) => MotivationParsers.parseHighlights(text, content),
|
|
9952
|
+
onChunk
|
|
9953
|
+
);
|
|
9863
9954
|
}
|
|
9864
9955
|
/**
|
|
9865
9956
|
* Detect assessments in content.
|
|
@@ -9868,11 +9959,16 @@ var AnnotationDetection = class {
|
|
|
9868
9959
|
* (annotation body locale). `sourceLanguage` is the locale of the content
|
|
9869
9960
|
* being analyzed (source-resource locale).
|
|
9870
9961
|
*/
|
|
9871
|
-
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage) {
|
|
9872
|
-
|
|
9873
|
-
|
|
9874
|
-
|
|
9875
|
-
|
|
9962
|
+
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onChunk) {
|
|
9963
|
+
return detectInChunks(
|
|
9964
|
+
client,
|
|
9965
|
+
content,
|
|
9966
|
+
(chunk) => MotivationPrompts.buildAssessmentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
|
|
9967
|
+
0.3,
|
|
9968
|
+
"assessment",
|
|
9969
|
+
(text) => MotivationParsers.parseAssessments(text, content),
|
|
9970
|
+
onChunk
|
|
9971
|
+
);
|
|
9876
9972
|
}
|
|
9877
9973
|
/**
|
|
9878
9974
|
* Detect tags in content for a specific category.
|
|
@@ -9886,28 +9982,33 @@ var AnnotationDetection = class {
|
|
|
9886
9982
|
* identifiers, not LLM-generated text — so it's consumed at the body-stamp
|
|
9887
9983
|
* site, not here.
|
|
9888
9984
|
*/
|
|
9889
|
-
static async detectTags(content, client, schema, category, sourceLanguage) {
|
|
9985
|
+
static async detectTags(content, client, schema, category, sourceLanguage, onChunk) {
|
|
9890
9986
|
const categoryInfo = schema.tags.find((t) => t.name === category);
|
|
9891
9987
|
if (!categoryInfo) {
|
|
9892
9988
|
throw new Error(`Invalid category "${category}" for schema ${schema.id}`);
|
|
9893
9989
|
}
|
|
9894
|
-
const
|
|
9990
|
+
const parsedTags = await detectInChunks(
|
|
9991
|
+
client,
|
|
9895
9992
|
content,
|
|
9896
|
-
|
|
9897
|
-
|
|
9898
|
-
|
|
9899
|
-
|
|
9900
|
-
|
|
9901
|
-
|
|
9902
|
-
|
|
9993
|
+
(chunk) => MotivationPrompts.buildTagPrompt(
|
|
9994
|
+
chunk,
|
|
9995
|
+
category,
|
|
9996
|
+
schema.name,
|
|
9997
|
+
schema.description,
|
|
9998
|
+
schema.domain,
|
|
9999
|
+
categoryInfo.description,
|
|
10000
|
+
categoryInfo.examples,
|
|
10001
|
+
sourceLanguage
|
|
10002
|
+
),
|
|
10003
|
+
0.2,
|
|
10004
|
+
"tag",
|
|
10005
|
+
(text) => MotivationParsers.parseTags(text),
|
|
10006
|
+
onChunk
|
|
9903
10007
|
);
|
|
9904
|
-
const response = await boundedGenerateWithMetadata(client, prompt, 4e3, 0.2, { format: "json" });
|
|
9905
|
-
assertNotTruncated(response, "tag");
|
|
9906
|
-
const parsedTags = MotivationParsers.parseTags(response.text);
|
|
9907
10008
|
return MotivationParsers.validateTagOffsets(parsedTags, content, category);
|
|
9908
10009
|
}
|
|
9909
10010
|
};
|
|
9910
|
-
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage) {
|
|
10011
|
+
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage, onChunk) {
|
|
9911
10012
|
const entityTypesDescription = entityTypes.map((et) => {
|
|
9912
10013
|
if (typeof et === "string") {
|
|
9913
10014
|
return et;
|
|
@@ -9939,11 +10040,11 @@ Find direct mentions only (names, proper nouns). Do not include pronouns or desc
|
|
|
9939
10040
|
const sourceLangGuidance = sourceLanguage ? `
|
|
9940
10041
|
Source text language: ${getLocaleEnglishName(sourceLanguage) || sourceLanguage}.
|
|
9941
10042
|
` : "";
|
|
9942
|
-
const
|
|
10043
|
+
const buildPrompt = (text) => `Identify entity references in the following text. Look for mentions of: ${entityTypesDescription}.
|
|
9943
10044
|
${descriptiveReferenceGuidance}${sourceLangGuidance}
|
|
9944
10045
|
Text to analyze:
|
|
9945
10046
|
"""
|
|
9946
|
-
${
|
|
10047
|
+
${text}
|
|
9947
10048
|
"""
|
|
9948
10049
|
|
|
9949
10050
|
Respond with a JSON array of entities found. Each entity should have:
|
|
@@ -9956,59 +10057,78 @@ If no entities are found, respond with an empty array [].
|
|
|
9956
10057
|
|
|
9957
10058
|
Example output:
|
|
9958
10059
|
[{"exact":"Alice","entityType":"Person","prefix":"","suffix":" went to"},{"exact":"Paris","entityType":"Location","prefix":"went to ","suffix":" yesterday"}]`;
|
|
9959
|
-
|
|
9960
|
-
const
|
|
9961
|
-
|
|
9962
|
-
|
|
9963
|
-
|
|
9964
|
-
|
|
9965
|
-
|
|
9966
|
-
|
|
9967
|
-
|
|
9968
|
-
|
|
9969
|
-
|
|
9970
|
-
|
|
9971
|
-
|
|
9972
|
-
|
|
9973
|
-
|
|
9974
|
-
|
|
9975
|
-
|
|
9976
|
-
|
|
9977
|
-
|
|
9978
|
-
|
|
9979
|
-
|
|
9980
|
-
|
|
9981
|
-
|
|
9982
|
-
|
|
9983
|
-
|
|
9984
|
-
|
|
9985
|
-
logger2.
|
|
9986
|
-
|
|
9987
|
-
|
|
9988
|
-
|
|
9989
|
-
throw new Error("Failed to parse entity extraction response", {
|
|
9990
|
-
cause: error instanceof Error ? error : new Error(String(error))
|
|
10060
|
+
const limits = await client.limits();
|
|
10061
|
+
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
10062
|
+
const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
|
|
10063
|
+
const chunks = chunkText(exact, chunking);
|
|
10064
|
+
logger2.debug("Sending entity extraction request", {
|
|
10065
|
+
entityTypes: entityTypesDescription,
|
|
10066
|
+
chunks: chunks.length,
|
|
10067
|
+
chunkSizeTokens: chunking.chunkSize,
|
|
10068
|
+
outputBudget
|
|
10069
|
+
});
|
|
10070
|
+
const collected = [];
|
|
10071
|
+
for (let i = 0; i < chunks.length; i++) {
|
|
10072
|
+
const response = await boundedGenerateWithMetadata(
|
|
10073
|
+
client,
|
|
10074
|
+
buildPrompt(chunks[i]),
|
|
10075
|
+
outputBudget,
|
|
10076
|
+
0.3,
|
|
10077
|
+
// Lower temperature for more consistent extraction
|
|
10078
|
+
// Force grammar-constrained JSON output. Without this, Ollama models
|
|
10079
|
+
// periodically emit malformed JSON (truncated brackets, mid-token
|
|
10080
|
+
// breaks at higher token counts) which silently parse-fails into
|
|
10081
|
+
// [] downstream. The prompt's schema (which keys, what types) still
|
|
10082
|
+
// governs *what* the JSON contains; `format: 'json'` governs that
|
|
10083
|
+
// it's syntactically valid.
|
|
10084
|
+
{ format: "json" }
|
|
10085
|
+
);
|
|
10086
|
+
logger2.debug("Got entity extraction response", {
|
|
10087
|
+
chunk: i + 1,
|
|
10088
|
+
chunks: chunks.length,
|
|
10089
|
+
responseLength: response.text.length
|
|
9991
10090
|
});
|
|
10091
|
+
if (response.stopReason === "max_tokens") {
|
|
10092
|
+
const errorMsg = `Entity extraction response truncated (max_tokens) on chunk ${i + 1}/${chunks.length} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than dropping annotations.`;
|
|
10093
|
+
logger2.error(errorMsg, { responseLength: response.text.length });
|
|
10094
|
+
throw new Error(errorMsg);
|
|
10095
|
+
}
|
|
10096
|
+
let entities;
|
|
10097
|
+
try {
|
|
10098
|
+
entities = JSON.parse(response.text.trim());
|
|
10099
|
+
} catch (error) {
|
|
10100
|
+
logger2.error("Failed to parse entity extraction response", {
|
|
10101
|
+
error: error instanceof Error ? error.message : String(error),
|
|
10102
|
+
response: response.text.slice(0, 500)
|
|
10103
|
+
});
|
|
10104
|
+
throw new Error("Failed to parse entity extraction response", {
|
|
10105
|
+
cause: error instanceof Error ? error : new Error(String(error))
|
|
10106
|
+
});
|
|
10107
|
+
}
|
|
10108
|
+
if (!isArray(entities)) {
|
|
10109
|
+
logger2.error("Failed to parse entity extraction response: expected a JSON array", {
|
|
10110
|
+
response: response.text.slice(0, 500)
|
|
10111
|
+
});
|
|
10112
|
+
throw new Error("Failed to parse entity extraction response: expected a JSON array");
|
|
10113
|
+
}
|
|
10114
|
+
logger2.debug("Parsed entities from AI response", { chunk: i + 1, count: entities.length });
|
|
10115
|
+
for (const e of entities) {
|
|
10116
|
+
if (isObject(e) && isString(e.exact) && isString(e.entityType)) {
|
|
10117
|
+
collected.push({
|
|
10118
|
+
exact: e.exact,
|
|
10119
|
+
entityType: e.entityType,
|
|
10120
|
+
...isString(e.prefix) ? { prefix: e.prefix } : {},
|
|
10121
|
+
...isString(e.suffix) ? { suffix: e.suffix } : {}
|
|
10122
|
+
});
|
|
10123
|
+
} else {
|
|
10124
|
+
logger2.debug("Dropped malformed LLM entity", { entity: e });
|
|
10125
|
+
}
|
|
10126
|
+
}
|
|
10127
|
+
if (i < chunks.length - 1) {
|
|
10128
|
+
onChunk?.(i + 1, chunks.length);
|
|
10129
|
+
}
|
|
9992
10130
|
}
|
|
9993
|
-
|
|
9994
|
-
logger2.error("Failed to parse entity extraction response: expected a JSON array", {
|
|
9995
|
-
response: response.text.slice(0, 500)
|
|
9996
|
-
});
|
|
9997
|
-
throw new Error("Failed to parse entity extraction response: expected a JSON array");
|
|
9998
|
-
}
|
|
9999
|
-
logger2.debug("Parsed entities from AI response", { count: entities.length });
|
|
10000
|
-
return entities.filter((e) => {
|
|
10001
|
-
const ok = isObject(e) && isString(e.exact) && isString(e.entityType);
|
|
10002
|
-
if (!ok) {
|
|
10003
|
-
logger2.debug("Dropped malformed LLM entity", { entity: e });
|
|
10004
|
-
}
|
|
10005
|
-
return ok;
|
|
10006
|
-
}).map((entity) => ({
|
|
10007
|
-
exact: entity.exact,
|
|
10008
|
-
entityType: entity.entityType,
|
|
10009
|
-
...isString(entity.prefix) ? { prefix: entity.prefix } : {},
|
|
10010
|
-
...isString(entity.suffix) ? { suffix: entity.suffix } : {}
|
|
10011
|
-
}));
|
|
10131
|
+
return collected;
|
|
10012
10132
|
}
|
|
10013
10133
|
function getLanguageName(locale) {
|
|
10014
10134
|
return getLocaleEnglishName(locale) || locale;
|
|
@@ -10019,7 +10139,7 @@ var SEMANTIC_MATCH_CHARS = 240;
|
|
|
10019
10139
|
function idLabel(resourceId, annotationId) {
|
|
10020
10140
|
return `[${resourceId}${annotationId ? `/${annotationId}` : ""}]`;
|
|
10021
10141
|
}
|
|
10022
|
-
async function generateResourceFromTopic(topic, entityTypes, client, logger2, userPrompt, locale, context, temperature, maxTokens, sourceLanguage, outputMediaType = "text/markdown", task = "resource", structure, cite = false) {
|
|
10142
|
+
async function generateResourceFromTopic(topic, entityTypes, client, logger2, userPrompt, locale, context, temperature, maxTokens, sourceLanguage, outputMediaType = "text/markdown", task = "resource", structure, cite = false, repair) {
|
|
10023
10143
|
logger2.debug("Generating resource from topic", {
|
|
10024
10144
|
topicPreview: topic.substring(0, 100),
|
|
10025
10145
|
entityTypes,
|
|
@@ -10135,11 +10255,12 @@ ${parts.join("\n")}`;
|
|
|
10135
10255
|
let semanticContextSection = "";
|
|
10136
10256
|
const similar = context?.semanticContext?.similar ?? [];
|
|
10137
10257
|
if (similar.length > 0) {
|
|
10138
|
-
const lines = [...similar].sort((a, b) => b.score - a.score).slice(0, SEMANTIC_MATCH_LIMIT).map((m) => `- ${idLabel(m.resourceId, m.annotationId)} (${m.score.toFixed(2)}) ${m.text.slice(0, SEMANTIC_MATCH_CHARS)}`);
|
|
10258
|
+
const lines = [...similar].sort((a, b) => b.score - a.score).slice(0, SEMANTIC_MATCH_LIMIT).map((m) => `- ${idLabel(m.resourceId, m.annotationId)} (${m.score.toFixed(2)})${m.machineRead ? " [OCR]" : ""} ${m.text.slice(0, SEMANTIC_MATCH_CHARS)}`);
|
|
10259
|
+
const ocrNote = similar.some((m) => m.machineRead) ? "\nPassages marked [OCR] were read from scanned images by character recognition; treat their exact wording and numbers as uncertain, and say so if you rely on one." : "";
|
|
10139
10260
|
semanticContextSection = `
|
|
10140
10261
|
|
|
10141
10262
|
Related passages from the knowledge base:
|
|
10142
|
-
${lines.join("\n")}`;
|
|
10263
|
+
${lines.join("\n")}${ocrNote}`;
|
|
10143
10264
|
}
|
|
10144
10265
|
let leadLine;
|
|
10145
10266
|
if (task === "resource") {
|
|
@@ -10154,11 +10275,12 @@ ${lines.join("\n")}`;
|
|
|
10154
10275
|
Topic: "${topic}"`;
|
|
10155
10276
|
}
|
|
10156
10277
|
const isPlainText = outputMediaType === "text/plain";
|
|
10278
|
+
const isPdf = outputMediaType === "application/pdf";
|
|
10157
10279
|
let structureRequirement = "";
|
|
10158
10280
|
let titleRequirement = "";
|
|
10159
10281
|
if (structure === "sections") {
|
|
10160
|
-
structureRequirement = isPlainText ? "\n- Organize the content into titled sections with well-structured paragraphs" : "\n- Organize the content into titled sections (## Section) with well-structured paragraphs";
|
|
10161
|
-
if (!isPlainText) {
|
|
10282
|
+
structureRequirement = isPdf ? "\n- Organize the content into titled sections (= Heading) with well-structured paragraphs" : isPlainText ? "\n- Organize the content into titled sections with well-structured paragraphs" : "\n- Organize the content into titled sections (## Section) with well-structured paragraphs";
|
|
10283
|
+
if (!isPlainText && !isPdf) {
|
|
10162
10284
|
titleRequirement = "\n- Start with a clear heading (# Title)";
|
|
10163
10285
|
}
|
|
10164
10286
|
} else if (structure === "prose") {
|
|
@@ -10171,11 +10293,20 @@ Topic: "${topic}"`;
|
|
|
10171
10293
|
- Organize the output as: ${structure}`;
|
|
10172
10294
|
}
|
|
10173
10295
|
const citeRequirement = cite ? "\n- Ground every claim in the provided context. Immediately after each claim, cite its source by emitting [[<id>]], where <id> is an id shown in square brackets in the context above (for a passage labeled [abc], emit [[abc]]). Cite only ids that appear in the context." : "";
|
|
10174
|
-
const formatRequirements =
|
|
10296
|
+
const formatRequirements = isPdf ? `- Write the response as Typst markup (the Typst typesetting language \u2014 not markdown, not LaTeX)
|
|
10297
|
+
- Headings are written as = Heading (deeper levels == Subheading); everything else is plain prose paragraphs
|
|
10298
|
+
- Do not emit markdown syntax or code fences` : isPlainText ? `- Write the response as plain text \u2014 no formatting markup (no #, *, backticks, headings, or links)
|
|
10175
10299
|
- Begin with the title on its own first line` : `- Use markdown formatting
|
|
10176
10300
|
- Write the response as markdown`;
|
|
10301
|
+
const repairSection = repair ? `
|
|
10302
|
+
|
|
10303
|
+
Your previous attempt failed to compile. Fix the error and return the complete corrected document \u2014 full source, not a diff.
|
|
10304
|
+
Compile error:
|
|
10305
|
+
${repair.error}
|
|
10306
|
+
Previous source:
|
|
10307
|
+
${repair.source}` : "";
|
|
10177
10308
|
const prompt = `${leadLine}
|
|
10178
|
-
${userPrompt ? `Instruction: ${userPrompt}` : ""}
|
|
10309
|
+
${userPrompt ? `Instruction: ${userPrompt}` : ""}${repairSection}
|
|
10179
10310
|
${entityTypes.length > 0 ? `Focus on these entity types: ${entityTypes.join(", ")}.` : ""}${annotationSection}${contextSection}${resourceSection}${graphSection}${semanticContextSection}${sourceLanguageInstruction}${languageInstruction}
|
|
10180
10311
|
|
|
10181
10312
|
Requirements:
|
|
@@ -10184,7 +10315,7 @@ Requirements:
|
|
|
10184
10315
|
${formatRequirements}`;
|
|
10185
10316
|
const parseResponse = (response2) => {
|
|
10186
10317
|
let content = response2.trim();
|
|
10187
|
-
if (content.startsWith("```markdown") || content.startsWith("```md")) {
|
|
10318
|
+
if (content.startsWith("```markdown") || content.startsWith("```md") || content.startsWith("```typst")) {
|
|
10188
10319
|
content = content.slice(content.indexOf("\n") + 1);
|
|
10189
10320
|
const endIndex = content.lastIndexOf("```");
|
|
10190
10321
|
if (endIndex !== -1) {
|
|
@@ -10219,6 +10350,29 @@ ${formatRequirements}`;
|
|
|
10219
10350
|
});
|
|
10220
10351
|
return result;
|
|
10221
10352
|
}
|
|
10353
|
+
var PINNED_CREATION_TIMESTAMP = 17e8;
|
|
10354
|
+
var MAX_COMPILE_REPAIRS = 2;
|
|
10355
|
+
function compileTypst(source) {
|
|
10356
|
+
const dir = mkdtempSync(join(tmpdir(), "typst-"));
|
|
10357
|
+
try {
|
|
10358
|
+
const inFile = join(dir, "doc.typ");
|
|
10359
|
+
const outFile = join(dir, "doc.pdf");
|
|
10360
|
+
writeFileSync(inFile, source);
|
|
10361
|
+
try {
|
|
10362
|
+
execFileSync(
|
|
10363
|
+
"typst",
|
|
10364
|
+
["compile", "--creation-timestamp", String(PINNED_CREATION_TIMESTAMP), inFile, outFile],
|
|
10365
|
+
{ stdio: ["ignore", "pipe", "pipe"] }
|
|
10366
|
+
);
|
|
10367
|
+
} catch (err) {
|
|
10368
|
+
const stderr = err.stderr;
|
|
10369
|
+
return { error: stderr?.length ? stderr.toString("utf8") : String(err) };
|
|
10370
|
+
}
|
|
10371
|
+
return { pdf: new Uint8Array(readFileSync(outFile)) };
|
|
10372
|
+
} finally {
|
|
10373
|
+
rmSync(dir, { recursive: true, force: true });
|
|
10374
|
+
}
|
|
10375
|
+
}
|
|
10222
10376
|
|
|
10223
10377
|
// src/workers/generation/citation-resolver.ts
|
|
10224
10378
|
var CITATION_TOKEN = /\[\[([^\s[\]/]+)(?:\/([^\s[\]/]+))?\]\]/g;
|
|
@@ -10363,9 +10517,9 @@ function buildTextAnnotation(content, resourceId, userId, generator, motivation,
|
|
|
10363
10517
|
...body !== void 0 ? { body } : {}
|
|
10364
10518
|
};
|
|
10365
10519
|
}
|
|
10366
|
-
function buildPdfAnnotation(
|
|
10367
|
-
const { rects, overlap } = locate(
|
|
10368
|
-
const coveredText = overlap.length ?
|
|
10520
|
+
function buildPdfAnnotation(anchored, resourceId, userId, generator, motivation, match, body) {
|
|
10521
|
+
const { rects, overlap } = locate(anchored, match.start, match.end);
|
|
10522
|
+
const coveredText = overlap.length ? anchored.text.substring(
|
|
10369
10523
|
Math.min(...overlap.map((i) => i.start)),
|
|
10370
10524
|
Math.max(...overlap.map((i) => i.end))
|
|
10371
10525
|
) : "";
|
|
@@ -10414,7 +10568,9 @@ async function processHighlightJob(content, inferenceClient, params, buildAnnota
|
|
|
10414
10568
|
inferenceClient,
|
|
10415
10569
|
params.instructions,
|
|
10416
10570
|
params.density,
|
|
10417
|
-
params.sourceLanguage
|
|
10571
|
+
params.sourceLanguage,
|
|
10572
|
+
// Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
|
|
10573
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
|
|
10418
10574
|
);
|
|
10419
10575
|
onProgress(60, `Creating ${highlights.length} annotations...`, "creating");
|
|
10420
10576
|
const annotations = dedupeAnnotations(highlights.map(
|
|
@@ -10436,7 +10592,9 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
10436
10592
|
params.tone,
|
|
10437
10593
|
params.density,
|
|
10438
10594
|
params.language,
|
|
10439
|
-
params.sourceLanguage
|
|
10595
|
+
params.sourceLanguage,
|
|
10596
|
+
// Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
|
|
10597
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
|
|
10440
10598
|
);
|
|
10441
10599
|
onProgress(60, `Creating ${comments.length} annotations...`, "creating");
|
|
10442
10600
|
const bodyLanguage = params.language ?? "en";
|
|
@@ -10466,7 +10624,9 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
10466
10624
|
params.tone,
|
|
10467
10625
|
params.density,
|
|
10468
10626
|
params.language,
|
|
10469
|
-
params.sourceLanguage
|
|
10627
|
+
params.sourceLanguage,
|
|
10628
|
+
// Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
|
|
10629
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
|
|
10470
10630
|
);
|
|
10471
10631
|
onProgress(60, `Creating ${assessments.length} annotations...`, "creating");
|
|
10472
10632
|
const bodyLanguage = params.language ?? "en";
|
|
@@ -10522,7 +10682,23 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
10522
10682
|
inferenceClient,
|
|
10523
10683
|
params.includeDescriptiveReferences ?? false,
|
|
10524
10684
|
logger2,
|
|
10525
|
-
params.sourceLanguage
|
|
10685
|
+
params.sourceLanguage,
|
|
10686
|
+
// Chunk-boundary heartbeat: progress is the worker's liveness signal
|
|
10687
|
+
// (stall watchdog + backend janitor), so multi-chunk extraction must
|
|
10688
|
+
// emit between inference calls. Percentage interpolates within this
|
|
10689
|
+
// entity type's band of the 20–80 range.
|
|
10690
|
+
(completed, total) => {
|
|
10691
|
+
const interpolated = 20 + Math.round((i + completed / total) / entityTypeNames.length * 60);
|
|
10692
|
+
onProgress(interpolated, `Detecting ${entityTypeName} entities...`, "analyzing", {
|
|
10693
|
+
currentEntityType: entityTypeName,
|
|
10694
|
+
processedEntityTypes: i,
|
|
10695
|
+
totalEntityTypes: entityTypeNames.length,
|
|
10696
|
+
entitiesFound: totalFound,
|
|
10697
|
+
entitiesEmitted: totalEmitted,
|
|
10698
|
+
completedEntityTypes: [...completedEntityTypes],
|
|
10699
|
+
requestParams
|
|
10700
|
+
});
|
|
10701
|
+
}
|
|
10526
10702
|
);
|
|
10527
10703
|
totalFound += extractedEntities.length;
|
|
10528
10704
|
completedEntityTypes.push({ entityType: entityTypeName, foundCount: extractedEntities.length });
|
|
@@ -10566,13 +10742,21 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
10566
10742
|
onProgress(10, "Loading resource...", "analyzing");
|
|
10567
10743
|
onProgress(30, "Analyzing text for tags...", "analyzing");
|
|
10568
10744
|
const allTags = [];
|
|
10569
|
-
for (
|
|
10745
|
+
for (let c = 0; c < params.categories.length; c++) {
|
|
10746
|
+
const category = params.categories[c];
|
|
10570
10747
|
const categoryTags = await AnnotationDetection.detectTags(
|
|
10571
10748
|
content,
|
|
10572
10749
|
inferenceClient,
|
|
10573
10750
|
params.schema,
|
|
10574
10751
|
category,
|
|
10575
|
-
params.sourceLanguage
|
|
10752
|
+
params.sourceLanguage,
|
|
10753
|
+
// Chunk-boundary heartbeat (liveness): interpolate within this
|
|
10754
|
+
// category's slice of the 30–60 band.
|
|
10755
|
+
(completed, total) => onProgress(
|
|
10756
|
+
30 + Math.round((c + completed / total) / params.categories.length * 30),
|
|
10757
|
+
"Analyzing text for tags...",
|
|
10758
|
+
"analyzing"
|
|
10759
|
+
)
|
|
10576
10760
|
);
|
|
10577
10761
|
allTags.push(...categoryTags);
|
|
10578
10762
|
}
|
|
@@ -10598,8 +10782,14 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
10598
10782
|
result: { tagsFound: tags.length, tagsCreated: annotations.length, byCategory }
|
|
10599
10783
|
};
|
|
10600
10784
|
}
|
|
10785
|
+
function assertWithinOutputBudget(byteLength) {
|
|
10786
|
+
if (!withinByteBudget(byteLength)) {
|
|
10787
|
+
throw new Error(
|
|
10788
|
+
`Generated artifact exceeds the output byte budget: ${byteLength} bytes > ${MAX_PDF_BYTES}. Refusing a runaway generation.`
|
|
10789
|
+
);
|
|
10790
|
+
}
|
|
10791
|
+
}
|
|
10601
10792
|
async function processGenerationJob(inferenceClient, params, onProgress, logger2) {
|
|
10602
|
-
const GENERATABLE_MEDIA_TYPES = ["text/markdown", "text/plain"];
|
|
10603
10793
|
const outputMediaType = params.outputMediaType ?? "text/markdown";
|
|
10604
10794
|
if (!GENERATABLE_MEDIA_TYPES.includes(outputMediaType)) {
|
|
10605
10795
|
throw new Error(
|
|
@@ -10608,6 +10798,84 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10608
10798
|
}
|
|
10609
10799
|
const title = params.title ?? "Untitled";
|
|
10610
10800
|
const entityTypes = (params.entityTypes ?? []).map(String);
|
|
10801
|
+
if (outputMediaType === "application/pdf") {
|
|
10802
|
+
onProgress(5, "Generating resource...", "generating");
|
|
10803
|
+
const validIds = params.cite === true ? collectContextResourceIds(params.context) : null;
|
|
10804
|
+
let generated2 = await generateResourceFromTopic(
|
|
10805
|
+
title,
|
|
10806
|
+
entityTypes,
|
|
10807
|
+
inferenceClient,
|
|
10808
|
+
logger2,
|
|
10809
|
+
params.prompt,
|
|
10810
|
+
params.language,
|
|
10811
|
+
params.context,
|
|
10812
|
+
params.temperature,
|
|
10813
|
+
params.maxTokens,
|
|
10814
|
+
params.sourceLanguage,
|
|
10815
|
+
outputMediaType,
|
|
10816
|
+
params.task,
|
|
10817
|
+
params.structure,
|
|
10818
|
+
params.cite
|
|
10819
|
+
);
|
|
10820
|
+
let source = generated2.content;
|
|
10821
|
+
let citations2 = [];
|
|
10822
|
+
if (validIds) {
|
|
10823
|
+
const resolved = resolveCitationTokens(generated2.content, validIds, logger2);
|
|
10824
|
+
source = resolved.content;
|
|
10825
|
+
citations2 = resolved.citations;
|
|
10826
|
+
}
|
|
10827
|
+
let compiled = compileTypst(source);
|
|
10828
|
+
let repairs = 0;
|
|
10829
|
+
while ("error" in compiled && repairs < MAX_COMPILE_REPAIRS) {
|
|
10830
|
+
repairs++;
|
|
10831
|
+
logger2.warn("Typst compile failed \u2014 feeding the error back for repair", {
|
|
10832
|
+
attempt: repairs,
|
|
10833
|
+
error: compiled.error.slice(0, 500)
|
|
10834
|
+
});
|
|
10835
|
+
generated2 = await generateResourceFromTopic(
|
|
10836
|
+
title,
|
|
10837
|
+
entityTypes,
|
|
10838
|
+
inferenceClient,
|
|
10839
|
+
logger2,
|
|
10840
|
+
params.prompt,
|
|
10841
|
+
params.language,
|
|
10842
|
+
params.context,
|
|
10843
|
+
params.temperature,
|
|
10844
|
+
params.maxTokens,
|
|
10845
|
+
params.sourceLanguage,
|
|
10846
|
+
outputMediaType,
|
|
10847
|
+
params.task,
|
|
10848
|
+
params.structure,
|
|
10849
|
+
params.cite,
|
|
10850
|
+
{ source, error: compiled.error }
|
|
10851
|
+
);
|
|
10852
|
+
if (validIds) {
|
|
10853
|
+
const resolved = resolveCitationTokens(generated2.content, validIds, logger2);
|
|
10854
|
+
source = resolved.content;
|
|
10855
|
+
citations2 = resolved.citations;
|
|
10856
|
+
} else {
|
|
10857
|
+
source = generated2.content;
|
|
10858
|
+
}
|
|
10859
|
+
compiled = compileTypst(source);
|
|
10860
|
+
}
|
|
10861
|
+
if ("error" in compiled) {
|
|
10862
|
+
throw new Error(
|
|
10863
|
+
`Typst compilation failed after ${MAX_COMPILE_REPAIRS} repair attempts: ${compiled.error}`
|
|
10864
|
+
);
|
|
10865
|
+
}
|
|
10866
|
+
assertWithinOutputBudget(compiled.pdf.byteLength);
|
|
10867
|
+
onProgress(95, "Creating resource...", "creating");
|
|
10868
|
+
return {
|
|
10869
|
+
content: compiled.pdf,
|
|
10870
|
+
title: generated2.title ?? title,
|
|
10871
|
+
format: outputMediaType,
|
|
10872
|
+
citations: citations2,
|
|
10873
|
+
result: {
|
|
10874
|
+
resourceId: "",
|
|
10875
|
+
resourceName: generated2.title ?? title
|
|
10876
|
+
}
|
|
10877
|
+
};
|
|
10878
|
+
}
|
|
10611
10879
|
onProgress(5, "Generating resource...", "generating");
|
|
10612
10880
|
const generated = await generateResourceFromTopic(
|
|
10613
10881
|
title,
|
|
@@ -10633,8 +10901,10 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10633
10901
|
citations = resolved.citations;
|
|
10634
10902
|
}
|
|
10635
10903
|
onProgress(95, "Creating resource...", "creating");
|
|
10904
|
+
const artifact = new TextEncoder().encode(content);
|
|
10905
|
+
assertWithinOutputBudget(artifact.byteLength);
|
|
10636
10906
|
return {
|
|
10637
|
-
content,
|
|
10907
|
+
content: artifact,
|
|
10638
10908
|
title: generated.title ?? title,
|
|
10639
10909
|
format: outputMediaType,
|
|
10640
10910
|
citations,
|
|
@@ -10646,28 +10916,37 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10646
10916
|
}
|
|
10647
10917
|
|
|
10648
10918
|
// src/workers/detection/prepare-detection.ts
|
|
10649
|
-
async function prepareDetection(
|
|
10650
|
-
|
|
10651
|
-
|
|
10652
|
-
|
|
10653
|
-
|
|
10654
|
-
|
|
10655
|
-
|
|
10656
|
-
|
|
10657
|
-
|
|
10658
|
-
|
|
10659
|
-
|
|
10660
|
-
|
|
10661
|
-
|
|
10662
|
-
|
|
10663
|
-
|
|
10664
|
-
|
|
10665
|
-
|
|
10666
|
-
}
|
|
10667
|
-
case "none":
|
|
10668
|
-
return null;
|
|
10919
|
+
async function prepareDetection(mediaType, session, resourceId, userId, generator, store) {
|
|
10920
|
+
const extractor = EXTRACTORS[textExtractionOf(mediaType)];
|
|
10921
|
+
if (!extractor) return { declined: "no-extractor" };
|
|
10922
|
+
const { data } = await session.client.browse.resourceRepresentation(resourceId);
|
|
10923
|
+
const bytes = Buffer.from(data);
|
|
10924
|
+
const extracted = await extractor.extract(bytes, mediaType, {
|
|
10925
|
+
key: calculateChecksum(bytes),
|
|
10926
|
+
store
|
|
10927
|
+
});
|
|
10928
|
+
if ("declined" in extracted) return extracted;
|
|
10929
|
+
if (!extracted.text.trim()) return { declined: "empty" };
|
|
10930
|
+
const items = extracted.items;
|
|
10931
|
+
if (items && items.length > 0) {
|
|
10932
|
+
const anchored = { text: extracted.text, items };
|
|
10933
|
+
return {
|
|
10934
|
+
text: extracted.text,
|
|
10935
|
+
buildAnnotation: (motivation, match, body) => buildPdfAnnotation(anchored, resourceId, userId, generator, motivation, match, body)
|
|
10936
|
+
};
|
|
10669
10937
|
}
|
|
10938
|
+
return {
|
|
10939
|
+
text: extracted.text,
|
|
10940
|
+
buildAnnotation: (motivation, match, body) => buildTextAnnotation(extracted.text, resourceId, userId, generator, motivation, match, body)
|
|
10941
|
+
};
|
|
10670
10942
|
}
|
|
10943
|
+
var DECLINE_MESSAGES = {
|
|
10944
|
+
"no-text-layer": "This PDF is a scan whose text could not be recognized; there is nothing to detect over.",
|
|
10945
|
+
"encrypted": "This PDF is password-protected, so its text cannot be read.",
|
|
10946
|
+
"corrupt": "This PDF could not be parsed \u2014 the file may be damaged or truncated.",
|
|
10947
|
+
"too-large": "This document is too large to extract text from.",
|
|
10948
|
+
"empty": "This document contains no text to detect over."
|
|
10949
|
+
};
|
|
10671
10950
|
async function emitEvent(session, channel, payload) {
|
|
10672
10951
|
await session.client.transport.emit(channel, payload);
|
|
10673
10952
|
}
|
|
@@ -10685,15 +10964,16 @@ function startWorkerProcess(config) {
|
|
|
10685
10964
|
const message = error instanceof Error ? error.message : String(error);
|
|
10686
10965
|
logger2.error("Job failed", { jobId: job.jobId, error: message, stack: error instanceof Error ? error.stack : void 0 });
|
|
10687
10966
|
const failAnnotationId = job.params.referenceId;
|
|
10688
|
-
|
|
10689
|
-
|
|
10690
|
-
|
|
10691
|
-
|
|
10692
|
-
|
|
10693
|
-
|
|
10694
|
-
|
|
10695
|
-
|
|
10696
|
-
|
|
10967
|
+
if (isJobType(job.type)) {
|
|
10968
|
+
emitEvent(session, "job:fail", {
|
|
10969
|
+
resourceId: job.resourceId,
|
|
10970
|
+
jobId: job.jobId,
|
|
10971
|
+
jobType: job.type,
|
|
10972
|
+
...failAnnotationId ? { annotationId: failAnnotationId } : {},
|
|
10973
|
+
error: message
|
|
10974
|
+
}).catch(() => {
|
|
10975
|
+
});
|
|
10976
|
+
}
|
|
10697
10977
|
adapter.failJob(job.jobId, message);
|
|
10698
10978
|
});
|
|
10699
10979
|
});
|
|
@@ -10725,11 +11005,16 @@ async function handleJob(adapter, config, job) {
|
|
|
10725
11005
|
}
|
|
10726
11006
|
async function handleJobInner(adapter, config, job) {
|
|
10727
11007
|
const { session, inferenceClient, generator } = config;
|
|
10728
|
-
const {
|
|
11008
|
+
const { userId, jobId } = job;
|
|
11009
|
+
if (!isJobType(job.type)) {
|
|
11010
|
+
adapter.failJob(jobId, `Unrecognized job type: ${job.type}`);
|
|
11011
|
+
return;
|
|
11012
|
+
}
|
|
11013
|
+
const jobType = job.type;
|
|
11014
|
+
const resourceId$1 = resourceId(job.resourceId);
|
|
10729
11015
|
const annotationId = job.params.referenceId;
|
|
10730
11016
|
const lifecycleBase = {
|
|
10731
|
-
resourceId,
|
|
10732
|
-
userId,
|
|
11017
|
+
resourceId: resourceId$1,
|
|
10733
11018
|
jobId,
|
|
10734
11019
|
jobType,
|
|
10735
11020
|
...annotationId ? { annotationId } : {}
|
|
@@ -10739,23 +11024,27 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10739
11024
|
adapter.failJob(jobId, `Worker not configured for job type: ${jobType}`);
|
|
10740
11025
|
return;
|
|
10741
11026
|
}
|
|
10742
|
-
let
|
|
11027
|
+
let ready = null;
|
|
10743
11028
|
if (jobType !== "generation") {
|
|
10744
|
-
const descriptor = await session.client.browse.resource(resourceId).fresh();
|
|
11029
|
+
const descriptor = await session.client.browse.resource(resourceId$1).fresh();
|
|
10745
11030
|
const mediaType = getPrimaryMediaType(descriptor);
|
|
10746
|
-
const
|
|
10747
|
-
if (
|
|
10748
|
-
|
|
10749
|
-
|
|
10750
|
-
|
|
10751
|
-
if (!source) {
|
|
11031
|
+
const source = await prepareDetection(mediaType ?? "", session, resourceId$1, userId, generator, config.anchoredTextStore);
|
|
11032
|
+
if ("declined" in source) {
|
|
11033
|
+
if (source.declined === "no-extractor") {
|
|
11034
|
+
throw new Error(`Cannot run ${jobType} on resource ${resourceId$1}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
|
|
11035
|
+
}
|
|
10752
11036
|
await emitEvent(session, "job:complete", {
|
|
10753
11037
|
...lifecycleBase,
|
|
10754
|
-
result: {
|
|
11038
|
+
result: {
|
|
11039
|
+
declined: true,
|
|
11040
|
+
reason: source.declined,
|
|
11041
|
+
message: DECLINE_MESSAGES[source.declined]
|
|
11042
|
+
}
|
|
10755
11043
|
});
|
|
10756
11044
|
adapter.completeJob();
|
|
10757
11045
|
return;
|
|
10758
11046
|
}
|
|
11047
|
+
ready = source;
|
|
10759
11048
|
}
|
|
10760
11049
|
const onProgress = (percentage, message, stage, extra) => {
|
|
10761
11050
|
adapter.touchActivity();
|
|
@@ -10774,14 +11063,14 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10774
11063
|
};
|
|
10775
11064
|
if (jobType === "highlight-annotation") {
|
|
10776
11065
|
const { annotations, result } = await processHighlightJob(
|
|
10777
|
-
|
|
11066
|
+
ready.text,
|
|
10778
11067
|
inferenceClient,
|
|
10779
|
-
job.params,
|
|
10780
|
-
|
|
11068
|
+
asJobParams(job.params),
|
|
11069
|
+
ready.buildAnnotation,
|
|
10781
11070
|
onProgress
|
|
10782
11071
|
);
|
|
10783
11072
|
for (const ann of annotations) {
|
|
10784
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11073
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10785
11074
|
}
|
|
10786
11075
|
await emitEvent(session, "job:complete", {
|
|
10787
11076
|
...lifecycleBase,
|
|
@@ -10790,14 +11079,14 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10790
11079
|
adapter.completeJob();
|
|
10791
11080
|
} else if (jobType === "comment-annotation") {
|
|
10792
11081
|
const { annotations, result } = await processCommentJob(
|
|
10793
|
-
|
|
11082
|
+
ready.text,
|
|
10794
11083
|
inferenceClient,
|
|
10795
|
-
job.params,
|
|
10796
|
-
|
|
11084
|
+
asJobParams(job.params),
|
|
11085
|
+
ready.buildAnnotation,
|
|
10797
11086
|
onProgress
|
|
10798
11087
|
);
|
|
10799
11088
|
for (const ann of annotations) {
|
|
10800
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11089
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10801
11090
|
}
|
|
10802
11091
|
await emitEvent(session, "job:complete", {
|
|
10803
11092
|
...lifecycleBase,
|
|
@@ -10806,14 +11095,14 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10806
11095
|
adapter.completeJob();
|
|
10807
11096
|
} else if (jobType === "assessment-annotation") {
|
|
10808
11097
|
const { annotations, result } = await processAssessmentJob(
|
|
10809
|
-
|
|
11098
|
+
ready.text,
|
|
10810
11099
|
inferenceClient,
|
|
10811
|
-
job.params,
|
|
10812
|
-
|
|
11100
|
+
asJobParams(job.params),
|
|
11101
|
+
ready.buildAnnotation,
|
|
10813
11102
|
onProgress
|
|
10814
11103
|
);
|
|
10815
11104
|
for (const ann of annotations) {
|
|
10816
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11105
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10817
11106
|
}
|
|
10818
11107
|
await emitEvent(session, "job:complete", {
|
|
10819
11108
|
...lifecycleBase,
|
|
@@ -10822,15 +11111,15 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10822
11111
|
adapter.completeJob();
|
|
10823
11112
|
} else if (jobType === "reference-annotation") {
|
|
10824
11113
|
const { annotations, result } = await processReferenceJob(
|
|
10825
|
-
|
|
11114
|
+
ready.text,
|
|
10826
11115
|
inferenceClient,
|
|
10827
|
-
job.params,
|
|
10828
|
-
|
|
11116
|
+
asJobParams(job.params),
|
|
11117
|
+
ready.buildAnnotation,
|
|
10829
11118
|
onProgress,
|
|
10830
11119
|
config.logger
|
|
10831
11120
|
);
|
|
10832
11121
|
for (const ann of annotations) {
|
|
10833
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11122
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10834
11123
|
}
|
|
10835
11124
|
await emitEvent(session, "job:complete", {
|
|
10836
11125
|
...lifecycleBase,
|
|
@@ -10839,14 +11128,14 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10839
11128
|
adapter.completeJob();
|
|
10840
11129
|
} else if (jobType === "tag-annotation") {
|
|
10841
11130
|
const { annotations, result } = await processTagJob(
|
|
10842
|
-
|
|
11131
|
+
ready.text,
|
|
10843
11132
|
inferenceClient,
|
|
10844
|
-
job.params,
|
|
10845
|
-
|
|
11133
|
+
asJobParams(job.params),
|
|
11134
|
+
ready.buildAnnotation,
|
|
10846
11135
|
onProgress
|
|
10847
11136
|
);
|
|
10848
11137
|
for (const ann of annotations) {
|
|
10849
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11138
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10850
11139
|
}
|
|
10851
11140
|
await emitEvent(session, "job:complete", {
|
|
10852
11141
|
...lifecycleBase,
|
|
@@ -10867,7 +11156,7 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10867
11156
|
file: Buffer.from(genResult.content),
|
|
10868
11157
|
format: genResult.format,
|
|
10869
11158
|
storageUri,
|
|
10870
|
-
sourceResourceId: resourceId,
|
|
11159
|
+
sourceResourceId: resourceId$1,
|
|
10871
11160
|
...genParams.referenceId ? { sourceAnnotationId: genParams.referenceId } : {},
|
|
10872
11161
|
...genParams.prompt ? { generationPrompt: genParams.prompt } : {},
|
|
10873
11162
|
...genParams.language ? { language: genParams.language } : {},
|
|
@@ -10878,29 +11167,63 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10878
11167
|
const { annotation: provenanceRef } = assembleAnnotation(
|
|
10879
11168
|
{
|
|
10880
11169
|
motivation: "linking",
|
|
10881
|
-
target: { source: String(resourceId) },
|
|
11170
|
+
target: { source: String(resourceId$1) },
|
|
10882
11171
|
body: { type: "SpecificResource", source: String(newResourceId), purpose: "linking" }
|
|
10883
11172
|
},
|
|
10884
11173
|
generator
|
|
10885
11174
|
);
|
|
10886
|
-
await emitEvent(session, "mark:create", { annotation: provenanceRef,
|
|
10887
|
-
}
|
|
10888
|
-
|
|
10889
|
-
const
|
|
10890
|
-
|
|
10891
|
-
|
|
10892
|
-
|
|
10893
|
-
|
|
10894
|
-
|
|
10895
|
-
|
|
10896
|
-
|
|
10897
|
-
|
|
11175
|
+
await emitEvent(session, "mark:create", { annotation: provenanceRef, resourceId: resourceId$1 });
|
|
11176
|
+
}
|
|
11177
|
+
if (genResult.format === "application/pdf" && genResult.citations.length > 0) {
|
|
11178
|
+
const layer = await extractPdfTextLayer(genResult.content);
|
|
11179
|
+
if (!layer) {
|
|
11180
|
+
config.logger.warn("PDF citations dropped \u2014 the generated artifact yielded no text layer", {
|
|
11181
|
+
jobId,
|
|
11182
|
+
resourceId: newResourceId,
|
|
11183
|
+
citations: genResult.citations.length
|
|
11184
|
+
});
|
|
11185
|
+
} else {
|
|
11186
|
+
for (const citation of genResult.citations) {
|
|
11187
|
+
const span = findClaimSpan(layer, citation.exact);
|
|
11188
|
+
if (!span) {
|
|
11189
|
+
config.logger.warn("PDF citation dropped \u2014 claim not found in the rendered text layer", {
|
|
11190
|
+
jobId,
|
|
11191
|
+
resourceId: newResourceId,
|
|
11192
|
+
citedResourceId: citation.resourceId,
|
|
11193
|
+
exactPreview: citation.exact.slice(0, 80)
|
|
11194
|
+
});
|
|
11195
|
+
continue;
|
|
11196
|
+
}
|
|
11197
|
+
const citationRef = buildPdfAnnotation(
|
|
11198
|
+
layer,
|
|
11199
|
+
resourceId(String(newResourceId)),
|
|
11200
|
+
userId,
|
|
11201
|
+
generator,
|
|
11202
|
+
"linking",
|
|
11203
|
+
{ exact: layer.text.slice(span.start, span.end), start: span.start, end: span.end },
|
|
11204
|
+
{ type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
|
|
11205
|
+
);
|
|
11206
|
+
await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
|
|
11207
|
+
}
|
|
11208
|
+
}
|
|
11209
|
+
} else {
|
|
11210
|
+
for (const citation of genResult.citations) {
|
|
11211
|
+
const { annotation: citationRef } = assembleAnnotation(
|
|
11212
|
+
{
|
|
11213
|
+
motivation: "linking",
|
|
11214
|
+
target: {
|
|
11215
|
+
source: String(newResourceId),
|
|
11216
|
+
selector: [
|
|
11217
|
+
{ type: "TextPositionSelector", start: citation.start, end: citation.end },
|
|
11218
|
+
{ type: "TextQuoteSelector", exact: citation.exact }
|
|
11219
|
+
]
|
|
11220
|
+
},
|
|
11221
|
+
body: { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
|
|
10898
11222
|
},
|
|
10899
|
-
|
|
10900
|
-
|
|
10901
|
-
|
|
10902
|
-
|
|
10903
|
-
await emitEvent(session, "mark:create", { annotation: citationRef, userId, resourceId: newResourceId });
|
|
11223
|
+
generator
|
|
11224
|
+
);
|
|
11225
|
+
await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
|
|
11226
|
+
}
|
|
10904
11227
|
}
|
|
10905
11228
|
await emitEvent(session, "job:complete", {
|
|
10906
11229
|
...lifecycleBase,
|
|
@@ -11052,6 +11375,10 @@ async function startAgentWorker(opts) {
|
|
|
11052
11375
|
jobTypes: group.jobTypes,
|
|
11053
11376
|
inferenceClient: group.client,
|
|
11054
11377
|
generator,
|
|
11378
|
+
// The extraction seam's cache, over this worker's content transport
|
|
11379
|
+
// (PERSIST-ANCHORS P2d). Built here because the client keeps its
|
|
11380
|
+
// content transport private — this is where it is in hand.
|
|
11381
|
+
anchoredTextStore: anchoredTextStoreOverTransport(content, logger2),
|
|
11055
11382
|
logger: logger2
|
|
11056
11383
|
});
|
|
11057
11384
|
logger2.info("Agent ready", {
|