@semiont/jobs 0.5.24 → 0.5.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +16 -20
- package/dist/index.js +412 -149
- package/dist/index.js.map +1 -1
- package/dist/worker-main.js +577 -240
- package/dist/worker-main.js.map +1 -1
- package/package.json +10 -9
package/dist/worker-main.js
CHANGED
|
@@ -1,14 +1,15 @@
|
|
|
1
|
-
import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest,
|
|
2
|
-
import { deriveStorageUri, extractPdfTextLayer,
|
|
1
|
+
import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText, isObject, isString, deriveViews } from '@semiont/core';
|
|
2
|
+
import { anchoredTextStoreOverTransport, deriveStorageUri, extractPdfTextLayer, EXTRACTORS, calculateChecksum, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
|
|
3
|
+
import { execFileSync } from 'child_process';
|
|
4
|
+
import { existsSync, readFileSync, mkdtempSync, writeFileSync, rmSync } from 'fs';
|
|
5
|
+
import { homedir, hostname, tmpdir } from 'os';
|
|
6
|
+
import { join } from 'path';
|
|
3
7
|
import { generateAnnotationId } from '@semiont/event-sourcing';
|
|
4
8
|
import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
|
|
5
|
-
import { homedir, hostname } from 'os';
|
|
6
9
|
import { InMemorySessionStorage, setStoredSession, kbBackendUrl, SemiontClient, SemiontSession } from '@semiont/sdk';
|
|
7
10
|
import { HttpTransport, HttpContentTransport } from '@semiont/http-transport';
|
|
8
11
|
import { createInferenceClient } from '@semiont/inference';
|
|
9
12
|
import { createServer } from 'http';
|
|
10
|
-
import { existsSync, readFileSync } from 'fs';
|
|
11
|
-
import { join } from 'path';
|
|
12
13
|
import { createProcessLogger } from '@semiont/observability/process-logger';
|
|
13
14
|
|
|
14
15
|
var __create = Object.create;
|
|
@@ -3765,9 +3766,9 @@ var require_mapOneOrManyArgs = __commonJS({
|
|
|
3765
3766
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3766
3767
|
exports.mapOneOrManyArgs = void 0;
|
|
3767
3768
|
var map_1 = require_map();
|
|
3768
|
-
var
|
|
3769
|
+
var isArray = Array.isArray;
|
|
3769
3770
|
function callOrApply(fn, args) {
|
|
3770
|
-
return
|
|
3771
|
+
return isArray(args) ? fn.apply(void 0, __spreadArray([], __read(args))) : fn(args);
|
|
3771
3772
|
}
|
|
3772
3773
|
function mapOneOrManyArgs(fn) {
|
|
3773
3774
|
return map_1.map(function(args) {
|
|
@@ -3912,14 +3913,14 @@ var require_argsArgArrayOrObject = __commonJS({
|
|
|
3912
3913
|
"../../node_modules/rxjs/dist/cjs/internal/util/argsArgArrayOrObject.js"(exports) {
|
|
3913
3914
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3914
3915
|
exports.argsArgArrayOrObject = void 0;
|
|
3915
|
-
var
|
|
3916
|
+
var isArray = Array.isArray;
|
|
3916
3917
|
var getPrototypeOf = Object.getPrototypeOf;
|
|
3917
3918
|
var objectProto = Object.prototype;
|
|
3918
3919
|
var getKeys = Object.keys;
|
|
3919
3920
|
function argsArgArrayOrObject(args) {
|
|
3920
3921
|
if (args.length === 1) {
|
|
3921
3922
|
var first_1 = args[0];
|
|
3922
|
-
if (
|
|
3923
|
+
if (isArray(first_1)) {
|
|
3923
3924
|
return { args: first_1, keys: null };
|
|
3924
3925
|
}
|
|
3925
3926
|
if (isPOJO(first_1)) {
|
|
@@ -4667,9 +4668,9 @@ var require_argsOrArgArray = __commonJS({
|
|
|
4667
4668
|
"../../node_modules/rxjs/dist/cjs/internal/util/argsOrArgArray.js"(exports) {
|
|
4668
4669
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
4669
4670
|
exports.argsOrArgArray = void 0;
|
|
4670
|
-
var
|
|
4671
|
+
var isArray = Array.isArray;
|
|
4671
4672
|
function argsOrArgArray(args) {
|
|
4672
|
-
return args.length === 1 &&
|
|
4673
|
+
return args.length === 1 && isArray(args[0]) ? args[0] : args;
|
|
4673
4674
|
}
|
|
4674
4675
|
exports.argsOrArgArray = argsOrArgArray;
|
|
4675
4676
|
}
|
|
@@ -9329,6 +9330,25 @@ function createJobClaimAdapter(options) {
|
|
|
9329
9330
|
};
|
|
9330
9331
|
}
|
|
9331
9332
|
|
|
9333
|
+
// src/types.ts
|
|
9334
|
+
var JOB_TYPES = /* @__PURE__ */ new Set([
|
|
9335
|
+
"reference-annotation",
|
|
9336
|
+
"generation",
|
|
9337
|
+
"highlight-annotation",
|
|
9338
|
+
"assessment-annotation",
|
|
9339
|
+
"comment-annotation",
|
|
9340
|
+
"tag-annotation"
|
|
9341
|
+
]);
|
|
9342
|
+
function isJobType(value) {
|
|
9343
|
+
return JOB_TYPES.has(value);
|
|
9344
|
+
}
|
|
9345
|
+
function asJobParams(params) {
|
|
9346
|
+
if (typeof params.resourceId !== "string") {
|
|
9347
|
+
throw new Error("Job params are missing a resourceId");
|
|
9348
|
+
}
|
|
9349
|
+
return params;
|
|
9350
|
+
}
|
|
9351
|
+
|
|
9332
9352
|
// src/workers/inference-call.ts
|
|
9333
9353
|
var INFERENCE_TIMEOUT_MS = 10 * 6e4;
|
|
9334
9354
|
async function withTimeout(work, label) {
|
|
@@ -9351,18 +9371,51 @@ async function withTimeout(work, label) {
|
|
|
9351
9371
|
clearTimeout(timer);
|
|
9352
9372
|
}
|
|
9353
9373
|
}
|
|
9354
|
-
function boundedGenerate(client, prompt, maxTokens, temperature
|
|
9374
|
+
function boundedGenerate(client, prompt, maxTokens, temperature) {
|
|
9355
9375
|
return withTimeout(
|
|
9356
|
-
client.generateText(prompt, maxTokens, temperature
|
|
9376
|
+
client.generateText(prompt, maxTokens, temperature),
|
|
9357
9377
|
`${client.type}:${client.modelId}`
|
|
9358
9378
|
);
|
|
9359
9379
|
}
|
|
9360
|
-
function
|
|
9380
|
+
function boundedGenerateStructured(client, prompt, maxTokens, temperature, elementSchema) {
|
|
9361
9381
|
return withTimeout(
|
|
9362
|
-
client.
|
|
9382
|
+
client.generateStructured(prompt, maxTokens, temperature, elementSchema),
|
|
9363
9383
|
`${client.type}:${client.modelId}`
|
|
9364
9384
|
);
|
|
9365
9385
|
}
|
|
9386
|
+
|
|
9387
|
+
// src/workers/detection/detection-chunking.ts
|
|
9388
|
+
var SELECTOR_CONTEXT_CHARS = 64;
|
|
9389
|
+
var OVERLAP_CHARS = SELECTOR_CONTEXT_CHARS + // prefix
|
|
9390
|
+
SELECTOR_CONTEXT_CHARS + // suffix
|
|
9391
|
+
2 * SELECTOR_CONTEXT_CHARS;
|
|
9392
|
+
var OVERLAP_TOKENS = Math.ceil(OVERLAP_CHARS / 4);
|
|
9393
|
+
function deriveDetectionBudget(limits, scaffoldTokens) {
|
|
9394
|
+
const { contextTokens, maxOutputTokens } = limits;
|
|
9395
|
+
const available = contextTokens - scaffoldTokens;
|
|
9396
|
+
let inputBudget;
|
|
9397
|
+
let outputBudget;
|
|
9398
|
+
if (maxOutputTokens >= contextTokens) {
|
|
9399
|
+
inputBudget = Math.floor(available / 3);
|
|
9400
|
+
outputBudget = available - inputBudget;
|
|
9401
|
+
} else {
|
|
9402
|
+
outputBudget = maxOutputTokens;
|
|
9403
|
+
inputBudget = contextTokens - outputBudget - scaffoldTokens;
|
|
9404
|
+
if (inputBudget <= 0) {
|
|
9405
|
+
inputBudget = Math.floor(available / 3);
|
|
9406
|
+
outputBudget = available - inputBudget;
|
|
9407
|
+
}
|
|
9408
|
+
}
|
|
9409
|
+
if (inputBudget <= OVERLAP_TOKENS) {
|
|
9410
|
+
throw new Error(
|
|
9411
|
+
`Inference window too small for detection: context ${contextTokens} tokens minus scaffold ${scaffoldTokens} leaves an input budget of ${inputBudget} (need > ${OVERLAP_TOKENS}). Use a model with a larger context window or reduce the prompt scaffold.`
|
|
9412
|
+
);
|
|
9413
|
+
}
|
|
9414
|
+
return {
|
|
9415
|
+
chunking: { chunkSize: inputBudget, overlap: OVERLAP_TOKENS },
|
|
9416
|
+
outputBudget
|
|
9417
|
+
};
|
|
9418
|
+
}
|
|
9366
9419
|
function languageName(tag) {
|
|
9367
9420
|
return getLocaleEnglishName(tag) || tag;
|
|
9368
9421
|
}
|
|
@@ -9382,7 +9435,9 @@ var MotivationPrompts = class {
|
|
|
9382
9435
|
/**
|
|
9383
9436
|
* Build a prompt for detecting comment-worthy passages
|
|
9384
9437
|
*
|
|
9385
|
-
* @param content - The text content to analyze
|
|
9438
|
+
* @param content - The text content to analyze — a chunk sized by the
|
|
9439
|
+
* caller from derived provider limits; NEVER re-truncated here (a
|
|
9440
|
+
* builder-level clip is silent input loss — see #738)
|
|
9386
9441
|
* @param instructions - Optional user-provided instructions
|
|
9387
9442
|
* @param tone - Optional tone guidance (e.g., "academic", "conversational")
|
|
9388
9443
|
* @param density - Optional target number of comments per 2000 words
|
|
@@ -9403,7 +9458,7 @@ ${instructions}${toneGuidance}${densityGuidance}${sourceLang}${bodyLang}
|
|
|
9403
9458
|
|
|
9404
9459
|
Text to analyze:
|
|
9405
9460
|
---
|
|
9406
|
-
${content
|
|
9461
|
+
${content}
|
|
9407
9462
|
---
|
|
9408
9463
|
|
|
9409
9464
|
Return a JSON array of comments. Each comment must have:
|
|
@@ -9437,7 +9492,7 @@ Guidelines:
|
|
|
9437
9492
|
|
|
9438
9493
|
Text to analyze:
|
|
9439
9494
|
---
|
|
9440
|
-
${content
|
|
9495
|
+
${content}
|
|
9441
9496
|
---
|
|
9442
9497
|
|
|
9443
9498
|
Return a JSON array of comments. Each comment should have:
|
|
@@ -9458,7 +9513,9 @@ Example format:
|
|
|
9458
9513
|
/**
|
|
9459
9514
|
* Build a prompt for detecting highlight-worthy passages
|
|
9460
9515
|
*
|
|
9461
|
-
* @param content - The text content to analyze
|
|
9516
|
+
* @param content - The text content to analyze — a chunk sized by the
|
|
9517
|
+
* caller from derived provider limits; NEVER re-truncated here (a
|
|
9518
|
+
* builder-level clip is silent input loss — see #738)
|
|
9462
9519
|
* @param instructions - Optional user-provided instructions
|
|
9463
9520
|
* @param density - Optional target number of highlights per 2000 words
|
|
9464
9521
|
* @returns Formatted prompt string
|
|
@@ -9476,7 +9533,7 @@ ${instructions}${densityGuidance}${sourceLang}
|
|
|
9476
9533
|
|
|
9477
9534
|
Text to analyze:
|
|
9478
9535
|
---
|
|
9479
|
-
${content
|
|
9536
|
+
${content}
|
|
9480
9537
|
---
|
|
9481
9538
|
|
|
9482
9539
|
Return a JSON array of highlights. Each highlight must have:
|
|
@@ -9507,7 +9564,7 @@ Guidelines:
|
|
|
9507
9564
|
|
|
9508
9565
|
Text to analyze:
|
|
9509
9566
|
---
|
|
9510
|
-
${content
|
|
9567
|
+
${content}
|
|
9511
9568
|
---
|
|
9512
9569
|
|
|
9513
9570
|
Return a JSON array of highlights. Each highlight should have:
|
|
@@ -9527,7 +9584,9 @@ Example format:
|
|
|
9527
9584
|
/**
|
|
9528
9585
|
* Build a prompt for detecting assessment-worthy passages
|
|
9529
9586
|
*
|
|
9530
|
-
* @param content - The text content to analyze
|
|
9587
|
+
* @param content - The text content to analyze — a chunk sized by the
|
|
9588
|
+
* caller from derived provider limits; NEVER re-truncated here (a
|
|
9589
|
+
* builder-level clip is silent input loss — see #738)
|
|
9531
9590
|
* @param instructions - Optional user-provided instructions
|
|
9532
9591
|
* @param tone - Optional tone guidance (e.g., "critical", "supportive")
|
|
9533
9592
|
* @param density - Optional target number of assessments per 2000 words
|
|
@@ -9548,7 +9607,7 @@ ${instructions}${toneGuidance}${densityGuidance}${sourceLang}${bodyLang}
|
|
|
9548
9607
|
|
|
9549
9608
|
Text to analyze:
|
|
9550
9609
|
---
|
|
9551
|
-
${content
|
|
9610
|
+
${content}
|
|
9552
9611
|
---
|
|
9553
9612
|
|
|
9554
9613
|
Return a JSON array of assessments. Each assessment must have:
|
|
@@ -9582,7 +9641,7 @@ Guidelines:
|
|
|
9582
9641
|
|
|
9583
9642
|
Text to analyze:
|
|
9584
9643
|
---
|
|
9585
|
-
${content
|
|
9644
|
+
${content}
|
|
9586
9645
|
---
|
|
9587
9646
|
|
|
9588
9647
|
Return a JSON array of assessments. Each assessment should have:
|
|
@@ -9654,32 +9713,57 @@ Example format:
|
|
|
9654
9713
|
return prompt;
|
|
9655
9714
|
}
|
|
9656
9715
|
};
|
|
9657
|
-
|
|
9658
|
-
|
|
9659
|
-
|
|
9660
|
-
|
|
9661
|
-
|
|
9662
|
-
|
|
9663
|
-
|
|
9664
|
-
|
|
9665
|
-
|
|
9666
|
-
|
|
9667
|
-
|
|
9668
|
-
|
|
9669
|
-
|
|
9670
|
-
|
|
9671
|
-
}
|
|
9716
|
+
var COMMENT_ELEMENT_SCHEMA = {
|
|
9717
|
+
type: "object",
|
|
9718
|
+
properties: {
|
|
9719
|
+
exact: { type: "string" },
|
|
9720
|
+
prefix: { type: "string" },
|
|
9721
|
+
suffix: { type: "string" },
|
|
9722
|
+
comment: { type: "string" }
|
|
9723
|
+
},
|
|
9724
|
+
required: ["exact", "comment"],
|
|
9725
|
+
additionalProperties: false
|
|
9726
|
+
};
|
|
9727
|
+
var HIGHLIGHT_ELEMENT_SCHEMA = {
|
|
9728
|
+
type: "object",
|
|
9729
|
+
properties: {
|
|
9730
|
+
exact: { type: "string" },
|
|
9731
|
+
prefix: { type: "string" },
|
|
9732
|
+
suffix: { type: "string" }
|
|
9733
|
+
},
|
|
9734
|
+
required: ["exact"],
|
|
9735
|
+
additionalProperties: false
|
|
9736
|
+
};
|
|
9737
|
+
var ASSESSMENT_ELEMENT_SCHEMA = {
|
|
9738
|
+
type: "object",
|
|
9739
|
+
properties: {
|
|
9740
|
+
exact: { type: "string" },
|
|
9741
|
+
prefix: { type: "string" },
|
|
9742
|
+
suffix: { type: "string" },
|
|
9743
|
+
assessment: { type: "string" }
|
|
9744
|
+
},
|
|
9745
|
+
required: ["exact", "assessment"],
|
|
9746
|
+
additionalProperties: false
|
|
9747
|
+
};
|
|
9748
|
+
var TAG_ELEMENT_SCHEMA = {
|
|
9749
|
+
type: "object",
|
|
9750
|
+
properties: {
|
|
9751
|
+
exact: { type: "string" },
|
|
9752
|
+
prefix: { type: "string" },
|
|
9753
|
+
suffix: { type: "string" }
|
|
9754
|
+
},
|
|
9755
|
+
required: ["exact"],
|
|
9756
|
+
additionalProperties: false
|
|
9757
|
+
};
|
|
9672
9758
|
var MotivationParsers = class {
|
|
9673
9759
|
/**
|
|
9674
|
-
*
|
|
9760
|
+
* Validate and reconcile structured comment elements.
|
|
9675
9761
|
*
|
|
9676
|
-
* @param
|
|
9762
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
9677
9763
|
* @param content - Original content to validate offsets against
|
|
9678
9764
|
* @returns Array of validated comment matches
|
|
9679
|
-
* @throws if the response is not a parseable JSON array
|
|
9680
9765
|
*/
|
|
9681
|
-
static parseComments(
|
|
9682
|
-
const parsed = parseJsonArray(response, "comment");
|
|
9766
|
+
static parseComments(parsed, content) {
|
|
9683
9767
|
const valid = parsed.filter(
|
|
9684
9768
|
(c) => isObject(c) && isString(c.exact) && isString(c.comment) && c.comment.trim().length > 0
|
|
9685
9769
|
);
|
|
@@ -9708,15 +9792,13 @@ var MotivationParsers = class {
|
|
|
9708
9792
|
return validatedComments;
|
|
9709
9793
|
}
|
|
9710
9794
|
/**
|
|
9711
|
-
*
|
|
9795
|
+
* Validate and reconcile structured highlight elements.
|
|
9712
9796
|
*
|
|
9713
|
-
* @param
|
|
9797
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
9714
9798
|
* @param content - Original content to validate offsets against
|
|
9715
9799
|
* @returns Array of validated highlight matches
|
|
9716
|
-
* @throws if the response is not a parseable JSON array
|
|
9717
9800
|
*/
|
|
9718
|
-
static parseHighlights(
|
|
9719
|
-
const parsed = parseJsonArray(response, "highlight");
|
|
9801
|
+
static parseHighlights(parsed, content) {
|
|
9720
9802
|
const highlights = parsed.filter(
|
|
9721
9803
|
(h) => isObject(h) && isString(h.exact)
|
|
9722
9804
|
);
|
|
@@ -9743,15 +9825,13 @@ var MotivationParsers = class {
|
|
|
9743
9825
|
return validatedHighlights;
|
|
9744
9826
|
}
|
|
9745
9827
|
/**
|
|
9746
|
-
*
|
|
9828
|
+
* Validate and reconcile structured assessment elements.
|
|
9747
9829
|
*
|
|
9748
|
-
* @param
|
|
9830
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
9749
9831
|
* @param content - Original content to validate offsets against
|
|
9750
9832
|
* @returns Array of validated assessment matches
|
|
9751
|
-
* @throws if the response is not a parseable JSON array
|
|
9752
9833
|
*/
|
|
9753
|
-
static parseAssessments(
|
|
9754
|
-
const parsed = parseJsonArray(response, "assessment");
|
|
9834
|
+
static parseAssessments(parsed, content) {
|
|
9755
9835
|
const assessments = parsed.filter(
|
|
9756
9836
|
(a) => isObject(a) && isString(a.exact) && isString(a.assessment)
|
|
9757
9837
|
);
|
|
@@ -9779,14 +9859,13 @@ var MotivationParsers = class {
|
|
|
9779
9859
|
return validatedAssessments;
|
|
9780
9860
|
}
|
|
9781
9861
|
/**
|
|
9782
|
-
*
|
|
9862
|
+
* Validate structured tag elements into raw, pre-reconciliation tag inputs.
|
|
9783
9863
|
* Reconciliation happens in `validateTagOffsets`, which adds `start`/`end`
|
|
9784
9864
|
* by anchoring `exact` against the source content.
|
|
9785
9865
|
*
|
|
9786
|
-
* @
|
|
9866
|
+
* @param parsed - Already-parsed elements from the structured surface
|
|
9787
9867
|
*/
|
|
9788
|
-
static parseTags(
|
|
9789
|
-
const parsed = parseJsonArray(response, "tag");
|
|
9868
|
+
static parseTags(parsed) {
|
|
9790
9869
|
const valid = parsed.filter(
|
|
9791
9870
|
(t) => isObject(t) && isString(t.exact) && t.exact.trim().length > 0
|
|
9792
9871
|
);
|
|
@@ -9828,10 +9907,32 @@ function logAnchorMethod(motivation, exact, anchorMethod) {
|
|
|
9828
9907
|
}
|
|
9829
9908
|
|
|
9830
9909
|
// src/workers/annotation-detection.ts
|
|
9831
|
-
function assertNotTruncated(response, motivation) {
|
|
9910
|
+
function assertNotTruncated(response, motivation, chunk, totalChunks, outputBudget) {
|
|
9832
9911
|
if (response.stopReason === "max_tokens") {
|
|
9833
|
-
throw new Error(`${motivation} detection response truncated (max_tokens)
|
|
9912
|
+
throw new Error(`${motivation} detection response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
|
|
9913
|
+
}
|
|
9914
|
+
}
|
|
9915
|
+
async function detectInChunks(client, content, buildPrompt, temperature, motivation, elementSchema, parse, onChunk) {
|
|
9916
|
+
const limits = await client.limits();
|
|
9917
|
+
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
9918
|
+
const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
|
|
9919
|
+
const chunks = chunkText(content, chunking);
|
|
9920
|
+
const collected = [];
|
|
9921
|
+
for (let i = 0; i < chunks.length; i++) {
|
|
9922
|
+
const response = await boundedGenerateStructured(
|
|
9923
|
+
client,
|
|
9924
|
+
buildPrompt(chunks[i]),
|
|
9925
|
+
outputBudget,
|
|
9926
|
+
temperature,
|
|
9927
|
+
elementSchema
|
|
9928
|
+
);
|
|
9929
|
+
assertNotTruncated(response, motivation, i + 1, chunks.length, outputBudget);
|
|
9930
|
+
collected.push(...parse(response.items));
|
|
9931
|
+
if (i < chunks.length - 1) {
|
|
9932
|
+
onChunk?.(i + 1, chunks.length);
|
|
9933
|
+
}
|
|
9834
9934
|
}
|
|
9935
|
+
return collected;
|
|
9835
9936
|
}
|
|
9836
9937
|
var AnnotationDetection = class {
|
|
9837
9938
|
/**
|
|
@@ -9842,11 +9943,17 @@ var AnnotationDetection = class {
|
|
|
9842
9943
|
* (source-resource locale). See `types.ts` "Locale conventions" for the
|
|
9843
9944
|
* full discussion.
|
|
9844
9945
|
*/
|
|
9845
|
-
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage) {
|
|
9846
|
-
|
|
9847
|
-
|
|
9848
|
-
|
|
9849
|
-
|
|
9946
|
+
static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onChunk) {
|
|
9947
|
+
return detectInChunks(
|
|
9948
|
+
client,
|
|
9949
|
+
content,
|
|
9950
|
+
(chunk) => MotivationPrompts.buildCommentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
|
|
9951
|
+
0.4,
|
|
9952
|
+
"comment",
|
|
9953
|
+
COMMENT_ELEMENT_SCHEMA,
|
|
9954
|
+
(items) => MotivationParsers.parseComments(items, content),
|
|
9955
|
+
onChunk
|
|
9956
|
+
);
|
|
9850
9957
|
}
|
|
9851
9958
|
/**
|
|
9852
9959
|
* Detect highlights in content.
|
|
@@ -9855,11 +9962,17 @@ var AnnotationDetection = class {
|
|
|
9855
9962
|
* applies, used in the prompt so the LLM analyzes non-English source
|
|
9856
9963
|
* correctly.
|
|
9857
9964
|
*/
|
|
9858
|
-
static async detectHighlights(content, client, instructions, density, sourceLanguage) {
|
|
9859
|
-
|
|
9860
|
-
|
|
9861
|
-
|
|
9862
|
-
|
|
9965
|
+
static async detectHighlights(content, client, instructions, density, sourceLanguage, onChunk) {
|
|
9966
|
+
return detectInChunks(
|
|
9967
|
+
client,
|
|
9968
|
+
content,
|
|
9969
|
+
(chunk) => MotivationPrompts.buildHighlightPrompt(chunk, instructions, density, sourceLanguage),
|
|
9970
|
+
0.3,
|
|
9971
|
+
"highlight",
|
|
9972
|
+
HIGHLIGHT_ELEMENT_SCHEMA,
|
|
9973
|
+
(items) => MotivationParsers.parseHighlights(items, content),
|
|
9974
|
+
onChunk
|
|
9975
|
+
);
|
|
9863
9976
|
}
|
|
9864
9977
|
/**
|
|
9865
9978
|
* Detect assessments in content.
|
|
@@ -9868,11 +9981,17 @@ var AnnotationDetection = class {
|
|
|
9868
9981
|
* (annotation body locale). `sourceLanguage` is the locale of the content
|
|
9869
9982
|
* being analyzed (source-resource locale).
|
|
9870
9983
|
*/
|
|
9871
|
-
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage) {
|
|
9872
|
-
|
|
9873
|
-
|
|
9874
|
-
|
|
9875
|
-
|
|
9984
|
+
static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onChunk) {
|
|
9985
|
+
return detectInChunks(
|
|
9986
|
+
client,
|
|
9987
|
+
content,
|
|
9988
|
+
(chunk) => MotivationPrompts.buildAssessmentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
|
|
9989
|
+
0.3,
|
|
9990
|
+
"assessment",
|
|
9991
|
+
ASSESSMENT_ELEMENT_SCHEMA,
|
|
9992
|
+
(items) => MotivationParsers.parseAssessments(items, content),
|
|
9993
|
+
onChunk
|
|
9994
|
+
);
|
|
9876
9995
|
}
|
|
9877
9996
|
/**
|
|
9878
9997
|
* Detect tags in content for a specific category.
|
|
@@ -9886,28 +10005,45 @@ var AnnotationDetection = class {
|
|
|
9886
10005
|
* identifiers, not LLM-generated text — so it's consumed at the body-stamp
|
|
9887
10006
|
* site, not here.
|
|
9888
10007
|
*/
|
|
9889
|
-
static async detectTags(content, client, schema, category, sourceLanguage) {
|
|
10008
|
+
static async detectTags(content, client, schema, category, sourceLanguage, onChunk) {
|
|
9890
10009
|
const categoryInfo = schema.tags.find((t) => t.name === category);
|
|
9891
10010
|
if (!categoryInfo) {
|
|
9892
10011
|
throw new Error(`Invalid category "${category}" for schema ${schema.id}`);
|
|
9893
10012
|
}
|
|
9894
|
-
const
|
|
10013
|
+
const parsedTags = await detectInChunks(
|
|
10014
|
+
client,
|
|
9895
10015
|
content,
|
|
9896
|
-
|
|
9897
|
-
|
|
9898
|
-
|
|
9899
|
-
|
|
9900
|
-
|
|
9901
|
-
|
|
9902
|
-
|
|
10016
|
+
(chunk) => MotivationPrompts.buildTagPrompt(
|
|
10017
|
+
chunk,
|
|
10018
|
+
category,
|
|
10019
|
+
schema.name,
|
|
10020
|
+
schema.description,
|
|
10021
|
+
schema.domain,
|
|
10022
|
+
categoryInfo.description,
|
|
10023
|
+
categoryInfo.examples,
|
|
10024
|
+
sourceLanguage
|
|
10025
|
+
),
|
|
10026
|
+
0.2,
|
|
10027
|
+
"tag",
|
|
10028
|
+
TAG_ELEMENT_SCHEMA,
|
|
10029
|
+
(items) => MotivationParsers.parseTags(items),
|
|
10030
|
+
onChunk
|
|
9903
10031
|
);
|
|
9904
|
-
const response = await boundedGenerateWithMetadata(client, prompt, 4e3, 0.2, { format: "json" });
|
|
9905
|
-
assertNotTruncated(response, "tag");
|
|
9906
|
-
const parsedTags = MotivationParsers.parseTags(response.text);
|
|
9907
10032
|
return MotivationParsers.validateTagOffsets(parsedTags, content, category);
|
|
9908
10033
|
}
|
|
9909
10034
|
};
|
|
9910
|
-
|
|
10035
|
+
var ENTITY_ELEMENT_SCHEMA = {
|
|
10036
|
+
type: "object",
|
|
10037
|
+
properties: {
|
|
10038
|
+
exact: { type: "string" },
|
|
10039
|
+
entityType: { type: "string" },
|
|
10040
|
+
prefix: { type: "string" },
|
|
10041
|
+
suffix: { type: "string" }
|
|
10042
|
+
},
|
|
10043
|
+
required: ["exact", "entityType"],
|
|
10044
|
+
additionalProperties: false
|
|
10045
|
+
};
|
|
10046
|
+
async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage, onChunk) {
|
|
9911
10047
|
const entityTypesDescription = entityTypes.map((et) => {
|
|
9912
10048
|
if (typeof et === "string") {
|
|
9913
10049
|
return et;
|
|
@@ -9939,11 +10075,11 @@ Find direct mentions only (names, proper nouns). Do not include pronouns or desc
|
|
|
9939
10075
|
const sourceLangGuidance = sourceLanguage ? `
|
|
9940
10076
|
Source text language: ${getLocaleEnglishName(sourceLanguage) || sourceLanguage}.
|
|
9941
10077
|
` : "";
|
|
9942
|
-
const
|
|
10078
|
+
const buildPrompt = (text) => `Identify entity references in the following text. Look for mentions of: ${entityTypesDescription}.
|
|
9943
10079
|
${descriptiveReferenceGuidance}${sourceLangGuidance}
|
|
9944
10080
|
Text to analyze:
|
|
9945
10081
|
"""
|
|
9946
|
-
${
|
|
10082
|
+
${text}
|
|
9947
10083
|
"""
|
|
9948
10084
|
|
|
9949
10085
|
Respond with a JSON array of entities found. Each entity should have:
|
|
@@ -9956,59 +10092,53 @@ If no entities are found, respond with an empty array [].
|
|
|
9956
10092
|
|
|
9957
10093
|
Example output:
|
|
9958
10094
|
[{"exact":"Alice","entityType":"Person","prefix":"","suffix":" went to"},{"exact":"Paris","entityType":"Location","prefix":"went to ","suffix":" yesterday"}]`;
|
|
9959
|
-
|
|
9960
|
-
const
|
|
9961
|
-
|
|
9962
|
-
|
|
9963
|
-
|
|
9964
|
-
|
|
9965
|
-
|
|
9966
|
-
|
|
9967
|
-
|
|
9968
|
-
|
|
9969
|
-
|
|
9970
|
-
|
|
9971
|
-
|
|
9972
|
-
|
|
9973
|
-
|
|
9974
|
-
|
|
9975
|
-
|
|
9976
|
-
|
|
9977
|
-
|
|
9978
|
-
|
|
9979
|
-
|
|
9980
|
-
|
|
9981
|
-
|
|
9982
|
-
|
|
9983
|
-
entities = JSON.parse(response.text.trim());
|
|
9984
|
-
} catch (error) {
|
|
9985
|
-
logger2.error("Failed to parse entity extraction response", {
|
|
9986
|
-
error: error instanceof Error ? error.message : String(error),
|
|
9987
|
-
response: response.text.slice(0, 500)
|
|
9988
|
-
});
|
|
9989
|
-
throw new Error("Failed to parse entity extraction response", {
|
|
9990
|
-
cause: error instanceof Error ? error : new Error(String(error))
|
|
10095
|
+
const limits = await client.limits();
|
|
10096
|
+
const scaffoldTokens = estimateTokens(buildPrompt(""));
|
|
10097
|
+
const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
|
|
10098
|
+
const chunks = chunkText(exact, chunking);
|
|
10099
|
+
logger2.debug("Sending entity extraction request", {
|
|
10100
|
+
entityTypes: entityTypesDescription,
|
|
10101
|
+
chunks: chunks.length,
|
|
10102
|
+
chunkSizeTokens: chunking.chunkSize,
|
|
10103
|
+
outputBudget
|
|
10104
|
+
});
|
|
10105
|
+
const collected = [];
|
|
10106
|
+
for (let i = 0; i < chunks.length; i++) {
|
|
10107
|
+
const response = await boundedGenerateStructured(
|
|
10108
|
+
client,
|
|
10109
|
+
buildPrompt(chunks[i]),
|
|
10110
|
+
outputBudget,
|
|
10111
|
+
0.3,
|
|
10112
|
+
// Lower temperature for more consistent extraction
|
|
10113
|
+
ENTITY_ELEMENT_SCHEMA
|
|
10114
|
+
);
|
|
10115
|
+
logger2.debug("Got entity extraction response", {
|
|
10116
|
+
chunk: i + 1,
|
|
10117
|
+
chunks: chunks.length,
|
|
10118
|
+
items: response.items.length
|
|
9991
10119
|
});
|
|
10120
|
+
if (response.stopReason === "max_tokens") {
|
|
10121
|
+
const errorMsg = `Entity extraction response truncated (max_tokens) on chunk ${i + 1}/${chunks.length} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than dropping annotations.`;
|
|
10122
|
+
logger2.error(errorMsg, { items: response.items.length });
|
|
10123
|
+
throw new Error(errorMsg);
|
|
10124
|
+
}
|
|
10125
|
+
for (const e of response.items) {
|
|
10126
|
+
if (isObject(e) && isString(e.exact) && isString(e.entityType)) {
|
|
10127
|
+
collected.push({
|
|
10128
|
+
exact: e.exact,
|
|
10129
|
+
entityType: e.entityType,
|
|
10130
|
+
...isString(e.prefix) ? { prefix: e.prefix } : {},
|
|
10131
|
+
...isString(e.suffix) ? { suffix: e.suffix } : {}
|
|
10132
|
+
});
|
|
10133
|
+
} else {
|
|
10134
|
+
logger2.debug("Dropped malformed LLM entity", { entity: e });
|
|
10135
|
+
}
|
|
10136
|
+
}
|
|
10137
|
+
if (i < chunks.length - 1) {
|
|
10138
|
+
onChunk?.(i + 1, chunks.length);
|
|
10139
|
+
}
|
|
9992
10140
|
}
|
|
9993
|
-
|
|
9994
|
-
logger2.error("Failed to parse entity extraction response: expected a JSON array", {
|
|
9995
|
-
response: response.text.slice(0, 500)
|
|
9996
|
-
});
|
|
9997
|
-
throw new Error("Failed to parse entity extraction response: expected a JSON array");
|
|
9998
|
-
}
|
|
9999
|
-
logger2.debug("Parsed entities from AI response", { count: entities.length });
|
|
10000
|
-
return entities.filter((e) => {
|
|
10001
|
-
const ok = isObject(e) && isString(e.exact) && isString(e.entityType);
|
|
10002
|
-
if (!ok) {
|
|
10003
|
-
logger2.debug("Dropped malformed LLM entity", { entity: e });
|
|
10004
|
-
}
|
|
10005
|
-
return ok;
|
|
10006
|
-
}).map((entity) => ({
|
|
10007
|
-
exact: entity.exact,
|
|
10008
|
-
entityType: entity.entityType,
|
|
10009
|
-
...isString(entity.prefix) ? { prefix: entity.prefix } : {},
|
|
10010
|
-
...isString(entity.suffix) ? { suffix: entity.suffix } : {}
|
|
10011
|
-
}));
|
|
10141
|
+
return collected;
|
|
10012
10142
|
}
|
|
10013
10143
|
function getLanguageName(locale) {
|
|
10014
10144
|
return getLocaleEnglishName(locale) || locale;
|
|
@@ -10019,7 +10149,7 @@ var SEMANTIC_MATCH_CHARS = 240;
|
|
|
10019
10149
|
function idLabel(resourceId, annotationId) {
|
|
10020
10150
|
return `[${resourceId}${annotationId ? `/${annotationId}` : ""}]`;
|
|
10021
10151
|
}
|
|
10022
|
-
async function generateResourceFromTopic(topic, entityTypes, client, logger2, userPrompt, locale, context, temperature, maxTokens, sourceLanguage, outputMediaType = "text/markdown", task = "resource", structure, cite = false) {
|
|
10152
|
+
async function generateResourceFromTopic(topic, entityTypes, client, logger2, userPrompt, locale, context, temperature, maxTokens, sourceLanguage, outputMediaType = "text/markdown", task = "resource", structure, cite = false, repair) {
|
|
10023
10153
|
logger2.debug("Generating resource from topic", {
|
|
10024
10154
|
topicPreview: topic.substring(0, 100),
|
|
10025
10155
|
entityTypes,
|
|
@@ -10135,11 +10265,12 @@ ${parts.join("\n")}`;
|
|
|
10135
10265
|
let semanticContextSection = "";
|
|
10136
10266
|
const similar = context?.semanticContext?.similar ?? [];
|
|
10137
10267
|
if (similar.length > 0) {
|
|
10138
|
-
const lines = [...similar].sort((a, b) => b.score - a.score).slice(0, SEMANTIC_MATCH_LIMIT).map((m) => `- ${idLabel(m.resourceId, m.annotationId)} (${m.score.toFixed(2)}) ${m.text.slice(0, SEMANTIC_MATCH_CHARS)}`);
|
|
10268
|
+
const lines = [...similar].sort((a, b) => b.score - a.score).slice(0, SEMANTIC_MATCH_LIMIT).map((m) => `- ${idLabel(m.resourceId, m.annotationId)} (${m.score.toFixed(2)})${m.machineRead ? " [OCR]" : ""} ${m.text.slice(0, SEMANTIC_MATCH_CHARS)}`);
|
|
10269
|
+
const ocrNote = similar.some((m) => m.machineRead) ? "\nPassages marked [OCR] were read from scanned images by character recognition; treat their exact wording and numbers as uncertain, and say so if you rely on one." : "";
|
|
10139
10270
|
semanticContextSection = `
|
|
10140
10271
|
|
|
10141
10272
|
Related passages from the knowledge base:
|
|
10142
|
-
${lines.join("\n")}`;
|
|
10273
|
+
${lines.join("\n")}${ocrNote}`;
|
|
10143
10274
|
}
|
|
10144
10275
|
let leadLine;
|
|
10145
10276
|
if (task === "resource") {
|
|
@@ -10154,11 +10285,12 @@ ${lines.join("\n")}`;
|
|
|
10154
10285
|
Topic: "${topic}"`;
|
|
10155
10286
|
}
|
|
10156
10287
|
const isPlainText = outputMediaType === "text/plain";
|
|
10288
|
+
const isPdf = outputMediaType === "application/pdf";
|
|
10157
10289
|
let structureRequirement = "";
|
|
10158
10290
|
let titleRequirement = "";
|
|
10159
10291
|
if (structure === "sections") {
|
|
10160
|
-
structureRequirement = isPlainText ? "\n- Organize the content into titled sections with well-structured paragraphs" : "\n- Organize the content into titled sections (## Section) with well-structured paragraphs";
|
|
10161
|
-
if (!isPlainText) {
|
|
10292
|
+
structureRequirement = isPdf ? "\n- Organize the content into titled sections (= Heading) with well-structured paragraphs" : isPlainText ? "\n- Organize the content into titled sections with well-structured paragraphs" : "\n- Organize the content into titled sections (## Section) with well-structured paragraphs";
|
|
10293
|
+
if (!isPlainText && !isPdf) {
|
|
10162
10294
|
titleRequirement = "\n- Start with a clear heading (# Title)";
|
|
10163
10295
|
}
|
|
10164
10296
|
} else if (structure === "prose") {
|
|
@@ -10171,11 +10303,20 @@ Topic: "${topic}"`;
|
|
|
10171
10303
|
- Organize the output as: ${structure}`;
|
|
10172
10304
|
}
|
|
10173
10305
|
const citeRequirement = cite ? "\n- Ground every claim in the provided context. Immediately after each claim, cite its source by emitting [[<id>]], where <id> is an id shown in square brackets in the context above (for a passage labeled [abc], emit [[abc]]). Cite only ids that appear in the context." : "";
|
|
10174
|
-
const formatRequirements =
|
|
10306
|
+
const formatRequirements = isPdf ? `- Write the response as Typst markup (the Typst typesetting language \u2014 not markdown, not LaTeX)
|
|
10307
|
+
- Headings are written as = Heading (deeper levels == Subheading); everything else is plain prose paragraphs
|
|
10308
|
+
- Do not emit markdown syntax or code fences` : isPlainText ? `- Write the response as plain text \u2014 no formatting markup (no #, *, backticks, headings, or links)
|
|
10175
10309
|
- Begin with the title on its own first line` : `- Use markdown formatting
|
|
10176
10310
|
- Write the response as markdown`;
|
|
10311
|
+
const repairSection = repair ? `
|
|
10312
|
+
|
|
10313
|
+
Your previous attempt failed to compile. Fix the error and return the complete corrected document \u2014 full source, not a diff.
|
|
10314
|
+
Compile error:
|
|
10315
|
+
${repair.error}
|
|
10316
|
+
Previous source:
|
|
10317
|
+
${repair.source}` : "";
|
|
10177
10318
|
const prompt = `${leadLine}
|
|
10178
|
-
${userPrompt ? `Instruction: ${userPrompt}` : ""}
|
|
10319
|
+
${userPrompt ? `Instruction: ${userPrompt}` : ""}${repairSection}
|
|
10179
10320
|
${entityTypes.length > 0 ? `Focus on these entity types: ${entityTypes.join(", ")}.` : ""}${annotationSection}${contextSection}${resourceSection}${graphSection}${semanticContextSection}${sourceLanguageInstruction}${languageInstruction}
|
|
10180
10321
|
|
|
10181
10322
|
Requirements:
|
|
@@ -10184,7 +10325,7 @@ Requirements:
|
|
|
10184
10325
|
${formatRequirements}`;
|
|
10185
10326
|
const parseResponse = (response2) => {
|
|
10186
10327
|
let content = response2.trim();
|
|
10187
|
-
if (content.startsWith("```markdown") || content.startsWith("```md")) {
|
|
10328
|
+
if (content.startsWith("```markdown") || content.startsWith("```md") || content.startsWith("```typst")) {
|
|
10188
10329
|
content = content.slice(content.indexOf("\n") + 1);
|
|
10189
10330
|
const endIndex = content.lastIndexOf("```");
|
|
10190
10331
|
if (endIndex !== -1) {
|
|
@@ -10219,6 +10360,29 @@ ${formatRequirements}`;
|
|
|
10219
10360
|
});
|
|
10220
10361
|
return result;
|
|
10221
10362
|
}
|
|
10363
|
+
var PINNED_CREATION_TIMESTAMP = 17e8;
|
|
10364
|
+
var MAX_COMPILE_REPAIRS = 2;
|
|
10365
|
+
function compileTypst(source) {
|
|
10366
|
+
const dir = mkdtempSync(join(tmpdir(), "typst-"));
|
|
10367
|
+
try {
|
|
10368
|
+
const inFile = join(dir, "doc.typ");
|
|
10369
|
+
const outFile = join(dir, "doc.pdf");
|
|
10370
|
+
writeFileSync(inFile, source);
|
|
10371
|
+
try {
|
|
10372
|
+
execFileSync(
|
|
10373
|
+
"typst",
|
|
10374
|
+
["compile", "--creation-timestamp", String(PINNED_CREATION_TIMESTAMP), inFile, outFile],
|
|
10375
|
+
{ stdio: ["ignore", "pipe", "pipe"] }
|
|
10376
|
+
);
|
|
10377
|
+
} catch (err) {
|
|
10378
|
+
const stderr = err.stderr;
|
|
10379
|
+
return { error: stderr?.length ? stderr.toString("utf8") : String(err) };
|
|
10380
|
+
}
|
|
10381
|
+
return { pdf: new Uint8Array(readFileSync(outFile)) };
|
|
10382
|
+
} finally {
|
|
10383
|
+
rmSync(dir, { recursive: true, force: true });
|
|
10384
|
+
}
|
|
10385
|
+
}
|
|
10222
10386
|
|
|
10223
10387
|
// src/workers/generation/citation-resolver.ts
|
|
10224
10388
|
var CITATION_TOKEN = /\[\[([^\s[\]/]+)(?:\/([^\s[\]/]+))?\]\]/g;
|
|
@@ -10363,9 +10527,9 @@ function buildTextAnnotation(content, resourceId, userId, generator, motivation,
|
|
|
10363
10527
|
...body !== void 0 ? { body } : {}
|
|
10364
10528
|
};
|
|
10365
10529
|
}
|
|
10366
|
-
function buildPdfAnnotation(
|
|
10367
|
-
const { rects, overlap } = locate(
|
|
10368
|
-
const coveredText = overlap.length ?
|
|
10530
|
+
function buildPdfAnnotation(anchored, resourceId, userId, generator, motivation, match, body) {
|
|
10531
|
+
const { rects, overlap } = locate(anchored, match.start, match.end);
|
|
10532
|
+
const coveredText = overlap.length ? anchored.text.substring(
|
|
10369
10533
|
Math.min(...overlap.map((i) => i.start)),
|
|
10370
10534
|
Math.max(...overlap.map((i) => i.end))
|
|
10371
10535
|
) : "";
|
|
@@ -10414,7 +10578,9 @@ async function processHighlightJob(content, inferenceClient, params, buildAnnota
|
|
|
10414
10578
|
inferenceClient,
|
|
10415
10579
|
params.instructions,
|
|
10416
10580
|
params.density,
|
|
10417
|
-
params.sourceLanguage
|
|
10581
|
+
params.sourceLanguage,
|
|
10582
|
+
// Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
|
|
10583
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
|
|
10418
10584
|
);
|
|
10419
10585
|
onProgress(60, `Creating ${highlights.length} annotations...`, "creating");
|
|
10420
10586
|
const annotations = dedupeAnnotations(highlights.map(
|
|
@@ -10436,7 +10602,9 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
|
|
|
10436
10602
|
params.tone,
|
|
10437
10603
|
params.density,
|
|
10438
10604
|
params.language,
|
|
10439
|
-
params.sourceLanguage
|
|
10605
|
+
params.sourceLanguage,
|
|
10606
|
+
// Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
|
|
10607
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
|
|
10440
10608
|
);
|
|
10441
10609
|
onProgress(60, `Creating ${comments.length} annotations...`, "creating");
|
|
10442
10610
|
const bodyLanguage = params.language ?? "en";
|
|
@@ -10466,7 +10634,9 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
|
|
|
10466
10634
|
params.tone,
|
|
10467
10635
|
params.density,
|
|
10468
10636
|
params.language,
|
|
10469
|
-
params.sourceLanguage
|
|
10637
|
+
params.sourceLanguage,
|
|
10638
|
+
// Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
|
|
10639
|
+
(completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
|
|
10470
10640
|
);
|
|
10471
10641
|
onProgress(60, `Creating ${assessments.length} annotations...`, "creating");
|
|
10472
10642
|
const bodyLanguage = params.language ?? "en";
|
|
@@ -10522,7 +10692,23 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
|
|
|
10522
10692
|
inferenceClient,
|
|
10523
10693
|
params.includeDescriptiveReferences ?? false,
|
|
10524
10694
|
logger2,
|
|
10525
|
-
params.sourceLanguage
|
|
10695
|
+
params.sourceLanguage,
|
|
10696
|
+
// Chunk-boundary heartbeat: progress is the worker's liveness signal
|
|
10697
|
+
// (stall watchdog + backend janitor), so multi-chunk extraction must
|
|
10698
|
+
// emit between inference calls. Percentage interpolates within this
|
|
10699
|
+
// entity type's band of the 20–80 range.
|
|
10700
|
+
(completed, total) => {
|
|
10701
|
+
const interpolated = 20 + Math.round((i + completed / total) / entityTypeNames.length * 60);
|
|
10702
|
+
onProgress(interpolated, `Detecting ${entityTypeName} entities...`, "analyzing", {
|
|
10703
|
+
currentEntityType: entityTypeName,
|
|
10704
|
+
processedEntityTypes: i,
|
|
10705
|
+
totalEntityTypes: entityTypeNames.length,
|
|
10706
|
+
entitiesFound: totalFound,
|
|
10707
|
+
entitiesEmitted: totalEmitted,
|
|
10708
|
+
completedEntityTypes: [...completedEntityTypes],
|
|
10709
|
+
requestParams
|
|
10710
|
+
});
|
|
10711
|
+
}
|
|
10526
10712
|
);
|
|
10527
10713
|
totalFound += extractedEntities.length;
|
|
10528
10714
|
completedEntityTypes.push({ entityType: entityTypeName, foundCount: extractedEntities.length });
|
|
@@ -10566,13 +10752,21 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
10566
10752
|
onProgress(10, "Loading resource...", "analyzing");
|
|
10567
10753
|
onProgress(30, "Analyzing text for tags...", "analyzing");
|
|
10568
10754
|
const allTags = [];
|
|
10569
|
-
for (
|
|
10755
|
+
for (let c = 0; c < params.categories.length; c++) {
|
|
10756
|
+
const category = params.categories[c];
|
|
10570
10757
|
const categoryTags = await AnnotationDetection.detectTags(
|
|
10571
10758
|
content,
|
|
10572
10759
|
inferenceClient,
|
|
10573
10760
|
params.schema,
|
|
10574
10761
|
category,
|
|
10575
|
-
params.sourceLanguage
|
|
10762
|
+
params.sourceLanguage,
|
|
10763
|
+
// Chunk-boundary heartbeat (liveness): interpolate within this
|
|
10764
|
+
// category's slice of the 30–60 band.
|
|
10765
|
+
(completed, total) => onProgress(
|
|
10766
|
+
30 + Math.round((c + completed / total) / params.categories.length * 30),
|
|
10767
|
+
"Analyzing text for tags...",
|
|
10768
|
+
"analyzing"
|
|
10769
|
+
)
|
|
10576
10770
|
);
|
|
10577
10771
|
allTags.push(...categoryTags);
|
|
10578
10772
|
}
|
|
@@ -10598,8 +10792,14 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
|
|
|
10598
10792
|
result: { tagsFound: tags.length, tagsCreated: annotations.length, byCategory }
|
|
10599
10793
|
};
|
|
10600
10794
|
}
|
|
10795
|
+
function assertWithinOutputBudget(byteLength) {
|
|
10796
|
+
if (!withinByteBudget(byteLength)) {
|
|
10797
|
+
throw new Error(
|
|
10798
|
+
`Generated artifact exceeds the output byte budget: ${byteLength} bytes > ${MAX_PDF_BYTES}. Refusing a runaway generation.`
|
|
10799
|
+
);
|
|
10800
|
+
}
|
|
10801
|
+
}
|
|
10601
10802
|
async function processGenerationJob(inferenceClient, params, onProgress, logger2) {
|
|
10602
|
-
const GENERATABLE_MEDIA_TYPES = ["text/markdown", "text/plain"];
|
|
10603
10803
|
const outputMediaType = params.outputMediaType ?? "text/markdown";
|
|
10604
10804
|
if (!GENERATABLE_MEDIA_TYPES.includes(outputMediaType)) {
|
|
10605
10805
|
throw new Error(
|
|
@@ -10608,6 +10808,84 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10608
10808
|
}
|
|
10609
10809
|
const title = params.title ?? "Untitled";
|
|
10610
10810
|
const entityTypes = (params.entityTypes ?? []).map(String);
|
|
10811
|
+
if (outputMediaType === "application/pdf") {
|
|
10812
|
+
onProgress(5, "Generating resource...", "generating");
|
|
10813
|
+
const validIds = params.cite === true ? collectContextResourceIds(params.context) : null;
|
|
10814
|
+
let generated2 = await generateResourceFromTopic(
|
|
10815
|
+
title,
|
|
10816
|
+
entityTypes,
|
|
10817
|
+
inferenceClient,
|
|
10818
|
+
logger2,
|
|
10819
|
+
params.prompt,
|
|
10820
|
+
params.language,
|
|
10821
|
+
params.context,
|
|
10822
|
+
params.temperature,
|
|
10823
|
+
params.maxTokens,
|
|
10824
|
+
params.sourceLanguage,
|
|
10825
|
+
outputMediaType,
|
|
10826
|
+
params.task,
|
|
10827
|
+
params.structure,
|
|
10828
|
+
params.cite
|
|
10829
|
+
);
|
|
10830
|
+
let source = generated2.content;
|
|
10831
|
+
let citations2 = [];
|
|
10832
|
+
if (validIds) {
|
|
10833
|
+
const resolved = resolveCitationTokens(generated2.content, validIds, logger2);
|
|
10834
|
+
source = resolved.content;
|
|
10835
|
+
citations2 = resolved.citations;
|
|
10836
|
+
}
|
|
10837
|
+
let compiled = compileTypst(source);
|
|
10838
|
+
let repairs = 0;
|
|
10839
|
+
while ("error" in compiled && repairs < MAX_COMPILE_REPAIRS) {
|
|
10840
|
+
repairs++;
|
|
10841
|
+
logger2.warn("Typst compile failed \u2014 feeding the error back for repair", {
|
|
10842
|
+
attempt: repairs,
|
|
10843
|
+
error: compiled.error.slice(0, 500)
|
|
10844
|
+
});
|
|
10845
|
+
generated2 = await generateResourceFromTopic(
|
|
10846
|
+
title,
|
|
10847
|
+
entityTypes,
|
|
10848
|
+
inferenceClient,
|
|
10849
|
+
logger2,
|
|
10850
|
+
params.prompt,
|
|
10851
|
+
params.language,
|
|
10852
|
+
params.context,
|
|
10853
|
+
params.temperature,
|
|
10854
|
+
params.maxTokens,
|
|
10855
|
+
params.sourceLanguage,
|
|
10856
|
+
outputMediaType,
|
|
10857
|
+
params.task,
|
|
10858
|
+
params.structure,
|
|
10859
|
+
params.cite,
|
|
10860
|
+
{ source, error: compiled.error }
|
|
10861
|
+
);
|
|
10862
|
+
if (validIds) {
|
|
10863
|
+
const resolved = resolveCitationTokens(generated2.content, validIds, logger2);
|
|
10864
|
+
source = resolved.content;
|
|
10865
|
+
citations2 = resolved.citations;
|
|
10866
|
+
} else {
|
|
10867
|
+
source = generated2.content;
|
|
10868
|
+
}
|
|
10869
|
+
compiled = compileTypst(source);
|
|
10870
|
+
}
|
|
10871
|
+
if ("error" in compiled) {
|
|
10872
|
+
throw new Error(
|
|
10873
|
+
`Typst compilation failed after ${MAX_COMPILE_REPAIRS} repair attempts: ${compiled.error}`
|
|
10874
|
+
);
|
|
10875
|
+
}
|
|
10876
|
+
assertWithinOutputBudget(compiled.pdf.byteLength);
|
|
10877
|
+
onProgress(95, "Creating resource...", "creating");
|
|
10878
|
+
return {
|
|
10879
|
+
content: compiled.pdf,
|
|
10880
|
+
title: generated2.title ?? title,
|
|
10881
|
+
format: outputMediaType,
|
|
10882
|
+
citations: citations2,
|
|
10883
|
+
result: {
|
|
10884
|
+
resourceId: "",
|
|
10885
|
+
resourceName: generated2.title ?? title
|
|
10886
|
+
}
|
|
10887
|
+
};
|
|
10888
|
+
}
|
|
10611
10889
|
onProgress(5, "Generating resource...", "generating");
|
|
10612
10890
|
const generated = await generateResourceFromTopic(
|
|
10613
10891
|
title,
|
|
@@ -10633,8 +10911,10 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10633
10911
|
citations = resolved.citations;
|
|
10634
10912
|
}
|
|
10635
10913
|
onProgress(95, "Creating resource...", "creating");
|
|
10914
|
+
const artifact = new TextEncoder().encode(content);
|
|
10915
|
+
assertWithinOutputBudget(artifact.byteLength);
|
|
10636
10916
|
return {
|
|
10637
|
-
content,
|
|
10917
|
+
content: artifact,
|
|
10638
10918
|
title: generated.title ?? title,
|
|
10639
10919
|
format: outputMediaType,
|
|
10640
10920
|
citations,
|
|
@@ -10646,28 +10926,37 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
|
|
|
10646
10926
|
}
|
|
10647
10927
|
|
|
10648
10928
|
// src/workers/detection/prepare-detection.ts
|
|
10649
|
-
async function prepareDetection(
|
|
10650
|
-
|
|
10651
|
-
|
|
10652
|
-
|
|
10653
|
-
|
|
10654
|
-
|
|
10655
|
-
|
|
10656
|
-
|
|
10657
|
-
|
|
10658
|
-
|
|
10659
|
-
|
|
10660
|
-
|
|
10661
|
-
|
|
10662
|
-
|
|
10663
|
-
|
|
10664
|
-
|
|
10665
|
-
|
|
10666
|
-
}
|
|
10667
|
-
case "none":
|
|
10668
|
-
return null;
|
|
10929
|
+
async function prepareDetection(mediaType, session, resourceId, userId, generator, store) {
|
|
10930
|
+
const extractor = EXTRACTORS[textExtractionOf(mediaType)];
|
|
10931
|
+
if (!extractor) return { declined: "no-extractor" };
|
|
10932
|
+
const { data } = await session.client.browse.resourceRepresentation(resourceId);
|
|
10933
|
+
const bytes = Buffer.from(data);
|
|
10934
|
+
const extracted = await extractor.extract(bytes, mediaType, {
|
|
10935
|
+
key: calculateChecksum(bytes),
|
|
10936
|
+
store
|
|
10937
|
+
});
|
|
10938
|
+
if ("declined" in extracted) return extracted;
|
|
10939
|
+
if (!extracted.text.trim()) return { declined: "empty" };
|
|
10940
|
+
const items = extracted.items;
|
|
10941
|
+
if (items && items.length > 0) {
|
|
10942
|
+
const anchored = { text: extracted.text, items };
|
|
10943
|
+
return {
|
|
10944
|
+
text: extracted.text,
|
|
10945
|
+
buildAnnotation: (motivation, match, body) => buildPdfAnnotation(anchored, resourceId, userId, generator, motivation, match, body)
|
|
10946
|
+
};
|
|
10669
10947
|
}
|
|
10948
|
+
return {
|
|
10949
|
+
text: extracted.text,
|
|
10950
|
+
buildAnnotation: (motivation, match, body) => buildTextAnnotation(extracted.text, resourceId, userId, generator, motivation, match, body)
|
|
10951
|
+
};
|
|
10670
10952
|
}
|
|
10953
|
+
var DECLINE_MESSAGES = {
|
|
10954
|
+
"no-text-layer": "This PDF is a scan whose text could not be recognized; there is nothing to detect over.",
|
|
10955
|
+
"encrypted": "This PDF is password-protected, so its text cannot be read.",
|
|
10956
|
+
"corrupt": "This PDF could not be parsed \u2014 the file may be damaged or truncated.",
|
|
10957
|
+
"too-large": "This document is too large to extract text from.",
|
|
10958
|
+
"empty": "This document contains no text to detect over."
|
|
10959
|
+
};
|
|
10671
10960
|
async function emitEvent(session, channel, payload) {
|
|
10672
10961
|
await session.client.transport.emit(channel, payload);
|
|
10673
10962
|
}
|
|
@@ -10685,15 +10974,16 @@ function startWorkerProcess(config) {
|
|
|
10685
10974
|
const message = error instanceof Error ? error.message : String(error);
|
|
10686
10975
|
logger2.error("Job failed", { jobId: job.jobId, error: message, stack: error instanceof Error ? error.stack : void 0 });
|
|
10687
10976
|
const failAnnotationId = job.params.referenceId;
|
|
10688
|
-
|
|
10689
|
-
|
|
10690
|
-
|
|
10691
|
-
|
|
10692
|
-
|
|
10693
|
-
|
|
10694
|
-
|
|
10695
|
-
|
|
10696
|
-
|
|
10977
|
+
if (isJobType(job.type)) {
|
|
10978
|
+
emitEvent(session, "job:fail", {
|
|
10979
|
+
resourceId: job.resourceId,
|
|
10980
|
+
jobId: job.jobId,
|
|
10981
|
+
jobType: job.type,
|
|
10982
|
+
...failAnnotationId ? { annotationId: failAnnotationId } : {},
|
|
10983
|
+
error: message
|
|
10984
|
+
}).catch(() => {
|
|
10985
|
+
});
|
|
10986
|
+
}
|
|
10697
10987
|
adapter.failJob(job.jobId, message);
|
|
10698
10988
|
});
|
|
10699
10989
|
});
|
|
@@ -10725,11 +11015,16 @@ async function handleJob(adapter, config, job) {
|
|
|
10725
11015
|
}
|
|
10726
11016
|
async function handleJobInner(adapter, config, job) {
|
|
10727
11017
|
const { session, inferenceClient, generator } = config;
|
|
10728
|
-
const {
|
|
11018
|
+
const { userId, jobId } = job;
|
|
11019
|
+
if (!isJobType(job.type)) {
|
|
11020
|
+
adapter.failJob(jobId, `Unrecognized job type: ${job.type}`);
|
|
11021
|
+
return;
|
|
11022
|
+
}
|
|
11023
|
+
const jobType = job.type;
|
|
11024
|
+
const resourceId$1 = resourceId(job.resourceId);
|
|
10729
11025
|
const annotationId = job.params.referenceId;
|
|
10730
11026
|
const lifecycleBase = {
|
|
10731
|
-
resourceId,
|
|
10732
|
-
userId,
|
|
11027
|
+
resourceId: resourceId$1,
|
|
10733
11028
|
jobId,
|
|
10734
11029
|
jobType,
|
|
10735
11030
|
...annotationId ? { annotationId } : {}
|
|
@@ -10739,23 +11034,27 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10739
11034
|
adapter.failJob(jobId, `Worker not configured for job type: ${jobType}`);
|
|
10740
11035
|
return;
|
|
10741
11036
|
}
|
|
10742
|
-
let
|
|
11037
|
+
let ready = null;
|
|
10743
11038
|
if (jobType !== "generation") {
|
|
10744
|
-
const descriptor = await session.client.browse.resource(resourceId).fresh();
|
|
11039
|
+
const descriptor = await session.client.browse.resource(resourceId$1).fresh();
|
|
10745
11040
|
const mediaType = getPrimaryMediaType(descriptor);
|
|
10746
|
-
const
|
|
10747
|
-
if (
|
|
10748
|
-
|
|
10749
|
-
|
|
10750
|
-
|
|
10751
|
-
if (!source) {
|
|
11041
|
+
const source = await prepareDetection(mediaType ?? "", session, resourceId$1, userId, generator, config.anchoredTextStore);
|
|
11042
|
+
if ("declined" in source) {
|
|
11043
|
+
if (source.declined === "no-extractor") {
|
|
11044
|
+
throw new Error(`Cannot run ${jobType} on resource ${resourceId$1}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
|
|
11045
|
+
}
|
|
10752
11046
|
await emitEvent(session, "job:complete", {
|
|
10753
11047
|
...lifecycleBase,
|
|
10754
|
-
result: {
|
|
11048
|
+
result: {
|
|
11049
|
+
declined: true,
|
|
11050
|
+
reason: source.declined,
|
|
11051
|
+
message: DECLINE_MESSAGES[source.declined]
|
|
11052
|
+
}
|
|
10755
11053
|
});
|
|
10756
11054
|
adapter.completeJob();
|
|
10757
11055
|
return;
|
|
10758
11056
|
}
|
|
11057
|
+
ready = source;
|
|
10759
11058
|
}
|
|
10760
11059
|
const onProgress = (percentage, message, stage, extra) => {
|
|
10761
11060
|
adapter.touchActivity();
|
|
@@ -10774,14 +11073,14 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10774
11073
|
};
|
|
10775
11074
|
if (jobType === "highlight-annotation") {
|
|
10776
11075
|
const { annotations, result } = await processHighlightJob(
|
|
10777
|
-
|
|
11076
|
+
ready.text,
|
|
10778
11077
|
inferenceClient,
|
|
10779
|
-
job.params,
|
|
10780
|
-
|
|
11078
|
+
asJobParams(job.params),
|
|
11079
|
+
ready.buildAnnotation,
|
|
10781
11080
|
onProgress
|
|
10782
11081
|
);
|
|
10783
11082
|
for (const ann of annotations) {
|
|
10784
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11083
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10785
11084
|
}
|
|
10786
11085
|
await emitEvent(session, "job:complete", {
|
|
10787
11086
|
...lifecycleBase,
|
|
@@ -10790,14 +11089,14 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10790
11089
|
adapter.completeJob();
|
|
10791
11090
|
} else if (jobType === "comment-annotation") {
|
|
10792
11091
|
const { annotations, result } = await processCommentJob(
|
|
10793
|
-
|
|
11092
|
+
ready.text,
|
|
10794
11093
|
inferenceClient,
|
|
10795
|
-
job.params,
|
|
10796
|
-
|
|
11094
|
+
asJobParams(job.params),
|
|
11095
|
+
ready.buildAnnotation,
|
|
10797
11096
|
onProgress
|
|
10798
11097
|
);
|
|
10799
11098
|
for (const ann of annotations) {
|
|
10800
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11099
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10801
11100
|
}
|
|
10802
11101
|
await emitEvent(session, "job:complete", {
|
|
10803
11102
|
...lifecycleBase,
|
|
@@ -10806,14 +11105,14 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10806
11105
|
adapter.completeJob();
|
|
10807
11106
|
} else if (jobType === "assessment-annotation") {
|
|
10808
11107
|
const { annotations, result } = await processAssessmentJob(
|
|
10809
|
-
|
|
11108
|
+
ready.text,
|
|
10810
11109
|
inferenceClient,
|
|
10811
|
-
job.params,
|
|
10812
|
-
|
|
11110
|
+
asJobParams(job.params),
|
|
11111
|
+
ready.buildAnnotation,
|
|
10813
11112
|
onProgress
|
|
10814
11113
|
);
|
|
10815
11114
|
for (const ann of annotations) {
|
|
10816
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11115
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10817
11116
|
}
|
|
10818
11117
|
await emitEvent(session, "job:complete", {
|
|
10819
11118
|
...lifecycleBase,
|
|
@@ -10822,15 +11121,15 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10822
11121
|
adapter.completeJob();
|
|
10823
11122
|
} else if (jobType === "reference-annotation") {
|
|
10824
11123
|
const { annotations, result } = await processReferenceJob(
|
|
10825
|
-
|
|
11124
|
+
ready.text,
|
|
10826
11125
|
inferenceClient,
|
|
10827
|
-
job.params,
|
|
10828
|
-
|
|
11126
|
+
asJobParams(job.params),
|
|
11127
|
+
ready.buildAnnotation,
|
|
10829
11128
|
onProgress,
|
|
10830
11129
|
config.logger
|
|
10831
11130
|
);
|
|
10832
11131
|
for (const ann of annotations) {
|
|
10833
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11132
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10834
11133
|
}
|
|
10835
11134
|
await emitEvent(session, "job:complete", {
|
|
10836
11135
|
...lifecycleBase,
|
|
@@ -10839,14 +11138,14 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10839
11138
|
adapter.completeJob();
|
|
10840
11139
|
} else if (jobType === "tag-annotation") {
|
|
10841
11140
|
const { annotations, result } = await processTagJob(
|
|
10842
|
-
|
|
11141
|
+
ready.text,
|
|
10843
11142
|
inferenceClient,
|
|
10844
|
-
job.params,
|
|
10845
|
-
|
|
11143
|
+
asJobParams(job.params),
|
|
11144
|
+
ready.buildAnnotation,
|
|
10846
11145
|
onProgress
|
|
10847
11146
|
);
|
|
10848
11147
|
for (const ann of annotations) {
|
|
10849
|
-
await emitEvent(session, "mark:create", { annotation: ann,
|
|
11148
|
+
await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
|
|
10850
11149
|
}
|
|
10851
11150
|
await emitEvent(session, "job:complete", {
|
|
10852
11151
|
...lifecycleBase,
|
|
@@ -10867,7 +11166,7 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10867
11166
|
file: Buffer.from(genResult.content),
|
|
10868
11167
|
format: genResult.format,
|
|
10869
11168
|
storageUri,
|
|
10870
|
-
sourceResourceId: resourceId,
|
|
11169
|
+
sourceResourceId: resourceId$1,
|
|
10871
11170
|
...genParams.referenceId ? { sourceAnnotationId: genParams.referenceId } : {},
|
|
10872
11171
|
...genParams.prompt ? { generationPrompt: genParams.prompt } : {},
|
|
10873
11172
|
...genParams.language ? { language: genParams.language } : {},
|
|
@@ -10878,29 +11177,63 @@ async function handleJobInner(adapter, config, job) {
|
|
|
10878
11177
|
const { annotation: provenanceRef } = assembleAnnotation(
|
|
10879
11178
|
{
|
|
10880
11179
|
motivation: "linking",
|
|
10881
|
-
target: { source: String(resourceId) },
|
|
11180
|
+
target: { source: String(resourceId$1) },
|
|
10882
11181
|
body: { type: "SpecificResource", source: String(newResourceId), purpose: "linking" }
|
|
10883
11182
|
},
|
|
10884
11183
|
generator
|
|
10885
11184
|
);
|
|
10886
|
-
await emitEvent(session, "mark:create", { annotation: provenanceRef,
|
|
10887
|
-
}
|
|
10888
|
-
|
|
10889
|
-
const
|
|
10890
|
-
|
|
10891
|
-
|
|
10892
|
-
|
|
10893
|
-
|
|
10894
|
-
|
|
10895
|
-
|
|
10896
|
-
|
|
10897
|
-
|
|
11185
|
+
await emitEvent(session, "mark:create", { annotation: provenanceRef, resourceId: resourceId$1 });
|
|
11186
|
+
}
|
|
11187
|
+
if (genResult.format === "application/pdf" && genResult.citations.length > 0) {
|
|
11188
|
+
const layer = await extractPdfTextLayer(genResult.content);
|
|
11189
|
+
if (!layer) {
|
|
11190
|
+
config.logger.warn("PDF citations dropped \u2014 the generated artifact yielded no text layer", {
|
|
11191
|
+
jobId,
|
|
11192
|
+
resourceId: newResourceId,
|
|
11193
|
+
citations: genResult.citations.length
|
|
11194
|
+
});
|
|
11195
|
+
} else {
|
|
11196
|
+
for (const citation of genResult.citations) {
|
|
11197
|
+
const span = findClaimSpan(layer, citation.exact);
|
|
11198
|
+
if (!span) {
|
|
11199
|
+
config.logger.warn("PDF citation dropped \u2014 claim not found in the rendered text layer", {
|
|
11200
|
+
jobId,
|
|
11201
|
+
resourceId: newResourceId,
|
|
11202
|
+
citedResourceId: citation.resourceId,
|
|
11203
|
+
exactPreview: citation.exact.slice(0, 80)
|
|
11204
|
+
});
|
|
11205
|
+
continue;
|
|
11206
|
+
}
|
|
11207
|
+
const citationRef = buildPdfAnnotation(
|
|
11208
|
+
layer,
|
|
11209
|
+
resourceId(String(newResourceId)),
|
|
11210
|
+
userId,
|
|
11211
|
+
generator,
|
|
11212
|
+
"linking",
|
|
11213
|
+
{ exact: layer.text.slice(span.start, span.end), start: span.start, end: span.end },
|
|
11214
|
+
{ type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
|
|
11215
|
+
);
|
|
11216
|
+
await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
|
|
11217
|
+
}
|
|
11218
|
+
}
|
|
11219
|
+
} else {
|
|
11220
|
+
for (const citation of genResult.citations) {
|
|
11221
|
+
const { annotation: citationRef } = assembleAnnotation(
|
|
11222
|
+
{
|
|
11223
|
+
motivation: "linking",
|
|
11224
|
+
target: {
|
|
11225
|
+
source: String(newResourceId),
|
|
11226
|
+
selector: [
|
|
11227
|
+
{ type: "TextPositionSelector", start: citation.start, end: citation.end },
|
|
11228
|
+
{ type: "TextQuoteSelector", exact: citation.exact }
|
|
11229
|
+
]
|
|
11230
|
+
},
|
|
11231
|
+
body: { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
|
|
10898
11232
|
},
|
|
10899
|
-
|
|
10900
|
-
|
|
10901
|
-
|
|
10902
|
-
|
|
10903
|
-
await emitEvent(session, "mark:create", { annotation: citationRef, userId, resourceId: newResourceId });
|
|
11233
|
+
generator
|
|
11234
|
+
);
|
|
11235
|
+
await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
|
|
11236
|
+
}
|
|
10904
11237
|
}
|
|
10905
11238
|
await emitEvent(session, "job:complete", {
|
|
10906
11239
|
...lifecycleBase,
|
|
@@ -11052,6 +11385,10 @@ async function startAgentWorker(opts) {
|
|
|
11052
11385
|
jobTypes: group.jobTypes,
|
|
11053
11386
|
inferenceClient: group.client,
|
|
11054
11387
|
generator,
|
|
11388
|
+
// The extraction seam's cache, over this worker's content transport
|
|
11389
|
+
// (PERSIST-ANCHORS P2d). Built here because the client keeps its
|
|
11390
|
+
// content transport private — this is where it is in hand.
|
|
11391
|
+
anchoredTextStore: anchoredTextStoreOverTransport(content, logger2),
|
|
11055
11392
|
logger: logger2
|
|
11056
11393
|
});
|
|
11057
11394
|
logger2.info("Agent ready", {
|