@semiont/jobs 0.5.24 → 0.5.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,15 @@
1
- import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, getPrimaryMediaType, textExtractionOf, assembleAnnotation, reconcileSelector, createFragmentSelector, getLocaleEnglishName, isArray, isObject, isString, deriveViews } from '@semiont/core';
2
- import { deriveStorageUri, extractPdfTextLayer, locate } from '@semiont/content';
1
+ import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText, isObject, isString, deriveViews } from '@semiont/core';
2
+ import { anchoredTextStoreOverTransport, deriveStorageUri, extractPdfTextLayer, EXTRACTORS, calculateChecksum, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
3
+ import { execFileSync } from 'child_process';
4
+ import { existsSync, readFileSync, mkdtempSync, writeFileSync, rmSync } from 'fs';
5
+ import { homedir, hostname, tmpdir } from 'os';
6
+ import { join } from 'path';
3
7
  import { generateAnnotationId } from '@semiont/event-sourcing';
4
8
  import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
5
- import { homedir, hostname } from 'os';
6
9
  import { InMemorySessionStorage, setStoredSession, kbBackendUrl, SemiontClient, SemiontSession } from '@semiont/sdk';
7
10
  import { HttpTransport, HttpContentTransport } from '@semiont/http-transport';
8
11
  import { createInferenceClient } from '@semiont/inference';
9
12
  import { createServer } from 'http';
10
- import { existsSync, readFileSync } from 'fs';
11
- import { join } from 'path';
12
13
  import { createProcessLogger } from '@semiont/observability/process-logger';
13
14
 
14
15
  var __create = Object.create;
@@ -3765,9 +3766,9 @@ var require_mapOneOrManyArgs = __commonJS({
3765
3766
  Object.defineProperty(exports, "__esModule", { value: true });
3766
3767
  exports.mapOneOrManyArgs = void 0;
3767
3768
  var map_1 = require_map();
3768
- var isArray2 = Array.isArray;
3769
+ var isArray = Array.isArray;
3769
3770
  function callOrApply(fn, args) {
3770
- return isArray2(args) ? fn.apply(void 0, __spreadArray([], __read(args))) : fn(args);
3771
+ return isArray(args) ? fn.apply(void 0, __spreadArray([], __read(args))) : fn(args);
3771
3772
  }
3772
3773
  function mapOneOrManyArgs(fn) {
3773
3774
  return map_1.map(function(args) {
@@ -3912,14 +3913,14 @@ var require_argsArgArrayOrObject = __commonJS({
3912
3913
  "../../node_modules/rxjs/dist/cjs/internal/util/argsArgArrayOrObject.js"(exports) {
3913
3914
  Object.defineProperty(exports, "__esModule", { value: true });
3914
3915
  exports.argsArgArrayOrObject = void 0;
3915
- var isArray2 = Array.isArray;
3916
+ var isArray = Array.isArray;
3916
3917
  var getPrototypeOf = Object.getPrototypeOf;
3917
3918
  var objectProto = Object.prototype;
3918
3919
  var getKeys = Object.keys;
3919
3920
  function argsArgArrayOrObject(args) {
3920
3921
  if (args.length === 1) {
3921
3922
  var first_1 = args[0];
3922
- if (isArray2(first_1)) {
3923
+ if (isArray(first_1)) {
3923
3924
  return { args: first_1, keys: null };
3924
3925
  }
3925
3926
  if (isPOJO(first_1)) {
@@ -4667,9 +4668,9 @@ var require_argsOrArgArray = __commonJS({
4667
4668
  "../../node_modules/rxjs/dist/cjs/internal/util/argsOrArgArray.js"(exports) {
4668
4669
  Object.defineProperty(exports, "__esModule", { value: true });
4669
4670
  exports.argsOrArgArray = void 0;
4670
- var isArray2 = Array.isArray;
4671
+ var isArray = Array.isArray;
4671
4672
  function argsOrArgArray(args) {
4672
- return args.length === 1 && isArray2(args[0]) ? args[0] : args;
4673
+ return args.length === 1 && isArray(args[0]) ? args[0] : args;
4673
4674
  }
4674
4675
  exports.argsOrArgArray = argsOrArgArray;
4675
4676
  }
@@ -9329,6 +9330,25 @@ function createJobClaimAdapter(options) {
9329
9330
  };
9330
9331
  }
9331
9332
 
9333
+ // src/types.ts
9334
+ var JOB_TYPES = /* @__PURE__ */ new Set([
9335
+ "reference-annotation",
9336
+ "generation",
9337
+ "highlight-annotation",
9338
+ "assessment-annotation",
9339
+ "comment-annotation",
9340
+ "tag-annotation"
9341
+ ]);
9342
+ function isJobType(value) {
9343
+ return JOB_TYPES.has(value);
9344
+ }
9345
+ function asJobParams(params) {
9346
+ if (typeof params.resourceId !== "string") {
9347
+ throw new Error("Job params are missing a resourceId");
9348
+ }
9349
+ return params;
9350
+ }
9351
+
9332
9352
  // src/workers/inference-call.ts
9333
9353
  var INFERENCE_TIMEOUT_MS = 10 * 6e4;
9334
9354
  async function withTimeout(work, label) {
@@ -9351,18 +9371,51 @@ async function withTimeout(work, label) {
9351
9371
  clearTimeout(timer);
9352
9372
  }
9353
9373
  }
9354
- function boundedGenerate(client, prompt, maxTokens, temperature, options) {
9374
+ function boundedGenerate(client, prompt, maxTokens, temperature) {
9355
9375
  return withTimeout(
9356
- client.generateText(prompt, maxTokens, temperature, options),
9376
+ client.generateText(prompt, maxTokens, temperature),
9357
9377
  `${client.type}:${client.modelId}`
9358
9378
  );
9359
9379
  }
9360
- function boundedGenerateWithMetadata(client, prompt, maxTokens, temperature, options) {
9380
+ function boundedGenerateStructured(client, prompt, maxTokens, temperature, elementSchema) {
9361
9381
  return withTimeout(
9362
- client.generateTextWithMetadata(prompt, maxTokens, temperature, options),
9382
+ client.generateStructured(prompt, maxTokens, temperature, elementSchema),
9363
9383
  `${client.type}:${client.modelId}`
9364
9384
  );
9365
9385
  }
9386
+
9387
+ // src/workers/detection/detection-chunking.ts
9388
+ var SELECTOR_CONTEXT_CHARS = 64;
9389
+ var OVERLAP_CHARS = SELECTOR_CONTEXT_CHARS + // prefix
9390
+ SELECTOR_CONTEXT_CHARS + // suffix
9391
+ 2 * SELECTOR_CONTEXT_CHARS;
9392
+ var OVERLAP_TOKENS = Math.ceil(OVERLAP_CHARS / 4);
9393
+ function deriveDetectionBudget(limits, scaffoldTokens) {
9394
+ const { contextTokens, maxOutputTokens } = limits;
9395
+ const available = contextTokens - scaffoldTokens;
9396
+ let inputBudget;
9397
+ let outputBudget;
9398
+ if (maxOutputTokens >= contextTokens) {
9399
+ inputBudget = Math.floor(available / 3);
9400
+ outputBudget = available - inputBudget;
9401
+ } else {
9402
+ outputBudget = maxOutputTokens;
9403
+ inputBudget = contextTokens - outputBudget - scaffoldTokens;
9404
+ if (inputBudget <= 0) {
9405
+ inputBudget = Math.floor(available / 3);
9406
+ outputBudget = available - inputBudget;
9407
+ }
9408
+ }
9409
+ if (inputBudget <= OVERLAP_TOKENS) {
9410
+ throw new Error(
9411
+ `Inference window too small for detection: context ${contextTokens} tokens minus scaffold ${scaffoldTokens} leaves an input budget of ${inputBudget} (need > ${OVERLAP_TOKENS}). Use a model with a larger context window or reduce the prompt scaffold.`
9412
+ );
9413
+ }
9414
+ return {
9415
+ chunking: { chunkSize: inputBudget, overlap: OVERLAP_TOKENS },
9416
+ outputBudget
9417
+ };
9418
+ }
9366
9419
  function languageName(tag) {
9367
9420
  return getLocaleEnglishName(tag) || tag;
9368
9421
  }
@@ -9382,7 +9435,9 @@ var MotivationPrompts = class {
9382
9435
  /**
9383
9436
  * Build a prompt for detecting comment-worthy passages
9384
9437
  *
9385
- * @param content - The text content to analyze (will be truncated to 8000 chars)
9438
+ * @param content - The text content to analyze — a chunk sized by the
9439
+ * caller from derived provider limits; NEVER re-truncated here (a
9440
+ * builder-level clip is silent input loss — see #738)
9386
9441
  * @param instructions - Optional user-provided instructions
9387
9442
  * @param tone - Optional tone guidance (e.g., "academic", "conversational")
9388
9443
  * @param density - Optional target number of comments per 2000 words
@@ -9403,7 +9458,7 @@ ${instructions}${toneGuidance}${densityGuidance}${sourceLang}${bodyLang}
9403
9458
 
9404
9459
  Text to analyze:
9405
9460
  ---
9406
- ${content.substring(0, 8e3)}
9461
+ ${content}
9407
9462
  ---
9408
9463
 
9409
9464
  Return a JSON array of comments. Each comment must have:
@@ -9437,7 +9492,7 @@ Guidelines:
9437
9492
 
9438
9493
  Text to analyze:
9439
9494
  ---
9440
- ${content.substring(0, 8e3)}
9495
+ ${content}
9441
9496
  ---
9442
9497
 
9443
9498
  Return a JSON array of comments. Each comment should have:
@@ -9458,7 +9513,9 @@ Example format:
9458
9513
  /**
9459
9514
  * Build a prompt for detecting highlight-worthy passages
9460
9515
  *
9461
- * @param content - The text content to analyze (will be truncated to 8000 chars)
9516
+ * @param content - The text content to analyze — a chunk sized by the
9517
+ * caller from derived provider limits; NEVER re-truncated here (a
9518
+ * builder-level clip is silent input loss — see #738)
9462
9519
  * @param instructions - Optional user-provided instructions
9463
9520
  * @param density - Optional target number of highlights per 2000 words
9464
9521
  * @returns Formatted prompt string
@@ -9476,7 +9533,7 @@ ${instructions}${densityGuidance}${sourceLang}
9476
9533
 
9477
9534
  Text to analyze:
9478
9535
  ---
9479
- ${content.substring(0, 8e3)}
9536
+ ${content}
9480
9537
  ---
9481
9538
 
9482
9539
  Return a JSON array of highlights. Each highlight must have:
@@ -9507,7 +9564,7 @@ Guidelines:
9507
9564
 
9508
9565
  Text to analyze:
9509
9566
  ---
9510
- ${content.substring(0, 8e3)}
9567
+ ${content}
9511
9568
  ---
9512
9569
 
9513
9570
  Return a JSON array of highlights. Each highlight should have:
@@ -9527,7 +9584,9 @@ Example format:
9527
9584
  /**
9528
9585
  * Build a prompt for detecting assessment-worthy passages
9529
9586
  *
9530
- * @param content - The text content to analyze (will be truncated to 8000 chars)
9587
+ * @param content - The text content to analyze — a chunk sized by the
9588
+ * caller from derived provider limits; NEVER re-truncated here (a
9589
+ * builder-level clip is silent input loss — see #738)
9531
9590
  * @param instructions - Optional user-provided instructions
9532
9591
  * @param tone - Optional tone guidance (e.g., "critical", "supportive")
9533
9592
  * @param density - Optional target number of assessments per 2000 words
@@ -9548,7 +9607,7 @@ ${instructions}${toneGuidance}${densityGuidance}${sourceLang}${bodyLang}
9548
9607
 
9549
9608
  Text to analyze:
9550
9609
  ---
9551
- ${content.substring(0, 8e3)}
9610
+ ${content}
9552
9611
  ---
9553
9612
 
9554
9613
  Return a JSON array of assessments. Each assessment must have:
@@ -9582,7 +9641,7 @@ Guidelines:
9582
9641
 
9583
9642
  Text to analyze:
9584
9643
  ---
9585
- ${content.substring(0, 8e3)}
9644
+ ${content}
9586
9645
  ---
9587
9646
 
9588
9647
  Return a JSON array of assessments. Each assessment should have:
@@ -9654,32 +9713,57 @@ Example format:
9654
9713
  return prompt;
9655
9714
  }
9656
9715
  };
9657
- function parseJsonArray(response, motivation) {
9658
- let parsed;
9659
- try {
9660
- parsed = JSON.parse(response.trim());
9661
- } catch (error) {
9662
- console.error(`[MotivationParsers] Failed to parse AI ${motivation} response:`, error);
9663
- console.error("Raw response:", response);
9664
- throw error instanceof Error ? error : new Error(String(error));
9665
- }
9666
- if (!Array.isArray(parsed)) {
9667
- console.error(`[MotivationParsers] Expected a JSON array for ${motivation} detection, got ${typeof parsed}:`, response);
9668
- throw new Error(`Expected a JSON array for ${motivation} detection, got ${typeof parsed}`);
9669
- }
9670
- return parsed;
9671
- }
9716
+ var COMMENT_ELEMENT_SCHEMA = {
9717
+ type: "object",
9718
+ properties: {
9719
+ exact: { type: "string" },
9720
+ prefix: { type: "string" },
9721
+ suffix: { type: "string" },
9722
+ comment: { type: "string" }
9723
+ },
9724
+ required: ["exact", "comment"],
9725
+ additionalProperties: false
9726
+ };
9727
+ var HIGHLIGHT_ELEMENT_SCHEMA = {
9728
+ type: "object",
9729
+ properties: {
9730
+ exact: { type: "string" },
9731
+ prefix: { type: "string" },
9732
+ suffix: { type: "string" }
9733
+ },
9734
+ required: ["exact"],
9735
+ additionalProperties: false
9736
+ };
9737
+ var ASSESSMENT_ELEMENT_SCHEMA = {
9738
+ type: "object",
9739
+ properties: {
9740
+ exact: { type: "string" },
9741
+ prefix: { type: "string" },
9742
+ suffix: { type: "string" },
9743
+ assessment: { type: "string" }
9744
+ },
9745
+ required: ["exact", "assessment"],
9746
+ additionalProperties: false
9747
+ };
9748
+ var TAG_ELEMENT_SCHEMA = {
9749
+ type: "object",
9750
+ properties: {
9751
+ exact: { type: "string" },
9752
+ prefix: { type: "string" },
9753
+ suffix: { type: "string" }
9754
+ },
9755
+ required: ["exact"],
9756
+ additionalProperties: false
9757
+ };
9672
9758
  var MotivationParsers = class {
9673
9759
  /**
9674
- * Parse and validate AI response for comment detection
9760
+ * Validate and reconcile structured comment elements.
9675
9761
  *
9676
- * @param response - Raw AI response text (a JSON array)
9762
+ * @param parsed - Already-parsed elements from the structured surface
9677
9763
  * @param content - Original content to validate offsets against
9678
9764
  * @returns Array of validated comment matches
9679
- * @throws if the response is not a parseable JSON array
9680
9765
  */
9681
- static parseComments(response, content) {
9682
- const parsed = parseJsonArray(response, "comment");
9766
+ static parseComments(parsed, content) {
9683
9767
  const valid = parsed.filter(
9684
9768
  (c) => isObject(c) && isString(c.exact) && isString(c.comment) && c.comment.trim().length > 0
9685
9769
  );
@@ -9708,15 +9792,13 @@ var MotivationParsers = class {
9708
9792
  return validatedComments;
9709
9793
  }
9710
9794
  /**
9711
- * Parse and validate AI response for highlight detection
9795
+ * Validate and reconcile structured highlight elements.
9712
9796
  *
9713
- * @param response - Raw AI response text (a JSON array)
9797
+ * @param parsed - Already-parsed elements from the structured surface
9714
9798
  * @param content - Original content to validate offsets against
9715
9799
  * @returns Array of validated highlight matches
9716
- * @throws if the response is not a parseable JSON array
9717
9800
  */
9718
- static parseHighlights(response, content) {
9719
- const parsed = parseJsonArray(response, "highlight");
9801
+ static parseHighlights(parsed, content) {
9720
9802
  const highlights = parsed.filter(
9721
9803
  (h) => isObject(h) && isString(h.exact)
9722
9804
  );
@@ -9743,15 +9825,13 @@ var MotivationParsers = class {
9743
9825
  return validatedHighlights;
9744
9826
  }
9745
9827
  /**
9746
- * Parse and validate AI response for assessment detection
9828
+ * Validate and reconcile structured assessment elements.
9747
9829
  *
9748
- * @param response - Raw AI response text (a JSON array)
9830
+ * @param parsed - Already-parsed elements from the structured surface
9749
9831
  * @param content - Original content to validate offsets against
9750
9832
  * @returns Array of validated assessment matches
9751
- * @throws if the response is not a parseable JSON array
9752
9833
  */
9753
- static parseAssessments(response, content) {
9754
- const parsed = parseJsonArray(response, "assessment");
9834
+ static parseAssessments(parsed, content) {
9755
9835
  const assessments = parsed.filter(
9756
9836
  (a) => isObject(a) && isString(a.exact) && isString(a.assessment)
9757
9837
  );
@@ -9779,14 +9859,13 @@ var MotivationParsers = class {
9779
9859
  return validatedAssessments;
9780
9860
  }
9781
9861
  /**
9782
- * Parse the LLM's tag response into raw, pre-reconciliation tag inputs.
9862
+ * Validate structured tag elements into raw, pre-reconciliation tag inputs.
9783
9863
  * Reconciliation happens in `validateTagOffsets`, which adds `start`/`end`
9784
9864
  * by anchoring `exact` against the source content.
9785
9865
  *
9786
- * @throws if the response is not a parseable JSON array
9866
+ * @param parsed - Already-parsed elements from the structured surface
9787
9867
  */
9788
- static parseTags(response) {
9789
- const parsed = parseJsonArray(response, "tag");
9868
+ static parseTags(parsed) {
9790
9869
  const valid = parsed.filter(
9791
9870
  (t) => isObject(t) && isString(t.exact) && t.exact.trim().length > 0
9792
9871
  );
@@ -9828,10 +9907,32 @@ function logAnchorMethod(motivation, exact, anchorMethod) {
9828
9907
  }
9829
9908
 
9830
9909
  // src/workers/annotation-detection.ts
9831
- function assertNotTruncated(response, motivation) {
9910
+ function assertNotTruncated(response, motivation, chunk, totalChunks, outputBudget) {
9832
9911
  if (response.stopReason === "max_tokens") {
9833
- throw new Error(`${motivation} detection response truncated (max_tokens) \u2014 increase max_tokens or reduce resource size; failing the job rather than under-reporting annotations.`);
9912
+ throw new Error(`${motivation} detection response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
9913
+ }
9914
+ }
9915
+ async function detectInChunks(client, content, buildPrompt, temperature, motivation, elementSchema, parse, onChunk) {
9916
+ const limits = await client.limits();
9917
+ const scaffoldTokens = estimateTokens(buildPrompt(""));
9918
+ const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
9919
+ const chunks = chunkText(content, chunking);
9920
+ const collected = [];
9921
+ for (let i = 0; i < chunks.length; i++) {
9922
+ const response = await boundedGenerateStructured(
9923
+ client,
9924
+ buildPrompt(chunks[i]),
9925
+ outputBudget,
9926
+ temperature,
9927
+ elementSchema
9928
+ );
9929
+ assertNotTruncated(response, motivation, i + 1, chunks.length, outputBudget);
9930
+ collected.push(...parse(response.items));
9931
+ if (i < chunks.length - 1) {
9932
+ onChunk?.(i + 1, chunks.length);
9933
+ }
9834
9934
  }
9935
+ return collected;
9835
9936
  }
9836
9937
  var AnnotationDetection = class {
9837
9938
  /**
@@ -9842,11 +9943,17 @@ var AnnotationDetection = class {
9842
9943
  * (source-resource locale). See `types.ts` "Locale conventions" for the
9843
9944
  * full discussion.
9844
9945
  */
9845
- static async detectComments(content, client, instructions, tone, density, language, sourceLanguage) {
9846
- const prompt = MotivationPrompts.buildCommentPrompt(content, instructions, tone, density, language, sourceLanguage);
9847
- const response = await boundedGenerateWithMetadata(client, prompt, 3e3, 0.4, { format: "json" });
9848
- assertNotTruncated(response, "comment");
9849
- return MotivationParsers.parseComments(response.text, content);
9946
+ static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onChunk) {
9947
+ return detectInChunks(
9948
+ client,
9949
+ content,
9950
+ (chunk) => MotivationPrompts.buildCommentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
9951
+ 0.4,
9952
+ "comment",
9953
+ COMMENT_ELEMENT_SCHEMA,
9954
+ (items) => MotivationParsers.parseComments(items, content),
9955
+ onChunk
9956
+ );
9850
9957
  }
9851
9958
  /**
9852
9959
  * Detect highlights in content.
@@ -9855,11 +9962,17 @@ var AnnotationDetection = class {
9855
9962
  * applies, used in the prompt so the LLM analyzes non-English source
9856
9963
  * correctly.
9857
9964
  */
9858
- static async detectHighlights(content, client, instructions, density, sourceLanguage) {
9859
- const prompt = MotivationPrompts.buildHighlightPrompt(content, instructions, density, sourceLanguage);
9860
- const response = await boundedGenerateWithMetadata(client, prompt, 2e3, 0.3, { format: "json" });
9861
- assertNotTruncated(response, "highlight");
9862
- return MotivationParsers.parseHighlights(response.text, content);
9965
+ static async detectHighlights(content, client, instructions, density, sourceLanguage, onChunk) {
9966
+ return detectInChunks(
9967
+ client,
9968
+ content,
9969
+ (chunk) => MotivationPrompts.buildHighlightPrompt(chunk, instructions, density, sourceLanguage),
9970
+ 0.3,
9971
+ "highlight",
9972
+ HIGHLIGHT_ELEMENT_SCHEMA,
9973
+ (items) => MotivationParsers.parseHighlights(items, content),
9974
+ onChunk
9975
+ );
9863
9976
  }
9864
9977
  /**
9865
9978
  * Detect assessments in content.
@@ -9868,11 +9981,17 @@ var AnnotationDetection = class {
9868
9981
  * (annotation body locale). `sourceLanguage` is the locale of the content
9869
9982
  * being analyzed (source-resource locale).
9870
9983
  */
9871
- static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage) {
9872
- const prompt = MotivationPrompts.buildAssessmentPrompt(content, instructions, tone, density, language, sourceLanguage);
9873
- const response = await boundedGenerateWithMetadata(client, prompt, 3e3, 0.3, { format: "json" });
9874
- assertNotTruncated(response, "assessment");
9875
- return MotivationParsers.parseAssessments(response.text, content);
9984
+ static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onChunk) {
9985
+ return detectInChunks(
9986
+ client,
9987
+ content,
9988
+ (chunk) => MotivationPrompts.buildAssessmentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
9989
+ 0.3,
9990
+ "assessment",
9991
+ ASSESSMENT_ELEMENT_SCHEMA,
9992
+ (items) => MotivationParsers.parseAssessments(items, content),
9993
+ onChunk
9994
+ );
9876
9995
  }
9877
9996
  /**
9878
9997
  * Detect tags in content for a specific category.
@@ -9886,28 +10005,45 @@ var AnnotationDetection = class {
9886
10005
  * identifiers, not LLM-generated text — so it's consumed at the body-stamp
9887
10006
  * site, not here.
9888
10007
  */
9889
- static async detectTags(content, client, schema, category, sourceLanguage) {
10008
+ static async detectTags(content, client, schema, category, sourceLanguage, onChunk) {
9890
10009
  const categoryInfo = schema.tags.find((t) => t.name === category);
9891
10010
  if (!categoryInfo) {
9892
10011
  throw new Error(`Invalid category "${category}" for schema ${schema.id}`);
9893
10012
  }
9894
- const prompt = MotivationPrompts.buildTagPrompt(
10013
+ const parsedTags = await detectInChunks(
10014
+ client,
9895
10015
  content,
9896
- category,
9897
- schema.name,
9898
- schema.description,
9899
- schema.domain,
9900
- categoryInfo.description,
9901
- categoryInfo.examples,
9902
- sourceLanguage
10016
+ (chunk) => MotivationPrompts.buildTagPrompt(
10017
+ chunk,
10018
+ category,
10019
+ schema.name,
10020
+ schema.description,
10021
+ schema.domain,
10022
+ categoryInfo.description,
10023
+ categoryInfo.examples,
10024
+ sourceLanguage
10025
+ ),
10026
+ 0.2,
10027
+ "tag",
10028
+ TAG_ELEMENT_SCHEMA,
10029
+ (items) => MotivationParsers.parseTags(items),
10030
+ onChunk
9903
10031
  );
9904
- const response = await boundedGenerateWithMetadata(client, prompt, 4e3, 0.2, { format: "json" });
9905
- assertNotTruncated(response, "tag");
9906
- const parsedTags = MotivationParsers.parseTags(response.text);
9907
10032
  return MotivationParsers.validateTagOffsets(parsedTags, content, category);
9908
10033
  }
9909
10034
  };
9910
- async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage) {
10035
+ var ENTITY_ELEMENT_SCHEMA = {
10036
+ type: "object",
10037
+ properties: {
10038
+ exact: { type: "string" },
10039
+ entityType: { type: "string" },
10040
+ prefix: { type: "string" },
10041
+ suffix: { type: "string" }
10042
+ },
10043
+ required: ["exact", "entityType"],
10044
+ additionalProperties: false
10045
+ };
10046
+ async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage, onChunk) {
9911
10047
  const entityTypesDescription = entityTypes.map((et) => {
9912
10048
  if (typeof et === "string") {
9913
10049
  return et;
@@ -9939,11 +10075,11 @@ Find direct mentions only (names, proper nouns). Do not include pronouns or desc
9939
10075
  const sourceLangGuidance = sourceLanguage ? `
9940
10076
  Source text language: ${getLocaleEnglishName(sourceLanguage) || sourceLanguage}.
9941
10077
  ` : "";
9942
- const prompt = `Identify entity references in the following text. Look for mentions of: ${entityTypesDescription}.
10078
+ const buildPrompt = (text) => `Identify entity references in the following text. Look for mentions of: ${entityTypesDescription}.
9943
10079
  ${descriptiveReferenceGuidance}${sourceLangGuidance}
9944
10080
  Text to analyze:
9945
10081
  """
9946
- ${exact}
10082
+ ${text}
9947
10083
  """
9948
10084
 
9949
10085
  Respond with a JSON array of entities found. Each entity should have:
@@ -9956,59 +10092,53 @@ If no entities are found, respond with an empty array [].
9956
10092
 
9957
10093
  Example output:
9958
10094
  [{"exact":"Alice","entityType":"Person","prefix":"","suffix":" went to"},{"exact":"Paris","entityType":"Location","prefix":"went to ","suffix":" yesterday"}]`;
9959
- logger2.debug("Sending entity extraction request", { entityTypes: entityTypesDescription });
9960
- const response = await boundedGenerateWithMetadata(
9961
- client,
9962
- prompt,
9963
- 4e3,
9964
- // Increased to handle many entities without truncation
9965
- 0.3,
9966
- // Lower temperature for more consistent extraction
9967
- // Force grammar-constrained JSON output. Without this, Ollama models
9968
- // periodically emit malformed JSON (truncated brackets, mid-token
9969
- // breaks at higher token counts) which silently parse-fails into
9970
- // [] downstream. The prompt's schema (which keys, what types) still
9971
- // governs *what* the JSON contains; `format: 'json'` governs that
9972
- // it's syntactically valid.
9973
- { format: "json" }
9974
- );
9975
- logger2.debug("Got entity extraction response", { responseLength: response.text.length });
9976
- if (response.stopReason === "max_tokens") {
9977
- const errorMsg = "Entity extraction response truncated (max_tokens) \u2014 increase max_tokens or reduce resource size; failing the job rather than dropping annotations.";
9978
- logger2.error(errorMsg, { responseLength: response.text.length });
9979
- throw new Error(errorMsg);
9980
- }
9981
- let entities;
9982
- try {
9983
- entities = JSON.parse(response.text.trim());
9984
- } catch (error) {
9985
- logger2.error("Failed to parse entity extraction response", {
9986
- error: error instanceof Error ? error.message : String(error),
9987
- response: response.text.slice(0, 500)
9988
- });
9989
- throw new Error("Failed to parse entity extraction response", {
9990
- cause: error instanceof Error ? error : new Error(String(error))
10095
+ const limits = await client.limits();
10096
+ const scaffoldTokens = estimateTokens(buildPrompt(""));
10097
+ const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
10098
+ const chunks = chunkText(exact, chunking);
10099
+ logger2.debug("Sending entity extraction request", {
10100
+ entityTypes: entityTypesDescription,
10101
+ chunks: chunks.length,
10102
+ chunkSizeTokens: chunking.chunkSize,
10103
+ outputBudget
10104
+ });
10105
+ const collected = [];
10106
+ for (let i = 0; i < chunks.length; i++) {
10107
+ const response = await boundedGenerateStructured(
10108
+ client,
10109
+ buildPrompt(chunks[i]),
10110
+ outputBudget,
10111
+ 0.3,
10112
+ // Lower temperature for more consistent extraction
10113
+ ENTITY_ELEMENT_SCHEMA
10114
+ );
10115
+ logger2.debug("Got entity extraction response", {
10116
+ chunk: i + 1,
10117
+ chunks: chunks.length,
10118
+ items: response.items.length
9991
10119
  });
10120
+ if (response.stopReason === "max_tokens") {
10121
+ const errorMsg = `Entity extraction response truncated (max_tokens) on chunk ${i + 1}/${chunks.length} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than dropping annotations.`;
10122
+ logger2.error(errorMsg, { items: response.items.length });
10123
+ throw new Error(errorMsg);
10124
+ }
10125
+ for (const e of response.items) {
10126
+ if (isObject(e) && isString(e.exact) && isString(e.entityType)) {
10127
+ collected.push({
10128
+ exact: e.exact,
10129
+ entityType: e.entityType,
10130
+ ...isString(e.prefix) ? { prefix: e.prefix } : {},
10131
+ ...isString(e.suffix) ? { suffix: e.suffix } : {}
10132
+ });
10133
+ } else {
10134
+ logger2.debug("Dropped malformed LLM entity", { entity: e });
10135
+ }
10136
+ }
10137
+ if (i < chunks.length - 1) {
10138
+ onChunk?.(i + 1, chunks.length);
10139
+ }
9992
10140
  }
9993
- if (!isArray(entities)) {
9994
- logger2.error("Failed to parse entity extraction response: expected a JSON array", {
9995
- response: response.text.slice(0, 500)
9996
- });
9997
- throw new Error("Failed to parse entity extraction response: expected a JSON array");
9998
- }
9999
- logger2.debug("Parsed entities from AI response", { count: entities.length });
10000
- return entities.filter((e) => {
10001
- const ok = isObject(e) && isString(e.exact) && isString(e.entityType);
10002
- if (!ok) {
10003
- logger2.debug("Dropped malformed LLM entity", { entity: e });
10004
- }
10005
- return ok;
10006
- }).map((entity) => ({
10007
- exact: entity.exact,
10008
- entityType: entity.entityType,
10009
- ...isString(entity.prefix) ? { prefix: entity.prefix } : {},
10010
- ...isString(entity.suffix) ? { suffix: entity.suffix } : {}
10011
- }));
10141
+ return collected;
10012
10142
  }
10013
10143
  function getLanguageName(locale) {
10014
10144
  return getLocaleEnglishName(locale) || locale;
@@ -10019,7 +10149,7 @@ var SEMANTIC_MATCH_CHARS = 240;
10019
10149
  function idLabel(resourceId, annotationId) {
10020
10150
  return `[${resourceId}${annotationId ? `/${annotationId}` : ""}]`;
10021
10151
  }
10022
- async function generateResourceFromTopic(topic, entityTypes, client, logger2, userPrompt, locale, context, temperature, maxTokens, sourceLanguage, outputMediaType = "text/markdown", task = "resource", structure, cite = false) {
10152
+ async function generateResourceFromTopic(topic, entityTypes, client, logger2, userPrompt, locale, context, temperature, maxTokens, sourceLanguage, outputMediaType = "text/markdown", task = "resource", structure, cite = false, repair) {
10023
10153
  logger2.debug("Generating resource from topic", {
10024
10154
  topicPreview: topic.substring(0, 100),
10025
10155
  entityTypes,
@@ -10135,11 +10265,12 @@ ${parts.join("\n")}`;
10135
10265
  let semanticContextSection = "";
10136
10266
  const similar = context?.semanticContext?.similar ?? [];
10137
10267
  if (similar.length > 0) {
10138
- const lines = [...similar].sort((a, b) => b.score - a.score).slice(0, SEMANTIC_MATCH_LIMIT).map((m) => `- ${idLabel(m.resourceId, m.annotationId)} (${m.score.toFixed(2)}) ${m.text.slice(0, SEMANTIC_MATCH_CHARS)}`);
10268
+ const lines = [...similar].sort((a, b) => b.score - a.score).slice(0, SEMANTIC_MATCH_LIMIT).map((m) => `- ${idLabel(m.resourceId, m.annotationId)} (${m.score.toFixed(2)})${m.machineRead ? " [OCR]" : ""} ${m.text.slice(0, SEMANTIC_MATCH_CHARS)}`);
10269
+ const ocrNote = similar.some((m) => m.machineRead) ? "\nPassages marked [OCR] were read from scanned images by character recognition; treat their exact wording and numbers as uncertain, and say so if you rely on one." : "";
10139
10270
  semanticContextSection = `
10140
10271
 
10141
10272
  Related passages from the knowledge base:
10142
- ${lines.join("\n")}`;
10273
+ ${lines.join("\n")}${ocrNote}`;
10143
10274
  }
10144
10275
  let leadLine;
10145
10276
  if (task === "resource") {
@@ -10154,11 +10285,12 @@ ${lines.join("\n")}`;
10154
10285
  Topic: "${topic}"`;
10155
10286
  }
10156
10287
  const isPlainText = outputMediaType === "text/plain";
10288
+ const isPdf = outputMediaType === "application/pdf";
10157
10289
  let structureRequirement = "";
10158
10290
  let titleRequirement = "";
10159
10291
  if (structure === "sections") {
10160
- structureRequirement = isPlainText ? "\n- Organize the content into titled sections with well-structured paragraphs" : "\n- Organize the content into titled sections (## Section) with well-structured paragraphs";
10161
- if (!isPlainText) {
10292
+ structureRequirement = isPdf ? "\n- Organize the content into titled sections (= Heading) with well-structured paragraphs" : isPlainText ? "\n- Organize the content into titled sections with well-structured paragraphs" : "\n- Organize the content into titled sections (## Section) with well-structured paragraphs";
10293
+ if (!isPlainText && !isPdf) {
10162
10294
  titleRequirement = "\n- Start with a clear heading (# Title)";
10163
10295
  }
10164
10296
  } else if (structure === "prose") {
@@ -10171,11 +10303,20 @@ Topic: "${topic}"`;
10171
10303
  - Organize the output as: ${structure}`;
10172
10304
  }
10173
10305
  const citeRequirement = cite ? "\n- Ground every claim in the provided context. Immediately after each claim, cite its source by emitting [[<id>]], where <id> is an id shown in square brackets in the context above (for a passage labeled [abc], emit [[abc]]). Cite only ids that appear in the context." : "";
10174
- const formatRequirements = isPlainText ? `- Write the response as plain text \u2014 no formatting markup (no #, *, backticks, headings, or links)
10306
+ const formatRequirements = isPdf ? `- Write the response as Typst markup (the Typst typesetting language \u2014 not markdown, not LaTeX)
10307
+ - Headings are written as = Heading (deeper levels == Subheading); everything else is plain prose paragraphs
10308
+ - Do not emit markdown syntax or code fences` : isPlainText ? `- Write the response as plain text \u2014 no formatting markup (no #, *, backticks, headings, or links)
10175
10309
  - Begin with the title on its own first line` : `- Use markdown formatting
10176
10310
  - Write the response as markdown`;
10311
+ const repairSection = repair ? `
10312
+
10313
+ Your previous attempt failed to compile. Fix the error and return the complete corrected document \u2014 full source, not a diff.
10314
+ Compile error:
10315
+ ${repair.error}
10316
+ Previous source:
10317
+ ${repair.source}` : "";
10177
10318
  const prompt = `${leadLine}
10178
- ${userPrompt ? `Instruction: ${userPrompt}` : ""}
10319
+ ${userPrompt ? `Instruction: ${userPrompt}` : ""}${repairSection}
10179
10320
  ${entityTypes.length > 0 ? `Focus on these entity types: ${entityTypes.join(", ")}.` : ""}${annotationSection}${contextSection}${resourceSection}${graphSection}${semanticContextSection}${sourceLanguageInstruction}${languageInstruction}
10180
10321
 
10181
10322
  Requirements:
@@ -10184,7 +10325,7 @@ Requirements:
10184
10325
  ${formatRequirements}`;
10185
10326
  const parseResponse = (response2) => {
10186
10327
  let content = response2.trim();
10187
- if (content.startsWith("```markdown") || content.startsWith("```md")) {
10328
+ if (content.startsWith("```markdown") || content.startsWith("```md") || content.startsWith("```typst")) {
10188
10329
  content = content.slice(content.indexOf("\n") + 1);
10189
10330
  const endIndex = content.lastIndexOf("```");
10190
10331
  if (endIndex !== -1) {
@@ -10219,6 +10360,29 @@ ${formatRequirements}`;
10219
10360
  });
10220
10361
  return result;
10221
10362
  }
10363
+ var PINNED_CREATION_TIMESTAMP = 17e8;
10364
+ var MAX_COMPILE_REPAIRS = 2;
10365
+ function compileTypst(source) {
10366
+ const dir = mkdtempSync(join(tmpdir(), "typst-"));
10367
+ try {
10368
+ const inFile = join(dir, "doc.typ");
10369
+ const outFile = join(dir, "doc.pdf");
10370
+ writeFileSync(inFile, source);
10371
+ try {
10372
+ execFileSync(
10373
+ "typst",
10374
+ ["compile", "--creation-timestamp", String(PINNED_CREATION_TIMESTAMP), inFile, outFile],
10375
+ { stdio: ["ignore", "pipe", "pipe"] }
10376
+ );
10377
+ } catch (err) {
10378
+ const stderr = err.stderr;
10379
+ return { error: stderr?.length ? stderr.toString("utf8") : String(err) };
10380
+ }
10381
+ return { pdf: new Uint8Array(readFileSync(outFile)) };
10382
+ } finally {
10383
+ rmSync(dir, { recursive: true, force: true });
10384
+ }
10385
+ }
10222
10386
 
10223
10387
  // src/workers/generation/citation-resolver.ts
10224
10388
  var CITATION_TOKEN = /\[\[([^\s[\]/]+)(?:\/([^\s[\]/]+))?\]\]/g;
@@ -10363,9 +10527,9 @@ function buildTextAnnotation(content, resourceId, userId, generator, motivation,
10363
10527
  ...body !== void 0 ? { body } : {}
10364
10528
  };
10365
10529
  }
10366
- function buildPdfAnnotation(layer, resourceId, userId, generator, motivation, match, body) {
10367
- const { rects, overlap } = locate(layer, match.start, match.end);
10368
- const coveredText = overlap.length ? layer.text.substring(
10530
+ function buildPdfAnnotation(anchored, resourceId, userId, generator, motivation, match, body) {
10531
+ const { rects, overlap } = locate(anchored, match.start, match.end);
10532
+ const coveredText = overlap.length ? anchored.text.substring(
10369
10533
  Math.min(...overlap.map((i) => i.start)),
10370
10534
  Math.max(...overlap.map((i) => i.end))
10371
10535
  ) : "";
@@ -10414,7 +10578,9 @@ async function processHighlightJob(content, inferenceClient, params, buildAnnota
10414
10578
  inferenceClient,
10415
10579
  params.instructions,
10416
10580
  params.density,
10417
- params.sourceLanguage
10581
+ params.sourceLanguage,
10582
+ // Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
10583
+ (completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
10418
10584
  );
10419
10585
  onProgress(60, `Creating ${highlights.length} annotations...`, "creating");
10420
10586
  const annotations = dedupeAnnotations(highlights.map(
@@ -10436,7 +10602,9 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
10436
10602
  params.tone,
10437
10603
  params.density,
10438
10604
  params.language,
10439
- params.sourceLanguage
10605
+ params.sourceLanguage,
10606
+ // Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
10607
+ (completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
10440
10608
  );
10441
10609
  onProgress(60, `Creating ${comments.length} annotations...`, "creating");
10442
10610
  const bodyLanguage = params.language ?? "en";
@@ -10466,7 +10634,9 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
10466
10634
  params.tone,
10467
10635
  params.density,
10468
10636
  params.language,
10469
- params.sourceLanguage
10637
+ params.sourceLanguage,
10638
+ // Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
10639
+ (completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
10470
10640
  );
10471
10641
  onProgress(60, `Creating ${assessments.length} annotations...`, "creating");
10472
10642
  const bodyLanguage = params.language ?? "en";
@@ -10522,7 +10692,23 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
10522
10692
  inferenceClient,
10523
10693
  params.includeDescriptiveReferences ?? false,
10524
10694
  logger2,
10525
- params.sourceLanguage
10695
+ params.sourceLanguage,
10696
+ // Chunk-boundary heartbeat: progress is the worker's liveness signal
10697
+ // (stall watchdog + backend janitor), so multi-chunk extraction must
10698
+ // emit between inference calls. Percentage interpolates within this
10699
+ // entity type's band of the 20–80 range.
10700
+ (completed, total) => {
10701
+ const interpolated = 20 + Math.round((i + completed / total) / entityTypeNames.length * 60);
10702
+ onProgress(interpolated, `Detecting ${entityTypeName} entities...`, "analyzing", {
10703
+ currentEntityType: entityTypeName,
10704
+ processedEntityTypes: i,
10705
+ totalEntityTypes: entityTypeNames.length,
10706
+ entitiesFound: totalFound,
10707
+ entitiesEmitted: totalEmitted,
10708
+ completedEntityTypes: [...completedEntityTypes],
10709
+ requestParams
10710
+ });
10711
+ }
10526
10712
  );
10527
10713
  totalFound += extractedEntities.length;
10528
10714
  completedEntityTypes.push({ entityType: entityTypeName, foundCount: extractedEntities.length });
@@ -10566,13 +10752,21 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
10566
10752
  onProgress(10, "Loading resource...", "analyzing");
10567
10753
  onProgress(30, "Analyzing text for tags...", "analyzing");
10568
10754
  const allTags = [];
10569
- for (const category of params.categories) {
10755
+ for (let c = 0; c < params.categories.length; c++) {
10756
+ const category = params.categories[c];
10570
10757
  const categoryTags = await AnnotationDetection.detectTags(
10571
10758
  content,
10572
10759
  inferenceClient,
10573
10760
  params.schema,
10574
10761
  category,
10575
- params.sourceLanguage
10762
+ params.sourceLanguage,
10763
+ // Chunk-boundary heartbeat (liveness): interpolate within this
10764
+ // category's slice of the 30–60 band.
10765
+ (completed, total) => onProgress(
10766
+ 30 + Math.round((c + completed / total) / params.categories.length * 30),
10767
+ "Analyzing text for tags...",
10768
+ "analyzing"
10769
+ )
10576
10770
  );
10577
10771
  allTags.push(...categoryTags);
10578
10772
  }
@@ -10598,8 +10792,14 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
10598
10792
  result: { tagsFound: tags.length, tagsCreated: annotations.length, byCategory }
10599
10793
  };
10600
10794
  }
10795
+ function assertWithinOutputBudget(byteLength) {
10796
+ if (!withinByteBudget(byteLength)) {
10797
+ throw new Error(
10798
+ `Generated artifact exceeds the output byte budget: ${byteLength} bytes > ${MAX_PDF_BYTES}. Refusing a runaway generation.`
10799
+ );
10800
+ }
10801
+ }
10601
10802
  async function processGenerationJob(inferenceClient, params, onProgress, logger2) {
10602
- const GENERATABLE_MEDIA_TYPES = ["text/markdown", "text/plain"];
10603
10803
  const outputMediaType = params.outputMediaType ?? "text/markdown";
10604
10804
  if (!GENERATABLE_MEDIA_TYPES.includes(outputMediaType)) {
10605
10805
  throw new Error(
@@ -10608,6 +10808,84 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
10608
10808
  }
10609
10809
  const title = params.title ?? "Untitled";
10610
10810
  const entityTypes = (params.entityTypes ?? []).map(String);
10811
+ if (outputMediaType === "application/pdf") {
10812
+ onProgress(5, "Generating resource...", "generating");
10813
+ const validIds = params.cite === true ? collectContextResourceIds(params.context) : null;
10814
+ let generated2 = await generateResourceFromTopic(
10815
+ title,
10816
+ entityTypes,
10817
+ inferenceClient,
10818
+ logger2,
10819
+ params.prompt,
10820
+ params.language,
10821
+ params.context,
10822
+ params.temperature,
10823
+ params.maxTokens,
10824
+ params.sourceLanguage,
10825
+ outputMediaType,
10826
+ params.task,
10827
+ params.structure,
10828
+ params.cite
10829
+ );
10830
+ let source = generated2.content;
10831
+ let citations2 = [];
10832
+ if (validIds) {
10833
+ const resolved = resolveCitationTokens(generated2.content, validIds, logger2);
10834
+ source = resolved.content;
10835
+ citations2 = resolved.citations;
10836
+ }
10837
+ let compiled = compileTypst(source);
10838
+ let repairs = 0;
10839
+ while ("error" in compiled && repairs < MAX_COMPILE_REPAIRS) {
10840
+ repairs++;
10841
+ logger2.warn("Typst compile failed \u2014 feeding the error back for repair", {
10842
+ attempt: repairs,
10843
+ error: compiled.error.slice(0, 500)
10844
+ });
10845
+ generated2 = await generateResourceFromTopic(
10846
+ title,
10847
+ entityTypes,
10848
+ inferenceClient,
10849
+ logger2,
10850
+ params.prompt,
10851
+ params.language,
10852
+ params.context,
10853
+ params.temperature,
10854
+ params.maxTokens,
10855
+ params.sourceLanguage,
10856
+ outputMediaType,
10857
+ params.task,
10858
+ params.structure,
10859
+ params.cite,
10860
+ { source, error: compiled.error }
10861
+ );
10862
+ if (validIds) {
10863
+ const resolved = resolveCitationTokens(generated2.content, validIds, logger2);
10864
+ source = resolved.content;
10865
+ citations2 = resolved.citations;
10866
+ } else {
10867
+ source = generated2.content;
10868
+ }
10869
+ compiled = compileTypst(source);
10870
+ }
10871
+ if ("error" in compiled) {
10872
+ throw new Error(
10873
+ `Typst compilation failed after ${MAX_COMPILE_REPAIRS} repair attempts: ${compiled.error}`
10874
+ );
10875
+ }
10876
+ assertWithinOutputBudget(compiled.pdf.byteLength);
10877
+ onProgress(95, "Creating resource...", "creating");
10878
+ return {
10879
+ content: compiled.pdf,
10880
+ title: generated2.title ?? title,
10881
+ format: outputMediaType,
10882
+ citations: citations2,
10883
+ result: {
10884
+ resourceId: "",
10885
+ resourceName: generated2.title ?? title
10886
+ }
10887
+ };
10888
+ }
10611
10889
  onProgress(5, "Generating resource...", "generating");
10612
10890
  const generated = await generateResourceFromTopic(
10613
10891
  title,
@@ -10633,8 +10911,10 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
10633
10911
  citations = resolved.citations;
10634
10912
  }
10635
10913
  onProgress(95, "Creating resource...", "creating");
10914
+ const artifact = new TextEncoder().encode(content);
10915
+ assertWithinOutputBudget(artifact.byteLength);
10636
10916
  return {
10637
- content,
10917
+ content: artifact,
10638
10918
  title: generated.title ?? title,
10639
10919
  format: outputMediaType,
10640
10920
  citations,
@@ -10646,28 +10926,37 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
10646
10926
  }
10647
10927
 
10648
10928
  // src/workers/detection/prepare-detection.ts
10649
- async function prepareDetection(strategy, session, resourceId, userId, generator) {
10650
- switch (strategy) {
10651
- case "decode": {
10652
- const text = await session.client.browse.resourceContent(resourceId);
10653
- return {
10654
- text,
10655
- buildAnnotation: (motivation, match, body) => buildTextAnnotation(text, resourceId, userId, generator, motivation, match, body)
10656
- };
10657
- }
10658
- case "pdf-text-layer": {
10659
- const { data } = await session.client.browse.resourceRepresentation(resourceId);
10660
- const layer = await extractPdfTextLayer(new Uint8Array(data));
10661
- if (!layer) return null;
10662
- return {
10663
- text: layer.text,
10664
- buildAnnotation: (motivation, match, body) => buildPdfAnnotation(layer, resourceId, userId, generator, motivation, match, body)
10665
- };
10666
- }
10667
- case "none":
10668
- return null;
10929
+ async function prepareDetection(mediaType, session, resourceId, userId, generator, store) {
10930
+ const extractor = EXTRACTORS[textExtractionOf(mediaType)];
10931
+ if (!extractor) return { declined: "no-extractor" };
10932
+ const { data } = await session.client.browse.resourceRepresentation(resourceId);
10933
+ const bytes = Buffer.from(data);
10934
+ const extracted = await extractor.extract(bytes, mediaType, {
10935
+ key: calculateChecksum(bytes),
10936
+ store
10937
+ });
10938
+ if ("declined" in extracted) return extracted;
10939
+ if (!extracted.text.trim()) return { declined: "empty" };
10940
+ const items = extracted.items;
10941
+ if (items && items.length > 0) {
10942
+ const anchored = { text: extracted.text, items };
10943
+ return {
10944
+ text: extracted.text,
10945
+ buildAnnotation: (motivation, match, body) => buildPdfAnnotation(anchored, resourceId, userId, generator, motivation, match, body)
10946
+ };
10669
10947
  }
10948
+ return {
10949
+ text: extracted.text,
10950
+ buildAnnotation: (motivation, match, body) => buildTextAnnotation(extracted.text, resourceId, userId, generator, motivation, match, body)
10951
+ };
10670
10952
  }
10953
+ var DECLINE_MESSAGES = {
10954
+ "no-text-layer": "This PDF is a scan whose text could not be recognized; there is nothing to detect over.",
10955
+ "encrypted": "This PDF is password-protected, so its text cannot be read.",
10956
+ "corrupt": "This PDF could not be parsed \u2014 the file may be damaged or truncated.",
10957
+ "too-large": "This document is too large to extract text from.",
10958
+ "empty": "This document contains no text to detect over."
10959
+ };
10671
10960
  async function emitEvent(session, channel, payload) {
10672
10961
  await session.client.transport.emit(channel, payload);
10673
10962
  }
@@ -10685,15 +10974,16 @@ function startWorkerProcess(config) {
10685
10974
  const message = error instanceof Error ? error.message : String(error);
10686
10975
  logger2.error("Job failed", { jobId: job.jobId, error: message, stack: error instanceof Error ? error.stack : void 0 });
10687
10976
  const failAnnotationId = job.params.referenceId;
10688
- emitEvent(session, "job:fail", {
10689
- resourceId: job.resourceId,
10690
- userId: job.userId,
10691
- jobId: job.jobId,
10692
- jobType: job.type,
10693
- ...failAnnotationId ? { annotationId: failAnnotationId } : {},
10694
- error: message
10695
- }).catch(() => {
10696
- });
10977
+ if (isJobType(job.type)) {
10978
+ emitEvent(session, "job:fail", {
10979
+ resourceId: job.resourceId,
10980
+ jobId: job.jobId,
10981
+ jobType: job.type,
10982
+ ...failAnnotationId ? { annotationId: failAnnotationId } : {},
10983
+ error: message
10984
+ }).catch(() => {
10985
+ });
10986
+ }
10697
10987
  adapter.failJob(job.jobId, message);
10698
10988
  });
10699
10989
  });
@@ -10725,11 +11015,16 @@ async function handleJob(adapter, config, job) {
10725
11015
  }
10726
11016
  async function handleJobInner(adapter, config, job) {
10727
11017
  const { session, inferenceClient, generator } = config;
10728
- const { resourceId, userId, jobId, type: jobType } = job;
11018
+ const { userId, jobId } = job;
11019
+ if (!isJobType(job.type)) {
11020
+ adapter.failJob(jobId, `Unrecognized job type: ${job.type}`);
11021
+ return;
11022
+ }
11023
+ const jobType = job.type;
11024
+ const resourceId$1 = resourceId(job.resourceId);
10729
11025
  const annotationId = job.params.referenceId;
10730
11026
  const lifecycleBase = {
10731
- resourceId,
10732
- userId,
11027
+ resourceId: resourceId$1,
10733
11028
  jobId,
10734
11029
  jobType,
10735
11030
  ...annotationId ? { annotationId } : {}
@@ -10739,23 +11034,27 @@ async function handleJobInner(adapter, config, job) {
10739
11034
  adapter.failJob(jobId, `Worker not configured for job type: ${jobType}`);
10740
11035
  return;
10741
11036
  }
10742
- let source = null;
11037
+ let ready = null;
10743
11038
  if (jobType !== "generation") {
10744
- const descriptor = await session.client.browse.resource(resourceId).fresh();
11039
+ const descriptor = await session.client.browse.resource(resourceId$1).fresh();
10745
11040
  const mediaType = getPrimaryMediaType(descriptor);
10746
- const strategy = mediaType ? textExtractionOf(mediaType) : "none";
10747
- if (strategy === "none") {
10748
- throw new Error(`Cannot run ${jobType} on resource ${resourceId}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
10749
- }
10750
- source = await prepareDetection(strategy, session, resourceId, userId, generator);
10751
- if (!source) {
11041
+ const source = await prepareDetection(mediaType ?? "", session, resourceId$1, userId, generator, config.anchoredTextStore);
11042
+ if ("declined" in source) {
11043
+ if (source.declined === "no-extractor") {
11044
+ throw new Error(`Cannot run ${jobType} on resource ${resourceId$1}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
11045
+ }
10752
11046
  await emitEvent(session, "job:complete", {
10753
11047
  ...lifecycleBase,
10754
- result: { declined: true, reason: "no-text-layer", message: "This PDF has no extractable text layer (scanned or image-only); detection is not supported." }
11048
+ result: {
11049
+ declined: true,
11050
+ reason: source.declined,
11051
+ message: DECLINE_MESSAGES[source.declined]
11052
+ }
10755
11053
  });
10756
11054
  adapter.completeJob();
10757
11055
  return;
10758
11056
  }
11057
+ ready = source;
10759
11058
  }
10760
11059
  const onProgress = (percentage, message, stage, extra) => {
10761
11060
  adapter.touchActivity();
@@ -10774,14 +11073,14 @@ async function handleJobInner(adapter, config, job) {
10774
11073
  };
10775
11074
  if (jobType === "highlight-annotation") {
10776
11075
  const { annotations, result } = await processHighlightJob(
10777
- source.text,
11076
+ ready.text,
10778
11077
  inferenceClient,
10779
- job.params,
10780
- source.buildAnnotation,
11078
+ asJobParams(job.params),
11079
+ ready.buildAnnotation,
10781
11080
  onProgress
10782
11081
  );
10783
11082
  for (const ann of annotations) {
10784
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11083
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10785
11084
  }
10786
11085
  await emitEvent(session, "job:complete", {
10787
11086
  ...lifecycleBase,
@@ -10790,14 +11089,14 @@ async function handleJobInner(adapter, config, job) {
10790
11089
  adapter.completeJob();
10791
11090
  } else if (jobType === "comment-annotation") {
10792
11091
  const { annotations, result } = await processCommentJob(
10793
- source.text,
11092
+ ready.text,
10794
11093
  inferenceClient,
10795
- job.params,
10796
- source.buildAnnotation,
11094
+ asJobParams(job.params),
11095
+ ready.buildAnnotation,
10797
11096
  onProgress
10798
11097
  );
10799
11098
  for (const ann of annotations) {
10800
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11099
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10801
11100
  }
10802
11101
  await emitEvent(session, "job:complete", {
10803
11102
  ...lifecycleBase,
@@ -10806,14 +11105,14 @@ async function handleJobInner(adapter, config, job) {
10806
11105
  adapter.completeJob();
10807
11106
  } else if (jobType === "assessment-annotation") {
10808
11107
  const { annotations, result } = await processAssessmentJob(
10809
- source.text,
11108
+ ready.text,
10810
11109
  inferenceClient,
10811
- job.params,
10812
- source.buildAnnotation,
11110
+ asJobParams(job.params),
11111
+ ready.buildAnnotation,
10813
11112
  onProgress
10814
11113
  );
10815
11114
  for (const ann of annotations) {
10816
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11115
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10817
11116
  }
10818
11117
  await emitEvent(session, "job:complete", {
10819
11118
  ...lifecycleBase,
@@ -10822,15 +11121,15 @@ async function handleJobInner(adapter, config, job) {
10822
11121
  adapter.completeJob();
10823
11122
  } else if (jobType === "reference-annotation") {
10824
11123
  const { annotations, result } = await processReferenceJob(
10825
- source.text,
11124
+ ready.text,
10826
11125
  inferenceClient,
10827
- job.params,
10828
- source.buildAnnotation,
11126
+ asJobParams(job.params),
11127
+ ready.buildAnnotation,
10829
11128
  onProgress,
10830
11129
  config.logger
10831
11130
  );
10832
11131
  for (const ann of annotations) {
10833
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11132
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10834
11133
  }
10835
11134
  await emitEvent(session, "job:complete", {
10836
11135
  ...lifecycleBase,
@@ -10839,14 +11138,14 @@ async function handleJobInner(adapter, config, job) {
10839
11138
  adapter.completeJob();
10840
11139
  } else if (jobType === "tag-annotation") {
10841
11140
  const { annotations, result } = await processTagJob(
10842
- source.text,
11141
+ ready.text,
10843
11142
  inferenceClient,
10844
- job.params,
10845
- source.buildAnnotation,
11143
+ asJobParams(job.params),
11144
+ ready.buildAnnotation,
10846
11145
  onProgress
10847
11146
  );
10848
11147
  for (const ann of annotations) {
10849
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11148
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10850
11149
  }
10851
11150
  await emitEvent(session, "job:complete", {
10852
11151
  ...lifecycleBase,
@@ -10867,7 +11166,7 @@ async function handleJobInner(adapter, config, job) {
10867
11166
  file: Buffer.from(genResult.content),
10868
11167
  format: genResult.format,
10869
11168
  storageUri,
10870
- sourceResourceId: resourceId,
11169
+ sourceResourceId: resourceId$1,
10871
11170
  ...genParams.referenceId ? { sourceAnnotationId: genParams.referenceId } : {},
10872
11171
  ...genParams.prompt ? { generationPrompt: genParams.prompt } : {},
10873
11172
  ...genParams.language ? { language: genParams.language } : {},
@@ -10878,29 +11177,63 @@ async function handleJobInner(adapter, config, job) {
10878
11177
  const { annotation: provenanceRef } = assembleAnnotation(
10879
11178
  {
10880
11179
  motivation: "linking",
10881
- target: { source: String(resourceId) },
11180
+ target: { source: String(resourceId$1) },
10882
11181
  body: { type: "SpecificResource", source: String(newResourceId), purpose: "linking" }
10883
11182
  },
10884
11183
  generator
10885
11184
  );
10886
- await emitEvent(session, "mark:create", { annotation: provenanceRef, userId, resourceId });
10887
- }
10888
- for (const citation of genResult.citations) {
10889
- const { annotation: citationRef } = assembleAnnotation(
10890
- {
10891
- motivation: "linking",
10892
- target: {
10893
- source: String(newResourceId),
10894
- selector: [
10895
- { type: "TextPositionSelector", start: citation.start, end: citation.end },
10896
- { type: "TextQuoteSelector", exact: citation.exact }
10897
- ]
11185
+ await emitEvent(session, "mark:create", { annotation: provenanceRef, resourceId: resourceId$1 });
11186
+ }
11187
+ if (genResult.format === "application/pdf" && genResult.citations.length > 0) {
11188
+ const layer = await extractPdfTextLayer(genResult.content);
11189
+ if (!layer) {
11190
+ config.logger.warn("PDF citations dropped \u2014 the generated artifact yielded no text layer", {
11191
+ jobId,
11192
+ resourceId: newResourceId,
11193
+ citations: genResult.citations.length
11194
+ });
11195
+ } else {
11196
+ for (const citation of genResult.citations) {
11197
+ const span = findClaimSpan(layer, citation.exact);
11198
+ if (!span) {
11199
+ config.logger.warn("PDF citation dropped \u2014 claim not found in the rendered text layer", {
11200
+ jobId,
11201
+ resourceId: newResourceId,
11202
+ citedResourceId: citation.resourceId,
11203
+ exactPreview: citation.exact.slice(0, 80)
11204
+ });
11205
+ continue;
11206
+ }
11207
+ const citationRef = buildPdfAnnotation(
11208
+ layer,
11209
+ resourceId(String(newResourceId)),
11210
+ userId,
11211
+ generator,
11212
+ "linking",
11213
+ { exact: layer.text.slice(span.start, span.end), start: span.start, end: span.end },
11214
+ { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
11215
+ );
11216
+ await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
11217
+ }
11218
+ }
11219
+ } else {
11220
+ for (const citation of genResult.citations) {
11221
+ const { annotation: citationRef } = assembleAnnotation(
11222
+ {
11223
+ motivation: "linking",
11224
+ target: {
11225
+ source: String(newResourceId),
11226
+ selector: [
11227
+ { type: "TextPositionSelector", start: citation.start, end: citation.end },
11228
+ { type: "TextQuoteSelector", exact: citation.exact }
11229
+ ]
11230
+ },
11231
+ body: { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
10898
11232
  },
10899
- body: { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
10900
- },
10901
- generator
10902
- );
10903
- await emitEvent(session, "mark:create", { annotation: citationRef, userId, resourceId: newResourceId });
11233
+ generator
11234
+ );
11235
+ await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
11236
+ }
10904
11237
  }
10905
11238
  await emitEvent(session, "job:complete", {
10906
11239
  ...lifecycleBase,
@@ -11052,6 +11385,10 @@ async function startAgentWorker(opts) {
11052
11385
  jobTypes: group.jobTypes,
11053
11386
  inferenceClient: group.client,
11054
11387
  generator,
11388
+ // The extraction seam's cache, over this worker's content transport
11389
+ // (PERSIST-ANCHORS P2d). Built here because the client keeps its
11390
+ // content transport private — this is where it is in hand.
11391
+ anchoredTextStore: anchoredTextStoreOverTransport(content, logger2),
11055
11392
  logger: logger2
11056
11393
  });
11057
11394
  logger2.info("Agent ready", {