@semiont/jobs 0.5.23 → 0.5.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,15 @@
1
- import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, getPrimaryMediaType, textExtractionOf, assembleAnnotation, reconcileSelector, createFragmentSelector, getLocaleEnglishName, isArray, isObject, isString, deriveViews } from '@semiont/core';
2
- import { deriveStorageUri, extractPdfTextLayer, locate } from '@semiont/content';
1
+ import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText, isArray, isObject, isString, deriveViews } from '@semiont/core';
2
+ import { anchoredTextStoreOverTransport, deriveStorageUri, extractPdfTextLayer, EXTRACTORS, calculateChecksum, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
3
+ import { execFileSync } from 'child_process';
4
+ import { existsSync, readFileSync, mkdtempSync, writeFileSync, rmSync } from 'fs';
5
+ import { homedir, hostname, tmpdir } from 'os';
6
+ import { join } from 'path';
3
7
  import { generateAnnotationId } from '@semiont/event-sourcing';
4
8
  import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
5
- import { homedir, hostname } from 'os';
6
9
  import { InMemorySessionStorage, setStoredSession, kbBackendUrl, SemiontClient, SemiontSession } from '@semiont/sdk';
7
10
  import { HttpTransport, HttpContentTransport } from '@semiont/http-transport';
8
11
  import { createInferenceClient } from '@semiont/inference';
9
12
  import { createServer } from 'http';
10
- import { existsSync, readFileSync } from 'fs';
11
- import { join } from 'path';
12
13
  import { createProcessLogger } from '@semiont/observability/process-logger';
13
14
 
14
15
  var __create = Object.create;
@@ -9329,6 +9330,25 @@ function createJobClaimAdapter(options) {
9329
9330
  };
9330
9331
  }
9331
9332
 
9333
+ // src/types.ts
9334
+ var JOB_TYPES = /* @__PURE__ */ new Set([
9335
+ "reference-annotation",
9336
+ "generation",
9337
+ "highlight-annotation",
9338
+ "assessment-annotation",
9339
+ "comment-annotation",
9340
+ "tag-annotation"
9341
+ ]);
9342
+ function isJobType(value) {
9343
+ return JOB_TYPES.has(value);
9344
+ }
9345
+ function asJobParams(params) {
9346
+ if (typeof params.resourceId !== "string") {
9347
+ throw new Error("Job params are missing a resourceId");
9348
+ }
9349
+ return params;
9350
+ }
9351
+
9332
9352
  // src/workers/inference-call.ts
9333
9353
  var INFERENCE_TIMEOUT_MS = 10 * 6e4;
9334
9354
  async function withTimeout(work, label) {
@@ -9363,6 +9383,39 @@ function boundedGenerateWithMetadata(client, prompt, maxTokens, temperature, opt
9363
9383
  `${client.type}:${client.modelId}`
9364
9384
  );
9365
9385
  }
9386
+
9387
+ // src/workers/detection/detection-chunking.ts
9388
+ var SELECTOR_CONTEXT_CHARS = 64;
9389
+ var OVERLAP_CHARS = SELECTOR_CONTEXT_CHARS + // prefix
9390
+ SELECTOR_CONTEXT_CHARS + // suffix
9391
+ 2 * SELECTOR_CONTEXT_CHARS;
9392
+ var OVERLAP_TOKENS = Math.ceil(OVERLAP_CHARS / 4);
9393
+ function deriveDetectionBudget(limits, scaffoldTokens) {
9394
+ const { contextTokens, maxOutputTokens } = limits;
9395
+ const available = contextTokens - scaffoldTokens;
9396
+ let inputBudget;
9397
+ let outputBudget;
9398
+ if (maxOutputTokens >= contextTokens) {
9399
+ inputBudget = Math.floor(available / 3);
9400
+ outputBudget = available - inputBudget;
9401
+ } else {
9402
+ outputBudget = maxOutputTokens;
9403
+ inputBudget = contextTokens - outputBudget - scaffoldTokens;
9404
+ if (inputBudget <= 0) {
9405
+ inputBudget = Math.floor(available / 3);
9406
+ outputBudget = available - inputBudget;
9407
+ }
9408
+ }
9409
+ if (inputBudget <= OVERLAP_TOKENS) {
9410
+ throw new Error(
9411
+ `Inference window too small for detection: context ${contextTokens} tokens minus scaffold ${scaffoldTokens} leaves an input budget of ${inputBudget} (need > ${OVERLAP_TOKENS}). Use a model with a larger context window or reduce the prompt scaffold.`
9412
+ );
9413
+ }
9414
+ return {
9415
+ chunking: { chunkSize: inputBudget, overlap: OVERLAP_TOKENS },
9416
+ outputBudget
9417
+ };
9418
+ }
9366
9419
  function languageName(tag) {
9367
9420
  return getLocaleEnglishName(tag) || tag;
9368
9421
  }
@@ -9382,7 +9435,9 @@ var MotivationPrompts = class {
9382
9435
  /**
9383
9436
  * Build a prompt for detecting comment-worthy passages
9384
9437
  *
9385
- * @param content - The text content to analyze (will be truncated to 8000 chars)
9438
+ * @param content - The text content to analyze — a chunk sized by the
9439
+ * caller from derived provider limits; NEVER re-truncated here (a
9440
+ * builder-level clip is silent input loss — see #738)
9386
9441
  * @param instructions - Optional user-provided instructions
9387
9442
  * @param tone - Optional tone guidance (e.g., "academic", "conversational")
9388
9443
  * @param density - Optional target number of comments per 2000 words
@@ -9403,7 +9458,7 @@ ${instructions}${toneGuidance}${densityGuidance}${sourceLang}${bodyLang}
9403
9458
 
9404
9459
  Text to analyze:
9405
9460
  ---
9406
- ${content.substring(0, 8e3)}
9461
+ ${content}
9407
9462
  ---
9408
9463
 
9409
9464
  Return a JSON array of comments. Each comment must have:
@@ -9437,7 +9492,7 @@ Guidelines:
9437
9492
 
9438
9493
  Text to analyze:
9439
9494
  ---
9440
- ${content.substring(0, 8e3)}
9495
+ ${content}
9441
9496
  ---
9442
9497
 
9443
9498
  Return a JSON array of comments. Each comment should have:
@@ -9458,7 +9513,9 @@ Example format:
9458
9513
  /**
9459
9514
  * Build a prompt for detecting highlight-worthy passages
9460
9515
  *
9461
- * @param content - The text content to analyze (will be truncated to 8000 chars)
9516
+ * @param content - The text content to analyze — a chunk sized by the
9517
+ * caller from derived provider limits; NEVER re-truncated here (a
9518
+ * builder-level clip is silent input loss — see #738)
9462
9519
  * @param instructions - Optional user-provided instructions
9463
9520
  * @param density - Optional target number of highlights per 2000 words
9464
9521
  * @returns Formatted prompt string
@@ -9476,7 +9533,7 @@ ${instructions}${densityGuidance}${sourceLang}
9476
9533
 
9477
9534
  Text to analyze:
9478
9535
  ---
9479
- ${content.substring(0, 8e3)}
9536
+ ${content}
9480
9537
  ---
9481
9538
 
9482
9539
  Return a JSON array of highlights. Each highlight must have:
@@ -9507,7 +9564,7 @@ Guidelines:
9507
9564
 
9508
9565
  Text to analyze:
9509
9566
  ---
9510
- ${content.substring(0, 8e3)}
9567
+ ${content}
9511
9568
  ---
9512
9569
 
9513
9570
  Return a JSON array of highlights. Each highlight should have:
@@ -9527,7 +9584,9 @@ Example format:
9527
9584
  /**
9528
9585
  * Build a prompt for detecting assessment-worthy passages
9529
9586
  *
9530
- * @param content - The text content to analyze (will be truncated to 8000 chars)
9587
+ * @param content - The text content to analyze — a chunk sized by the
9588
+ * caller from derived provider limits; NEVER re-truncated here (a
9589
+ * builder-level clip is silent input loss — see #738)
9531
9590
  * @param instructions - Optional user-provided instructions
9532
9591
  * @param tone - Optional tone guidance (e.g., "critical", "supportive")
9533
9592
  * @param density - Optional target number of assessments per 2000 words
@@ -9548,7 +9607,7 @@ ${instructions}${toneGuidance}${densityGuidance}${sourceLang}${bodyLang}
9548
9607
 
9549
9608
  Text to analyze:
9550
9609
  ---
9551
- ${content.substring(0, 8e3)}
9610
+ ${content}
9552
9611
  ---
9553
9612
 
9554
9613
  Return a JSON array of assessments. Each assessment must have:
@@ -9582,7 +9641,7 @@ Guidelines:
9582
9641
 
9583
9642
  Text to analyze:
9584
9643
  ---
9585
- ${content.substring(0, 8e3)}
9644
+ ${content}
9586
9645
  ---
9587
9646
 
9588
9647
  Return a JSON array of assessments. Each assessment should have:
@@ -9828,10 +9887,32 @@ function logAnchorMethod(motivation, exact, anchorMethod) {
9828
9887
  }
9829
9888
 
9830
9889
  // src/workers/annotation-detection.ts
9831
- function assertNotTruncated(response, motivation) {
9890
+ function assertNotTruncated(response, motivation, chunk, totalChunks, outputBudget) {
9832
9891
  if (response.stopReason === "max_tokens") {
9833
- throw new Error(`${motivation} detection response truncated (max_tokens) \u2014 increase max_tokens or reduce resource size; failing the job rather than under-reporting annotations.`);
9892
+ throw new Error(`${motivation} detection response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
9893
+ }
9894
+ }
9895
+ async function detectInChunks(client, content, buildPrompt, temperature, motivation, parse, onChunk) {
9896
+ const limits = await client.limits();
9897
+ const scaffoldTokens = estimateTokens(buildPrompt(""));
9898
+ const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
9899
+ const chunks = chunkText(content, chunking);
9900
+ const collected = [];
9901
+ for (let i = 0; i < chunks.length; i++) {
9902
+ const response = await boundedGenerateWithMetadata(
9903
+ client,
9904
+ buildPrompt(chunks[i]),
9905
+ outputBudget,
9906
+ temperature,
9907
+ { format: "json" }
9908
+ );
9909
+ assertNotTruncated(response, motivation, i + 1, chunks.length, outputBudget);
9910
+ collected.push(...parse(response.text));
9911
+ if (i < chunks.length - 1) {
9912
+ onChunk?.(i + 1, chunks.length);
9913
+ }
9834
9914
  }
9915
+ return collected;
9835
9916
  }
9836
9917
  var AnnotationDetection = class {
9837
9918
  /**
@@ -9842,11 +9923,16 @@ var AnnotationDetection = class {
9842
9923
  * (source-resource locale). See `types.ts` "Locale conventions" for the
9843
9924
  * full discussion.
9844
9925
  */
9845
- static async detectComments(content, client, instructions, tone, density, language, sourceLanguage) {
9846
- const prompt = MotivationPrompts.buildCommentPrompt(content, instructions, tone, density, language, sourceLanguage);
9847
- const response = await boundedGenerateWithMetadata(client, prompt, 3e3, 0.4, { format: "json" });
9848
- assertNotTruncated(response, "comment");
9849
- return MotivationParsers.parseComments(response.text, content);
9926
+ static async detectComments(content, client, instructions, tone, density, language, sourceLanguage, onChunk) {
9927
+ return detectInChunks(
9928
+ client,
9929
+ content,
9930
+ (chunk) => MotivationPrompts.buildCommentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
9931
+ 0.4,
9932
+ "comment",
9933
+ (text) => MotivationParsers.parseComments(text, content),
9934
+ onChunk
9935
+ );
9850
9936
  }
9851
9937
  /**
9852
9938
  * Detect highlights in content.
@@ -9855,11 +9941,16 @@ var AnnotationDetection = class {
9855
9941
  * applies, used in the prompt so the LLM analyzes non-English source
9856
9942
  * correctly.
9857
9943
  */
9858
- static async detectHighlights(content, client, instructions, density, sourceLanguage) {
9859
- const prompt = MotivationPrompts.buildHighlightPrompt(content, instructions, density, sourceLanguage);
9860
- const response = await boundedGenerateWithMetadata(client, prompt, 2e3, 0.3, { format: "json" });
9861
- assertNotTruncated(response, "highlight");
9862
- return MotivationParsers.parseHighlights(response.text, content);
9944
+ static async detectHighlights(content, client, instructions, density, sourceLanguage, onChunk) {
9945
+ return detectInChunks(
9946
+ client,
9947
+ content,
9948
+ (chunk) => MotivationPrompts.buildHighlightPrompt(chunk, instructions, density, sourceLanguage),
9949
+ 0.3,
9950
+ "highlight",
9951
+ (text) => MotivationParsers.parseHighlights(text, content),
9952
+ onChunk
9953
+ );
9863
9954
  }
9864
9955
  /**
9865
9956
  * Detect assessments in content.
@@ -9868,11 +9959,16 @@ var AnnotationDetection = class {
9868
9959
  * (annotation body locale). `sourceLanguage` is the locale of the content
9869
9960
  * being analyzed (source-resource locale).
9870
9961
  */
9871
- static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage) {
9872
- const prompt = MotivationPrompts.buildAssessmentPrompt(content, instructions, tone, density, language, sourceLanguage);
9873
- const response = await boundedGenerateWithMetadata(client, prompt, 3e3, 0.3, { format: "json" });
9874
- assertNotTruncated(response, "assessment");
9875
- return MotivationParsers.parseAssessments(response.text, content);
9962
+ static async detectAssessments(content, client, instructions, tone, density, language, sourceLanguage, onChunk) {
9963
+ return detectInChunks(
9964
+ client,
9965
+ content,
9966
+ (chunk) => MotivationPrompts.buildAssessmentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
9967
+ 0.3,
9968
+ "assessment",
9969
+ (text) => MotivationParsers.parseAssessments(text, content),
9970
+ onChunk
9971
+ );
9876
9972
  }
9877
9973
  /**
9878
9974
  * Detect tags in content for a specific category.
@@ -9886,28 +9982,33 @@ var AnnotationDetection = class {
9886
9982
  * identifiers, not LLM-generated text — so it's consumed at the body-stamp
9887
9983
  * site, not here.
9888
9984
  */
9889
- static async detectTags(content, client, schema, category, sourceLanguage) {
9985
+ static async detectTags(content, client, schema, category, sourceLanguage, onChunk) {
9890
9986
  const categoryInfo = schema.tags.find((t) => t.name === category);
9891
9987
  if (!categoryInfo) {
9892
9988
  throw new Error(`Invalid category "${category}" for schema ${schema.id}`);
9893
9989
  }
9894
- const prompt = MotivationPrompts.buildTagPrompt(
9990
+ const parsedTags = await detectInChunks(
9991
+ client,
9895
9992
  content,
9896
- category,
9897
- schema.name,
9898
- schema.description,
9899
- schema.domain,
9900
- categoryInfo.description,
9901
- categoryInfo.examples,
9902
- sourceLanguage
9993
+ (chunk) => MotivationPrompts.buildTagPrompt(
9994
+ chunk,
9995
+ category,
9996
+ schema.name,
9997
+ schema.description,
9998
+ schema.domain,
9999
+ categoryInfo.description,
10000
+ categoryInfo.examples,
10001
+ sourceLanguage
10002
+ ),
10003
+ 0.2,
10004
+ "tag",
10005
+ (text) => MotivationParsers.parseTags(text),
10006
+ onChunk
9903
10007
  );
9904
- const response = await boundedGenerateWithMetadata(client, prompt, 4e3, 0.2, { format: "json" });
9905
- assertNotTruncated(response, "tag");
9906
- const parsedTags = MotivationParsers.parseTags(response.text);
9907
10008
  return MotivationParsers.validateTagOffsets(parsedTags, content, category);
9908
10009
  }
9909
10010
  };
9910
- async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage) {
10011
+ async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage, onChunk) {
9911
10012
  const entityTypesDescription = entityTypes.map((et) => {
9912
10013
  if (typeof et === "string") {
9913
10014
  return et;
@@ -9939,11 +10040,11 @@ Find direct mentions only (names, proper nouns). Do not include pronouns or desc
9939
10040
  const sourceLangGuidance = sourceLanguage ? `
9940
10041
  Source text language: ${getLocaleEnglishName(sourceLanguage) || sourceLanguage}.
9941
10042
  ` : "";
9942
- const prompt = `Identify entity references in the following text. Look for mentions of: ${entityTypesDescription}.
10043
+ const buildPrompt = (text) => `Identify entity references in the following text. Look for mentions of: ${entityTypesDescription}.
9943
10044
  ${descriptiveReferenceGuidance}${sourceLangGuidance}
9944
10045
  Text to analyze:
9945
10046
  """
9946
- ${exact}
10047
+ ${text}
9947
10048
  """
9948
10049
 
9949
10050
  Respond with a JSON array of entities found. Each entity should have:
@@ -9956,59 +10057,78 @@ If no entities are found, respond with an empty array [].
9956
10057
 
9957
10058
  Example output:
9958
10059
  [{"exact":"Alice","entityType":"Person","prefix":"","suffix":" went to"},{"exact":"Paris","entityType":"Location","prefix":"went to ","suffix":" yesterday"}]`;
9959
- logger2.debug("Sending entity extraction request", { entityTypes: entityTypesDescription });
9960
- const response = await boundedGenerateWithMetadata(
9961
- client,
9962
- prompt,
9963
- 4e3,
9964
- // Increased to handle many entities without truncation
9965
- 0.3,
9966
- // Lower temperature for more consistent extraction
9967
- // Force grammar-constrained JSON output. Without this, Ollama models
9968
- // periodically emit malformed JSON (truncated brackets, mid-token
9969
- // breaks at higher token counts) which silently parse-fails into
9970
- // [] downstream. The prompt's schema (which keys, what types) still
9971
- // governs *what* the JSON contains; `format: 'json'` governs that
9972
- // it's syntactically valid.
9973
- { format: "json" }
9974
- );
9975
- logger2.debug("Got entity extraction response", { responseLength: response.text.length });
9976
- if (response.stopReason === "max_tokens") {
9977
- const errorMsg = "Entity extraction response truncated (max_tokens) \u2014 increase max_tokens or reduce resource size; failing the job rather than dropping annotations.";
9978
- logger2.error(errorMsg, { responseLength: response.text.length });
9979
- throw new Error(errorMsg);
9980
- }
9981
- let entities;
9982
- try {
9983
- entities = JSON.parse(response.text.trim());
9984
- } catch (error) {
9985
- logger2.error("Failed to parse entity extraction response", {
9986
- error: error instanceof Error ? error.message : String(error),
9987
- response: response.text.slice(0, 500)
9988
- });
9989
- throw new Error("Failed to parse entity extraction response", {
9990
- cause: error instanceof Error ? error : new Error(String(error))
10060
+ const limits = await client.limits();
10061
+ const scaffoldTokens = estimateTokens(buildPrompt(""));
10062
+ const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
10063
+ const chunks = chunkText(exact, chunking);
10064
+ logger2.debug("Sending entity extraction request", {
10065
+ entityTypes: entityTypesDescription,
10066
+ chunks: chunks.length,
10067
+ chunkSizeTokens: chunking.chunkSize,
10068
+ outputBudget
10069
+ });
10070
+ const collected = [];
10071
+ for (let i = 0; i < chunks.length; i++) {
10072
+ const response = await boundedGenerateWithMetadata(
10073
+ client,
10074
+ buildPrompt(chunks[i]),
10075
+ outputBudget,
10076
+ 0.3,
10077
+ // Lower temperature for more consistent extraction
10078
+ // Force grammar-constrained JSON output. Without this, Ollama models
10079
+ // periodically emit malformed JSON (truncated brackets, mid-token
10080
+ // breaks at higher token counts) which silently parse-fails into
10081
+ // [] downstream. The prompt's schema (which keys, what types) still
10082
+ // governs *what* the JSON contains; `format: 'json'` governs that
10083
+ // it's syntactically valid.
10084
+ { format: "json" }
10085
+ );
10086
+ logger2.debug("Got entity extraction response", {
10087
+ chunk: i + 1,
10088
+ chunks: chunks.length,
10089
+ responseLength: response.text.length
9991
10090
  });
10091
+ if (response.stopReason === "max_tokens") {
10092
+ const errorMsg = `Entity extraction response truncated (max_tokens) on chunk ${i + 1}/${chunks.length} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than dropping annotations.`;
10093
+ logger2.error(errorMsg, { responseLength: response.text.length });
10094
+ throw new Error(errorMsg);
10095
+ }
10096
+ let entities;
10097
+ try {
10098
+ entities = JSON.parse(response.text.trim());
10099
+ } catch (error) {
10100
+ logger2.error("Failed to parse entity extraction response", {
10101
+ error: error instanceof Error ? error.message : String(error),
10102
+ response: response.text.slice(0, 500)
10103
+ });
10104
+ throw new Error("Failed to parse entity extraction response", {
10105
+ cause: error instanceof Error ? error : new Error(String(error))
10106
+ });
10107
+ }
10108
+ if (!isArray(entities)) {
10109
+ logger2.error("Failed to parse entity extraction response: expected a JSON array", {
10110
+ response: response.text.slice(0, 500)
10111
+ });
10112
+ throw new Error("Failed to parse entity extraction response: expected a JSON array");
10113
+ }
10114
+ logger2.debug("Parsed entities from AI response", { chunk: i + 1, count: entities.length });
10115
+ for (const e of entities) {
10116
+ if (isObject(e) && isString(e.exact) && isString(e.entityType)) {
10117
+ collected.push({
10118
+ exact: e.exact,
10119
+ entityType: e.entityType,
10120
+ ...isString(e.prefix) ? { prefix: e.prefix } : {},
10121
+ ...isString(e.suffix) ? { suffix: e.suffix } : {}
10122
+ });
10123
+ } else {
10124
+ logger2.debug("Dropped malformed LLM entity", { entity: e });
10125
+ }
10126
+ }
10127
+ if (i < chunks.length - 1) {
10128
+ onChunk?.(i + 1, chunks.length);
10129
+ }
9992
10130
  }
9993
- if (!isArray(entities)) {
9994
- logger2.error("Failed to parse entity extraction response: expected a JSON array", {
9995
- response: response.text.slice(0, 500)
9996
- });
9997
- throw new Error("Failed to parse entity extraction response: expected a JSON array");
9998
- }
9999
- logger2.debug("Parsed entities from AI response", { count: entities.length });
10000
- return entities.filter((e) => {
10001
- const ok = isObject(e) && isString(e.exact) && isString(e.entityType);
10002
- if (!ok) {
10003
- logger2.debug("Dropped malformed LLM entity", { entity: e });
10004
- }
10005
- return ok;
10006
- }).map((entity) => ({
10007
- exact: entity.exact,
10008
- entityType: entity.entityType,
10009
- ...isString(entity.prefix) ? { prefix: entity.prefix } : {},
10010
- ...isString(entity.suffix) ? { suffix: entity.suffix } : {}
10011
- }));
10131
+ return collected;
10012
10132
  }
10013
10133
  function getLanguageName(locale) {
10014
10134
  return getLocaleEnglishName(locale) || locale;
@@ -10019,7 +10139,7 @@ var SEMANTIC_MATCH_CHARS = 240;
10019
10139
  function idLabel(resourceId, annotationId) {
10020
10140
  return `[${resourceId}${annotationId ? `/${annotationId}` : ""}]`;
10021
10141
  }
10022
- async function generateResourceFromTopic(topic, entityTypes, client, logger2, userPrompt, locale, context, temperature, maxTokens, sourceLanguage, outputMediaType = "text/markdown", task = "resource", structure, cite = false) {
10142
+ async function generateResourceFromTopic(topic, entityTypes, client, logger2, userPrompt, locale, context, temperature, maxTokens, sourceLanguage, outputMediaType = "text/markdown", task = "resource", structure, cite = false, repair) {
10023
10143
  logger2.debug("Generating resource from topic", {
10024
10144
  topicPreview: topic.substring(0, 100),
10025
10145
  entityTypes,
@@ -10135,11 +10255,12 @@ ${parts.join("\n")}`;
10135
10255
  let semanticContextSection = "";
10136
10256
  const similar = context?.semanticContext?.similar ?? [];
10137
10257
  if (similar.length > 0) {
10138
- const lines = [...similar].sort((a, b) => b.score - a.score).slice(0, SEMANTIC_MATCH_LIMIT).map((m) => `- ${idLabel(m.resourceId, m.annotationId)} (${m.score.toFixed(2)}) ${m.text.slice(0, SEMANTIC_MATCH_CHARS)}`);
10258
+ const lines = [...similar].sort((a, b) => b.score - a.score).slice(0, SEMANTIC_MATCH_LIMIT).map((m) => `- ${idLabel(m.resourceId, m.annotationId)} (${m.score.toFixed(2)})${m.machineRead ? " [OCR]" : ""} ${m.text.slice(0, SEMANTIC_MATCH_CHARS)}`);
10259
+ const ocrNote = similar.some((m) => m.machineRead) ? "\nPassages marked [OCR] were read from scanned images by character recognition; treat their exact wording and numbers as uncertain, and say so if you rely on one." : "";
10139
10260
  semanticContextSection = `
10140
10261
 
10141
10262
  Related passages from the knowledge base:
10142
- ${lines.join("\n")}`;
10263
+ ${lines.join("\n")}${ocrNote}`;
10143
10264
  }
10144
10265
  let leadLine;
10145
10266
  if (task === "resource") {
@@ -10154,11 +10275,12 @@ ${lines.join("\n")}`;
10154
10275
  Topic: "${topic}"`;
10155
10276
  }
10156
10277
  const isPlainText = outputMediaType === "text/plain";
10278
+ const isPdf = outputMediaType === "application/pdf";
10157
10279
  let structureRequirement = "";
10158
10280
  let titleRequirement = "";
10159
10281
  if (structure === "sections") {
10160
- structureRequirement = isPlainText ? "\n- Organize the content into titled sections with well-structured paragraphs" : "\n- Organize the content into titled sections (## Section) with well-structured paragraphs";
10161
- if (!isPlainText) {
10282
+ structureRequirement = isPdf ? "\n- Organize the content into titled sections (= Heading) with well-structured paragraphs" : isPlainText ? "\n- Organize the content into titled sections with well-structured paragraphs" : "\n- Organize the content into titled sections (## Section) with well-structured paragraphs";
10283
+ if (!isPlainText && !isPdf) {
10162
10284
  titleRequirement = "\n- Start with a clear heading (# Title)";
10163
10285
  }
10164
10286
  } else if (structure === "prose") {
@@ -10171,11 +10293,20 @@ Topic: "${topic}"`;
10171
10293
  - Organize the output as: ${structure}`;
10172
10294
  }
10173
10295
  const citeRequirement = cite ? "\n- Ground every claim in the provided context. Immediately after each claim, cite its source by emitting [[<id>]], where <id> is an id shown in square brackets in the context above (for a passage labeled [abc], emit [[abc]]). Cite only ids that appear in the context." : "";
10174
- const formatRequirements = isPlainText ? `- Write the response as plain text \u2014 no formatting markup (no #, *, backticks, headings, or links)
10296
+ const formatRequirements = isPdf ? `- Write the response as Typst markup (the Typst typesetting language \u2014 not markdown, not LaTeX)
10297
+ - Headings are written as = Heading (deeper levels == Subheading); everything else is plain prose paragraphs
10298
+ - Do not emit markdown syntax or code fences` : isPlainText ? `- Write the response as plain text \u2014 no formatting markup (no #, *, backticks, headings, or links)
10175
10299
  - Begin with the title on its own first line` : `- Use markdown formatting
10176
10300
  - Write the response as markdown`;
10301
+ const repairSection = repair ? `
10302
+
10303
+ Your previous attempt failed to compile. Fix the error and return the complete corrected document \u2014 full source, not a diff.
10304
+ Compile error:
10305
+ ${repair.error}
10306
+ Previous source:
10307
+ ${repair.source}` : "";
10177
10308
  const prompt = `${leadLine}
10178
- ${userPrompt ? `Instruction: ${userPrompt}` : ""}
10309
+ ${userPrompt ? `Instruction: ${userPrompt}` : ""}${repairSection}
10179
10310
  ${entityTypes.length > 0 ? `Focus on these entity types: ${entityTypes.join(", ")}.` : ""}${annotationSection}${contextSection}${resourceSection}${graphSection}${semanticContextSection}${sourceLanguageInstruction}${languageInstruction}
10180
10311
 
10181
10312
  Requirements:
@@ -10184,7 +10315,7 @@ Requirements:
10184
10315
  ${formatRequirements}`;
10185
10316
  const parseResponse = (response2) => {
10186
10317
  let content = response2.trim();
10187
- if (content.startsWith("```markdown") || content.startsWith("```md")) {
10318
+ if (content.startsWith("```markdown") || content.startsWith("```md") || content.startsWith("```typst")) {
10188
10319
  content = content.slice(content.indexOf("\n") + 1);
10189
10320
  const endIndex = content.lastIndexOf("```");
10190
10321
  if (endIndex !== -1) {
@@ -10219,6 +10350,29 @@ ${formatRequirements}`;
10219
10350
  });
10220
10351
  return result;
10221
10352
  }
10353
+ var PINNED_CREATION_TIMESTAMP = 17e8;
10354
+ var MAX_COMPILE_REPAIRS = 2;
10355
+ function compileTypst(source) {
10356
+ const dir = mkdtempSync(join(tmpdir(), "typst-"));
10357
+ try {
10358
+ const inFile = join(dir, "doc.typ");
10359
+ const outFile = join(dir, "doc.pdf");
10360
+ writeFileSync(inFile, source);
10361
+ try {
10362
+ execFileSync(
10363
+ "typst",
10364
+ ["compile", "--creation-timestamp", String(PINNED_CREATION_TIMESTAMP), inFile, outFile],
10365
+ { stdio: ["ignore", "pipe", "pipe"] }
10366
+ );
10367
+ } catch (err) {
10368
+ const stderr = err.stderr;
10369
+ return { error: stderr?.length ? stderr.toString("utf8") : String(err) };
10370
+ }
10371
+ return { pdf: new Uint8Array(readFileSync(outFile)) };
10372
+ } finally {
10373
+ rmSync(dir, { recursive: true, force: true });
10374
+ }
10375
+ }
10222
10376
 
10223
10377
  // src/workers/generation/citation-resolver.ts
10224
10378
  var CITATION_TOKEN = /\[\[([^\s[\]/]+)(?:\/([^\s[\]/]+))?\]\]/g;
@@ -10363,9 +10517,9 @@ function buildTextAnnotation(content, resourceId, userId, generator, motivation,
10363
10517
  ...body !== void 0 ? { body } : {}
10364
10518
  };
10365
10519
  }
10366
- function buildPdfAnnotation(layer, resourceId, userId, generator, motivation, match, body) {
10367
- const { rects, overlap } = locate(layer, match.start, match.end);
10368
- const coveredText = overlap.length ? layer.text.substring(
10520
+ function buildPdfAnnotation(anchored, resourceId, userId, generator, motivation, match, body) {
10521
+ const { rects, overlap } = locate(anchored, match.start, match.end);
10522
+ const coveredText = overlap.length ? anchored.text.substring(
10369
10523
  Math.min(...overlap.map((i) => i.start)),
10370
10524
  Math.max(...overlap.map((i) => i.end))
10371
10525
  ) : "";
@@ -10414,7 +10568,9 @@ async function processHighlightJob(content, inferenceClient, params, buildAnnota
10414
10568
  inferenceClient,
10415
10569
  params.instructions,
10416
10570
  params.density,
10417
- params.sourceLanguage
10571
+ params.sourceLanguage,
10572
+ // Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
10573
+ (completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
10418
10574
  );
10419
10575
  onProgress(60, `Creating ${highlights.length} annotations...`, "creating");
10420
10576
  const annotations = dedupeAnnotations(highlights.map(
@@ -10436,7 +10592,9 @@ async function processCommentJob(content, inferenceClient, params, buildAnnotati
10436
10592
  params.tone,
10437
10593
  params.density,
10438
10594
  params.language,
10439
- params.sourceLanguage
10595
+ params.sourceLanguage,
10596
+ // Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
10597
+ (completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
10440
10598
  );
10441
10599
  onProgress(60, `Creating ${comments.length} annotations...`, "creating");
10442
10600
  const bodyLanguage = params.language ?? "en";
@@ -10466,7 +10624,9 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
10466
10624
  params.tone,
10467
10625
  params.density,
10468
10626
  params.language,
10469
- params.sourceLanguage
10627
+ params.sourceLanguage,
10628
+ // Chunk-boundary heartbeat (liveness): interpolate within the 30–60 band.
10629
+ (completed, total) => onProgress(30 + Math.round(completed / total * 30), "Analyzing text...", "analyzing")
10470
10630
  );
10471
10631
  onProgress(60, `Creating ${assessments.length} annotations...`, "creating");
10472
10632
  const bodyLanguage = params.language ?? "en";
@@ -10522,7 +10682,23 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
10522
10682
  inferenceClient,
10523
10683
  params.includeDescriptiveReferences ?? false,
10524
10684
  logger2,
10525
- params.sourceLanguage
10685
+ params.sourceLanguage,
10686
+ // Chunk-boundary heartbeat: progress is the worker's liveness signal
10687
+ // (stall watchdog + backend janitor), so multi-chunk extraction must
10688
+ // emit between inference calls. Percentage interpolates within this
10689
+ // entity type's band of the 20–80 range.
10690
+ (completed, total) => {
10691
+ const interpolated = 20 + Math.round((i + completed / total) / entityTypeNames.length * 60);
10692
+ onProgress(interpolated, `Detecting ${entityTypeName} entities...`, "analyzing", {
10693
+ currentEntityType: entityTypeName,
10694
+ processedEntityTypes: i,
10695
+ totalEntityTypes: entityTypeNames.length,
10696
+ entitiesFound: totalFound,
10697
+ entitiesEmitted: totalEmitted,
10698
+ completedEntityTypes: [...completedEntityTypes],
10699
+ requestParams
10700
+ });
10701
+ }
10526
10702
  );
10527
10703
  totalFound += extractedEntities.length;
10528
10704
  completedEntityTypes.push({ entityType: entityTypeName, foundCount: extractedEntities.length });
@@ -10566,13 +10742,21 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
10566
10742
  onProgress(10, "Loading resource...", "analyzing");
10567
10743
  onProgress(30, "Analyzing text for tags...", "analyzing");
10568
10744
  const allTags = [];
10569
- for (const category of params.categories) {
10745
+ for (let c = 0; c < params.categories.length; c++) {
10746
+ const category = params.categories[c];
10570
10747
  const categoryTags = await AnnotationDetection.detectTags(
10571
10748
  content,
10572
10749
  inferenceClient,
10573
10750
  params.schema,
10574
10751
  category,
10575
- params.sourceLanguage
10752
+ params.sourceLanguage,
10753
+ // Chunk-boundary heartbeat (liveness): interpolate within this
10754
+ // category's slice of the 30–60 band.
10755
+ (completed, total) => onProgress(
10756
+ 30 + Math.round((c + completed / total) / params.categories.length * 30),
10757
+ "Analyzing text for tags...",
10758
+ "analyzing"
10759
+ )
10576
10760
  );
10577
10761
  allTags.push(...categoryTags);
10578
10762
  }
@@ -10598,8 +10782,14 @@ async function processTagJob(content, inferenceClient, params, buildAnnotation,
10598
10782
  result: { tagsFound: tags.length, tagsCreated: annotations.length, byCategory }
10599
10783
  };
10600
10784
  }
10785
+ function assertWithinOutputBudget(byteLength) {
10786
+ if (!withinByteBudget(byteLength)) {
10787
+ throw new Error(
10788
+ `Generated artifact exceeds the output byte budget: ${byteLength} bytes > ${MAX_PDF_BYTES}. Refusing a runaway generation.`
10789
+ );
10790
+ }
10791
+ }
10601
10792
  async function processGenerationJob(inferenceClient, params, onProgress, logger2) {
10602
- const GENERATABLE_MEDIA_TYPES = ["text/markdown", "text/plain"];
10603
10793
  const outputMediaType = params.outputMediaType ?? "text/markdown";
10604
10794
  if (!GENERATABLE_MEDIA_TYPES.includes(outputMediaType)) {
10605
10795
  throw new Error(
@@ -10608,6 +10798,84 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
10608
10798
  }
10609
10799
  const title = params.title ?? "Untitled";
10610
10800
  const entityTypes = (params.entityTypes ?? []).map(String);
10801
+ if (outputMediaType === "application/pdf") {
10802
+ onProgress(5, "Generating resource...", "generating");
10803
+ const validIds = params.cite === true ? collectContextResourceIds(params.context) : null;
10804
+ let generated2 = await generateResourceFromTopic(
10805
+ title,
10806
+ entityTypes,
10807
+ inferenceClient,
10808
+ logger2,
10809
+ params.prompt,
10810
+ params.language,
10811
+ params.context,
10812
+ params.temperature,
10813
+ params.maxTokens,
10814
+ params.sourceLanguage,
10815
+ outputMediaType,
10816
+ params.task,
10817
+ params.structure,
10818
+ params.cite
10819
+ );
10820
+ let source = generated2.content;
10821
+ let citations2 = [];
10822
+ if (validIds) {
10823
+ const resolved = resolveCitationTokens(generated2.content, validIds, logger2);
10824
+ source = resolved.content;
10825
+ citations2 = resolved.citations;
10826
+ }
10827
+ let compiled = compileTypst(source);
10828
+ let repairs = 0;
10829
+ while ("error" in compiled && repairs < MAX_COMPILE_REPAIRS) {
10830
+ repairs++;
10831
+ logger2.warn("Typst compile failed \u2014 feeding the error back for repair", {
10832
+ attempt: repairs,
10833
+ error: compiled.error.slice(0, 500)
10834
+ });
10835
+ generated2 = await generateResourceFromTopic(
10836
+ title,
10837
+ entityTypes,
10838
+ inferenceClient,
10839
+ logger2,
10840
+ params.prompt,
10841
+ params.language,
10842
+ params.context,
10843
+ params.temperature,
10844
+ params.maxTokens,
10845
+ params.sourceLanguage,
10846
+ outputMediaType,
10847
+ params.task,
10848
+ params.structure,
10849
+ params.cite,
10850
+ { source, error: compiled.error }
10851
+ );
10852
+ if (validIds) {
10853
+ const resolved = resolveCitationTokens(generated2.content, validIds, logger2);
10854
+ source = resolved.content;
10855
+ citations2 = resolved.citations;
10856
+ } else {
10857
+ source = generated2.content;
10858
+ }
10859
+ compiled = compileTypst(source);
10860
+ }
10861
+ if ("error" in compiled) {
10862
+ throw new Error(
10863
+ `Typst compilation failed after ${MAX_COMPILE_REPAIRS} repair attempts: ${compiled.error}`
10864
+ );
10865
+ }
10866
+ assertWithinOutputBudget(compiled.pdf.byteLength);
10867
+ onProgress(95, "Creating resource...", "creating");
10868
+ return {
10869
+ content: compiled.pdf,
10870
+ title: generated2.title ?? title,
10871
+ format: outputMediaType,
10872
+ citations: citations2,
10873
+ result: {
10874
+ resourceId: "",
10875
+ resourceName: generated2.title ?? title
10876
+ }
10877
+ };
10878
+ }
10611
10879
  onProgress(5, "Generating resource...", "generating");
10612
10880
  const generated = await generateResourceFromTopic(
10613
10881
  title,
@@ -10633,8 +10901,10 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
10633
10901
  citations = resolved.citations;
10634
10902
  }
10635
10903
  onProgress(95, "Creating resource...", "creating");
10904
+ const artifact = new TextEncoder().encode(content);
10905
+ assertWithinOutputBudget(artifact.byteLength);
10636
10906
  return {
10637
- content,
10907
+ content: artifact,
10638
10908
  title: generated.title ?? title,
10639
10909
  format: outputMediaType,
10640
10910
  citations,
@@ -10646,28 +10916,37 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
10646
10916
  }
10647
10917
 
10648
10918
  // src/workers/detection/prepare-detection.ts
10649
- async function prepareDetection(strategy, session, resourceId, userId, generator) {
10650
- switch (strategy) {
10651
- case "decode": {
10652
- const text = await session.client.browse.resourceContent(resourceId);
10653
- return {
10654
- text,
10655
- buildAnnotation: (motivation, match, body) => buildTextAnnotation(text, resourceId, userId, generator, motivation, match, body)
10656
- };
10657
- }
10658
- case "pdf-text-layer": {
10659
- const { data } = await session.client.browse.resourceRepresentation(resourceId);
10660
- const layer = await extractPdfTextLayer(new Uint8Array(data));
10661
- if (!layer) return null;
10662
- return {
10663
- text: layer.text,
10664
- buildAnnotation: (motivation, match, body) => buildPdfAnnotation(layer, resourceId, userId, generator, motivation, match, body)
10665
- };
10666
- }
10667
- case "none":
10668
- return null;
10919
+ async function prepareDetection(mediaType, session, resourceId, userId, generator, store) {
10920
+ const extractor = EXTRACTORS[textExtractionOf(mediaType)];
10921
+ if (!extractor) return { declined: "no-extractor" };
10922
+ const { data } = await session.client.browse.resourceRepresentation(resourceId);
10923
+ const bytes = Buffer.from(data);
10924
+ const extracted = await extractor.extract(bytes, mediaType, {
10925
+ key: calculateChecksum(bytes),
10926
+ store
10927
+ });
10928
+ if ("declined" in extracted) return extracted;
10929
+ if (!extracted.text.trim()) return { declined: "empty" };
10930
+ const items = extracted.items;
10931
+ if (items && items.length > 0) {
10932
+ const anchored = { text: extracted.text, items };
10933
+ return {
10934
+ text: extracted.text,
10935
+ buildAnnotation: (motivation, match, body) => buildPdfAnnotation(anchored, resourceId, userId, generator, motivation, match, body)
10936
+ };
10669
10937
  }
10938
+ return {
10939
+ text: extracted.text,
10940
+ buildAnnotation: (motivation, match, body) => buildTextAnnotation(extracted.text, resourceId, userId, generator, motivation, match, body)
10941
+ };
10670
10942
  }
10943
+ var DECLINE_MESSAGES = {
10944
+ "no-text-layer": "This PDF is a scan whose text could not be recognized; there is nothing to detect over.",
10945
+ "encrypted": "This PDF is password-protected, so its text cannot be read.",
10946
+ "corrupt": "This PDF could not be parsed \u2014 the file may be damaged or truncated.",
10947
+ "too-large": "This document is too large to extract text from.",
10948
+ "empty": "This document contains no text to detect over."
10949
+ };
10671
10950
  async function emitEvent(session, channel, payload) {
10672
10951
  await session.client.transport.emit(channel, payload);
10673
10952
  }
@@ -10685,15 +10964,16 @@ function startWorkerProcess(config) {
10685
10964
  const message = error instanceof Error ? error.message : String(error);
10686
10965
  logger2.error("Job failed", { jobId: job.jobId, error: message, stack: error instanceof Error ? error.stack : void 0 });
10687
10966
  const failAnnotationId = job.params.referenceId;
10688
- emitEvent(session, "job:fail", {
10689
- resourceId: job.resourceId,
10690
- userId: job.userId,
10691
- jobId: job.jobId,
10692
- jobType: job.type,
10693
- ...failAnnotationId ? { annotationId: failAnnotationId } : {},
10694
- error: message
10695
- }).catch(() => {
10696
- });
10967
+ if (isJobType(job.type)) {
10968
+ emitEvent(session, "job:fail", {
10969
+ resourceId: job.resourceId,
10970
+ jobId: job.jobId,
10971
+ jobType: job.type,
10972
+ ...failAnnotationId ? { annotationId: failAnnotationId } : {},
10973
+ error: message
10974
+ }).catch(() => {
10975
+ });
10976
+ }
10697
10977
  adapter.failJob(job.jobId, message);
10698
10978
  });
10699
10979
  });
@@ -10725,11 +11005,16 @@ async function handleJob(adapter, config, job) {
10725
11005
  }
10726
11006
  async function handleJobInner(adapter, config, job) {
10727
11007
  const { session, inferenceClient, generator } = config;
10728
- const { resourceId, userId, jobId, type: jobType } = job;
11008
+ const { userId, jobId } = job;
11009
+ if (!isJobType(job.type)) {
11010
+ adapter.failJob(jobId, `Unrecognized job type: ${job.type}`);
11011
+ return;
11012
+ }
11013
+ const jobType = job.type;
11014
+ const resourceId$1 = resourceId(job.resourceId);
10729
11015
  const annotationId = job.params.referenceId;
10730
11016
  const lifecycleBase = {
10731
- resourceId,
10732
- userId,
11017
+ resourceId: resourceId$1,
10733
11018
  jobId,
10734
11019
  jobType,
10735
11020
  ...annotationId ? { annotationId } : {}
@@ -10739,23 +11024,27 @@ async function handleJobInner(adapter, config, job) {
10739
11024
  adapter.failJob(jobId, `Worker not configured for job type: ${jobType}`);
10740
11025
  return;
10741
11026
  }
10742
- let source = null;
11027
+ let ready = null;
10743
11028
  if (jobType !== "generation") {
10744
- const descriptor = await session.client.browse.resource(resourceId).fresh();
11029
+ const descriptor = await session.client.browse.resource(resourceId$1).fresh();
10745
11030
  const mediaType = getPrimaryMediaType(descriptor);
10746
- const strategy = mediaType ? textExtractionOf(mediaType) : "none";
10747
- if (strategy === "none") {
10748
- throw new Error(`Cannot run ${jobType} on resource ${resourceId}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
10749
- }
10750
- source = await prepareDetection(strategy, session, resourceId, userId, generator);
10751
- if (!source) {
11031
+ const source = await prepareDetection(mediaType ?? "", session, resourceId$1, userId, generator, config.anchoredTextStore);
11032
+ if ("declined" in source) {
11033
+ if (source.declined === "no-extractor") {
11034
+ throw new Error(`Cannot run ${jobType} on resource ${resourceId$1}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
11035
+ }
10752
11036
  await emitEvent(session, "job:complete", {
10753
11037
  ...lifecycleBase,
10754
- result: { declined: true, reason: "no-text-layer", message: "This PDF has no extractable text layer (scanned or image-only); detection is not supported." }
11038
+ result: {
11039
+ declined: true,
11040
+ reason: source.declined,
11041
+ message: DECLINE_MESSAGES[source.declined]
11042
+ }
10755
11043
  });
10756
11044
  adapter.completeJob();
10757
11045
  return;
10758
11046
  }
11047
+ ready = source;
10759
11048
  }
10760
11049
  const onProgress = (percentage, message, stage, extra) => {
10761
11050
  adapter.touchActivity();
@@ -10774,14 +11063,14 @@ async function handleJobInner(adapter, config, job) {
10774
11063
  };
10775
11064
  if (jobType === "highlight-annotation") {
10776
11065
  const { annotations, result } = await processHighlightJob(
10777
- source.text,
11066
+ ready.text,
10778
11067
  inferenceClient,
10779
- job.params,
10780
- source.buildAnnotation,
11068
+ asJobParams(job.params),
11069
+ ready.buildAnnotation,
10781
11070
  onProgress
10782
11071
  );
10783
11072
  for (const ann of annotations) {
10784
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11073
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10785
11074
  }
10786
11075
  await emitEvent(session, "job:complete", {
10787
11076
  ...lifecycleBase,
@@ -10790,14 +11079,14 @@ async function handleJobInner(adapter, config, job) {
10790
11079
  adapter.completeJob();
10791
11080
  } else if (jobType === "comment-annotation") {
10792
11081
  const { annotations, result } = await processCommentJob(
10793
- source.text,
11082
+ ready.text,
10794
11083
  inferenceClient,
10795
- job.params,
10796
- source.buildAnnotation,
11084
+ asJobParams(job.params),
11085
+ ready.buildAnnotation,
10797
11086
  onProgress
10798
11087
  );
10799
11088
  for (const ann of annotations) {
10800
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11089
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10801
11090
  }
10802
11091
  await emitEvent(session, "job:complete", {
10803
11092
  ...lifecycleBase,
@@ -10806,14 +11095,14 @@ async function handleJobInner(adapter, config, job) {
10806
11095
  adapter.completeJob();
10807
11096
  } else if (jobType === "assessment-annotation") {
10808
11097
  const { annotations, result } = await processAssessmentJob(
10809
- source.text,
11098
+ ready.text,
10810
11099
  inferenceClient,
10811
- job.params,
10812
- source.buildAnnotation,
11100
+ asJobParams(job.params),
11101
+ ready.buildAnnotation,
10813
11102
  onProgress
10814
11103
  );
10815
11104
  for (const ann of annotations) {
10816
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11105
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10817
11106
  }
10818
11107
  await emitEvent(session, "job:complete", {
10819
11108
  ...lifecycleBase,
@@ -10822,15 +11111,15 @@ async function handleJobInner(adapter, config, job) {
10822
11111
  adapter.completeJob();
10823
11112
  } else if (jobType === "reference-annotation") {
10824
11113
  const { annotations, result } = await processReferenceJob(
10825
- source.text,
11114
+ ready.text,
10826
11115
  inferenceClient,
10827
- job.params,
10828
- source.buildAnnotation,
11116
+ asJobParams(job.params),
11117
+ ready.buildAnnotation,
10829
11118
  onProgress,
10830
11119
  config.logger
10831
11120
  );
10832
11121
  for (const ann of annotations) {
10833
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11122
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10834
11123
  }
10835
11124
  await emitEvent(session, "job:complete", {
10836
11125
  ...lifecycleBase,
@@ -10839,14 +11128,14 @@ async function handleJobInner(adapter, config, job) {
10839
11128
  adapter.completeJob();
10840
11129
  } else if (jobType === "tag-annotation") {
10841
11130
  const { annotations, result } = await processTagJob(
10842
- source.text,
11131
+ ready.text,
10843
11132
  inferenceClient,
10844
- job.params,
10845
- source.buildAnnotation,
11133
+ asJobParams(job.params),
11134
+ ready.buildAnnotation,
10846
11135
  onProgress
10847
11136
  );
10848
11137
  for (const ann of annotations) {
10849
- await emitEvent(session, "mark:create", { annotation: ann, userId, resourceId });
11138
+ await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
10850
11139
  }
10851
11140
  await emitEvent(session, "job:complete", {
10852
11141
  ...lifecycleBase,
@@ -10867,7 +11156,7 @@ async function handleJobInner(adapter, config, job) {
10867
11156
  file: Buffer.from(genResult.content),
10868
11157
  format: genResult.format,
10869
11158
  storageUri,
10870
- sourceResourceId: resourceId,
11159
+ sourceResourceId: resourceId$1,
10871
11160
  ...genParams.referenceId ? { sourceAnnotationId: genParams.referenceId } : {},
10872
11161
  ...genParams.prompt ? { generationPrompt: genParams.prompt } : {},
10873
11162
  ...genParams.language ? { language: genParams.language } : {},
@@ -10878,29 +11167,63 @@ async function handleJobInner(adapter, config, job) {
10878
11167
  const { annotation: provenanceRef } = assembleAnnotation(
10879
11168
  {
10880
11169
  motivation: "linking",
10881
- target: { source: String(resourceId) },
11170
+ target: { source: String(resourceId$1) },
10882
11171
  body: { type: "SpecificResource", source: String(newResourceId), purpose: "linking" }
10883
11172
  },
10884
11173
  generator
10885
11174
  );
10886
- await emitEvent(session, "mark:create", { annotation: provenanceRef, userId, resourceId });
10887
- }
10888
- for (const citation of genResult.citations) {
10889
- const { annotation: citationRef } = assembleAnnotation(
10890
- {
10891
- motivation: "linking",
10892
- target: {
10893
- source: String(newResourceId),
10894
- selector: [
10895
- { type: "TextPositionSelector", start: citation.start, end: citation.end },
10896
- { type: "TextQuoteSelector", exact: citation.exact }
10897
- ]
11175
+ await emitEvent(session, "mark:create", { annotation: provenanceRef, resourceId: resourceId$1 });
11176
+ }
11177
+ if (genResult.format === "application/pdf" && genResult.citations.length > 0) {
11178
+ const layer = await extractPdfTextLayer(genResult.content);
11179
+ if (!layer) {
11180
+ config.logger.warn("PDF citations dropped \u2014 the generated artifact yielded no text layer", {
11181
+ jobId,
11182
+ resourceId: newResourceId,
11183
+ citations: genResult.citations.length
11184
+ });
11185
+ } else {
11186
+ for (const citation of genResult.citations) {
11187
+ const span = findClaimSpan(layer, citation.exact);
11188
+ if (!span) {
11189
+ config.logger.warn("PDF citation dropped \u2014 claim not found in the rendered text layer", {
11190
+ jobId,
11191
+ resourceId: newResourceId,
11192
+ citedResourceId: citation.resourceId,
11193
+ exactPreview: citation.exact.slice(0, 80)
11194
+ });
11195
+ continue;
11196
+ }
11197
+ const citationRef = buildPdfAnnotation(
11198
+ layer,
11199
+ resourceId(String(newResourceId)),
11200
+ userId,
11201
+ generator,
11202
+ "linking",
11203
+ { exact: layer.text.slice(span.start, span.end), start: span.start, end: span.end },
11204
+ { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
11205
+ );
11206
+ await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
11207
+ }
11208
+ }
11209
+ } else {
11210
+ for (const citation of genResult.citations) {
11211
+ const { annotation: citationRef } = assembleAnnotation(
11212
+ {
11213
+ motivation: "linking",
11214
+ target: {
11215
+ source: String(newResourceId),
11216
+ selector: [
11217
+ { type: "TextPositionSelector", start: citation.start, end: citation.end },
11218
+ { type: "TextQuoteSelector", exact: citation.exact }
11219
+ ]
11220
+ },
11221
+ body: { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
10898
11222
  },
10899
- body: { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
10900
- },
10901
- generator
10902
- );
10903
- await emitEvent(session, "mark:create", { annotation: citationRef, userId, resourceId: newResourceId });
11223
+ generator
11224
+ );
11225
+ await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
11226
+ }
10904
11227
  }
10905
11228
  await emitEvent(session, "job:complete", {
10906
11229
  ...lifecycleBase,
@@ -11052,6 +11375,10 @@ async function startAgentWorker(opts) {
11052
11375
  jobTypes: group.jobTypes,
11053
11376
  inferenceClient: group.client,
11054
11377
  generator,
11378
+ // The extraction seam's cache, over this worker's content transport
11379
+ // (PERSIST-ANCHORS P2d). Built here because the client keeps its
11380
+ // content transport private — this is where it is in hand.
11381
+ anchoredTextStore: anchoredTextStoreOverTransport(content, logger2),
11055
11382
  logger: logger2
11056
11383
  });
11057
11384
  logger2.info("Agent ready", {