@semiont/jobs 0.5.28 → 0.5.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,14 @@
1
- import { createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, busRequest, resourceId, getPrimaryMediaType, isGenerationJobParams, assembleAnnotation, findClaimSpan, textExtractionOf, reconcileSelector, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, getLocaleEnglishName, estimateTokens, chunkText, isObject, isString, deriveViews } from '@semiont/core';
2
- import { anchoredTextStoreOverTransport, deriveStorageUri, extractPdfTextLayer, EXTRACTORS, calculateChecksum, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
3
- import { withSpan, SpanKind, recordJobOutcome } from '@semiont/observability';
1
+ import { replyChannelsFor, createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, isTransientFetchError, isObject, isString, isNumber, busRequest, isArray, resourceId, getPrimaryMediaType, isGenerationJobParams, capabilitiesOf, assembleAnnotation, findClaimSpan, textExtractionOf, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, deriveViews, estimateTokens, chunkText, reconcileSelector, getLocaleEnglishName } from '@semiont/core';
2
+ import { archivistContentReads, extractPdfTextLayer, EXTRACTORS, calculateChecksum, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
3
+ import { withSpan, SpanKind, recordJobOutcome, recordAnchorOutcome, recordDetectionCall } from '@semiont/observability';
4
+ import { createInferenceClient, StructuredReadError } from '@semiont/inference';
4
5
  import { execFileSync } from 'child_process';
5
6
  import { existsSync, readFileSync, mkdtempSync, writeFileSync, rmSync } from 'fs';
6
7
  import { homedir, hostname, tmpdir } from 'os';
7
8
  import { join } from 'path';
8
- import { generateAnnotationId } from '@semiont/event-sourcing';
9
- import { InMemorySessionStorage, setStoredSession, kbBackendUrl, SemiontClient, SemiontSession } from '@semiont/sdk';
9
+ import { annotationIdFor } from '@semiont/event-sourcing';
10
+ import { InMemorySessionStorage, setStoredSession, kbGatewayUrl, SemiontClient, SemiontSession } from '@semiont/sdk';
10
11
  import { HttpTransport, HttpContentTransport } from '@semiont/http-transport';
11
- import { createInferenceClient } from '@semiont/inference';
12
12
  import { createServer } from 'http';
13
13
  import { createProcessLogger } from '@semiont/observability/process-logger';
14
14
 
@@ -3766,9 +3766,9 @@ var require_mapOneOrManyArgs = __commonJS({
3766
3766
  Object.defineProperty(exports, "__esModule", { value: true });
3767
3767
  exports.mapOneOrManyArgs = void 0;
3768
3768
  var map_1 = require_map();
3769
- var isArray = Array.isArray;
3769
+ var isArray2 = Array.isArray;
3770
3770
  function callOrApply(fn, args) {
3771
- return isArray(args) ? fn.apply(void 0, __spreadArray([], __read(args))) : fn(args);
3771
+ return isArray2(args) ? fn.apply(void 0, __spreadArray([], __read(args))) : fn(args);
3772
3772
  }
3773
3773
  function mapOneOrManyArgs(fn) {
3774
3774
  return map_1.map(function(args) {
@@ -3913,14 +3913,14 @@ var require_argsArgArrayOrObject = __commonJS({
3913
3913
  "../../node_modules/rxjs/dist/cjs/internal/util/argsArgArrayOrObject.js"(exports) {
3914
3914
  Object.defineProperty(exports, "__esModule", { value: true });
3915
3915
  exports.argsArgArrayOrObject = void 0;
3916
- var isArray = Array.isArray;
3916
+ var isArray2 = Array.isArray;
3917
3917
  var getPrototypeOf = Object.getPrototypeOf;
3918
3918
  var objectProto = Object.prototype;
3919
3919
  var getKeys = Object.keys;
3920
3920
  function argsArgArrayOrObject(args) {
3921
3921
  if (args.length === 1) {
3922
3922
  var first_1 = args[0];
3923
- if (isArray(first_1)) {
3923
+ if (isArray2(first_1)) {
3924
3924
  return { args: first_1, keys: null };
3925
3925
  }
3926
3926
  if (isPOJO(first_1)) {
@@ -4668,9 +4668,9 @@ var require_argsOrArgArray = __commonJS({
4668
4668
  "../../node_modules/rxjs/dist/cjs/internal/util/argsOrArgArray.js"(exports) {
4669
4669
  Object.defineProperty(exports, "__esModule", { value: true });
4670
4670
  exports.argsOrArgArray = void 0;
4671
- var isArray = Array.isArray;
4671
+ var isArray2 = Array.isArray;
4672
4672
  function argsOrArgArray(args) {
4673
- return args.length === 1 && isArray(args[0]) ? args[0] : args;
4673
+ return args.length === 1 && isArray2(args[0]) ? args[0] : args;
4674
4674
  }
4675
4675
  exports.argsOrArgArray = argsOrArgArray;
4676
4676
  }
@@ -9208,6 +9208,8 @@ var require_cjs = __commonJS({
9208
9208
 
9209
9209
  // src/job-claim-adapter.ts
9210
9210
  var import_rxjs = __toESM(require_cjs());
9211
+
9212
+ // src/worker-bus-primitive.ts
9211
9213
  function workerBusAsPrimitive(bus) {
9212
9214
  return {
9213
9215
  emit(channel, payload) {
@@ -9219,9 +9221,15 @@ function workerBusAsPrimitive(bus) {
9219
9221
  // Pass the real connection state through: a worker's first `job:claim`
9220
9222
  // fires right after connect — exactly the attach-window emit the gate
9221
9223
  // exists for (.plans/BUS-ATTACH-GATE.md).
9222
- state$: bus.state$
9224
+ state$: bus.state$,
9225
+ // Pass the subscription probe through so a request against a transport
9226
+ // whose narrowed channel set omits the replies fails fast
9227
+ // (`bus.unsubscribed`) instead of timing out.
9228
+ ...bus.isSubscribed ? { isSubscribed: (channel) => bus.isSubscribed(channel) } : {}
9223
9229
  };
9224
9230
  }
9231
+
9232
+ // src/job-claim-adapter.ts
9225
9233
  function createJobClaimAdapter(options) {
9226
9234
  const { bus, jobTypes } = options;
9227
9235
  const requestBus = workerBusAsPrimitive(bus);
@@ -9240,12 +9248,18 @@ function createJobClaimAdapter(options) {
9240
9248
  const claimJob = async (assignment) => {
9241
9249
  try {
9242
9250
  const job = await busRequest(requestBus, "job:claim", { jobId: assignment.jobId }, 1e4);
9251
+ const completedUnits = isArray(job.metadata?.completedUnits) ? job.metadata.completedUnits.filter(isString) : [];
9243
9252
  return {
9244
9253
  jobId: assignment.jobId,
9245
9254
  type: assignment.type,
9246
9255
  resourceId: assignment.resourceId,
9247
9256
  userId: job.metadata?.userId ?? "",
9248
- params: job.params ?? {}
9257
+ params: job.params ?? {},
9258
+ completedUnits,
9259
+ // Absent or malformed metadata reads as "no budget left" — a worker
9260
+ // that cannot see the budget must not claim a retry is coming.
9261
+ retryCount: isNumber(job.metadata?.retryCount) ? job.metadata.retryCount : 0,
9262
+ maxRetries: isNumber(job.metadata?.maxRetries) ? job.metadata.maxRetries : 0
9249
9263
  };
9250
9264
  } catch {
9251
9265
  return null;
@@ -9330,6 +9344,11 @@ function createJobClaimAdapter(options) {
9330
9344
  };
9331
9345
  }
9332
9346
 
9347
+ // src/will-retry.ts
9348
+ function willRetryAfter(metadata, failureClass) {
9349
+ return failureClass !== "deterministic" && metadata.retryCount < metadata.maxRetries;
9350
+ }
9351
+
9333
9352
  // src/types.ts
9334
9353
  var JOB_TYPES = /* @__PURE__ */ new Set([
9335
9354
  "reference-annotation",
@@ -9349,6 +9368,9 @@ function asJobParams(params) {
9349
9368
  return params;
9350
9369
  }
9351
9370
  var INFERENCE_TIMEOUT_MS = 10 * 6e4;
9371
+ var InferenceTimeoutError = class extends Error {
9372
+ name = "InferenceTimeoutError";
9373
+ };
9352
9374
  var INFERENCE_HEARTBEAT_MS = 15e3;
9353
9375
  function spanned(client, kind, maxTokens, work) {
9354
9376
  return withSpan(`inference:${kind}`, work, {
@@ -9359,12 +9381,20 @@ function spanned(client, kind, maxTokens, work) {
9359
9381
  }
9360
9382
  });
9361
9383
  }
9362
- async function withTimeout(work, label, onHeartbeat) {
9384
+ async function withTimeout(work, meta, onHeartbeat, logger2) {
9385
+ const controller = new AbortController();
9363
9386
  let timer;
9364
9387
  const timedOut = new Promise((_, reject) => {
9365
9388
  timer = setTimeout(() => {
9366
- reject(new Error(
9367
- `Inference call timed out after ${INFERENCE_TIMEOUT_MS / 6e4} minutes (${label}) \u2014 failing the job to keep the claim loop live`
9389
+ logger2?.warn("Aborting in-flight inference call at the timeout bound", {
9390
+ provider: meta.provider,
9391
+ model: meta.model,
9392
+ label: meta.label,
9393
+ boundMs: INFERENCE_TIMEOUT_MS
9394
+ });
9395
+ controller.abort();
9396
+ reject(new InferenceTimeoutError(
9397
+ `Inference call timed out after ${INFERENCE_TIMEOUT_MS / 6e4} minutes (${meta.label}) \u2014 failing the job to keep the claim loop live`
9368
9398
  ));
9369
9399
  }, INFERENCE_TIMEOUT_MS);
9370
9400
  timer.unref?.();
@@ -9379,10 +9409,11 @@ async function withTimeout(work, label, onHeartbeat) {
9379
9409
  }, INFERENCE_HEARTBEAT_MS);
9380
9410
  heartbeat.unref?.();
9381
9411
  }
9412
+ const pending = work(controller.signal);
9382
9413
  try {
9383
- return await Promise.race([work, timedOut]);
9414
+ return await Promise.race([pending, timedOut]);
9384
9415
  } catch (err) {
9385
- work.catch(() => {
9416
+ pending.catch(() => {
9386
9417
  });
9387
9418
  throw err;
9388
9419
  } finally {
@@ -9390,28 +9421,70 @@ async function withTimeout(work, label, onHeartbeat) {
9390
9421
  if (heartbeat) clearInterval(heartbeat);
9391
9422
  }
9392
9423
  }
9393
- function boundedGenerateWithMetadata(client, prompt, maxTokens, temperature, onHeartbeat) {
9424
+ function boundedGenerateWithMetadata(client, prompt, maxTokens, temperature, onHeartbeat, logger2) {
9394
9425
  return spanned(client, "text", maxTokens, () => withTimeout(
9395
- client.generateTextWithMetadata(prompt, maxTokens, temperature),
9396
- `${client.type}:${client.modelId}`,
9397
- onHeartbeat
9426
+ (signal) => client.generateTextWithMetadata(prompt, maxTokens, temperature, signal),
9427
+ { provider: client.type, model: client.modelId, label: `${client.type}:${client.modelId}` },
9428
+ onHeartbeat,
9429
+ logger2
9398
9430
  ));
9399
9431
  }
9400
- function boundedGenerateStructured(client, prompt, maxTokens, temperature, elementSchema, onHeartbeat) {
9432
+ function boundedGenerateStructured(client, prompt, maxTokens, temperature, elementSchema, onHeartbeat, logger2) {
9401
9433
  return spanned(client, "structured", maxTokens, () => withTimeout(
9402
- client.generateStructured(prompt, maxTokens, temperature, elementSchema),
9403
- `${client.type}:${client.modelId}`,
9404
- onHeartbeat
9434
+ (signal) => client.generateStructured(prompt, maxTokens, temperature, elementSchema, signal),
9435
+ { provider: client.type, model: client.modelId, label: `${client.type}:${client.modelId}` },
9436
+ onHeartbeat,
9437
+ logger2
9405
9438
  ));
9406
9439
  }
9440
+ var DeterministicJobError = class extends Error {
9441
+ // Typed string, not the literal: subclasses (YieldCollapseError) carry
9442
+ // their own name — classification is instanceof, never name-matching.
9443
+ name = "DeterministicJobError";
9444
+ };
9445
+ function classifyFailure(error) {
9446
+ if (error instanceof DeterministicJobError) return "deterministic";
9447
+ if (error instanceof InferenceTimeoutError) return "transient";
9448
+ if (error instanceof StructuredReadError && error.stopReason === "max_tokens") return "deterministic";
9449
+ if (!isObject(error)) return void 0;
9450
+ const name = isString(error.name) ? error.name : "";
9451
+ if (name === "DeterministicJobError") return "deterministic";
9452
+ if (name === "APIUserAbortError" || name === "AbortError") return "transient";
9453
+ const status = isNumber(error.status) ? error.status : void 0;
9454
+ if (status !== void 0) {
9455
+ if (status === 408 || status === 429 || status >= 500) return "transient";
9456
+ if (status >= 400) return "deterministic";
9457
+ }
9458
+ return void 0;
9459
+ }
9407
9460
 
9408
9461
  // src/workers/detection/detection-chunking.ts
9462
+ function assertNotTruncated(response, label, chunk, totalChunks, outputBudget) {
9463
+ if (response.stopReason === "max_tokens") {
9464
+ throw new DeterministicJobError(`${label} response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
9465
+ }
9466
+ }
9409
9467
  var SELECTOR_CONTEXT_CHARS = 64;
9410
9468
  var OVERLAP_CHARS = SELECTOR_CONTEXT_CHARS + // prefix
9411
9469
  SELECTOR_CONTEXT_CHARS + // suffix
9412
9470
  2 * SELECTOR_CONTEXT_CHARS;
9413
9471
  var OVERLAP_TOKENS = Math.ceil(OVERLAP_CHARS / 4);
9414
- function deriveDetectionBudget(limits, scaffoldTokens) {
9472
+ var DETECTION_TEMPERATURE = 0;
9473
+ var ASSUMED_OUTPUT_TOKENS_PER_HOUR = 108e3;
9474
+ var YIELD_COLLAPSE_BAND = 2;
9475
+ var YieldCollapseError = class extends DeterministicJobError {
9476
+ /** What the flagged extraction DID find — every span write-time-verified,
9477
+ * so discarding it at the floor would add loss on top of the under-report.
9478
+ * Carried on the error because the flag site cannot know whether descent
9479
+ * remains possible; the floor is the subdivider's knowledge. */
9480
+ constructor(message, salvage = []) {
9481
+ super(message);
9482
+ this.salvage = salvage;
9483
+ }
9484
+ salvage;
9485
+ name = "YieldCollapseError";
9486
+ };
9487
+ function deriveDetectionBudget(limits, scaffoldTokens, typesPerCall) {
9415
9488
  const { contextTokens, maxOutputTokens } = limits;
9416
9489
  const available = contextTokens - scaffoldTokens;
9417
9490
  let inputBudget;
@@ -9427,6 +9500,15 @@ function deriveDetectionBudget(limits, scaffoldTokens) {
9427
9500
  outputBudget = available - inputBudget;
9428
9501
  }
9429
9502
  }
9503
+ const outputTokensPerHour = limits.outputTokensPerHour ?? ASSUMED_OUTPUT_TOKENS_PER_HOUR;
9504
+ {
9505
+ const durationSafeOutput = Math.floor(outputTokensPerHour * (INFERENCE_TIMEOUT_MS / 2) / 36e5);
9506
+ if (outputBudget > durationSafeOutput) {
9507
+ inputBudget = Math.floor(inputBudget * (durationSafeOutput / outputBudget));
9508
+ outputBudget = durationSafeOutput;
9509
+ }
9510
+ }
9511
+ inputBudget = Math.min(inputBudget, Math.floor(outputBudget / (2 * typesPerCall)));
9430
9512
  if (inputBudget <= OVERLAP_TOKENS) {
9431
9513
  throw new Error(
9432
9514
  `Inference window too small for detection: context ${contextTokens} tokens minus scaffold ${scaffoldTokens} leaves an input budget of ${inputBudget} (need > ${OVERLAP_TOKENS}). Use a model with a larger context window or reduce the prompt scaffold.`
@@ -9437,6 +9519,91 @@ function deriveDetectionBudget(limits, scaffoldTokens) {
9437
9519
  outputBudget
9438
9520
  };
9439
9521
  }
9522
+ var MAX_SUBDIVISION_DEPTH = 2;
9523
+ function unknownUnreadable(error) {
9524
+ return error instanceof StructuredReadError && error.stopReason === "unknown";
9525
+ }
9526
+ function subdividable(error) {
9527
+ return error instanceof InferenceTimeoutError || truncation(error) || unknownUnreadable(error);
9528
+ }
9529
+ function truncation(error) {
9530
+ return error instanceof DeterministicJobError || error instanceof StructuredReadError && error.stopReason === "max_tokens";
9531
+ }
9532
+ function outcomeOf(error) {
9533
+ if (error instanceof YieldCollapseError) return "collapsed";
9534
+ if (truncation(error)) return "truncated";
9535
+ if (error instanceof InferenceTimeoutError) return "timeout";
9536
+ return "error";
9537
+ }
9538
+ async function callChunkSubdividing(label, chunk, chunking, call, logger2) {
9539
+ async function recorded(piece, depth, reroll) {
9540
+ const start = performance.now();
9541
+ try {
9542
+ const result = await call(piece);
9543
+ recordDetectionCall({
9544
+ label,
9545
+ pieceChars: piece.length,
9546
+ durationMs: performance.now() - start,
9547
+ items: result.items.length,
9548
+ depth,
9549
+ reroll,
9550
+ outcome: "success",
9551
+ ...result.usage ? { inputTokens: result.usage.inputTokens, outputTokens: result.usage.outputTokens } : {}
9552
+ });
9553
+ return result;
9554
+ } catch (error) {
9555
+ recordDetectionCall({
9556
+ label,
9557
+ pieceChars: piece.length,
9558
+ durationMs: performance.now() - start,
9559
+ items: 0,
9560
+ depth,
9561
+ reroll,
9562
+ outcome: outcomeOf(error)
9563
+ });
9564
+ throw error;
9565
+ }
9566
+ }
9567
+ async function attempt(piece, chunkSize, depth) {
9568
+ try {
9569
+ return (await recorded(piece, depth, false)).items;
9570
+ } catch (error) {
9571
+ if (!subdividable(error)) throw error;
9572
+ const half = Math.floor(chunkSize / 2);
9573
+ const pieces = chunkText(piece, { chunkSize: half, overlap: chunking.overlap });
9574
+ const shrinks = pieces.length > 1 || pieces[0] !== piece;
9575
+ const canDescend = shrinks && (truncation(error) || unknownUnreadable(error) ? half > 2 * OVERLAP_TOKENS : depth < MAX_SUBDIVISION_DEPTH);
9576
+ if (!canDescend) {
9577
+ if (error instanceof YieldCollapseError) {
9578
+ logger2?.warn("Floor-size piece still flagged as collapsed \u2014 accepting its under-reported salvage and continuing", {
9579
+ pieceChars: piece.length,
9580
+ salvaged: error.salvage.length,
9581
+ error: error.message
9582
+ });
9583
+ return error.salvage;
9584
+ }
9585
+ if (!truncation(error)) throw error;
9586
+ logger2?.warn("Floor-size piece truncated \u2014 re-rolling once before giving up", {
9587
+ pieceChars: piece.length,
9588
+ error: error instanceof Error ? error.message : String(error)
9589
+ });
9590
+ return (await recorded(piece, depth, true)).items;
9591
+ }
9592
+ logger2?.warn("Chunk call failed at a size-shaped bound \u2014 subdividing and retrying smaller", {
9593
+ depth: depth + 1,
9594
+ pieceChars: piece.length,
9595
+ nextChunkSizeTokens: half,
9596
+ error: error instanceof Error ? error.message : String(error)
9597
+ });
9598
+ const collected = [];
9599
+ for (const p of pieces) {
9600
+ collected.push(...await attempt(p, half, depth + 1));
9601
+ }
9602
+ return collected;
9603
+ }
9604
+ }
9605
+ return attempt(chunk, chunking.chunkSize, 0);
9606
+ }
9440
9607
  function languageName(tag) {
9441
9608
  return getLocaleEnglishName(tag) || tag;
9442
9609
  }
@@ -9734,6 +9901,16 @@ Example format:
9734
9901
  return prompt;
9735
9902
  }
9736
9903
  };
9904
+ var DEGRADED = /* @__PURE__ */ new Set(["first-of-many", "fuzzy-match"]);
9905
+ function noteAnchor(label, exact, method, logger2) {
9906
+ recordAnchorOutcome(label, method);
9907
+ if (!DEGRADED.has(method)) return;
9908
+ const detail = { text: exact, anchorMethod: method };
9909
+ if (logger2) logger2.warn("Annotation anchored via degraded method", { label, ...detail });
9910
+ else console.warn(`[${label}] anchored via ${method}: "${exact}"`);
9911
+ }
9912
+
9913
+ // src/workers/detection/motivation-parsers.ts
9737
9914
  var COMMENT_ELEMENT_SCHEMA = {
9738
9915
  type: "object",
9739
9916
  properties: {
@@ -9922,35 +10099,31 @@ var MotivationParsers = class {
9922
10099
  }
9923
10100
  };
9924
10101
  function logAnchorMethod(motivation, exact, anchorMethod) {
9925
- if (anchorMethod === "first-of-many" || anchorMethod === "fuzzy-match") {
9926
- console.warn(`[MotivationParsers] ${motivation} anchored via ${anchorMethod}: "${exact}"`);
9927
- }
10102
+ noteAnchor(motivation, exact, anchorMethod);
9928
10103
  }
9929
10104
 
9930
10105
  // src/workers/annotation-detection.ts
9931
- function assertNotTruncated(response, motivation, chunk, totalChunks, outputBudget) {
9932
- if (response.stopReason === "max_tokens") {
9933
- throw new Error(`${motivation} detection response truncated (max_tokens) on chunk ${chunk}/${totalChunks} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than under-reporting annotations.`);
9934
- }
9935
- }
9936
- async function detectInChunks(client, content, buildPrompt, temperature, motivation, elementSchema, parse, onActivity) {
10106
+ async function detectInChunks(client, content, buildPrompt, motivation, elementSchema, parse, onActivity) {
9937
10107
  const limits = await client.limits();
9938
10108
  const scaffoldTokens = estimateTokens(buildPrompt(""));
9939
- const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
10109
+ const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens, 1);
9940
10110
  const chunks = chunkText(content, chunking);
9941
10111
  const collected = [];
9942
10112
  for (let i = 0; i < chunks.length; i++) {
9943
- const response = await boundedGenerateStructured(
9944
- client,
9945
- buildPrompt(chunks[i]),
9946
- outputBudget,
9947
- temperature,
9948
- elementSchema,
9949
- // Still alive, same position (a long single call is otherwise silent).
9950
- () => onActivity?.(i, chunks.length)
9951
- );
9952
- assertNotTruncated(response, motivation, i + 1, chunks.length, outputBudget);
9953
- collected.push(...parse(response.items));
10113
+ const items = await callChunkSubdividing(motivation, chunks[i], chunking, async (piece) => {
10114
+ const response = await boundedGenerateStructured(
10115
+ client,
10116
+ buildPrompt(piece),
10117
+ outputBudget,
10118
+ DETECTION_TEMPERATURE,
10119
+ elementSchema,
10120
+ // Still alive, same position (a long single call is otherwise silent).
10121
+ () => onActivity?.(i, chunks.length)
10122
+ );
10123
+ assertNotTruncated(response, `${motivation} detection`, i + 1, chunks.length, outputBudget);
10124
+ return { items: response.items, ...response.usage ? { usage: response.usage } : {} };
10125
+ });
10126
+ collected.push(...parse(items));
9954
10127
  if (i < chunks.length - 1) {
9955
10128
  onActivity?.(i + 1, chunks.length);
9956
10129
  }
@@ -9971,7 +10144,6 @@ var AnnotationDetection = class {
9971
10144
  client,
9972
10145
  content,
9973
10146
  (chunk) => MotivationPrompts.buildCommentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
9974
- 0.4,
9975
10147
  "comment",
9976
10148
  COMMENT_ELEMENT_SCHEMA,
9977
10149
  (items) => MotivationParsers.parseComments(items, content),
@@ -9990,7 +10162,6 @@ var AnnotationDetection = class {
9990
10162
  client,
9991
10163
  content,
9992
10164
  (chunk) => MotivationPrompts.buildHighlightPrompt(chunk, instructions, density, sourceLanguage),
9993
- 0.3,
9994
10165
  "highlight",
9995
10166
  HIGHLIGHT_ELEMENT_SCHEMA,
9996
10167
  (items) => MotivationParsers.parseHighlights(items, content),
@@ -10009,7 +10180,6 @@ var AnnotationDetection = class {
10009
10180
  client,
10010
10181
  content,
10011
10182
  (chunk) => MotivationPrompts.buildAssessmentPrompt(chunk, instructions, tone, density, language, sourceLanguage),
10012
- 0.3,
10013
10183
  "assessment",
10014
10184
  ASSESSMENT_ELEMENT_SCHEMA,
10015
10185
  (items) => MotivationParsers.parseAssessments(items, content),
@@ -10046,7 +10216,6 @@ var AnnotationDetection = class {
10046
10216
  categoryInfo.examples,
10047
10217
  sourceLanguage
10048
10218
  ),
10049
- 0.2,
10050
10219
  "tag",
10051
10220
  TAG_ELEMENT_SCHEMA,
10052
10221
  (items) => MotivationParsers.parseTags(items),
@@ -10066,6 +10235,40 @@ var ENTITY_ELEMENT_SCHEMA = {
10066
10235
  required: ["exact", "entityType"],
10067
10236
  additionalProperties: false
10068
10237
  };
10238
+ var COUNT_MAX_TOKENS = 16;
10239
+ function parseCount(text) {
10240
+ const m = text.trim().match(/\d+/);
10241
+ return m ? Number(m[0]) : void 0;
10242
+ }
10243
+ async function assertYieldNotCollapsed(client, piece, items, entityTypesDescription, logger2) {
10244
+ const prompt = `Count every mention of: ${entityTypesDescription} in the following text. Repeated mentions of the same entity count separately. Respond with only the number.
10245
+
10246
+ Text:
10247
+ """
10248
+ ${piece}
10249
+ """`;
10250
+ let counted;
10251
+ try {
10252
+ const response = await boundedGenerateWithMetadata(client, prompt, COUNT_MAX_TOKENS, DETECTION_TEMPERATURE, void 0, logger2);
10253
+ counted = parseCount(response.text);
10254
+ } catch (err) {
10255
+ logger2.warn("Count-verifier call failed \u2014 yield check skipped for this chunk", {
10256
+ pieceChars: piece.length,
10257
+ error: err instanceof Error ? err.message : String(err)
10258
+ });
10259
+ return;
10260
+ }
10261
+ if (counted === void 0) {
10262
+ logger2.warn("Count-verifier answer carried no number \u2014 yield check skipped for this chunk", { pieceChars: piece.length });
10263
+ return;
10264
+ }
10265
+ if (items.length * YIELD_COLLAPSE_BAND < counted) {
10266
+ throw new YieldCollapseError(
10267
+ `Extraction found ${items.length} entities where a count call reports ~${counted} mentions (band \xD7${YIELD_COLLAPSE_BAND}) on a ${piece.length}-char chunk \u2014 silent yield collapse (F7): deterministic \u2014 a same-size retry returns the identical under-report.`,
10268
+ [...items]
10269
+ );
10270
+ }
10271
+ }
10069
10272
  async function extractEntities(exact, entityTypes, client, includeDescriptiveReferences, logger2, sourceLanguage, onActivity) {
10070
10273
  const entityTypesDescription = entityTypes.map((et) => {
10071
10274
  if (typeof et === "string") {
@@ -10116,8 +10319,9 @@ If no entities are found, respond with an empty array [].
10116
10319
  Example output:
10117
10320
  [{"exact":"Alice","entityType":"Person","prefix":"","suffix":" went to"},{"exact":"Paris","entityType":"Location","prefix":"went to ","suffix":" yesterday"}]`;
10118
10321
  const limits = await client.limits();
10322
+ const verifyYield = client.verifyDetectionYield;
10119
10323
  const scaffoldTokens = estimateTokens(buildPrompt(""));
10120
- const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens);
10324
+ const { chunking, outputBudget } = deriveDetectionBudget(limits, scaffoldTokens, entityTypes.length);
10121
10325
  const chunks = chunkText(exact, chunking);
10122
10326
  logger2.debug("Sending entity extraction request", {
10123
10327
  entityTypes: entityTypesDescription,
@@ -10127,28 +10331,31 @@ Example output:
10127
10331
  });
10128
10332
  const collected = [];
10129
10333
  for (let i = 0; i < chunks.length; i++) {
10130
- const response = await boundedGenerateStructured(
10131
- client,
10132
- buildPrompt(chunks[i]),
10133
- outputBudget,
10134
- 0.3,
10135
- // Lower temperature for more consistent extraction
10136
- ENTITY_ELEMENT_SCHEMA,
10137
- // Still alive, same position: a long single call would otherwise emit
10138
- // nothing at all between start and finish.
10139
- () => onActivity?.(i, chunks.length)
10140
- );
10141
- logger2.debug("Got entity extraction response", {
10142
- chunk: i + 1,
10143
- chunks: chunks.length,
10144
- items: response.items.length
10145
- });
10146
- if (response.stopReason === "max_tokens") {
10147
- const errorMsg = `Entity extraction response truncated (max_tokens) on chunk ${i + 1}/${chunks.length} despite the derived output budget of ${outputBudget} tokens \u2014 failing the job rather than dropping annotations.`;
10148
- logger2.error(errorMsg, { items: response.items.length });
10149
- throw new Error(errorMsg);
10150
- }
10151
- for (const e of response.items) {
10334
+ const items = await callChunkSubdividing("reference", chunks[i], chunking, async (piece) => {
10335
+ const response = await boundedGenerateStructured(
10336
+ client,
10337
+ buildPrompt(piece),
10338
+ outputBudget,
10339
+ DETECTION_TEMPERATURE,
10340
+ ENTITY_ELEMENT_SCHEMA,
10341
+ // Still alive, same position: a long single call would otherwise emit
10342
+ // nothing at all between start and finish.
10343
+ () => onActivity?.(i, chunks.length),
10344
+ logger2
10345
+ );
10346
+ logger2.debug("Got entity extraction response", {
10347
+ chunk: i + 1,
10348
+ chunks: chunks.length,
10349
+ pieceChars: piece.length,
10350
+ items: response.items.length
10351
+ });
10352
+ assertNotTruncated(response, "Entity extraction", i + 1, chunks.length, outputBudget);
10353
+ if (verifyYield) {
10354
+ await assertYieldNotCollapsed(client, piece, response.items, entityTypesDescription, logger2);
10355
+ }
10356
+ return { items: response.items, ...response.usage ? { usage: response.usage } : {} };
10357
+ }, logger2);
10358
+ for (const e of items) {
10152
10359
  if (isObject(e) && isString(e.exact) && isString(e.entityType)) {
10153
10360
  collected.push({
10154
10361
  exact: e.exact,
@@ -10379,7 +10586,7 @@ ${formatRequirements}`;
10379
10586
  temperature: finalTemperature,
10380
10587
  maxTokens: finalMaxTokens
10381
10588
  });
10382
- const response = await boundedGenerateWithMetadata(client, prompt, finalMaxTokens, finalTemperature);
10589
+ const response = await boundedGenerateWithMetadata(client, prompt, finalMaxTokens, finalTemperature, void 0, logger2);
10383
10590
  logger2.debug("Got response from inference", { responseLength: response.text.length, stopReason: response.stopReason });
10384
10591
  const result = parseResponse(response.text);
10385
10592
  logger2.debug("Parsed response", {
@@ -10474,6 +10681,27 @@ function resolveCitationTokens(content, validResourceIds, logger2) {
10474
10681
  clean += content.slice(last);
10475
10682
  return { content: clean, citations };
10476
10683
  }
10684
+
10685
+ // src/workers/detection/bounded-concurrency.ts
10686
+ async function runBounded(items, limit, worker) {
10687
+ const results = new Array(items.length);
10688
+ let next = 0;
10689
+ async function pump() {
10690
+ while (true) {
10691
+ const i = next++;
10692
+ if (i >= items.length) return;
10693
+ results[i] = await worker(items[i], i);
10694
+ }
10695
+ }
10696
+ const poolSize = Math.max(1, Math.min(limit, items.length));
10697
+ await Promise.all(Array.from({ length: poolSize }, () => pump()));
10698
+ return results;
10699
+ }
10700
+
10701
+ // src/processors.ts
10702
+ function spanAnchor(match) {
10703
+ return `${match.start}:${match.end}:${match.exact}`;
10704
+ }
10477
10705
  function toMatch(r) {
10478
10706
  return {
10479
10707
  exact: r.exact,
@@ -10535,7 +10763,7 @@ function buildTextAnnotation(content, resourceId, userId, generator, motivation,
10535
10763
  return {
10536
10764
  "@context": "http://www.w3.org/ns/anno.jsonld",
10537
10765
  "type": "Annotation",
10538
- "id": generateAnnotationId(),
10766
+ "id": annotationIdFor({ resourceId, motivation, anchor: spanAnchor(match), body }),
10539
10767
  motivation,
10540
10768
  creator,
10541
10769
  generator,
@@ -10574,7 +10802,7 @@ function buildPdfAnnotation(anchored, resourceId, userId, generator, motivation,
10574
10802
  return {
10575
10803
  "@context": "http://www.w3.org/ns/anno.jsonld",
10576
10804
  "type": "Annotation",
10577
- "id": generateAnnotationId(),
10805
+ "id": annotationIdFor({ resourceId, motivation, anchor: spanAnchor(match), body }),
10578
10806
  motivation,
10579
10807
  creator,
10580
10808
  generator,
@@ -10703,31 +10931,33 @@ async function processAssessmentJob(content, inferenceClient, params, buildAnnot
10703
10931
  result: { kind: "assessment-annotation", assessmentsFound: assessments.length, assessmentsCreated: annotations.length }
10704
10932
  };
10705
10933
  }
10706
- async function processReferenceJob(content, inferenceClient, params, buildAnnotation, onProgress, logger2) {
10934
+ async function processReferenceJob(content, inferenceClient, params, buildAnnotation, onProgress, logger2, onUnitComplete, signal) {
10707
10935
  const entityTypeNames = params.entityTypes.map(String);
10708
10936
  const requestParams = [{ label: "entity-types", value: entityTypeNames.join(", ") }];
10709
10937
  const completedItems = [];
10710
10938
  let totalFound = 0;
10711
10939
  let totalEmitted = 0;
10712
10940
  let errors = 0;
10713
- const allAnnotations = [];
10714
10941
  onProgress(10, { code: "loading" }, { requestParams });
10715
10942
  const bodyLanguage = params.language ?? "en";
10716
- for (let i = 0; i < entityTypeNames.length; i++) {
10717
- const entityTypeName = entityTypeNames[i];
10718
- if (!entityTypeName) continue;
10719
- const pct = 20 + Math.round(i / entityTypeNames.length * 60);
10943
+ let completed = 0;
10944
+ const total = entityTypeNames.length;
10945
+ const emitTypeProgress = (entityTypeName) => {
10946
+ const pct = 20 + Math.round(completed / total * 60);
10720
10947
  onProgress(pct, { code: "detecting-entities", entityType: entityTypeName }, {
10721
- // One vocabulary for "what is in flight" (CLEAN-PROGRESS D2): the entity
10722
- // type is KB data, `kind` is the code the client localizes around it.
10723
10948
  current: { kind: "entity-type", value: entityTypeName },
10724
- processed: i,
10725
- total: entityTypeNames.length,
10949
+ processed: completed,
10950
+ total,
10726
10951
  entitiesFound: totalFound,
10727
10952
  entitiesEmitted: totalEmitted,
10728
10953
  completedItems: [...completedItems],
10729
10954
  requestParams
10730
10955
  });
10956
+ };
10957
+ await runBounded(entityTypeNames, inferenceClient.maxConcurrency, async (entityTypeName) => {
10958
+ if (!entityTypeName) return;
10959
+ if (signal?.aborted) return;
10960
+ emitTypeProgress(entityTypeName);
10731
10961
  const extractedEntities = await extractEntities(
10732
10962
  content,
10733
10963
  [entityTypeName],
@@ -10735,31 +10965,17 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
10735
10965
  params.includeDescriptiveReferences ?? false,
10736
10966
  logger2,
10737
10967
  params.sourceLanguage,
10738
- // Liveness: fires at chunk boundaries AND every ~15 s while a single
10739
- // inference call is in flight (DETECTION-HEARTBEAT). Progress feeds the
10740
- // stall watchdog, the janitor, AND the client's inter-emission timeout,
10741
- // so a long single-chunk call must not be silent. Percentage
10742
- // interpolates within this entity type's band of the 20–80 range; a
10743
- // heartbeat repeats the current position rather than inventing an
10744
- // advance.
10745
- (completed, total) => {
10746
- const interpolated = 20 + Math.round((i + completed / total) / entityTypeNames.length * 60);
10747
- onProgress(interpolated, { code: "detecting-entities", entityType: entityTypeName }, {
10748
- current: { kind: "entity-type", value: entityTypeName },
10749
- processed: i,
10750
- total: entityTypeNames.length,
10751
- entitiesFound: totalFound,
10752
- entitiesEmitted: totalEmitted,
10753
- completedItems: [...completedItems],
10754
- requestParams
10755
- });
10756
- }
10968
+ // Liveness heartbeat (DETECTION-HEARTBEAT): fires at chunk boundaries and
10969
+ // every ~15 s while a call is in flight, so a long single-chunk call is
10970
+ // not silent. It repeats the current position rather than inventing an
10971
+ // advance — the stall watchdog, janitor and client timeout need a signal,
10972
+ // not a monotone.
10973
+ () => emitTypeProgress(entityTypeName)
10757
10974
  );
10758
- totalFound += extractedEntities.length;
10759
- completedItems.push({ value: entityTypeName, foundCount: extractedEntities.length });
10760
10975
  const unresolvedBody = [
10761
10976
  { type: "TextualBody", value: entityTypeName, purpose: "tagging", format: "text/plain", language: bodyLanguage }
10762
10977
  ];
10978
+ const built = [];
10763
10979
  for (const entity of extractedEntities) {
10764
10980
  const reconciled = reconcileSelector(content, {
10765
10981
  exact: entity.exact,
@@ -10774,23 +10990,28 @@ async function processReferenceJob(content, inferenceClient, params, buildAnnota
10774
10990
  errors++;
10775
10991
  continue;
10776
10992
  }
10777
- if (reconciled.anchorMethod === "first-of-many" || reconciled.anchorMethod === "fuzzy-match") {
10778
- logger2.warn("Entity anchored via degraded method", {
10779
- text: entity.exact,
10780
- entityType: entity.entityType,
10781
- anchorMethod: reconciled.anchorMethod
10782
- });
10783
- }
10993
+ noteAnchor("reference", entity.exact, reconciled.anchorMethod, logger2);
10784
10994
  const ann = buildAnnotation("linking", toMatch(reconciled), unresolvedBody);
10785
- allAnnotations.push(ann);
10786
- totalEmitted++;
10995
+ built.push(ann);
10787
10996
  }
10788
- }
10789
- const annotations = dedupeAnnotations(allAnnotations);
10790
- onProgress(100, { code: "complete-created", count: annotations.length, kind: "reference" }, { requestParams });
10997
+ const unitAnnotations = dedupeAnnotations(built);
10998
+ await onUnitComplete(entityTypeName, unitAnnotations);
10999
+ totalEmitted += unitAnnotations.length;
11000
+ totalFound += extractedEntities.length;
11001
+ completedItems.push({
11002
+ value: entityTypeName,
11003
+ foundCount: extractedEntities.length,
11004
+ persistedCount: unitAnnotations.length
11005
+ });
11006
+ completed++;
11007
+ emitTypeProgress(entityTypeName);
11008
+ });
11009
+ onProgress(100, { code: "complete-created", count: totalEmitted, kind: "reference" }, {
11010
+ completedItems: [...completedItems],
11011
+ requestParams
11012
+ });
10791
11013
  return {
10792
- annotations,
10793
- result: { kind: "reference-annotation", totalFound, totalEmitted: annotations.length, errors }
11014
+ result: { kind: "reference-annotation", totalFound, totalEmitted, errors }
10794
11015
  };
10795
11016
  }
10796
11017
  async function processTagJob(content, inferenceClient, params, buildAnnotation, onProgress) {
@@ -10995,10 +11216,10 @@ async function processGenerationJob(inferenceClient, params, onProgress, logger2
10995
11216
  }
10996
11217
 
10997
11218
  // src/workers/detection/prepare-detection.ts
10998
- async function prepareDetection(mediaType, session, resourceId, userId, generator, store) {
11219
+ async function prepareDetection(mediaType, content, resourceId, userId, generator, store) {
10999
11220
  const extractor = EXTRACTORS[textExtractionOf(mediaType)];
11000
11221
  if (!extractor) return { declined: "no-extractor" };
11001
- const { data } = await session.client.browse.resourceRepresentation(resourceId);
11222
+ const { data } = await content.getBinary(resourceId);
11002
11223
  const bytes = Buffer.from(data);
11003
11224
  const extracted = await extractor.extract(bytes, mediaType, {
11004
11225
  key: calculateChecksum(bytes),
@@ -11031,6 +11252,16 @@ function referenceIdOf(job) {
11031
11252
  const ref = job.params.referenceId;
11032
11253
  return typeof ref === "string" ? ref : void 0;
11033
11254
  }
11255
+ var MARK_COMMIT_TIMEOUT_MS = 6e4;
11256
+ async function commitAnnotations(session, resourceId, annotations) {
11257
+ if (annotations.length === 0) return;
11258
+ await busRequest(
11259
+ workerBusAsPrimitive(session.client.transport.actor),
11260
+ "mark:commit",
11261
+ { resourceId, annotations },
11262
+ MARK_COMMIT_TIMEOUT_MS
11263
+ );
11264
+ }
11034
11265
  async function emitEvent(session, channel, payload) {
11035
11266
  await session.client.transport.emit(channel, payload);
11036
11267
  }
@@ -11041,12 +11272,29 @@ function startWorkerProcess(config) {
11041
11272
  bus: httpTransport.actor,
11042
11273
  jobTypes: config.jobTypes
11043
11274
  });
11275
+ const completedUnitsByJob = /* @__PURE__ */ new Map();
11276
+ let activeCancel = null;
11277
+ httpTransport.actor.addChannels?.(["job:cancel-requested"]);
11278
+ httpTransport.on("job:cancel-requested", (event) => {
11279
+ const targetId = event.jobId;
11280
+ if (targetId && activeCancel?.jobId === targetId) {
11281
+ logger2.info("Cancel requested for active job \u2014 stopping at next unit boundary", { jobId: targetId });
11282
+ activeCancel.controller.abort();
11283
+ }
11284
+ });
11044
11285
  adapter.activeJob$.subscribe((job) => {
11045
11286
  if (!job) return;
11046
11287
  logger2.info("Processing job", { jobId: job.jobId, type: job.type, resourceId: job.resourceId });
11047
- handleJob(adapter, config, job).catch((error) => {
11288
+ const controller = new AbortController();
11289
+ activeCancel = { jobId: job.jobId, controller };
11290
+ handleJob(adapter, config, job, completedUnitsByJob, controller.signal).then(() => {
11291
+ completedUnitsByJob.delete(job.jobId);
11292
+ }).catch((error) => {
11048
11293
  const message = error instanceof Error ? error.message : String(error);
11049
- logger2.error("Job failed", { jobId: job.jobId, error: message, stack: error instanceof Error ? error.stack : void 0 });
11294
+ const failureClass = classifyFailure(error);
11295
+ logger2.error("Job failed", { jobId: job.jobId, error: message, failureClass, stack: error instanceof Error ? error.stack : void 0 });
11296
+ const completedUnits = completedUnitsByJob.get(job.jobId);
11297
+ completedUnitsByJob.delete(job.jobId);
11050
11298
  const failAnnotationId = referenceIdOf(job);
11051
11299
  if (isJobType(job.type)) {
11052
11300
  emitEvent(session, "job:fail", {
@@ -11054,23 +11302,33 @@ function startWorkerProcess(config) {
11054
11302
  jobId: job.jobId,
11055
11303
  jobType: job.type,
11056
11304
  ...failAnnotationId ? { annotationId: failAnnotationId } : {},
11057
- error: message
11305
+ error: message,
11306
+ ...completedUnits && completedUnits.length > 0 ? { completedUnits } : {},
11307
+ ...failureClass !== void 0 ? { failureClass } : {},
11308
+ // Whether this failure is the END, answered by the same predicate
11309
+ // the queue applies at failJob (JOB-RESTART-SAFETY P5). Without
11310
+ // it a client cannot tell a recovering run from a dead one: it
11311
+ // sees job:fail either way and would end its stream on a job the
11312
+ // queue is about to re-run.
11313
+ willRetry: willRetryAfter(job, failureClass)
11058
11314
  }).catch(() => {
11059
11315
  });
11060
11316
  }
11061
11317
  adapter.failJob(job.jobId, message);
11318
+ }).finally(() => {
11319
+ if (activeCancel?.jobId === job.jobId) activeCancel = null;
11062
11320
  });
11063
11321
  });
11064
11322
  adapter.start();
11065
11323
  return adapter;
11066
11324
  }
11067
- async function handleJob(adapter, config, job) {
11325
+ async function handleJob(adapter, config, job, completedUnitsByJob = /* @__PURE__ */ new Map(), signal) {
11068
11326
  const start = performance.now();
11069
11327
  let outcome = "completed";
11070
11328
  try {
11071
11329
  return await withSpan(
11072
11330
  `job:${job.type}`,
11073
- () => handleJobInner(adapter, config, job),
11331
+ () => handleJobInner(adapter, config, job, completedUnitsByJob, signal),
11074
11332
  {
11075
11333
  kind: SpanKind.CONSUMER,
11076
11334
  attrs: {
@@ -11087,7 +11345,7 @@ async function handleJob(adapter, config, job) {
11087
11345
  recordJobOutcome(job.type, outcome, performance.now() - start);
11088
11346
  }
11089
11347
  }
11090
- async function handleJobInner(adapter, config, job) {
11348
+ async function handleJobInner(adapter, config, job, completedUnitsByJob, signal) {
11091
11349
  const { session, inferenceClient, generator } = config;
11092
11350
  const { userId, jobId } = job;
11093
11351
  if (!isJobType(job.type)) {
@@ -11114,12 +11372,12 @@ async function handleJobInner(adapter, config, job) {
11114
11372
  const mediaType = getPrimaryMediaType(descriptor);
11115
11373
  const source = await withSpan(
11116
11374
  "detection:prepare",
11117
- () => prepareDetection(mediaType ?? "", session, resourceId$1, userId, generator, config.anchoredTextStore),
11375
+ () => prepareDetection(mediaType ?? "", config.contentReads, resourceId$1, userId, generator, config.anchoredTextStore),
11118
11376
  { attrs: { "resource.id": resourceId$1, "media.type": mediaType ?? "unknown" } }
11119
11377
  );
11120
11378
  if ("declined" in source) {
11121
11379
  if (source.declined === "no-extractor") {
11122
- throw new Error(`Cannot run ${jobType} on resource ${resourceId$1}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
11380
+ throw new DeterministicJobError(`Cannot run ${jobType} on resource ${resourceId$1}: media type '${mediaType ?? "unknown"}' has no extractable text to analyze`);
11123
11381
  }
11124
11382
  await emitEvent(session, "job:complete", {
11125
11383
  ...lifecycleBase,
@@ -11156,9 +11414,7 @@ async function handleJobInner(adapter, config, job) {
11156
11414
  ready.buildAnnotation,
11157
11415
  onProgress
11158
11416
  );
11159
- for (const ann of annotations) {
11160
- await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
11161
- }
11417
+ await commitAnnotations(session, String(resourceId$1), annotations);
11162
11418
  await emitEvent(session, "job:complete", {
11163
11419
  ...lifecycleBase,
11164
11420
  result
@@ -11172,9 +11428,7 @@ async function handleJobInner(adapter, config, job) {
11172
11428
  ready.buildAnnotation,
11173
11429
  onProgress
11174
11430
  );
11175
- for (const ann of annotations) {
11176
- await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
11177
- }
11431
+ await commitAnnotations(session, String(resourceId$1), annotations);
11178
11432
  await emitEvent(session, "job:complete", {
11179
11433
  ...lifecycleBase,
11180
11434
  result
@@ -11188,25 +11442,45 @@ async function handleJobInner(adapter, config, job) {
11188
11442
  ready.buildAnnotation,
11189
11443
  onProgress
11190
11444
  );
11191
- for (const ann of annotations) {
11192
- await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
11193
- }
11445
+ await commitAnnotations(session, String(resourceId$1), annotations);
11194
11446
  await emitEvent(session, "job:complete", {
11195
11447
  ...lifecycleBase,
11196
11448
  result
11197
11449
  });
11198
11450
  adapter.completeJob();
11199
11451
  } else if (jobType === "reference-annotation") {
11200
- const { annotations, result } = await processReferenceJob(
11452
+ const params = asJobParams(job.params);
11453
+ const skip = new Set(job.completedUnits);
11454
+ const remaining = {
11455
+ ...params,
11456
+ entityTypes: params.entityTypes.filter((t) => !skip.has(String(t)))
11457
+ };
11458
+ const committed = [];
11459
+ completedUnitsByJob.set(job.jobId, committed);
11460
+ const { result } = await processReferenceJob(
11201
11461
  ready.text,
11202
11462
  inferenceClient,
11203
- asJobParams(job.params),
11463
+ remaining,
11204
11464
  ready.buildAnnotation,
11205
11465
  onProgress,
11206
- config.logger
11466
+ config.logger,
11467
+ async (unit, annotations) => {
11468
+ await commitAnnotations(session, String(resourceId$1), annotations);
11469
+ committed.push(unit);
11470
+ await emitEvent(session, "job:checkpoint", {
11471
+ jobId: job.jobId,
11472
+ completedUnits: [...committed]
11473
+ });
11474
+ },
11475
+ signal
11207
11476
  );
11208
- for (const ann of annotations) {
11209
- await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
11477
+ if (signal?.aborted) {
11478
+ await emitEvent(session, "job:cancel", {
11479
+ ...lifecycleBase,
11480
+ ...committed.length > 0 ? { completedUnits: [...committed] } : {}
11481
+ });
11482
+ adapter.completeJob();
11483
+ return;
11210
11484
  }
11211
11485
  await emitEvent(session, "job:complete", {
11212
11486
  ...lifecycleBase,
@@ -11221,9 +11495,7 @@ async function handleJobInner(adapter, config, job) {
11221
11495
  ready.buildAnnotation,
11222
11496
  onProgress
11223
11497
  );
11224
- for (const ann of annotations) {
11225
- await emitEvent(session, "mark:create", { annotation: ann, resourceId: resourceId$1 });
11226
- }
11498
+ await commitAnnotations(session, String(resourceId$1), annotations);
11227
11499
  await emitEvent(session, "job:complete", {
11228
11500
  ...lifecycleBase,
11229
11501
  result
@@ -11243,7 +11515,16 @@ async function handleJobInner(adapter, config, job) {
11243
11515
  );
11244
11516
  const genParams = job.params;
11245
11517
  const genReferenceId = referenceIdOf(job);
11246
- const storageUri = deriveStorageUri(genResult.title, genResult.format);
11518
+ const storageUri = job.params.storageUri;
11519
+ const expectedExtension = capabilitiesOf(genResult.format)?.extension;
11520
+ if (expectedExtension && !storageUri.toLowerCase().endsWith(expectedExtension)) {
11521
+ config.logger.warn("Storage URI extension does not match the generated format \u2014 writing it as requested", {
11522
+ jobId,
11523
+ storageUri,
11524
+ format: genResult.format,
11525
+ expectedExtension
11526
+ });
11527
+ }
11247
11528
  const { resourceId: newResourceId } = await session.client.yield.resource({
11248
11529
  name: genResult.title,
11249
11530
  file: Buffer.from(genResult.content),
@@ -11265,8 +11546,9 @@ async function handleJobInner(adapter, config, job) {
11265
11546
  },
11266
11547
  generator
11267
11548
  );
11268
- await emitEvent(session, "mark:create", { annotation: provenanceRef, resourceId: resourceId$1 });
11549
+ await commitAnnotations(session, String(resourceId$1), [provenanceRef]);
11269
11550
  }
11551
+ const citationRefs = [];
11270
11552
  if (genResult.format === "application/pdf" && genResult.citations.length > 0) {
11271
11553
  const layer = await extractPdfTextLayer(genResult.content);
11272
11554
  if (!layer) {
@@ -11296,7 +11578,7 @@ async function handleJobInner(adapter, config, job) {
11296
11578
  { exact: layer.text.slice(span.start, span.end), start: span.start, end: span.end },
11297
11579
  { type: "SpecificResource", source: citation.resourceId, purpose: "linking" }
11298
11580
  );
11299
- await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
11581
+ citationRefs.push(citationRef);
11300
11582
  }
11301
11583
  }
11302
11584
  } else {
@@ -11315,9 +11597,10 @@ async function handleJobInner(adapter, config, job) {
11315
11597
  },
11316
11598
  generator
11317
11599
  );
11318
- await emitEvent(session, "mark:create", { annotation: citationRef, resourceId: newResourceId });
11600
+ citationRefs.push(citationRef);
11319
11601
  }
11320
11602
  }
11603
+ await commitAnnotations(session, String(newResourceId), citationRefs);
11321
11604
  await emitEvent(session, "job:complete", {
11322
11605
  ...lifecycleBase,
11323
11606
  result: { kind: "generation", resourceId: newResourceId, resourceName: genResult.title, truncated: genResult.result.truncated }
@@ -11328,6 +11611,29 @@ async function handleJobInner(adapter, config, job) {
11328
11611
  }
11329
11612
  }
11330
11613
 
11614
+ // src/anchored-text-over-bus.ts
11615
+ function anchoredTextOverBus(client, logger2) {
11616
+ return {
11617
+ async read(key) {
11618
+ try {
11619
+ return await client.browse.anchoredTextByChecksum(key);
11620
+ } catch (error) {
11621
+ logger2?.debug("Anchored-text consult failed \u2014 treating as miss", {
11622
+ key,
11623
+ reason: error instanceof Error ? error.message : String(error)
11624
+ });
11625
+ return null;
11626
+ }
11627
+ },
11628
+ async write(key) {
11629
+ logger2?.debug("Anchored-text write skipped \u2014 the Smelter is the sole writer", { key });
11630
+ },
11631
+ async list() {
11632
+ return [];
11633
+ }
11634
+ };
11635
+ }
11636
+
11331
11637
  // src/worker-runtime.ts
11332
11638
  var import_rxjs2 = __toESM(require_cjs());
11333
11639
  function buildHealthPayload(workers) {
@@ -11367,7 +11673,18 @@ function startStallWatchdog(opts) {
11367
11673
  timer.unref?.();
11368
11674
  return { dispose: () => clearInterval(timer) };
11369
11675
  }
11370
- function parseBackendUrl(url) {
11676
+ var WORKER_AWAITED_OPERATIONS = [
11677
+ "job:claim",
11678
+ "browse:resource-requested",
11679
+ "browse:anchored-text-by-checksum-requested",
11680
+ // Durability acknowledgement for a unit's annotations (JOB-RESTART-SAFETY
11681
+ // P6). The worker AWAITS this one — a unit may not advance until its
11682
+ // annotations are in the event log — so its replies must be in the narrow
11683
+ // channel set or every commit fails fast with `bus.unsubscribed`.
11684
+ "mark:commit"
11685
+ ];
11686
+ var WORKER_CHANNELS = replyChannelsFor(WORKER_AWAITED_OPERATIONS);
11687
+ function parseGatewayUrl(url) {
11371
11688
  const parsed = new URL(url);
11372
11689
  const protocol = parsed.protocol.replace(":", "") === "https" ? "https" : "http";
11373
11690
  const host = parsed.hostname;
@@ -11375,13 +11692,13 @@ function parseBackendUrl(url) {
11375
11692
  return { protocol, host, port };
11376
11693
  }
11377
11694
  async function authenticateAgent(opts) {
11378
- const { backendBaseUrl: backendBaseUrl2, workerSecret: workerSecret2, provider, model, logger: logger2, retry = STARTUP_FETCH_RETRY } = opts;
11695
+ const { gatewayBaseUrl: gatewayBaseUrl2, workerSecret: workerSecret2, provider, model, logger: logger2, retry = STARTUP_FETCH_RETRY } = opts;
11379
11696
  if (!workerSecret2) {
11380
11697
  throw new Error("SEMIONT_WORKER_SECRET is required to authenticate worker agents");
11381
11698
  }
11382
11699
  return retryWithBackoff(
11383
11700
  async () => {
11384
- const response = await fetch(`${backendBaseUrl2}/api/tokens/agent`, {
11701
+ const response = await fetch(`${gatewayBaseUrl2}/api/tokens/agent`, {
11385
11702
  method: "POST",
11386
11703
  headers: { "Content-Type": "application/json" },
11387
11704
  body: JSON.stringify({ secret: workerSecret2, provider, model })
@@ -11394,7 +11711,7 @@ async function authenticateAgent(opts) {
11394
11711
  isTransientFetchError,
11395
11712
  retry,
11396
11713
  ({ attempt, attempts, delayMs, error }) => {
11397
- logger2?.warn("Backend unreachable, retrying agent authentication", {
11714
+ logger2?.warn("Gateway unreachable, retrying agent authentication", {
11398
11715
  agent: `${provider}:${model}`,
11399
11716
  attempt,
11400
11717
  attempts,
@@ -11405,11 +11722,11 @@ async function authenticateAgent(opts) {
11405
11722
  );
11406
11723
  }
11407
11724
  async function startAgentWorker(opts) {
11408
- const { group, backendBaseUrl: backendBaseUrl2, workerSecret: workerSecret2, logger: logger2 } = opts;
11725
+ const { group, gatewayBaseUrl: gatewayBaseUrl2, workerSecret: workerSecret2, contentReads: contentReads2, logger: logger2 } = opts;
11409
11726
  const { inference } = group;
11410
- const { protocol, host, port } = parseBackendUrl(backendBaseUrl2);
11727
+ const { protocol, host, port } = parseGatewayUrl(gatewayBaseUrl2);
11411
11728
  const { token: initialToken, did } = await authenticateAgent({
11412
- backendBaseUrl: backendBaseUrl2,
11729
+ gatewayBaseUrl: gatewayBaseUrl2,
11413
11730
  workerSecret: workerSecret2,
11414
11731
  provider: inference.type,
11415
11732
  model: inference.model,
@@ -11429,9 +11746,12 @@ async function startAgentWorker(opts) {
11429
11746
  const token$ = new import_rxjs2.BehaviorSubject(null);
11430
11747
  let session;
11431
11748
  const transport = new HttpTransport({
11432
- baseUrl: baseUrl(kbBackendUrl(endpoint)),
11749
+ baseUrl: baseUrl(kbGatewayUrl(endpoint)),
11433
11750
  token$,
11434
- tokenRefresher: () => session.refresh().then((t) => t ?? null)
11751
+ tokenRefresher: () => session.refresh().then((t) => t ?? null),
11752
+ // Only the reply channels this process awaits — not the full bridged
11753
+ // set. See WORKER_AWAITED_OPERATIONS.
11754
+ channels: WORKER_CHANNELS
11435
11755
  });
11436
11756
  const content = new HttpContentTransport(transport);
11437
11757
  const client = new SemiontClient(transport, content, transport);
@@ -11443,7 +11763,7 @@ async function startAgentWorker(opts) {
11443
11763
  refresh: async () => {
11444
11764
  try {
11445
11765
  const { token } = await authenticateAgent({
11446
- backendBaseUrl: backendBaseUrl2,
11766
+ gatewayBaseUrl: gatewayBaseUrl2,
11447
11767
  workerSecret: workerSecret2,
11448
11768
  provider: inference.type,
11449
11769
  model: inference.model,
@@ -11468,10 +11788,12 @@ async function startAgentWorker(opts) {
11468
11788
  jobTypes: group.jobTypes,
11469
11789
  inferenceClient: group.client,
11470
11790
  generator,
11471
- // The extraction seam's cache, over this worker's content transport
11472
- // (PERSIST-ANCHORS P2d). Built here because the client keeps its
11473
- // content transport private — this is where it is in hand.
11474
- anchoredTextStore: anchoredTextStoreOverTransport(content, logger2),
11791
+ // The extraction seam's cache, consulted over the bus and never written
11792
+ // (ANCHORED-TEXT-TO-SMELTER D2): the Smelter owns this store, the
11793
+ // Archivist answers the checksum-addressed read, and a worker that
11794
+ // misses extracts locally and discards.
11795
+ anchoredTextStore: anchoredTextOverBus(client, logger2),
11796
+ contentReads: contentReads2,
11475
11797
  logger: logger2
11476
11798
  });
11477
11799
  logger2.info("Agent ready", {
@@ -11527,13 +11849,14 @@ function resolveWorker(jobType) {
11527
11849
  `No inference config for worker '${jobType}' and no workers.default in ~/.semiontconfig.`
11528
11850
  );
11529
11851
  }
11530
- var backendPublicURL = envConfig.services?.backend?.publicURL;
11531
- if (!backendPublicURL) {
11532
- throw new Error("services.backend.publicURL is required in ~/.semiontconfig");
11852
+ var gatewayPublicURL = envConfig.services?.gateway?.publicURL;
11853
+ if (!gatewayPublicURL) {
11854
+ throw new Error("services.gateway.publicURL is required in ~/.semiontconfig");
11533
11855
  }
11534
- var backendBaseUrl = backendPublicURL;
11856
+ var gatewayBaseUrl = gatewayPublicURL;
11535
11857
  var workerSecret = process.env.SEMIONT_WORKER_SECRET ?? "";
11536
- var healthPort = 9090;
11858
+ var contentReads = archivistContentReads(envConfig);
11859
+ var healthPort = 24100;
11537
11860
  var logger = createProcessLogger("worker");
11538
11861
  function clientKey(w) {
11539
11862
  return [w.type, w.model, w.apiKey ?? "", w.endpoint ?? "", w.baseURL ?? ""].join("|");
@@ -11566,7 +11889,7 @@ async function main() {
11566
11889
  const { initObservabilityNode } = await import('@semiont/observability/node');
11567
11890
  initObservabilityNode({ serviceName: "semiont-worker" });
11568
11891
  logger.info("Starting agents", {
11569
- baseUrl: backendBaseUrl,
11892
+ baseUrl: gatewayBaseUrl,
11570
11893
  agents: Array.from(groups.values()).map((g) => ({
11571
11894
  provider: g.inference.type,
11572
11895
  model: g.inference.model,
@@ -11575,7 +11898,7 @@ async function main() {
11575
11898
  });
11576
11899
  const workers = await Promise.all(
11577
11900
  Array.from(groups.values()).map(
11578
- (group) => startAgentWorker({ group, backendBaseUrl, workerSecret, logger })
11901
+ (group) => startAgentWorker({ group, gatewayBaseUrl, workerSecret, contentReads, logger })
11579
11902
  )
11580
11903
  );
11581
11904
  const health = createServer((req, res) => {