@semiont/jobs 0.6.6 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  import { replyChannelsFor, createTomlConfigLoader, didToAgent, baseUrl, STARTUP_FETCH_RETRY, retryWithBackoff, serviceAccountToken, isTransientFetchError, isObject, isString, isNumber, RETRY_RULES, resourceId, getPrimaryMediaType, isGenerationJobParams, capabilitiesOf, assembleAnnotation, findClaimSpan, busRequest, BusRequestError, isArray, textSourceOf, yieldsGeometryOf, decodeRepresentation, GENERATABLE_MEDIA_TYPES, locate, createFragmentSelector, deriveViews, annotationId, estimateTokens, reconcileSelector, getLocaleEnglishName, cutChunk, chunkText } from '@semiont/core';
2
2
  import { archivistContentReads, extractPdfTextLayer, withinByteBudget, MAX_PDF_BYTES } from '@semiont/content';
3
3
  import { withSpan, SpanKind, recordJobOutcome, recordAnchorOutcome, recordDetectionCall } from '@semiont/observability';
4
- import { createInferenceClient, StructuredReadError } from '@semiont/inference';
4
+ import { createInferenceClient, answerLimitsRequests, StructuredReadError } from '@semiont/inference';
5
5
  import { execFileSync } from 'child_process';
6
6
  import { existsSync, readFileSync, mkdtempSync, writeFileSync, rmSync } from 'fs';
7
7
  import { homedir, hostname, tmpdir } from 'os';
@@ -12016,6 +12016,9 @@ var WORKER_CONSUMED_BROADCASTS = [
12016
12016
  // Cooperative cancellation of the ACTIVE job (JOB-RESTART-SAFETY P4).
12017
12017
  "job:cancel-requested"
12018
12018
  ];
12019
+ var WORKER_ANSWERED_OPERATIONS = [
12020
+ "job:limits-requested"
12021
+ ];
12019
12022
  var WORKER_CHANNELS = [
12020
12023
  ...replyChannelsFor(WORKER_AWAITED_OPERATIONS),
12021
12024
  ...WORKER_CONSUMED_BROADCASTS
@@ -12059,7 +12062,7 @@ async function authenticateAgent(opts) {
12059
12062
  );
12060
12063
  }
12061
12064
  async function startAgentWorker(opts) {
12062
- const { group, gatewayBaseUrl: gatewayBaseUrl2, credential: credential2, contentReads: contentReads2, logger: logger2 } = opts;
12065
+ const { group, gatewayBaseUrl: gatewayBaseUrl2, credential: credential2, contentReads: contentReads2, reportsLimitsOf, logger: logger2 } = opts;
12063
12066
  const { inference } = group;
12064
12067
  const { protocol, host, port } = parseGatewayUrl(gatewayBaseUrl2);
12065
12068
  const { token: initialToken, did } = await authenticateAgent({
@@ -12091,8 +12094,9 @@ async function startAgentWorker(opts) {
12091
12094
  token$,
12092
12095
  tokenRefresher: () => session.refresh().then((t) => t ?? null),
12093
12096
  // Only the reply channels this process awaits — not the full bridged
12094
- // set. See WORKER_AWAITED_OPERATIONS.
12095
- channels: WORKER_CHANNELS
12097
+ // set. See WORKER_AWAITED_OPERATIONS. The agent that reports its pool's
12098
+ // limits also subscribes the requests it answers.
12099
+ channels: reportsLimitsOf.length > 0 ? [...WORKER_CHANNELS, ...WORKER_ANSWERED_OPERATIONS] : WORKER_CHANNELS
12096
12100
  });
12097
12101
  const content = new HttpContentTransport(transport);
12098
12102
  const client = new SemiontClient(transport, content, transport);
@@ -12136,6 +12140,7 @@ async function startAgentWorker(opts) {
12136
12140
  contentReads: contentReads2,
12137
12141
  logger: logger2
12138
12142
  });
12143
+ const limitsResponder = reportsLimitsOf.length > 0 ? answerLimitsRequests(transport, "job:limits-requested", reportsLimitsOf, logger2) : void 0;
12139
12144
  logger2.info("Agent ready", {
12140
12145
  did,
12141
12146
  provider: inference.type,
@@ -12152,6 +12157,7 @@ async function startAgentWorker(opts) {
12152
12157
  ...adapter.vitals()
12153
12158
  }),
12154
12159
  dispose: async () => {
12160
+ limitsResponder?.unsubscribe();
12155
12161
  adapter.dispose();
12156
12162
  await session.dispose();
12157
12163
  }
@@ -12172,7 +12178,8 @@ var tomlReader = {
12172
12178
  var envConfig = createTomlConfigLoader(
12173
12179
  tomlReader,
12174
12180
  configPath,
12175
- process.env
12181
+ process.env,
12182
+ "worker"
12176
12183
  )(null);
12177
12184
  var workerInferenceMap = envConfig._metadata?.workers;
12178
12185
  if (!workerInferenceMap || Object.keys(workerInferenceMap).length === 0) {
@@ -12247,8 +12254,17 @@ async function main() {
12247
12254
  }))
12248
12255
  });
12249
12256
  const workers = await Promise.all(
12257
+ // The first agent reports every group's limits: the gateway delivers only
12258
+ // the first reply to a request, so one agent answers for the pool.
12250
12259
  Array.from(groups.values()).map(
12251
- (group) => startAgentWorker({ group, gatewayBaseUrl, credential, contentReads, logger })
12260
+ (group, i, all) => startAgentWorker({
12261
+ group,
12262
+ gatewayBaseUrl,
12263
+ credential,
12264
+ contentReads,
12265
+ logger,
12266
+ reportsLimitsOf: i === 0 ? all.map((g) => g.client) : []
12267
+ })
12252
12268
  )
12253
12269
  );
12254
12270
  const health = createServer((req, res) => {