@opengeni/worker-bundle 2.0.0 → 2.0.1-canary.35916963870001

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/dist/activities/agent-turn/agent-build.d.ts +1 -0
  2. package/dist/activities/agent-turn/failure-settlement.d.ts +1 -1
  3. package/dist/activities/agent-turn/tool-environment.d.ts +34 -0
  4. package/dist/activities/codemode-dispatcher.d.ts +3 -1
  5. package/dist/activities/sandbox-lease.d.ts +3 -4
  6. package/dist/{activities-control-BOLIY3H4.js → activities-control-UF2CR2LB.js} +199 -23
  7. package/dist/activities-control-UF2CR2LB.js.map +1 -0
  8. package/dist/{activities-turn-2B3WMHBB.js → activities-turn-KBLXHBWS.js} +109 -25
  9. package/dist/activities-turn-KBLXHBWS.js.map +1 -0
  10. package/dist/{chunk-TVEE2UUC.js → chunk-K3BPAI4R.js} +12 -6
  11. package/dist/chunk-K3BPAI4R.js.map +1 -0
  12. package/dist/index.js +7 -2
  13. package/dist/index.js.map +1 -1
  14. package/dist/workflow-bundle.js +4 -1
  15. package/package.json +19 -19
  16. package/src/activities/agent-turn/agent-build.ts +5 -1
  17. package/src/activities/agent-turn/claim.ts +12 -2
  18. package/src/activities/agent-turn/errors.ts +1 -0
  19. package/src/activities/agent-turn/failure-settlement.ts +40 -8
  20. package/src/activities/agent-turn/run.ts +1 -0
  21. package/src/activities/agent-turn/sandbox-runtime.ts +25 -9
  22. package/src/activities/agent-turn/stream-attempt.ts +14 -1
  23. package/src/activities/agent-turn/tool-environment.ts +5 -0
  24. package/src/activities/codemode-dispatcher.ts +6 -0
  25. package/src/activities/knowledge-indexing.ts +124 -6
  26. package/src/activities/retained-screenshots.ts +42 -4
  27. package/src/activities/sandbox-lease.ts +141 -21
  28. package/src/sandbox-resume.ts +6 -0
  29. package/src/sandbox-routing.ts +7 -3
  30. package/src/workflows/activities.ts +14 -1
  31. package/dist/activities-control-BOLIY3H4.js.map +0 -1
  32. package/dist/activities-turn-2B3WMHBB.js.map +0 -1
  33. package/dist/chunk-TVEE2UUC.js.map +0 -1
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@opengeni/worker-bundle",
3
- "version": "2.0.0",
3
+ "version": "2.0.1-canary.35916963870001",
4
4
  "description": "OpenGeni worker entry and reusable embedded lifecycle, shipped with a release-coherent pre-bundled Temporal workflow artifact.",
5
5
  "license": "Apache-2.0",
6
6
  "repository": {
@@ -54,24 +54,24 @@
54
54
  },
55
55
  "dependencies": {
56
56
  "@llamaindex/liteparse": "2.14.2",
57
- "@opengeni/agent-proto": "^0.6.0",
58
- "@opengeni/capabilities": "^0.3.4",
59
- "@opengeni/codemode": "^0.6.0",
60
- "@opengeni/codex": "^0.2.24",
61
- "@opengeni/config": "^2.0.0",
62
- "@opengeni/contracts": "^5.0.0",
63
- "@opengeni/core": "^4.0.0",
64
- "@opengeni/db": "^6.0.0",
65
- "@opengeni/documents": "^0.8.31",
66
- "@opengeni/events": "^0.4.29",
67
- "@opengeni/github": "^0.7.14",
68
- "@opengeni/network": "^0.3.1",
69
- "@opengeni/observability": "^0.8.30",
70
- "@opengeni/runtime": "^4.0.0",
71
- "@opengeni/sdk": "^7.0.0",
72
- "@opengeni/storage": "^0.2.131",
73
- "@opengeni/tool-gateway": "^0.1.11",
74
- "@opengeni/xai-subscription": "^0.1.4",
57
+ "@opengeni/agent-proto": "^0.6.0-canary.35916963870001",
58
+ "@opengeni/capabilities": "^0.3.4-canary.35916963870001",
59
+ "@opengeni/codemode": "^0.6.1-canary.35916963870001",
60
+ "@opengeni/codex": "^0.2.25-canary.35916963870001",
61
+ "@opengeni/config": "^2.1.0-canary.35916963870001",
62
+ "@opengeni/contracts": "^5.1.0-canary.35916963870001",
63
+ "@opengeni/core": "^4.0.1-canary.35916963870001",
64
+ "@opengeni/db": "^6.0.1-canary.35916963870001",
65
+ "@opengeni/documents": "^0.8.32-canary.35916963870001",
66
+ "@opengeni/events": "^0.4.30-canary.35916963870001",
67
+ "@opengeni/github": "^0.7.15-canary.35916963870001",
68
+ "@opengeni/network": "^0.3.1-canary.35916963870001",
69
+ "@opengeni/observability": "^0.8.31-canary.35916963870001",
70
+ "@opengeni/runtime": "^4.0.1-canary.35916963870001",
71
+ "@opengeni/sdk": "^7.1.0-canary.35916963870001",
72
+ "@opengeni/storage": "^0.2.132-canary.35916963870001",
73
+ "@opengeni/tool-gateway": "^0.1.12-canary.35916963870001",
74
+ "@opengeni/xai-subscription": "^0.1.4-canary.35916963870001",
75
75
  "@temporalio/activity": "^1.17.0",
76
76
  "@temporalio/client": "^1.17.0",
77
77
  "@temporalio/worker": "^1.17.0",
@@ -79,6 +79,7 @@ import { resolveVideoReferenceSandboxAccess } from "./video-reference-sandbox";
79
79
 
80
80
  export type BuildTurnAgentDeps = {
81
81
  skillCatalog: NonNullable<BuildAgentOptions["skillCatalog"]>;
82
+ mcpServers: Settings["mcpServers"];
82
83
  input: RunAgentTurnInput;
83
84
  db: ActivityServices["db"];
84
85
  runtime: ActivityServices["runtime"];
@@ -139,6 +140,7 @@ export type BuildTurnAgentDeps = {
139
140
 
140
141
  export async function buildTurnAgent(deps: BuildTurnAgentDeps) {
141
142
  const {
143
+ mcpServers,
142
144
  input,
143
145
  db,
144
146
  runtime,
@@ -621,7 +623,9 @@ export async function buildTurnAgent(deps: BuildTurnAgentDeps) {
621
623
  const agentConstructionStartedAt = performance.now();
622
624
  let agentConstructionOutcome: "completed" | "failed" = "completed";
623
625
  try {
624
- return runtime.buildAgent(eventing.modelRunSettings, runtimeResources, {
626
+ // Approval policy must use the same accepted account identities as tool
627
+ // preparation. Keep the separately resolved model/sandbox settings intact.
628
+ return runtime.buildAgent({ ...eventing.modelRunSettings, mcpServers }, runtimeResources, {
625
629
  ...(linkedToolAuthority
626
630
  ? {
627
631
  authorizeAttemptExecution: async () => {
@@ -19,6 +19,7 @@ import { linkCurrentSpanToAdmission, turnExecutionTelemetryKey } from "@opengeni
19
19
  import { deliverChildRequiresActionToParent } from "../parent-wake";
20
20
  import {
21
21
  assertTurnExecutionPolicyMatchesConfigV1,
22
+ settingsForAcceptedSubscriptionTurn,
22
23
  resolveTurnExecutionPolicyV1,
23
24
  type Settings,
24
25
  } from "@opengeni/config";
@@ -277,7 +278,7 @@ export async function claimTurnAttempt(deps: ClaimTurnDeps): Promise<ClaimTurnOu
277
278
  gatewaySettings,
278
279
  claimedPolicy.kind === "valid" ? claimedPolicy.policy.productModelId : turn.model,
279
280
  );
280
- const capabilitySettings = await settingsWithOrganizationProviderCredentials(
281
+ let capabilitySettings = await settingsWithOrganizationProviderCredentials(
281
282
  db,
282
283
  input.accountId,
283
284
  input.workspaceId,
@@ -287,7 +288,6 @@ export async function claimTurnAttempt(deps: ClaimTurnDeps): Promise<ClaimTurnOu
287
288
  const codexAppsCredentialId = capabilitySettings.codexConnectedAppsEnabled
288
289
  ? await resolveCodexAppsCredentialIdForRun(db, input.workspaceId)
289
290
  : null;
290
- runtime.configure(capabilitySettings);
291
291
  const policyForAbsent =
292
292
  claimedPolicy.kind === "valid"
293
293
  ? claimedPolicy.policy
@@ -304,6 +304,16 @@ export async function claimTurnAttempt(deps: ClaimTurnDeps): Promise<ClaimTurnOu
304
304
  if (!installedPolicy.accepted) {
305
305
  throw new TurnAttemptFencedError(`turn execution policy was fenced: ${installedPolicy.reason}`);
306
306
  }
307
+ capabilitySettings = settingsForAcceptedSubscriptionTurn(
308
+ capabilitySettings,
309
+ installedPolicy.policy,
310
+ {
311
+ modelId: turn.model,
312
+ reasoningEffort: turn.reasoningEffort,
313
+ latencyMode: turn.latencyMode,
314
+ },
315
+ );
316
+ runtime.configure(capabilitySettings);
307
317
  const verifiedExecutionPolicy = assertTurnExecutionPolicyMatchesConfigV1(
308
318
  capabilitySettings,
309
319
  installedPolicy.policy,
@@ -119,6 +119,7 @@ export function providerRecoveryResult(input: {
119
119
  input.failureCode === "sandbox_command_start_unavailable" ||
120
120
  input.failureCode === "mcp_transport_timeout" ||
121
121
  input.failureCode === "mcp_transport_unavailable" ||
122
+ input.failureCode === "turn_execution_policy_definition_mismatch" ||
122
123
  input.failureCode === POST_COMPACTION_CONTINUATION_EMPTY_CODE
123
124
  ? Math.max(
124
125
  providerDelay ?? 0,
@@ -16,14 +16,14 @@ import {
16
16
  } from "@opengeni/db";
17
17
  import { publishDurableSessionEvents } from "@opengeni/events";
18
18
  import { maxTurnsExceededRunState } from "@opengeni/runtime";
19
- import { CancelledFailure } from "@temporalio/activity";
19
+ import { ApplicationFailure, CancelledFailure } from "@temporalio/activity";
20
20
  import {
21
21
  authoritativeCodexCapacityResetAt,
22
22
  classifyCodexPin,
23
23
  selectCodexCredentialLeaseForTurn,
24
24
  type CodexRotationStrategy,
25
25
  } from "../codex-rotation";
26
- import type { Settings } from "@opengeni/config";
26
+ import { TurnExecutionPolicyDefinitionMismatchError, type Settings } from "@opengeni/config";
27
27
  import {
28
28
  classifyCodexEncryptedArtifactRejection,
29
29
  classifyCodexUsageLimitError,
@@ -1499,10 +1499,28 @@ export async function settleTurnFailure(deps: TurnFailureDeps): Promise<RunAgent
1499
1499
  // truth, recover this SAME accepted turn, then let the workflow re-claim
1500
1500
  // it after a pacing delay. This is independent of goal state and never
1501
1501
  // relies on a synthetic continuation prompt.
1502
- let failure = agentRunFailurePayload(error, {
1503
- isCodexTurn: billingState.isCodexTurn,
1504
- }) as ReturnType<typeof agentRunFailurePayload>;
1505
- if (failure.retryable && eventing.publish && attempt.turnId && eventing.turnStartedPublished) {
1502
+ // A rolling-deployment definition mismatch is a separate configuration
1503
+ // class: only the exact typed setup error can use this checkpoint before
1504
+ // eventing exists. No generic setup/credential failure gains retry authority.
1505
+ const earlyDefinitionMismatch =
1506
+ error instanceof TurnExecutionPolicyDefinitionMismatchError &&
1507
+ !attempt.modelRequestStarted &&
1508
+ !eventing.turnStartedPublished &&
1509
+ !!attempt.turnId &&
1510
+ !!attempt.triggerEventId &&
1511
+ attempt.executionGeneration > 0;
1512
+ let failure = (
1513
+ earlyDefinitionMismatch
1514
+ ? { error: error.message, code: error.code, retryable: true }
1515
+ : agentRunFailurePayload(error, {
1516
+ isCodexTurn: billingState.isCodexTurn,
1517
+ })
1518
+ ) as ReturnType<typeof agentRunFailurePayload>;
1519
+ if (
1520
+ attempt.turnId &&
1521
+ (earlyDefinitionMismatch ||
1522
+ (failure.retryable && eventing.publish && eventing.turnStartedPublished))
1523
+ ) {
1506
1524
  const nextProviderRecoveryCount = attempt.providerRecoveryCount + 1;
1507
1525
  const recoveryResult = providerRecoveryResult({
1508
1526
  failureCode: failure.code,
@@ -1511,8 +1529,10 @@ export async function settleTurnFailure(deps: TurnFailureDeps): Promise<RunAgent
1511
1529
  });
1512
1530
  try {
1513
1531
  if (recoveryResult.status === "recovering") {
1514
- await flushRuntimeBatcher();
1515
- await historySink.reconcileConversationTruth({ requireDurable: true });
1532
+ if (!earlyDefinitionMismatch) {
1533
+ await flushRuntimeBatcher();
1534
+ await historySink.reconcileConversationTruth({ requireDurable: true });
1535
+ }
1516
1536
  const recovery = await requestSessionTurnRecovery(db, input.workspaceId, {
1517
1537
  sessionId: input.sessionId,
1518
1538
  turnId: attempt.turnId,
@@ -1540,6 +1560,18 @@ export async function settleTurnFailure(deps: TurnFailureDeps): Promise<RunAgent
1540
1560
  return claimedResult(recoveryResult);
1541
1561
  }
1542
1562
  failure = providerRecoveryExhaustedFailure(failure, recoveryResult);
1563
+ if (earlyDefinitionMismatch) {
1564
+ // Setup has no eventing sink yet. Carry only the fixed, safe diagnostic
1565
+ // through Temporal into exact-attempt workflow failure settlement.
1566
+ control.activityStatus = "failed";
1567
+ control.turnMetricOutcome = "failed";
1568
+ control.activityError = error;
1569
+ throw ApplicationFailure.create({
1570
+ message: `${error.message}. Automatic same-turn configuration recovery exhausted after ${recoveryResult.providerRecoveryCount} retries.`,
1571
+ type: "TurnExecutionPolicyDefinitionMismatchError",
1572
+ nonRetryable: true,
1573
+ });
1574
+ }
1543
1575
  } catch (recoveryError) {
1544
1576
  const escaped =
1545
1577
  recoveryResult.status === "recovering"
@@ -1302,6 +1302,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
1302
1302
 
1303
1303
  const builtAgent = await buildTurnAgent({
1304
1304
  skillCatalog: toolRuntime.skillCatalog,
1305
+ mcpServers: toolRuntime.mcpServers,
1305
1306
  input,
1306
1307
  db,
1307
1308
  runtime,
@@ -1,7 +1,7 @@
1
1
  import {
2
2
  advanceWorkspaceGeneration,
3
3
  verifyWorkspaceMutationSettlement,
4
- heartbeatLeaseHolder,
4
+ heartbeatLeaseHolderStatus,
5
5
  readLease,
6
6
  accrueWarmSeconds,
7
7
  SandboxWorkspaceMutationFencedError,
@@ -526,12 +526,15 @@ export function createSandboxTurnRuntime(deps: SandboxTurnRuntimeDeps) {
526
526
  // home turn that degraded to the cloud group box (swap-away / flag-off), that
527
527
  // is the deployment default (modal), so the fallback box is warm-metered at
528
528
  // the cloud rate instead of selfhosted's rate-0 (which would under-bill).
529
- const warmRate = sandboxWarmRateMicrosPerSecond(
530
- settings,
531
- warmBackend ?? (sandbox.established.backendId as Settings["sandboxBackend"]),
532
- );
529
+ const warmRate =
530
+ settings.sandboxWarmBillingMode === "usage_only"
531
+ ? 0
532
+ : sandboxWarmRateMicrosPerSecond(
533
+ settings,
534
+ warmBackend ?? (sandbox.established.backendId as Settings["sandboxBackend"]),
535
+ );
533
536
  sandboxState.leaseHeartbeatTimer = setInterval(() => {
534
- void heartbeatLeaseHolder(db, {
537
+ void heartbeatLeaseHolderStatus(db, {
535
538
  accountId: input.accountId,
536
539
  workspaceId: input.workspaceId,
537
540
  sandboxGroupId: heartbeatGroupId,
@@ -539,9 +542,17 @@ export function createSandboxTurnRuntime(deps: SandboxTurnRuntimeDeps) {
539
542
  holderId: heartbeatHolderId,
540
543
  leaseTtlMs: settings.sandboxLeaseTtlMs,
541
544
  expectedEpoch: heartbeatEpoch,
545
+ billingMode: settings.sandboxWarmBillingMode,
542
546
  })
543
- .then(async (alive) => {
544
- if (alive) return;
547
+ .then(async (status) => {
548
+ if (status.fence === "funding") {
549
+ stopLeaseHeartbeat();
550
+ sandboxRotationController.abort(
551
+ new Error("Insufficient OpenGeni credits to extend paid sandbox compute"),
552
+ );
553
+ return;
554
+ }
555
+ if (status.leaseExtended) return;
545
556
  const rotation = await beginRotationPreemption(sandbox, heartbeatEpoch, heartbeatGroupId);
546
557
  if (rotation === "not_rotating") {
547
558
  // The holder was reaped, the exact attempt closed, the epoch was
@@ -557,9 +568,14 @@ export function createSandboxTurnRuntime(deps: SandboxTurnRuntimeDeps) {
557
568
  sandboxGroupId: heartbeatGroupId,
558
569
  expectedEpoch: heartbeatEpoch,
559
570
  warmRateMicrosPerSecond: warmRate,
571
+ billingMode: settings.sandboxWarmBillingMode,
560
572
  subjectId: input.sessionId,
561
573
  })
562
- .then((result) => recordCreditMicros(observability, "usage", result.costMicros))
574
+ .then((result) => {
575
+ if (settings.sandboxWarmBillingMode === "credits") {
576
+ recordCreditMicros(observability, "usage", result.costMicros);
577
+ }
578
+ })
563
579
  .catch(() => undefined);
564
580
  // MID-SESSION snapshot (sandbox-file-persistence): while the turn holds
565
581
  // the box, fold a fresh /workspace snapshot onto the lease every
@@ -17,6 +17,7 @@ import { publishDurableSessionEvents } from "@opengeni/events";
17
17
  import {
18
18
  normalizeModelCallUsage,
19
19
  normalizeSdkEvent,
20
+ withMcpToolDisplayMetadata,
20
21
  extractOpenSuffixFromRunState,
21
22
  assertOpenSuffixResumable,
22
23
  interruptionKindForCallItem,
@@ -1304,6 +1305,11 @@ export async function runTurnStreamAttempt(
1304
1305
  : {},
1305
1306
  );
1306
1307
  for (const event of normalized) {
1308
+ if (event.type === "agent.toolCall.created")
1309
+ event.payload = withMcpToolDisplayMetadata(
1310
+ eventing.preparedTools?.mcpServers ?? [],
1311
+ event.payload,
1312
+ );
1307
1313
  streamTiming.onEvent(event.type);
1308
1314
  await eventing.batcher.push(event);
1309
1315
  }
@@ -1641,7 +1647,14 @@ export async function runTurnStreamAttempt(
1641
1647
  ? [
1642
1648
  {
1643
1649
  type: "session.requiresAction" as const,
1644
- payload: { approvals },
1650
+ payload: {
1651
+ approvals: approvals.map((approval) =>
1652
+ withMcpToolDisplayMetadata(
1653
+ eventing.preparedTools?.mcpServers ?? [],
1654
+ approval,
1655
+ ),
1656
+ ),
1657
+ },
1645
1658
  },
1646
1659
  ]
1647
1660
  : []),
@@ -31,6 +31,7 @@ import {
31
31
  type ConnectorAttachmentMaterializationRequest,
32
32
  type ConnectorActionPolicyHooks,
33
33
  createFirstPartyInteractionAttemptToolDefinitions,
34
+ mcpToolDisplayMetadata,
34
35
  } from "@opengeni/runtime";
35
36
  import {
36
37
  createGoogleDrivePublicationAttemptTool,
@@ -1095,6 +1096,9 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
1095
1096
  executionGeneration: attempt.executionGeneration,
1096
1097
  },
1097
1098
  cancellationSignal,
1099
+ undefined,
1100
+ {},
1101
+ (name) => mcpToolDisplayMetadata(tools.mcpServers, name),
1098
1102
  );
1099
1103
  eventing.codemodeDispatcher.start();
1100
1104
  };
@@ -1113,6 +1117,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
1113
1117
  return {
1114
1118
  attemptConnectorActionBindings,
1115
1119
  connectorActionPolicy,
1120
+ mcpServers: runSettings.mcpServers,
1116
1121
  generateSessionTitleInParallel: titleToolPlan.generateTitleInParallel,
1117
1122
  postToolPreparationStartedAt,
1118
1123
  preparationIndependentToolNames: [
@@ -16,6 +16,7 @@ import {
16
16
  CODEMODE_MAX_CONCURRENT_CALLS_PER_ATTEMPT,
17
17
  type CodemodeOperation,
18
18
  type SessionEvent,
19
+ type ToolDisplayMetadata,
19
20
  } from "@opengeni/contracts";
20
21
  import {
21
22
  cancelQueuedCodemodeOperationsForAttempt,
@@ -70,6 +71,7 @@ export class CodemodeAttemptDispatcher {
70
71
  private readonly turnSignal?: AbortSignal,
71
72
  private readonly maxConcurrentCalls = CODEMODE_MAX_CONCURRENT_CALLS_PER_ATTEMPT,
72
73
  timings: CodemodeDispatcherTimings = {},
74
+ private readonly toolDisplayMetadata?: (modelName: string) => ToolDisplayMetadata | undefined,
73
75
  ) {
74
76
  if (
75
77
  environment.catalog.accountId !== scope.accountId ||
@@ -371,6 +373,10 @@ export class CodemodeAttemptDispatcher {
371
373
  payload: {
372
374
  id: operation.operationId,
373
375
  name: entry.modelName,
376
+ display: this.toolDisplayMetadata?.(entry.modelName) ?? {
377
+ toolName: entry.identity.toolName,
378
+ ...(entry.title ? { title: entry.title } : {}),
379
+ },
374
380
  arguments: operation.arguments,
375
381
  origin: "codemode",
376
382
  subjectId: operation.caller.subjectId,
@@ -1,15 +1,21 @@
1
1
  import { configuredStaticUsageLimits } from "@opengeni/config";
2
+ import { documentEmbeddingCostMicros, paidDocumentEmbedding } from "@opengeni/core";
2
3
  import {
4
+ applyCreditDebitAfterUse,
3
5
  appendKnowledgeIndexChunks,
4
6
  claimKnowledgeIndexJobs,
5
7
  completeKnowledgeIndexJob,
6
8
  continueKnowledgeIndexJob,
7
9
  deferKnowledgeIndexJob,
10
+ freezeKnowledgeIndexBillingMode,
8
11
  getBillingBalance,
12
+ guardPaidKnowledgeIndexPublication,
13
+ knowledgeIndexBillingActivationTime,
9
14
  readKnowledgeIndexSource,
10
15
  recordUsageEvent,
11
16
  sumUsageQuantity,
12
17
  withWorkspaceUsageLock,
18
+ waitKnowledgeIndexForFunding,
13
19
  } from "@opengeni/db";
14
20
  import type { DocumentServices } from "@opengeni/documents";
15
21
  import type { ControlActivityServices } from "./types";
@@ -18,11 +24,26 @@ export function createKnowledgeIndexingActivities(
18
24
  services: () => Promise<ControlActivityServices>,
19
25
  resolveDocumentServices?: () => Promise<DocumentServices>,
20
26
  ) {
27
+ // Unpaid modes retain a conservative first-poll cutoff; paid mode uses the
28
+ // operator's explicit timestamp so process restarts cannot change eligibility.
29
+ let activationTime: Promise<Date> | undefined;
21
30
  return {
22
31
  indexKnowledge: async () => {
23
32
  const result = { completed: 0, advanced: 0, deferred: 0, unavailable: 0 };
24
33
  if (!resolveDocumentServices) return result;
25
34
  const { db, settings, observability } = await services();
35
+ let policyActivatedAt: Date;
36
+ if (paidDocumentEmbedding(settings)) {
37
+ if (!settings.documentEmbeddingCreditsActivatedAt)
38
+ throw new Error("paid Knowledge embedding requires an activation cutoff");
39
+ policyActivatedAt = new Date(settings.documentEmbeddingCreditsActivatedAt);
40
+ } else {
41
+ activationTime ??= knowledgeIndexBillingActivationTime(db).catch((error) => {
42
+ activationTime = undefined;
43
+ throw error;
44
+ });
45
+ policyActivatedAt = await activationTime;
46
+ }
26
47
  const { embedder } = await resolveDocumentServices();
27
48
  const { knowledgeIndexChunks } = await import("@opengeni/documents");
28
49
  const claims = await claimKnowledgeIndexJobs(db, {
@@ -37,9 +58,9 @@ export function createKnowledgeIndexingActivities(
37
58
  result.unavailable++;
38
59
  continue;
39
60
  }
40
- // Reuse the existing embedding budget and usage ledger. Each batch's
41
- // checkpoint and charge commit together, so a restart never charges
42
- // twice for an accepted projection. Canonical text is never changed.
61
+ // The checkpoint, usage and post-use debit commit together. A new
62
+ // generation requires funding once; committed batches may finish even
63
+ // if their accumulated cost takes the balance below zero.
43
64
  await withWorkspaceUsageLock(db, source.billingWorkspaceId, async (lockedDb) => {
44
65
  const current = await readKnowledgeIndexSource(lockedDb, claim);
45
66
  if (!current) {
@@ -57,9 +78,34 @@ export function createKnowledgeIndexingActivities(
57
78
  chunks.push(chunk);
58
79
  }
59
80
  if (chunks.length) {
60
- if (settings.billingMode === "stripe" || settings.usageLimitsMode === "managed") {
81
+ const frozenPolicy = await freezeKnowledgeIndexBillingMode(
82
+ lockedDb,
83
+ claim,
84
+ // Deterministic embeddings incur no provider charge; a valid
85
+ // credits-mode config with zero tariff must still index them.
86
+ // OpenAI shadow keeps its price snapshot for internal estimates.
87
+ settings.documentEmbeddingProvider === "openai"
88
+ ? (settings.documentEmbeddingBillingMode ?? "usage_only")
89
+ : "usage_only",
90
+ policyActivatedAt,
91
+ settings.documentEmbeddingRateMicrosPerMillionBytes ?? 0,
92
+ );
93
+ if (frozenPolicy.mode === "awaiting_review") {
94
+ result.deferred++;
95
+ return;
96
+ }
97
+ if (frozenPolicy.mode === "obsolete") {
98
+ result.unavailable++;
99
+ return;
100
+ }
101
+ const paid = frozenPolicy.mode === "credits" && paidDocumentEmbedding(settings);
102
+ if (paid && current.nextIndex === 0) {
61
103
  const balance = await getBillingBalance(lockedDb, claim.accountId);
62
- if (balance.balanceMicros <= 0) throw new Error("insufficient OpenGeni credits");
104
+ if (balance.balanceMicros <= 0) {
105
+ await waitKnowledgeIndexForFunding(lockedDb, claim);
106
+ result.deferred++;
107
+ return;
108
+ }
63
109
  }
64
110
  if (settings.usageLimitsMode === "static" || settings.usageLimitsMode === "managed") {
65
111
  const limit =
@@ -75,9 +121,24 @@ export function createKnowledgeIndexingActivities(
75
121
  throw new Error("monthly document indexing limit reached");
76
122
  }
77
123
  }
78
- const vectors = await embedder.embedMany(chunks.map((chunk) => chunk.embeddingInput));
124
+ const inputs = chunks.map((chunk) => chunk.embeddingInput);
125
+ const bytes = inputs.reduce(
126
+ (sum, input) => sum + Buffer.byteLength(input, "utf8"),
127
+ 0,
128
+ );
129
+ const vectors = await embedder.embedMany(inputs);
79
130
  if (vectors.length !== chunks.length)
80
131
  throw new Error("Incomplete Knowledge embeddings");
132
+ // A reviewer may have rejected this revision during the provider
133
+ // call. The DB guard holds its publication row through settlement.
134
+ if (paid) {
135
+ const publication = await guardPaidKnowledgeIndexPublication(lockedDb, claim);
136
+ if (publication !== "published") {
137
+ if (publication === "obsolete") result.unavailable++;
138
+ else result.deferred++;
139
+ return;
140
+ }
141
+ }
81
142
  const appended = await appendKnowledgeIndexChunks(
82
143
  lockedDb,
83
144
  claim,
@@ -98,6 +159,63 @@ export function createKnowledgeIndexingActivities(
98
159
  sourceResourceId: claim.revisionId,
99
160
  idempotencyKey: `knowledge.indexed:${claim.revisionId}:${claim.generation}:${current.nextIndex}`,
100
161
  });
162
+ await recordUsageEvent(lockedDb, {
163
+ accountId: claim.accountId,
164
+ workspaceId: current.billingWorkspaceId,
165
+ eventType: "document.embedding_bytes",
166
+ quantity: bytes,
167
+ unit: "byte",
168
+ sourceResourceType: "knowledge_revision",
169
+ sourceResourceId: claim.revisionId,
170
+ idempotencyKey: `knowledge.embedding_bytes:${claim.revisionId}:${claim.generation}:${current.nextIndex}`,
171
+ });
172
+ if (frozenPolicy.mode === "shadow" && frozenPolicy.rateMicrosPerMillionBytes > 0) {
173
+ const estimate = documentEmbeddingCostMicros(
174
+ {
175
+ ...settings,
176
+ documentEmbeddingRateMicrosPerMillionBytes:
177
+ frozenPolicy.rateMicrosPerMillionBytes,
178
+ },
179
+ bytes,
180
+ );
181
+ if (estimate > 0)
182
+ await recordUsageEvent(lockedDb, {
183
+ accountId: claim.accountId,
184
+ workspaceId: current.billingWorkspaceId,
185
+ eventType: "document.embedding_shadow_estimate",
186
+ quantity: estimate,
187
+ unit: "micro_usd",
188
+ sourceResourceType: "knowledge_revision",
189
+ sourceResourceId: claim.revisionId,
190
+ idempotencyKey: `knowledge.embedding_shadow:${claim.revisionId}:${claim.generation}:${current.nextIndex}`,
191
+ });
192
+ }
193
+ if (paid) {
194
+ const cost = documentEmbeddingCostMicros(
195
+ {
196
+ ...settings,
197
+ documentEmbeddingRateMicrosPerMillionBytes:
198
+ frozenPolicy.rateMicrosPerMillionBytes,
199
+ },
200
+ bytes,
201
+ );
202
+ if (cost > 0)
203
+ await applyCreditDebitAfterUse(lockedDb, {
204
+ accountId: claim.accountId,
205
+ workspaceId: current.billingWorkspaceId,
206
+ type: "document_embedding_debit",
207
+ amountMicros: cost,
208
+ sourceType: "knowledge_revision",
209
+ sourceId: claim.revisionId,
210
+ idempotencyKey: `knowledge.embedding:${claim.revisionId}:${claim.generation}:${current.nextIndex}`,
211
+ metadata: {
212
+ model: claim.model,
213
+ bytes,
214
+ chunks: chunks.length,
215
+ rateMicrosPerMillionBytes: frozenPolicy.rateMicrosPerMillionBytes,
216
+ },
217
+ });
218
+ }
101
219
  }
102
220
  if (more) {
103
221
  await continueKnowledgeIndexJob(lockedDb, claim);
@@ -13,6 +13,7 @@ import {
13
13
  import {
14
14
  RetainedScreenshotQuotaExceededError,
15
15
  getRetainedScreenshotArtifact,
16
+ getRetainedScreenshotArtifactForToolCall,
16
17
  isDatabasePersistenceFailure,
17
18
  isSessionEventPersistenceError,
18
19
  prepareRetainedScreenshotArtifact,
@@ -874,6 +875,26 @@ async function materializeRetainedScreenshotHistoryWithCache(
874
875
  cache: Map<string, string>,
875
876
  ): Promise<Array<Record<string, unknown>>> {
876
877
  const now = input.now ?? new Date();
878
+ const receiptForMarker = async (
879
+ marker: unknown,
880
+ callId: string | null,
881
+ ): Promise<RetainedArtifactMetadata | null> => {
882
+ if (!isRetainedImageMarker(marker)) return null;
883
+ const receipt = retainedReceipt(marker.artifact);
884
+ if (receipt) return receipt;
885
+ // Older canonical history could truncate the artifact UUID and reason as
886
+ // ordinary text. Recover only from this exact session and an unambiguous
887
+ // tool call; otherwise fail before constructing a malformed model image.
888
+ if (!callId) throw new Error("Retained screenshot receipt has no tool call identity");
889
+ const artifact = await getRetainedScreenshotArtifactForToolCall(
890
+ input.db,
891
+ input.workspaceId,
892
+ input.sessionId,
893
+ callId,
894
+ );
895
+ if (!artifact) throw new Error("Retained screenshot receipt cannot be recovered");
896
+ return reference(artifact);
897
+ };
877
898
  const dataUrlForReceipt = async (receipt: RetainedArtifactMetadata): Promise<string> => {
878
899
  let dataUrl = cache.get(receipt.artifactId);
879
900
  if (!dataUrl) {
@@ -921,8 +942,12 @@ async function materializeRetainedScreenshotHistoryWithCache(
921
942
  }
922
943
  return dataUrl;
923
944
  };
924
- const materializeEntry = async (entry: unknown): Promise<unknown> => {
925
- const receipt = retainedReceiptFromImageContent(entry);
945
+ const materializeEntry = async (entry: unknown, callId: string | null): Promise<unknown> => {
946
+ const image =
947
+ entry && typeof entry === "object" && !Array.isArray(entry)
948
+ ? (entry as Record<string, unknown>).image
949
+ : null;
950
+ const receipt = await receiptForMarker(image, callId);
926
951
  if (!receipt) return entry;
927
952
  const dataUrl = await dataUrlForReceipt(receipt);
928
953
  return { ...(entry as Record<string, unknown>), image: dataUrl };
@@ -930,7 +955,8 @@ async function materializeRetainedScreenshotHistoryWithCache(
930
955
 
931
956
  const materialized: Array<Record<string, unknown>> = [];
932
957
  for (const item of input.history) {
933
- const directReceipt = retainedReceiptFromDirectOutput(item.output);
958
+ const callId = historyCallId(item);
959
+ const directReceipt = await receiptForMarker(item.output, callId);
934
960
  if (directReceipt) {
935
961
  materialized.push({
936
962
  ...item,
@@ -954,7 +980,7 @@ async function materializeRetainedScreenshotHistoryWithCache(
954
980
  let changed = false;
955
981
  const output: unknown[] = [];
956
982
  for (const entry of outputEntries) {
957
- const next = await materializeEntry(entry);
983
+ const next = await materializeEntry(entry, callId);
958
984
  changed ||= next !== entry;
959
985
  output.push(next);
960
986
  }
@@ -1110,6 +1136,18 @@ function retainedReceiptFromImageContent(entry: unknown): RetainedArtifactMetada
1110
1136
  return retainedReceipt(marker.artifact);
1111
1137
  }
1112
1138
 
1139
+ function isRetainedImageMarker(value: unknown): value is {
1140
+ type: "retained_artifact";
1141
+ artifact: unknown;
1142
+ } {
1143
+ return Boolean(
1144
+ value &&
1145
+ typeof value === "object" &&
1146
+ !Array.isArray(value) &&
1147
+ (value as Record<string, unknown>).type === RETAINED_IMAGE_MARKER,
1148
+ );
1149
+ }
1150
+
1113
1151
  function retainedReceiptFromDirectOutput(output: unknown): RetainedArtifactMetadata | null {
1114
1152
  if (!output || typeof output !== "object" || Array.isArray(output)) return null;
1115
1153
  const marker = output as Record<string, unknown>;