@opengeni/worker-bundle 2.2.0-canary.36855269964001 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/dist/activities/agent-run-admission.d.ts +1 -3
  2. package/dist/activities/agent-turn/admission.d.ts +3 -24
  3. package/dist/activities/agent-turn/claude-usage-observer.d.ts +1 -13
  4. package/dist/activities/agent-turn/run-credentials.d.ts +0 -2
  5. package/dist/activities/agent-turn/stream-attempt.d.ts +0 -1
  6. package/dist/activities/agent-turn/turn-context.d.ts +0 -1
  7. package/dist/activities/goals.d.ts +0 -8
  8. package/dist/activities/run-credentials.d.ts +2 -4
  9. package/dist/activities/types.d.ts +1 -1
  10. package/dist/{activities-control-YKXRDVMB.js → activities-control-MHNA3CYV.js} +66 -28
  11. package/dist/activities-control-MHNA3CYV.js.map +1 -0
  12. package/dist/{activities-turn-V37B7CQY.js → activities-turn-FQ5OO2WW.js} +197 -540
  13. package/dist/activities-turn-FQ5OO2WW.js.map +1 -0
  14. package/dist/{chunk-ZCCLJK3V.js → chunk-RZAAGRFQ.js} +65 -145
  15. package/dist/chunk-RZAAGRFQ.js.map +1 -0
  16. package/dist/index.js +1 -1
  17. package/dist/workflow-bundle.js +1 -1
  18. package/package.json +20 -20
  19. package/src/activities/agent-run-admission.ts +1 -17
  20. package/src/activities/agent-turn/admission.ts +1 -64
  21. package/src/activities/agent-turn/claim.ts +14 -56
  22. package/src/activities/agent-turn/claude-usage-observer.ts +1 -33
  23. package/src/activities/agent-turn/errors.ts +0 -46
  24. package/src/activities/agent-turn/failure-settlement.ts +0 -5
  25. package/src/activities/agent-turn/run-credentials.ts +3 -15
  26. package/src/activities/agent-turn/run.ts +3 -32
  27. package/src/activities/agent-turn/stream-attempt.ts +35 -138
  28. package/src/activities/agent-turn/tool-environment.ts +20 -13
  29. package/src/activities/agent-turn/turn-context.ts +0 -1
  30. package/src/activities/capabilities.ts +14 -52
  31. package/src/activities/goals.ts +78 -38
  32. package/src/activities/knowledge-indexing.ts +0 -29
  33. package/src/activities/run-credentials.ts +16 -22
  34. package/src/activities/scheduled-tasks.ts +0 -1
  35. package/src/activities/types.ts +0 -1
  36. package/src/activities/workspace-credential-provider.ts +0 -12
  37. package/src/sandbox-routing.ts +2 -4
  38. package/dist/activities/agent-turn/final-reply.d.ts +0 -18
  39. package/dist/activities/agent-turn/model-call-admission.d.ts +0 -15
  40. package/dist/activities-control-YKXRDVMB.js.map +0 -1
  41. package/dist/activities-turn-V37B7CQY.js.map +0 -1
  42. package/dist/chunk-ZCCLJK3V.js.map +0 -1
  43. package/src/activities/agent-turn/final-reply.ts +0 -46
  44. package/src/activities/agent-turn/model-call-admission.ts +0 -94
@@ -1,8 +1,5 @@
1
1
  import { measureMcpPhase, withMcpCallIdentity } from "@opengeni/observability";
2
2
  import {
3
- appendSessionHistoryItems,
4
- sessionTurnFinalReplyFacts,
5
- sessionTurnHasFinalReplyNudge,
6
3
  getSessionEvent,
7
4
  getHumanInputResumeForEvent,
8
5
  getSessionHumanInputRequest,
@@ -93,14 +90,14 @@ import {
93
90
  type SessionTurn,
94
91
  } from "@opengeni/contracts";
95
92
  import { createModelCheckpointMemoryCollector } from "../../model-checkpoint-memory-collector";
96
- import { createModelCallAdmission } from "./model-call-admission";
97
93
 
98
94
  import {
99
95
  assertWorkspaceHumanInputAllowed,
100
96
  stableHumanInputRequestId,
101
97
  stableInteractionInterventionId,
102
98
  stableInteractionInterventionOperationId,
103
- ensureRunAllowedBetweenModelCalls,
99
+ BudgetExhaustedError,
100
+ ensureRunAllowed,
104
101
  } from "./admission";
105
102
  import {
106
103
  compactionFailureReason,
@@ -144,7 +141,6 @@ import {
144
141
  import { inputWaitReply, latestDurableTurnMessageText } from "./input-wait-reply";
145
142
  import { waitForTurnOperation } from "./sandbox-provision";
146
143
  import { createSharedRigSetupCoordinator } from "./sandbox-shared-preparation";
147
- import { finalReplyNudge, needsFinalReply } from "./final-reply";
148
144
 
149
145
  import type { CompactionSummarizer } from "../context-compaction";
150
146
  import type { TurnExecutionPolicyV1 } from "@opengeni/contracts";
@@ -239,7 +235,6 @@ export type TurnStreamAttemptDeps = {
239
235
  groupBoxBackend: Settings["sandboxBackend"];
240
236
  turnExecutionPolicy: TurnExecutionPolicyV1;
241
237
  turn: Pick<SessionTurn, "initiator" | "initiatorContext"> & {
242
- initiatingHumanSubjectId: string | null;
243
238
  id: string;
244
239
  executionGeneration: number;
245
240
  model: string;
@@ -588,41 +583,14 @@ export async function runTurnStreamAttempt(
588
583
  // durably: the reply a wait-ended human turn records on turn.completed.
589
584
  let latestAssistantMessageText: string | null = null;
590
585
  let workerPreparationTotalRecorded = false;
591
- let toolsExecuted = false;
592
- let finalReplyNudged = false;
593
- const revalidateModelCallAdmission = async () => {
594
- await historySink.reconcileConversationTruth({ requireDurable: true });
595
- await ensureRunAllowedBetweenModelCalls({
596
- settings,
597
- db,
598
- accountId: input.accountId,
599
- workspaceId: input.workspaceId,
600
- isExternallyBilledTurn: billingState.isExternallyBilledTurn,
601
- entitlements,
602
- chargesOpenGeniCredits: billingState.chargesOpenGeniCredits,
603
- countsTowardTokenCap: billingState.countsTowardTokenCap,
604
- initiatingHumanSubjectId: turn.initiatingHumanSubjectId,
605
- });
606
- };
607
586
  const runStreamAttempt = async (options: {
608
587
  requireTerminalModelResponse: boolean;
609
- }): Promise<RunAgentTurnResult | "reply_nudge"> => {
588
+ }): Promise<RunAgentTurnResult> => {
610
589
  if (!runInput) {
611
590
  throw new Error("Run input was not prepared");
612
591
  }
613
- // The previous stream was persisted before compaction; the sink is now
614
- // seeded from its durable replacement. Do not reconcile the old prefix
615
- // against that replacement while checking the next stream's admission.
616
- eventing.stream = undefined;
617
- // Compaction commits its paid usage and replacement history before this
618
- // boundary, both during preparation and in-activity recovery. Revalidate
619
- // the accepted turn's frozen human before another stream can dispatch.
620
- if (options.requireTerminalModelResponse) await revalidateModelCallAdmission();
621
- const modelCallAdmission = createModelCallAdmission({
622
- signal: runtimeCancellationSignal,
623
- admit: revalidateModelCallAdmission,
624
- });
625
592
  const responseCountBeforeStream = modelResponseState.responseCount;
593
+ eventing.stream = undefined;
626
594
  eventing.batcher = null;
627
595
  // The SDK emits every processed call item for one model response before
628
596
  // it emits any result for that response. Keep that response-local batch
@@ -838,8 +806,6 @@ export async function runTurnStreamAttempt(
838
806
  }
839
807
  attempt.modelRequestStarted = true;
840
808
  return await runtime.runStream(agent, runInput!, eventing.modelRunSettings, {
841
- beforeModelRequest: modelCallAdmission.beforeModelRequest,
842
- onModelResponse: modelCallAdmission.onModelResponse,
843
809
  signal: runtimeCancellationSignal,
844
810
  sandboxEnvironment,
845
811
  onModelVisibleContext: async (snapshot) => {
@@ -975,13 +941,7 @@ export async function runTurnStreamAttempt(
975
941
  if (leases.xai.lost) {
976
942
  throw new Error("xAI credential lease expired before the model run");
977
943
  }
978
- try {
979
- eventing.stream = await withProviderRequestContext(runStreamOnce);
980
- } catch (error) {
981
- modelCallAdmission.fail(error);
982
- modelCallAdmission.close();
983
- throw error;
984
- }
944
+ eventing.stream = await withProviderRequestContext(runStreamOnce);
985
945
  // Bounded provider label for the streaming SLIs — the resolved registry
986
946
  // provider id (or the built-in OpenAI/Azure provider), never a raw
987
947
  // user-supplied model string.
@@ -1134,22 +1094,35 @@ export async function runTurnStreamAttempt(
1134
1094
  await historySink.reconcileConversationTruth();
1135
1095
  turnLifecycleMetricsFor(observability).progress({ attemptId: input.attemptId });
1136
1096
  modelCheckpointMemoryCollector.schedule(observability);
1137
- await ensureRunAllowedBetweenModelCalls({
1138
- settings,
1139
- db,
1140
- accountId: input.accountId,
1141
- workspaceId: input.workspaceId,
1142
- isExternallyBilledTurn: billingState.isExternallyBilledTurn,
1143
- entitlements,
1144
- chargesOpenGeniCredits: billingState.chargesOpenGeniCredits,
1145
- countsTowardTokenCap: billingState.countsTowardTokenCap,
1146
- initiatingHumanSubjectId: turn.initiatingHumanSubjectId,
1147
- serializedRunState: () =>
1148
- media.compactMediaRunState(String(eventing.stream!.state.toString())),
1149
- });
1097
+ try {
1098
+ await ensureRunAllowed(
1099
+ settings,
1100
+ db,
1101
+ input.accountId,
1102
+ input.workspaceId,
1103
+ billingState.isExternallyBilledTurn,
1104
+ entitlements,
1105
+ billingState.chargesOpenGeniCredits,
1106
+ billingState.countsTowardTokenCap,
1107
+ );
1108
+ } catch (limitError) {
1109
+ // Capture the run state at the boundary so the budget valve in
1110
+ // the outer catch can end this segment gracefully with full
1111
+ // conversation context preserved for the post-top-up resume.
1112
+ let serializedRunState: string | null = null;
1113
+ try {
1114
+ serializedRunState = media.compactMediaRunState(
1115
+ String(eventing.stream.state.toString()),
1116
+ );
1117
+ } catch {
1118
+ serializedRunState = null;
1119
+ }
1120
+ throw new BudgetExhaustedError(
1121
+ limitError instanceof Error ? limitError.message : String(limitError),
1122
+ serializedRunState,
1123
+ );
1124
+ }
1150
1125
  }
1151
- // Release only after both the debit and frozen-human admission finish.
1152
- modelCallAdmission.settle(next.value);
1153
1126
  const durableSdkEvent = generatedImageReceipt
1154
1127
  ? compactGeneratedImageSdkEvent(next.value, generatedImageReceipt)
1155
1128
  : next.value;
@@ -1186,7 +1159,6 @@ export async function runTurnStreamAttempt(
1186
1159
  }
1187
1160
  const completedToolCall = completedToolCallFromSdkEvent(durableSdkEvent);
1188
1161
  if (completedToolCall) {
1189
- toolsExecuted = true;
1190
1162
  retainedScreenshotMetadata =
1191
1163
  media.retainedScreenshotReceiptsByCallId.get(completedToolCall.callId) ?? null;
1192
1164
  const typedScreenshot = retainedScreenshotMetadata
@@ -1424,7 +1396,6 @@ export async function runTurnStreamAttempt(
1424
1396
  }
1425
1397
  }
1426
1398
  } catch (error) {
1427
- modelCallAdmission.fail(error);
1428
1399
  // Event processing can fail while SDK completion is still pending.
1429
1400
  // Close this stream before any failure publication; a legitimate
1430
1401
  // compaction retry may start a new, independently fenced generation.
@@ -1480,7 +1451,6 @@ export async function runTurnStreamAttempt(
1480
1451
  }
1481
1452
  throw error;
1482
1453
  } finally {
1483
- modelCallAdmission.close();
1484
1454
  if (!streamDone) {
1485
1455
  // ReadableStream cancellation synchronously trips the Agents SDK's
1486
1456
  // abort controller, but its returned promise may wait for an
@@ -1502,10 +1472,7 @@ export async function runTurnStreamAttempt(
1502
1472
  // External Codemode stays reachable until finalization. Close wait
1503
1473
  // admission before any terminal output/history decision, and drain an
1504
1474
  // already-admitted wait before consulting the actual runner-yield latch.
1505
- // Drain trusted receipts before deciding whether an empty stream may need
1506
- // a handoff. Do not terminally seal the attempt until that decision: the
1507
- // one same-turn stream still uses the ordinary input-wait admission gate.
1508
- await eventing.preparedTools?.inputWaitYield?.drainForHandoff(runtimeCancellationSignal);
1475
+ await eventing.preparedTools?.inputWaitYield?.sealForSettlement(runtimeCancellationSignal);
1509
1476
  if (
1510
1477
  options.requireTerminalModelResponse &&
1511
1478
  !eventing.preparedTools?.inputWaitYield?.yielded &&
@@ -1618,7 +1585,6 @@ export async function runTurnStreamAttempt(
1618
1585
  }
1619
1586
  }
1620
1587
  if (eventing.stream.interruptions.length > 0) {
1621
- await eventing.preparedTools?.inputWaitYield?.sealForSettlement(runtimeCancellationSignal);
1622
1588
  await historySink.reconcileConversationTruth({ requireDurable: true });
1623
1589
  const approvals = runtime.serializeApprovals(eventing.stream.interruptions);
1624
1590
  const humanInputInterruptions =
@@ -1773,60 +1739,6 @@ export async function runTurnStreamAttempt(
1773
1739
  const finalOutput = String(
1774
1740
  requireAgentStreamFinalOutput(eventing.stream.finalOutput, inputWaitYielded),
1775
1741
  );
1776
- const durableReplyFacts =
1777
- !inputWaitYielded && finalOutput.trim().length === 0
1778
- ? await sessionTurnFinalReplyFacts(db, input.workspaceId, input.sessionId, activeTurnId)
1779
- : { toolsExecuted: false, completedGoal: false };
1780
- let emptyFinalReply = false;
1781
- if (
1782
- needsFinalReply({
1783
- output: finalOutput,
1784
- inputWaitYielded:
1785
- inputWaitYielded || eventing.preparedTools?.inputWaitYield?.requested === true,
1786
- interrupted: false, // interruption settlement returned above
1787
- maintenance: turn.source === "compaction",
1788
- toolsExecuted: toolsExecuted || finalReplyNudged || durableReplyFacts.toolsExecuted,
1789
- completedGoal: durableReplyFacts.completedGoal,
1790
- })
1791
- ) {
1792
- await historySink.reconcileConversationTruth({ requireDurable: true });
1793
- // Consult retained truth, including inactive compacted rows, so an
1794
- // attempt replacement or compaction cannot spend this bound again.
1795
- if (
1796
- finalReplyNudged ||
1797
- (await sessionTurnHasFinalReplyNudge(
1798
- db,
1799
- input.workspaceId,
1800
- input.sessionId,
1801
- activeTurnId,
1802
- finalReplyNudge(activeTurnId).content[0]!.text,
1803
- ))
1804
- ) {
1805
- // A second empty response is a delivery-quality notice, not failed
1806
- // execution: preserve goal continuation and later machine-input wakes.
1807
- emptyFinalReply = true;
1808
- } else {
1809
- const appended = await appendSessionHistoryItems(db, {
1810
- accountId: input.accountId,
1811
- workspaceId: input.workspaceId,
1812
- sessionId: input.sessionId,
1813
- turnId: activeTurnId,
1814
- expectedExecutionGeneration: attempt.executionGeneration,
1815
- expectedAttemptId: input.attemptId,
1816
- items: [
1817
- {
1818
- position: await nextSessionHistoryPosition(db, input.workspaceId, input.sessionId),
1819
- item: finalReplyNudge(activeTurnId),
1820
- },
1821
- ],
1822
- });
1823
- if (!appended) throw new TurnAttemptFencedError("turn ended before final reply handoff");
1824
- finalReplyNudged = true;
1825
- await prepareRunAttemptInput();
1826
- return "reply_nudge";
1827
- }
1828
- }
1829
- await eventing.preparedTools?.inputWaitYield?.sealForSettlement(runtimeCancellationSignal);
1830
1742
  // The final output is the newest message this stream completed, already
1831
1743
  // durable with its provider identity and phase. A phase-less settlement
1832
1744
  // copy is published only when this stream did not complete that text.
@@ -1861,11 +1773,7 @@ export async function runTurnStreamAttempt(
1861
1773
  : [{ type: "agent.message.completed" as const, payload: { text: finalOutput } }]),
1862
1774
  {
1863
1775
  type: "turn.completed",
1864
- payload: {
1865
- output: finalOutput,
1866
- ...(reply === null ? {} : { reply }),
1867
- ...(emptyFinalReply ? { emptyFinalReply: true } : {}),
1868
- },
1776
+ payload: { output: finalOutput, ...(reply === null ? {} : { reply }) },
1869
1777
  },
1870
1778
  { type: "session.status.changed", payload: { status: "idle" } },
1871
1779
  ],
@@ -1926,9 +1834,6 @@ export async function runTurnStreamAttempt(
1926
1834
  ) {
1927
1835
  return claimedResult({ status: "cancelled" });
1928
1836
  }
1929
- // Preparation may have spent the last allowance on a completed summary.
1930
- // Neither the title sidecar nor ordinary inference may dispatch afterward.
1931
- await revalidateModelCallAdmission();
1932
1837
  if (
1933
1838
  turn.source !== "compaction" &&
1934
1839
  generateSessionTitleInParallel &&
@@ -1966,19 +1871,11 @@ export async function runTurnStreamAttempt(
1966
1871
  }
1967
1872
  try {
1968
1873
  let retriedAfterCompaction = false;
1969
- finalReplyNudged = await sessionTurnHasFinalReplyNudge(
1970
- db,
1971
- input.workspaceId,
1972
- input.sessionId,
1973
- activeTurnId,
1974
- finalReplyNudge(activeTurnId).content[0]!.text,
1975
- );
1976
1874
  while (true) {
1977
1875
  try {
1978
1876
  const result = await runStreamAttempt({
1979
1877
  requireTerminalModelResponse: retriedAfterCompaction,
1980
1878
  });
1981
- if (result === "reply_nudge") continue;
1982
1879
  if (retriedAfterCompaction) {
1983
1880
  observability.info("context compaction recovery succeeded after in-activity retry", {
1984
1881
  sessionId: input.sessionId,
@@ -68,10 +68,8 @@ import {
68
68
  buildApiIntegrationMcpServers,
69
69
  resolveCatalogSettings,
70
70
  resolveWorkspaceModelSelection,
71
- loadWorkspaceCodexModelAvailability,
72
71
  withFrozenPersonalConnectionDelegations,
73
- resolveTurnToolPolicy,
74
- scheduledTurnMcpServerIds,
72
+ resolveSessionToolPolicy,
75
73
  hasPermission,
76
74
  } from "@opengeni/core";
77
75
  import { loadWorkspaceEnvironmentForRunWithCredentials } from "../environment";
@@ -225,20 +223,32 @@ export async function prepareTurnToolPolicy(deps: PrepareTurnToolPolicyDeps) {
225
223
  // sessions follow the current configured MCP set,
226
224
  // while explicit, inherited-fixed, and legacy sessions remain narrowed
227
225
  // to their stored materialized allow-list.
228
- const resolvedToolPolicy = resolveTurnToolPolicy({
226
+ const scheduledEffectiveMcpServerIds = (() => {
227
+ const value =
228
+ turn.metadata && typeof turn.metadata === "object" && !Array.isArray(turn.metadata)
229
+ ? (turn.metadata as Record<string, unknown>).scheduledEffectiveMcpServerIds
230
+ : null;
231
+ return Array.isArray(value) && value.every((id) => typeof id === "string")
232
+ ? [...new Set(value)].sort()
233
+ : null;
234
+ })();
235
+ const currentMcpServerIds = new Set(runSettings.mcpServers.map((server) => server.id));
236
+ const resolvedToolPolicy = resolveSessionToolPolicy({
229
237
  toolPolicy: session.toolPolicy,
230
- session,
231
- turn,
232
- availableMcpServerIds: runSettings.mcpServers.map((server) => server.id),
238
+ sessionTools: scheduledEffectiveMcpServerIds ? turn.tools : session.tools,
239
+ availableMcpServerIds: scheduledEffectiveMcpServerIds
240
+ ? scheduledEffectiveMcpServerIds.filter((id) => currentMcpServerIds.has(id))
241
+ : [...currentMcpServerIds],
233
242
  defaultMcpServerIds:
234
- scheduledTurnMcpServerIds(turn) === null && session.toolPolicy.mode === "workspace_default"
243
+ scheduledEffectiveMcpServerIds ??
244
+ (session.toolPolicy.mode === "workspace_default"
235
245
  ? await workspaceSessionToolPolicyDefaultServerIds(
236
246
  db,
237
247
  input.workspaceId,
238
248
  capabilitySettings,
239
249
  fileAuthoritySubjectId ?? undefined,
240
250
  )
241
- : [],
251
+ : []),
242
252
  });
243
253
  const mcpAvailabilityNote = unavailableMcpOperationalContext({
244
254
  droppedIds: resolvedToolPolicy.effectivePolicy.droppedIds,
@@ -898,7 +908,6 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
898
908
  organizationGatewayCustomModels,
899
909
  organizationOpenRouterConnectionActive,
900
910
  organizationOpenRouterCustomModels,
901
- codexModelAvailability,
902
911
  ] = await Promise.all([
903
912
  getWorkspaceConnectionModelRestrictions(
904
913
  db,
@@ -944,11 +953,9 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
944
953
  workspaceId: input.workspaceId,
945
954
  providerKind: "openrouter",
946
955
  }),
947
- loadWorkspaceCodexModelAvailability(db, currentSettings, input.workspaceId),
948
956
  ]);
949
957
  return {
950
958
  selections: resolveWorkspaceModelSelection({
951
- observations: codexModelAvailability,
952
959
  connectionModelRestrictions,
953
960
  settings: currentSettings,
954
961
  policy,
@@ -1103,7 +1110,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
1103
1110
  sessionAttachedRemoteMcpTargets: selectedSessionRemoteMcpTargets(
1104
1111
  githubRestMcp.settings,
1105
1112
  session.mcpServers ?? [],
1106
- githubRestMcp.tools,
1113
+ turn.tools ?? [],
1107
1114
  localMcpServers,
1108
1115
  ),
1109
1116
  ...(deps.runMcpCredentials ? { runMcpCredentials: deps.runMcpCredentials } : {}),
@@ -48,7 +48,6 @@ export type TurnSettleFn = (input: {
48
48
  sessionStatus: SessionStatus;
49
49
  activeTurnId: string | null;
50
50
  suppressGoalContinuation?: boolean;
51
- allowanceGoalPause?: ApplySessionTurnSettlementInput["allowanceGoalPause"];
52
51
  consumeRequestedCompactionFailure?: boolean;
53
52
  runState?: ApplySessionTurnSettlementInput["runState"];
54
53
  }) => Promise<boolean>;
@@ -40,8 +40,6 @@ import {
40
40
  listWorkspaceProviderCustomModels,
41
41
  getWorkspaceProviderCustomModelForExecution,
42
42
  loadWorkspaceProviderApiKey,
43
- loadClaudeSubscriptionUsageCredential,
44
- assertModelConnectionAllowsTurn,
45
43
  type Database,
46
44
  type SessionMcpServerForRun,
47
45
  } from "@opengeni/db";
@@ -305,30 +303,12 @@ export async function settingsWithOrganizationProviderCredentials(
305
303
  if (kind === "claude_subscription" && !settings.claudeSubscriptionEnabled) continue;
306
304
  const models = await buildModels(kind, claudeProviderId(kind) + "/");
307
305
  result = withClaudeConnectionCatalog(result, { [kind]: { models } });
308
- const binding =
309
- kind === "claude_subscription"
310
- ? await loadClaudeSubscriptionUsageCredential(db, settings, {
311
- accountId,
312
- workspaceId,
313
- scope: "organization",
314
- })
315
- : null;
316
- const credential =
317
- kind === "claude_subscription"
318
- ? binding?.serializedCredential
319
- : await loadOrganizationModelProviderApiKey(db, settings, {
320
- accountId,
321
- workspaceId,
322
- providerKind: kind,
323
- });
324
- if (credential)
325
- result = withClaudeConnectionCredential(
326
- result,
327
- kind,
328
- credential,
329
- "organization",
330
- binding ?? undefined,
331
- );
306
+ const credential = await loadOrganizationModelProviderApiKey(db, settings, {
307
+ accountId,
308
+ workspaceId,
309
+ providerKind: kind,
310
+ });
311
+ if (credential) result = withClaudeConnectionCredential(result, kind, credential);
332
312
  const workspaceModels = await listWorkspaceProviderCustomModels(db, {
333
313
  accountId,
334
314
  workspaceId,
@@ -353,33 +333,15 @@ export async function settingsWithOrganizationProviderCredentials(
353
333
  { [kind]: { models: workspaceModels } },
354
334
  "workspace",
355
335
  );
356
- const workspaceBinding =
357
- kind === "claude_subscription"
358
- ? await loadClaudeSubscriptionUsageCredential(db, settings, {
359
- accountId,
360
- workspaceId,
361
- scope: "workspace",
362
- })
363
- : null;
364
- if (workspaceBinding && workspaceModelId)
365
- await assertModelConnectionAllowsTurn(db, {
366
- workspaceId,
367
- subjectId: "worker:model-access",
368
- modelId: workspaceModelId,
369
- workspaceProviderConnectionId: workspaceBinding.connectionId,
370
- });
371
- const workspaceCredential =
372
- kind === "claude_subscription"
373
- ? workspaceBinding?.serializedCredential
374
- : await loadWorkspaceProviderApiKey(db, settings, workspaceId, kind, workspaceModelId);
336
+ const workspaceCredential = await loadWorkspaceProviderApiKey(
337
+ db,
338
+ settings,
339
+ workspaceId,
340
+ kind,
341
+ workspaceModelId,
342
+ );
375
343
  if (workspaceCredential)
376
- result = withClaudeConnectionCredential(
377
- result,
378
- kind,
379
- workspaceCredential,
380
- "workspace",
381
- workspaceBinding ?? undefined,
382
- );
344
+ result = withClaudeConnectionCredential(result, kind, workspaceCredential, "workspace");
383
345
  }
384
346
  return result;
385
347
  }
@@ -19,10 +19,14 @@ import {
19
19
  import { isCodexBilledModel } from "@opengeni/codex";
20
20
  import {
21
21
  enqueueSessionWorkflowWakeIfRunnable,
22
+ getBillingBalance,
22
23
  getWorkspaceModelPolicy,
23
24
  getSessionGoal,
25
+ isCodexBilledTurn,
24
26
  materializeGoalContinuation,
25
27
  requireSession,
28
+ sumUsageQuantity,
29
+ type Database,
26
30
  } from "@opengeni/db";
27
31
  import type {
28
32
  ControlActivityServices,
@@ -34,7 +38,6 @@ import {
34
38
  resolveCatalogSettings,
35
39
  resolveWorkspaceCatalogSettings,
36
40
  } from "@opengeni/core";
37
- import { agentRunAdmissionDenial } from "./agent-run-admission";
38
41
 
39
42
  export function createGoalActivities(services: () => Promise<ControlActivityServices>) {
40
43
  async function enqueueGoalRetryWake(input: MaybeContinueGoalInput): Promise<void> {
@@ -106,6 +109,30 @@ export function createGoalActivities(services: () => Promise<ControlActivityServ
106
109
  ) {
107
110
  modelPolicyBlocked = `session is locked to Codex remote compaction v2; model "${continuationModel}" is not a Codex subscription model`;
108
111
  }
112
+ // A codex-model goal continuation is paid by the user's ChatGPT/Codex plan,
113
+ // so it must not be budget-paused for zero OpenGeni credits. This file uses
114
+ // BASE settings (no codex overlay); the predicate does its own credential read.
115
+ const isCodexRun = await isCodexBilledTurn({
116
+ db,
117
+ settings,
118
+ workspaceId: input.workspaceId,
119
+ model: continuationModel,
120
+ });
121
+ const fundedWithoutCredits = goalContinuationFundedWithoutCredits(
122
+ settings,
123
+ continuationModel,
124
+ isCodexRun,
125
+ );
126
+ // Budget exhaustion pauses the goal visibly instead of failing the
127
+ // session. Computed up front and applied inside the locked decision so a
128
+ // limits pause never consumes continuation budget.
129
+ const budgetBlocked = await goalRunBudgetBlocked(
130
+ settings,
131
+ db,
132
+ input.accountId,
133
+ input.workspaceId,
134
+ fundedWithoutCredits,
135
+ );
109
136
  const turnExecutionPolicy = resolveTurnExecutionPolicyV1(settings, {
110
137
  modelId: continuationModel,
111
138
  requestedModelId: null,
@@ -130,25 +157,7 @@ export function createGoalActivities(services: () => Promise<ControlActivityServ
130
157
  // A model-policy block takes precedence: it is deterministic (a budget
131
158
  // pause can clear on its own; a policy pause needs a model/policy change)
132
159
  // and rides the same visible-pause channel.
133
- admission: async (tx, causalTurn) => {
134
- const budgetBlocked = modelPolicyBlocked
135
- ? null
136
- : await goalRunBudgetBlocked(
137
- { ...service, settings, db: tx },
138
- {
139
- accountId: input.accountId,
140
- workspaceId: input.workspaceId,
141
- model: continuationModel,
142
- initiatingHumanSubjectId: causalTurn?.initiatingHumanSubjectId ?? null,
143
- },
144
- );
145
- return {
146
- budgetBlocked: modelPolicyBlocked ?? budgetBlocked?.message ?? null,
147
- budgetPausedReason: modelPolicyBlocked
148
- ? "limits"
149
- : (budgetBlocked?.pausedReason ?? "limits"),
150
- };
151
- },
160
+ budgetBlocked: modelPolicyBlocked ?? budgetBlocked,
152
161
  policy: {
153
162
  model: continuationModel,
154
163
  reasoningEffort: continuationReasoningEffort,
@@ -298,8 +307,6 @@ export function goalContinuationPrompt(
298
307
  "- For document report deliverables (a document the user asked for, or a large report meant to be kept or shared), follow the Documents Skill: create the durable native document first, inspect its relevant final head after the last edit, and provide the returned artifact reference. Declare report requirements through the available goal tools before authoring and satisfy every persisted report requirement with verified artifact delivery evidence before completion. Sandbox paths and raw file IDs do not prove report delivery. If artifact tooling or access is unavailable, keep that deliverable incomplete and state the blocker; never invent proof or silently substitute a local report. Ordinary chat answers, short progress updates, source-code links, and explicitly requested local-file work remain outside this report contract.",
299
308
  "",
300
309
  "Do not rely on intent, partial progress, memory of earlier work, or a plausible final answer as proof of completion. Call opengeni__goal_complete with concrete evidence only when the full objective is actually achieved and no required work remains.",
301
- "Goal evidence is a short proof for the ledger, not the deliverable. After goal_complete succeeds, finish this same turn with the requested user-facing answer, or a concise summary and retained artifact link. Goal completion stops future automatic continuations; it does not send the answer or end this turn. Never compress a report into evidence or omit the final reply.",
302
- "Goal progress notes are short human-readable milestone statuses, not raw transcripts or continuation instructions. Keep normal spaces and summarize detail instead of squeezing words into a ledger field. The text and successCriteria fields each allow 8192 UTF-8 bytes, progressNote allows 8192 UTF-8 bytes, rationale allows 2048 UTF-8 bytes, and evidence allows 8192 characters.",
303
310
  "",
304
311
  ...waitingGuidance,
305
312
  ...childNoticeGuidance,
@@ -329,22 +336,55 @@ export function withFirstPartyTools(settings: Settings, tools: ToolRef[]): ToolR
329
336
  }
330
337
 
331
338
  /**
332
- * Goals share scheduled admission and pause visibly without synthesizing work.
339
+ * Non-throwing variant of the scheduled-run admission check: returns a human
340
+ * readable reason when balance or monthly caps block another agent run.
333
341
  */
334
- export async function goalRunBudgetBlocked(
335
- services: Parameters<typeof agentRunAdmissionDenial>[0],
336
- input: Omit<Parameters<typeof agentRunAdmissionDenial>[1], "requestedAgentRuns">,
337
- ): Promise<{ pausedReason: "limits" | "allowance"; message: string } | null> {
338
- const denial = await agentRunAdmissionDenial(services, { ...input, requestedAgentRuns: 1 });
339
- if (denial === null) return null;
340
- if (denial === "allowance_exhausted") {
341
- return { pausedReason: "allowance", message: "OpenGeni usage allowance exhausted" };
342
+ async function goalRunBudgetBlocked(
343
+ settings: Settings,
344
+ db: Database,
345
+ accountId: string,
346
+ workspaceId: string,
347
+ fundedWithoutCredits: boolean,
348
+ ): Promise<string | null> {
349
+ // Free, subscription, and workspace-funded continuations skip OpenGeni's
350
+ // credit-balance gate and monthly model-cost cap. The agent-run COUNT cap
351
+ // below is a volume quota (not a credit/cost gate) and remains enforced.
352
+ if (
353
+ !fundedWithoutCredits &&
354
+ (settings.billingMode === "stripe" || settings.usageLimitsMode === "managed")
355
+ ) {
356
+ const balance = await getBillingBalance(db, accountId);
357
+ if (balance.balanceMicros <= 0) {
358
+ return "insufficient OpenGeni credits";
359
+ }
342
360
  }
343
- const limits = configuredStaticUsageLimits(services.settings);
344
- const messages = {
345
- insufficient_credits: "insufficient OpenGeni credits",
346
- monthly_model_cost_limit: `monthly model cost limit reached (${limits.maxMonthlyCostMicrosPerAccount} micros)`,
347
- monthly_agent_run_limit: `monthly agent run limit reached (${limits.maxMonthlyAgentRunsPerWorkspace})`,
348
- };
349
- return { pausedReason: "limits", message: messages[denial] };
361
+ if (settings.usageLimitsMode === "static" || settings.usageLimitsMode === "managed") {
362
+ const limits = configuredStaticUsageLimits(settings);
363
+ if (!fundedWithoutCredits && limits.maxMonthlyCostMicrosPerAccount) {
364
+ const used = await sumUsageQuantity(db, {
365
+ accountId,
366
+ eventType: "model.cost",
367
+ since: startOfUtcMonth(),
368
+ });
369
+ if (used >= limits.maxMonthlyCostMicrosPerAccount) {
370
+ return `monthly model cost limit reached (${limits.maxMonthlyCostMicrosPerAccount} micros)`;
371
+ }
372
+ }
373
+ if (limits.maxMonthlyAgentRunsPerWorkspace) {
374
+ const used = await sumUsageQuantity(db, {
375
+ workspaceId,
376
+ eventType: "agent_run.created",
377
+ since: startOfUtcMonth(),
378
+ });
379
+ if (used + 1 > limits.maxMonthlyAgentRunsPerWorkspace) {
380
+ return `monthly agent run limit reached (${limits.maxMonthlyAgentRunsPerWorkspace})`;
381
+ }
382
+ }
383
+ }
384
+ return null;
385
+ }
386
+
387
+ function startOfUtcMonth(): Date {
388
+ const now = new Date();
389
+ return new Date(Date.UTC(now.getUTCFullYear(), now.getUTCMonth(), 1));
350
390
  }