@opengeni/worker-bundle 2.0.1 → 2.0.3-canary.36199476632001

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/dist/activities/agent-turn/agent-build.d.ts +2 -0
  2. package/dist/activities/agent-turn/code-search.d.ts +60 -0
  3. package/dist/activities/agent-turn/codex-capacity.d.ts +13 -0
  4. package/dist/activities/agent-turn/errors.d.ts +12 -1
  5. package/dist/activities/agent-turn/failure-settlement.d.ts +1 -1
  6. package/dist/activities/agent-turn/governance-model.d.ts +2 -0
  7. package/dist/activities/agent-turn/history.d.ts +2 -0
  8. package/dist/activities/agent-turn/media-artifacts.d.ts +5 -3
  9. package/dist/activities/agent-turn/session-title.d.ts +32 -2
  10. package/dist/activities/agent-turn/tool-environment.d.ts +4 -0
  11. package/dist/activities/context-compaction.d.ts +10 -2
  12. package/dist/activities/knowledge-indexing.d.ts +19 -0
  13. package/dist/activities/retained-screenshots.d.ts +7 -2
  14. package/dist/activities/sandbox-lease.d.ts +3 -4
  15. package/dist/{activities-control-QA6OLRQY.js → activities-control-I72CRMAE.js} +283 -35
  16. package/dist/activities-control-I72CRMAE.js.map +1 -0
  17. package/dist/{activities-turn-YVVKCGAK.js → activities-turn-CSDKK4N3.js} +531 -77
  18. package/dist/activities-turn-CSDKK4N3.js.map +1 -0
  19. package/dist/artifact-outbox-entry.js +3 -0
  20. package/dist/artifact-outbox-entry.js.map +1 -1
  21. package/dist/{chunk-M3NVMF2V.js → chunk-7M6G2SCF.js} +172 -17
  22. package/dist/chunk-7M6G2SCF.js.map +1 -0
  23. package/dist/{chunk-I7HKKMUJ.js → chunk-DSPP6CZL.js} +48 -3
  24. package/dist/chunk-DSPP6CZL.js.map +1 -0
  25. package/dist/index.js +10 -5
  26. package/dist/index.js.map +1 -1
  27. package/dist/observability-metrics.d.ts +11 -1
  28. package/dist/sandbox-resume.d.ts +46 -0
  29. package/dist/workflow-bundle.js +80 -2
  30. package/dist/workflows/session.d.ts +30 -0
  31. package/package.json +20 -19
  32. package/src/activities/agent-turn/agent-build.ts +4 -0
  33. package/src/activities/agent-turn/code-search.ts +275 -0
  34. package/src/activities/agent-turn/codex-capacity.ts +37 -2
  35. package/src/activities/agent-turn/compaction-prep.ts +52 -16
  36. package/src/activities/agent-turn/errors.ts +67 -4
  37. package/src/activities/agent-turn/failure-settlement.ts +40 -8
  38. package/src/activities/agent-turn/governance-model.ts +17 -1
  39. package/src/activities/agent-turn/history.ts +14 -1
  40. package/src/activities/agent-turn/media-artifacts.ts +2 -1
  41. package/src/activities/agent-turn/run.ts +4 -0
  42. package/src/activities/agent-turn/sandbox-establish.ts +12 -1
  43. package/src/activities/agent-turn/sandbox-runtime.ts +25 -9
  44. package/src/activities/agent-turn/session-title.ts +73 -3
  45. package/src/activities/agent-turn/stream-attempt.ts +27 -15
  46. package/src/activities/agent-turn/tool-environment.ts +65 -0
  47. package/src/activities/context-compaction.ts +49 -14
  48. package/src/activities/knowledge-indexing.ts +240 -15
  49. package/src/activities/retained-screenshots.ts +82 -11
  50. package/src/activities/sandbox-lease.ts +141 -21
  51. package/src/activities/scheduled-tasks.ts +18 -2
  52. package/src/activity-services.ts +16 -4
  53. package/src/editable-artifact-hint-broker.ts +3 -0
  54. package/src/index.ts +2 -2
  55. package/src/observability-metrics.ts +70 -1
  56. package/src/personal-github-git-credentials.ts +2 -0
  57. package/src/sandbox-resume.ts +211 -5
  58. package/src/workflows/activities.ts +14 -1
  59. package/src/workflows/session.ts +102 -6
  60. package/dist/activities-control-QA6OLRQY.js.map +0 -1
  61. package/dist/activities-turn-YVVKCGAK.js.map +0 -1
  62. package/dist/chunk-I7HKKMUJ.js.map +0 -1
  63. package/dist/chunk-M3NVMF2V.js.map +0 -1
@@ -16,11 +16,17 @@ import {
16
16
  type CompactionProviderRejection,
17
17
  isMcpRequestTimeoutError,
18
18
  isMcpTransportConnectivityError,
19
- isModalTaskExecStartDnsResolutionError,
19
+ isModalTaskExecStartPreDispatchUnavailableError,
20
+ isRoutingMutationOutcomeUnknownError,
20
21
  RoutingWorkspaceRootChangedError,
21
22
  SandboxMaterializationVerificationError,
22
23
  materializationVerificationDiagnostic,
23
24
  type MaterializationVerificationDiagnostic,
25
+ PROVIDER_QUOTA_EXHAUSTED_CODE,
26
+ type ProviderQuotaExhaustion,
27
+ type ProviderQuotaScope,
28
+ classifyProviderQuotaError,
29
+ providerQuotaExhaustedMessage,
24
30
  SelfhostedWorkspaceRootChangedError,
25
31
  UNKNOWN_MODEL_FINISH_REASON_CODE,
26
32
  } from "@opengeni/runtime";
@@ -33,6 +39,7 @@ import { CODEX_USAGE_EXHAUSTED_PCT } from "../codex-rotation";
33
39
  import { RetainedAttachmentTransportLimitError } from "../run-input";
34
40
  import type { CodexAccountStatus } from "@opengeni/db";
35
41
  import {
42
+ CODEX_USAGE_LIMIT_ERROR_TYPE,
36
43
  CodexReloginRequired,
37
44
  classifyCodexEncryptedArtifactRejection,
38
45
  classifyCodexResponseTimeoutError,
@@ -119,6 +126,7 @@ export function providerRecoveryResult(input: {
119
126
  input.failureCode === "sandbox_command_start_unavailable" ||
120
127
  input.failureCode === "mcp_transport_timeout" ||
121
128
  input.failureCode === "mcp_transport_unavailable" ||
129
+ input.failureCode === "turn_execution_policy_definition_mismatch" ||
122
130
  input.failureCode === POST_COMPACTION_CONTINUATION_EMPTY_CODE
123
131
  ? Math.max(
124
132
  providerDelay ?? 0,
@@ -612,6 +620,14 @@ export function compactionFailureReasonFromError(error: unknown): string {
612
620
  `the model provider rejected the compaction request (${describeCompactionProviderRejection(rejection)}). Active history was preserved. ${COMPACTION_PROVIDER_REJECTION_GUIDANCE}`,
613
621
  );
614
622
  }
623
+ // An exhausted provider quota is not retried (see agentRunFailurePayload),
624
+ // so name the refusal plainly instead of the raw diagnostic envelope.
625
+ const quota = classifyProviderQuotaExhaustionError(error);
626
+ if (quota) {
627
+ return compactionFailureReason(
628
+ `${providerQuotaExhaustedMessage(quota.scope)} Active history was preserved.`,
629
+ );
630
+ }
615
631
  if (
616
632
  error instanceof CompactionProviderResponseError ||
617
633
  error instanceof EmptyCompactionSummaryError
@@ -638,8 +654,12 @@ export function compactionFailureTurnEventPayload(
638
654
  recovery: "user_message";
639
655
  compacted: false;
640
656
  providerRejection?: CompactionProviderRejection;
657
+ quotaScope?: ProviderQuotaScope;
641
658
  } {
642
659
  const rejection = compactionProviderRejection(error);
660
+ // The same closed marker as a `provider_quota_exhausted` turn failure, so
661
+ // clients can name the exhausted limit and offer another model here too.
662
+ const quota = rejection ? null : classifyProviderQuotaExhaustionError(error);
643
663
  return {
644
664
  error: overrides.error ?? compactionFailureReasonFromError(error),
645
665
  code: "context_compaction_failed",
@@ -647,6 +667,7 @@ export function compactionFailureTurnEventPayload(
647
667
  recovery: "user_message",
648
668
  compacted: false,
649
669
  ...(rejection ? { providerRejection: rejection } : {}),
670
+ ...(quota ? { quotaScope: quota.scope } : {}),
650
671
  };
651
672
  }
652
673
 
@@ -844,6 +865,27 @@ export function isTransientProviderError(error: unknown): boolean {
844
865
  );
845
866
  }
846
867
 
868
+ /** An explicit `usage_limit_reached` type or code string anywhere on the error chain. */
869
+ function hasCodexUsageLimitType(error: unknown): boolean {
870
+ return collectErrorStrings(error).some((value) => value.includes(CODEX_USAGE_LIMIT_ERROR_TYPE));
871
+ }
872
+
873
+ /**
874
+ * Recognize an exhausted API-key provider quota (a daily or monthly allowance,
875
+ * a free-tier day cap, or an account out of credits) as distinct from an
876
+ * ordinary per-minute rate limit. Retrying within the bounded same-turn budget
877
+ * cannot succeed, so the turn fails promptly instead. Subscription transports
878
+ * own their quota semantics through credential rotation and durable capacity
879
+ * waits, so a Codex or SuperGrok transport error never classifies here.
880
+ */
881
+ export function classifyProviderQuotaExhaustionError(
882
+ error: unknown,
883
+ ): ProviderQuotaExhaustion | null {
884
+ if (isCodexTransportError(error) || isXaiSubscriptionTransportError(error)) return null;
885
+ // The same reader the OpenAI SDK retry veto uses, so the two never disagree.
886
+ return classifyProviderQuotaError(error);
887
+ }
888
+
847
889
  export type XaiCredentialFailure = {
848
890
  kind: "auth" | "forbidden" | "rate_limit";
849
891
  cooldownMs: number | null;
@@ -950,6 +992,7 @@ function baseAgentRunFailurePayload(
950
992
  historyPersistenceStage?: MandatoryHistoryPersistenceStage;
951
993
  mcpTransportDiagnostic?: McpTransportRequestFailureDiagnostic;
952
994
  materializationDiagnostic?: MaterializationVerificationDiagnostic;
995
+ quotaScope?: ProviderQuotaScope;
953
996
  } {
954
997
  if (error instanceof SandboxMaterializationVerificationError) {
955
998
  return {
@@ -1021,10 +1064,13 @@ function baseAgentRunFailurePayload(
1021
1064
  retryable: true,
1022
1065
  };
1023
1066
  }
1024
- if (isModalTaskExecStartDnsResolutionError(error)) {
1067
+ if (
1068
+ !isRoutingMutationOutcomeUnknownError(error) &&
1069
+ isModalTaskExecStartPreDispatchUnavailableError(error)
1070
+ ) {
1025
1071
  return {
1026
1072
  error:
1027
- "The managed sandbox command transport was temporarily unreachable before the command started. The same turn will retry after a short delay.",
1073
+ "The managed sandbox command router was not ready before the command was sent. The same turn will retry after a short delay.",
1028
1074
  code: "sandbox_command_start_unavailable",
1029
1075
  retryable: true,
1030
1076
  };
@@ -1099,8 +1145,11 @@ function baseAgentRunFailurePayload(
1099
1145
  // `usage_limit_reached` shape must still outrank generic 429 retryability.
1100
1146
  // Credential quarantine/failover remains separately provenance-gated by
1101
1147
  // `isCodexTransportError`; this branch only chooses the truthful user payload.
1148
+ // The looser "429 ... usage limit" wording counts only on a Codex transport
1149
+ // error: an API-key provider's 429 that says "usage limit" is provider quota
1150
+ // evidence, not a ChatGPT/Codex subscription cap.
1102
1151
  const usageLimit = classifyCodexUsageLimitError(error);
1103
- if (usageLimit) {
1152
+ if (usageLimit && (isCodexTransportError(error) || hasCodexUsageLimitType(error))) {
1104
1153
  return codexUsageLimitFailurePayload(usageLimit, message);
1105
1154
  }
1106
1155
  const codexTimeout = classifyCodexResponseTimeoutError(error, {
@@ -1152,6 +1201,20 @@ function baseAgentRunFailurePayload(
1152
1201
  retryable: true,
1153
1202
  };
1154
1203
  }
1204
+ // An exhausted quota also arrives as HTTP 429 (or 402), but no retry within
1205
+ // the finite same-turn budget can succeed. Fail the turn promptly with a
1206
+ // distinct code so the client can offer another model; ordinary short rate
1207
+ // limits fall through to the retryable branch below.
1208
+ const quota = classifyProviderQuotaExhaustionError(error);
1209
+ if (quota) {
1210
+ return {
1211
+ error: providerQuotaExhaustedMessage(quota.scope),
1212
+ code: PROVIDER_QUOTA_EXHAUSTED_CODE,
1213
+ retryable: false,
1214
+ quotaScope: quota.scope,
1215
+ ...(message ? { detail: message } : {}),
1216
+ };
1217
+ }
1155
1218
  if (
1156
1219
  status === 429 ||
1157
1220
  code === "rate_limit_exceeded" ||
@@ -16,14 +16,14 @@ import {
16
16
  } from "@opengeni/db";
17
17
  import { publishDurableSessionEvents } from "@opengeni/events";
18
18
  import { maxTurnsExceededRunState } from "@opengeni/runtime";
19
- import { CancelledFailure } from "@temporalio/activity";
19
+ import { ApplicationFailure, CancelledFailure } from "@temporalio/activity";
20
20
  import {
21
21
  authoritativeCodexCapacityResetAt,
22
22
  classifyCodexPin,
23
23
  selectCodexCredentialLeaseForTurn,
24
24
  type CodexRotationStrategy,
25
25
  } from "../codex-rotation";
26
- import type { Settings } from "@opengeni/config";
26
+ import { TurnExecutionPolicyDefinitionMismatchError, type Settings } from "@opengeni/config";
27
27
  import {
28
28
  classifyCodexEncryptedArtifactRejection,
29
29
  classifyCodexUsageLimitError,
@@ -1499,10 +1499,28 @@ export async function settleTurnFailure(deps: TurnFailureDeps): Promise<RunAgent
1499
1499
  // truth, recover this SAME accepted turn, then let the workflow re-claim
1500
1500
  // it after a pacing delay. This is independent of goal state and never
1501
1501
  // relies on a synthetic continuation prompt.
1502
- let failure = agentRunFailurePayload(error, {
1503
- isCodexTurn: billingState.isCodexTurn,
1504
- }) as ReturnType<typeof agentRunFailurePayload>;
1505
- if (failure.retryable && eventing.publish && attempt.turnId && eventing.turnStartedPublished) {
1502
+ // A rolling-deployment definition mismatch is a separate configuration
1503
+ // class: only the exact typed setup error can use this checkpoint before
1504
+ // eventing exists. No generic setup/credential failure gains retry authority.
1505
+ const earlyDefinitionMismatch =
1506
+ error instanceof TurnExecutionPolicyDefinitionMismatchError &&
1507
+ !attempt.modelRequestStarted &&
1508
+ !eventing.turnStartedPublished &&
1509
+ !!attempt.turnId &&
1510
+ !!attempt.triggerEventId &&
1511
+ attempt.executionGeneration > 0;
1512
+ let failure = (
1513
+ earlyDefinitionMismatch
1514
+ ? { error: error.message, code: error.code, retryable: true }
1515
+ : agentRunFailurePayload(error, {
1516
+ isCodexTurn: billingState.isCodexTurn,
1517
+ })
1518
+ ) as ReturnType<typeof agentRunFailurePayload>;
1519
+ if (
1520
+ attempt.turnId &&
1521
+ (earlyDefinitionMismatch ||
1522
+ (failure.retryable && eventing.publish && eventing.turnStartedPublished))
1523
+ ) {
1506
1524
  const nextProviderRecoveryCount = attempt.providerRecoveryCount + 1;
1507
1525
  const recoveryResult = providerRecoveryResult({
1508
1526
  failureCode: failure.code,
@@ -1511,8 +1529,10 @@ export async function settleTurnFailure(deps: TurnFailureDeps): Promise<RunAgent
1511
1529
  });
1512
1530
  try {
1513
1531
  if (recoveryResult.status === "recovering") {
1514
- await flushRuntimeBatcher();
1515
- await historySink.reconcileConversationTruth({ requireDurable: true });
1532
+ if (!earlyDefinitionMismatch) {
1533
+ await flushRuntimeBatcher();
1534
+ await historySink.reconcileConversationTruth({ requireDurable: true });
1535
+ }
1516
1536
  const recovery = await requestSessionTurnRecovery(db, input.workspaceId, {
1517
1537
  sessionId: input.sessionId,
1518
1538
  turnId: attempt.turnId,
@@ -1540,6 +1560,18 @@ export async function settleTurnFailure(deps: TurnFailureDeps): Promise<RunAgent
1540
1560
  return claimedResult(recoveryResult);
1541
1561
  }
1542
1562
  failure = providerRecoveryExhaustedFailure(failure, recoveryResult);
1563
+ if (earlyDefinitionMismatch) {
1564
+ // Setup has no eventing sink yet. Carry only the fixed, safe diagnostic
1565
+ // through Temporal into exact-attempt workflow failure settlement.
1566
+ control.activityStatus = "failed";
1567
+ control.turnMetricOutcome = "failed";
1568
+ control.activityError = error;
1569
+ throw ApplicationFailure.create({
1570
+ message: `${error.message}. Automatic same-turn configuration recovery exhausted after ${recoveryResult.providerRecoveryCount} retries.`,
1571
+ type: "TurnExecutionPolicyDefinitionMismatchError",
1572
+ nonRetryable: true,
1573
+ });
1574
+ }
1543
1575
  } catch (recoveryError) {
1544
1576
  const escaped =
1545
1577
  recoveryResult.status === "recovering"
@@ -20,7 +20,11 @@ import {
20
20
  renderWorkspaceGovernanceContext,
21
21
  type OpenGeniRuntime,
22
22
  } from "@opengeni/runtime";
23
- import { settingsWithResolvedModelContext, type Settings } from "@opengeni/config";
23
+ import {
24
+ codeSearchDeploymentPolicy,
25
+ settingsWithResolvedModelContext,
26
+ type Settings,
27
+ } from "@opengeni/config";
24
28
  import { projectReasoningConfigurations, supportsReasoningConfiguration } from "@opengeni/codex";
25
29
  import { settingsWithSessionMcpServersForRun } from "../capabilities";
26
30
  import { resolveRigProviderImageForRun } from "@opengeni/core";
@@ -46,6 +50,7 @@ import {
46
50
  resolveWorkspaceAgentHumanInputEnabled,
47
51
  type MediaGenerationResult,
48
52
  } from "@opengeni/contracts";
53
+ import { codeSearchEnabledForTurn } from "@opengeni/contracts/code-search";
49
54
 
50
55
  import { assertWorkspaceHumanInputAllowed } from "./admission";
51
56
  import {
@@ -89,6 +94,8 @@ export type GovernanceModelOk = {
89
94
  | null;
90
95
  rigName: string | null;
91
96
  agentHumanInputEnabled: boolean;
97
+ /** Deployment and workspace allow the Jev-backed code_search tool. */
98
+ codeSearchEnabled: boolean;
92
99
  workspaceAgentInstructions: string | null | undefined;
93
100
  workspaceGovernance: ReturnType<typeof renderWorkspaceGovernanceContext>;
94
101
  structuredWorkspacePolicyActive: boolean;
@@ -230,6 +237,14 @@ export async function prepareGovernanceAndModel(
230
237
  workspaceRefs.rigVersionId = session.rigVersionId ?? "";
231
238
  if (!workspace) throw new Error(`Workspace not found: ${input.workspaceId}`);
232
239
  const agentHumanInputEnabled = resolveWorkspaceAgentHumanInputEnabled(workspace.settings);
240
+ // The session's decision was frozen when it was created, so only a
241
+ // deliberate switch-off (deployment or workspace Off), or undoing one,
242
+ // changes its tool list.
243
+ const codeSearchEnabled = codeSearchEnabledForTurn(
244
+ session.codeSearchEnabled,
245
+ workspace.settings,
246
+ codeSearchDeploymentPolicy(capabilitySettings),
247
+ );
233
248
  const contextSelection = await resolveCompanyBrainContextSelection(db, governanceClaims);
234
249
  const workspaceAgentInstructions = contextSelection.legacyWorkspaceInstructions;
235
250
  const memoryPromptMode = contextSelection.receipt.memoryPromptMode;
@@ -460,6 +475,7 @@ export async function prepareGovernanceAndModel(
460
475
  rigVersion,
461
476
  rigName,
462
477
  agentHumanInputEnabled,
478
+ codeSearchEnabled,
463
479
  workspaceAgentInstructions,
464
480
  workspaceGovernance,
465
481
  structuredWorkspacePolicyActive,
@@ -186,7 +186,20 @@ export function toolCallProducesRetainableSessionImage(name: string | null): boo
186
186
  return (
187
187
  name === "computer_screenshot" ||
188
188
  name === "view_image" ||
189
- name === "interaction__computer_observe"
189
+ name === "interaction__computer_observe" ||
190
+ retainableBrowserScreenshotToolCall(name)
191
+ );
192
+ }
193
+
194
+ /** Browser interactions can carry an image block, including opt-in observe frames. */
195
+ export function retainableBrowserScreenshotToolCall(name: string | null): boolean {
196
+ return (
197
+ name === "browser_screenshot" ||
198
+ name === "interaction__browser_screenshot" ||
199
+ name === "browser_observe" ||
200
+ name === "interaction__browser_observe" ||
201
+ name === "browser_act" ||
202
+ name === "interaction__browser_act"
190
203
  );
191
204
  }
192
205
 
@@ -12,7 +12,7 @@ import {
12
12
  type TurnToolCancellationFence,
13
13
  } from "@opengeni/runtime";
14
14
  import type { Settings } from "@opengeni/config";
15
- import type { RetainedArtifactMetadata } from "@opengeni/contracts";
15
+ import type { RetainedArtifactMetadata, RetainedSessionScreenshotKind } from "@opengeni/contracts";
16
16
  import type { ResumedTurnSandbox } from "../../sandbox-resume";
17
17
  import type { SharedActivityServices } from "../types";
18
18
  import {
@@ -79,6 +79,7 @@ export class TurnMediaArtifacts {
79
79
  sandboxFileDownloadBackend: Settings["sandboxBackend"];
80
80
  nativeImageGenerationRetention: NativeImageGenerationRetention | null = null;
81
81
  readonly retainedSessionImageCallIds = new Set<string>();
82
+ readonly retainedSessionImageKindsByCallId = new Map<string, RetainedSessionScreenshotKind>();
82
83
 
83
84
  constructor(private readonly deps: TurnMediaArtifactDeps) {
84
85
  this.sandboxFileDownloadBackend = deps.getModelRunSettings().sandboxBackend;
@@ -535,6 +535,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
535
535
  rigVersion,
536
536
  rigName,
537
537
  agentHumanInputEnabled,
538
+ codeSearchEnabled,
538
539
  workspaceAgentInstructions,
539
540
  workspaceGovernance,
540
541
  structuredWorkspacePolicyActive,
@@ -1278,6 +1279,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
1278
1279
  turnExecutionPolicy,
1279
1280
  trigger,
1280
1281
  runSettings,
1282
+ resolvedModel,
1281
1283
  lazyToolTransport,
1282
1284
  turnTools,
1283
1285
  connectionScope,
@@ -1289,6 +1291,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
1289
1291
  credentialSubjectId,
1290
1292
  interactionInterventionResume,
1291
1293
  runWorkspaceMutationForSandbox,
1294
+ codeSearchEnabled,
1292
1295
  throwIfWorkerShuttingDown,
1293
1296
  throwIfTurnCancelled,
1294
1297
  });
@@ -1352,6 +1355,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
1352
1355
  connectorActionPolicy,
1353
1356
  trigger,
1354
1357
  preparationIndependentToolNames,
1358
+ codeSearchAvailable: toolRuntime.codeSearchAvailable,
1355
1359
  videoGenerationAcceptancesByCallId,
1356
1360
  activeSandboxBackend,
1357
1361
  groupBoxBackend,
@@ -35,7 +35,11 @@ import {
35
35
  import { rigProviderImageSourceImage } from "../sandbox-images";
36
36
  import type { TurnActivityServices as ActivityServices, RunAgentTurnInput } from "../types";
37
37
  import type { currentActivityContext } from "../streaming";
38
- import { resumeBoxForTurn, type ResumedTurnSandbox } from "../../sandbox-resume";
38
+ import {
39
+ createFreshSandboxReadinessReplacementBudget,
40
+ resumeBoxForTurn,
41
+ type ResumedTurnSandbox,
42
+ } from "../../sandbox-resume";
39
43
  import {
40
44
  wrapTurnBoxWithRouting,
41
45
  wrapLazyTurnBoxWithRouting,
@@ -298,6 +302,9 @@ export async function resolveSandboxRoute(deps: SandboxRouteDeps): Promise<Sandb
298
302
  }
299
303
 
300
304
  export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Promise<void> {
305
+ // One fresh-box readiness replacement per turn attempt, shared by the eager
306
+ // establish and every lazy provisioner retry of this attempt.
307
+ const freshSandboxReadinessReplacementBudget = createFreshSandboxReadinessReplacementBudget();
301
308
  const {
302
309
  input,
303
310
  settings,
@@ -592,6 +599,8 @@ export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Prom
592
599
  logicalFallbackSettings: logicalSandboxSettings,
593
600
  cancellationSignal: sandboxResumeSignal,
594
601
  sandboxMetrics: runtimeMetricsHooksForObservability(observability),
602
+ observability,
603
+ freshSandboxReadinessReplacementBudget,
595
604
  onSandboxLost: publishSandboxLost,
596
605
  objectStorage,
597
606
  },
@@ -667,6 +676,8 @@ export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Prom
667
676
  logicalFallbackSettings: logicalSandboxSettings,
668
677
  cancellationSignal: sandboxResumeSignal,
669
678
  sandboxMetrics: runtimeMetricsHooksForObservability(observability),
679
+ observability,
680
+ freshSandboxReadinessReplacementBudget,
670
681
  onSandboxLost: publishSandboxLost,
671
682
  objectStorage,
672
683
  },
@@ -1,7 +1,7 @@
1
1
  import {
2
2
  advanceWorkspaceGeneration,
3
3
  verifyWorkspaceMutationSettlement,
4
- heartbeatLeaseHolder,
4
+ heartbeatLeaseHolderStatus,
5
5
  readLease,
6
6
  accrueWarmSeconds,
7
7
  SandboxWorkspaceMutationFencedError,
@@ -526,12 +526,15 @@ export function createSandboxTurnRuntime(deps: SandboxTurnRuntimeDeps) {
526
526
  // home turn that degraded to the cloud group box (swap-away / flag-off), that
527
527
  // is the deployment default (modal), so the fallback box is warm-metered at
528
528
  // the cloud rate instead of selfhosted's rate-0 (which would under-bill).
529
- const warmRate = sandboxWarmRateMicrosPerSecond(
530
- settings,
531
- warmBackend ?? (sandbox.established.backendId as Settings["sandboxBackend"]),
532
- );
529
+ const warmRate =
530
+ settings.sandboxWarmBillingMode === "usage_only"
531
+ ? 0
532
+ : sandboxWarmRateMicrosPerSecond(
533
+ settings,
534
+ warmBackend ?? (sandbox.established.backendId as Settings["sandboxBackend"]),
535
+ );
533
536
  sandboxState.leaseHeartbeatTimer = setInterval(() => {
534
- void heartbeatLeaseHolder(db, {
537
+ void heartbeatLeaseHolderStatus(db, {
535
538
  accountId: input.accountId,
536
539
  workspaceId: input.workspaceId,
537
540
  sandboxGroupId: heartbeatGroupId,
@@ -539,9 +542,17 @@ export function createSandboxTurnRuntime(deps: SandboxTurnRuntimeDeps) {
539
542
  holderId: heartbeatHolderId,
540
543
  leaseTtlMs: settings.sandboxLeaseTtlMs,
541
544
  expectedEpoch: heartbeatEpoch,
545
+ billingMode: settings.sandboxWarmBillingMode,
542
546
  })
543
- .then(async (alive) => {
544
- if (alive) return;
547
+ .then(async (status) => {
548
+ if (status.fence === "funding") {
549
+ stopLeaseHeartbeat();
550
+ sandboxRotationController.abort(
551
+ new Error("Insufficient OpenGeni credits to extend paid sandbox compute"),
552
+ );
553
+ return;
554
+ }
555
+ if (status.leaseExtended) return;
545
556
  const rotation = await beginRotationPreemption(sandbox, heartbeatEpoch, heartbeatGroupId);
546
557
  if (rotation === "not_rotating") {
547
558
  // The holder was reaped, the exact attempt closed, the epoch was
@@ -557,9 +568,14 @@ export function createSandboxTurnRuntime(deps: SandboxTurnRuntimeDeps) {
557
568
  sandboxGroupId: heartbeatGroupId,
558
569
  expectedEpoch: heartbeatEpoch,
559
570
  warmRateMicrosPerSecond: warmRate,
571
+ billingMode: settings.sandboxWarmBillingMode,
560
572
  subjectId: input.sessionId,
561
573
  })
562
- .then((result) => recordCreditMicros(observability, "usage", result.costMicros))
574
+ .then((result) => {
575
+ if (settings.sandboxWarmBillingMode === "credits") {
576
+ recordCreditMicros(observability, "usage", result.costMicros);
577
+ }
578
+ })
563
579
  .catch(() => undefined);
564
580
  // MID-SESSION snapshot (sandbox-file-persistence): while the turn holds
565
581
  // the box, fold a fresh /workspace snapshot onto the lease every
@@ -1,9 +1,15 @@
1
1
  import { hasPermission } from "@opengeni/core";
2
2
  import type { AttemptToolDefinition } from "@opengeni/codemode";
3
- import type { GeneratedSessionTitle } from "@opengeni/runtime";
3
+ import { isManagedOpenRouterFreeRoute, type ModelCapabilitiesV1 } from "@opengeni/config";
4
+ import type {
5
+ GeneratedSessionTitle,
6
+ GenerateSessionTitleOptions,
7
+ OpenGeniRuntime,
8
+ } from "@opengeni/runtime";
4
9
  import {
5
10
  AUTOMATIC_SESSION_TITLE_FALLBACK,
6
11
  DEFAULT_FIRST_PARTY_MCP_PERMISSIONS,
12
+ ReasoningEffort,
7
13
  type FirstPartyMcpToolName,
8
14
  type Permission,
9
15
  type ToolRef,
@@ -29,11 +35,26 @@ export function shouldRequestMissingSessionTitle(input: {
29
35
  return hasPermission([...permissions], "sessions:control");
30
36
  }
31
37
 
38
+ /**
39
+ * Whether the turn's route can afford a model request spent only on a title.
40
+ * The managed OpenRouter free route draws on one deployment-wide per-minute
41
+ * and per-day request quota that users' turns need, so an untitled session on
42
+ * it gets no title sidecar and no title tool (whose call would cost a
43
+ * follow-up request). Clients keep showing the prompt preview, and a later
44
+ * turn on another route titles the session.
45
+ */
46
+ export function routeAllowsSessionTitleRequests(
47
+ resolvedModel: ReturnType<OpenGeniRuntime["resolveTurnModel"]>,
48
+ ): boolean {
49
+ return !resolvedModel || !isManagedOpenRouterFreeRoute(resolvedModel);
50
+ }
51
+
32
52
  export function sessionTitleToolPlan(input: {
33
53
  tools: readonly ToolRef[];
34
54
  selectedFirstPartyMcpTools: readonly FirstPartyMcpToolName[];
35
55
  shouldRequestTitle: boolean;
36
56
  parallelGenerationAvailable: boolean;
57
+ routeAllowsTitleRequests: boolean;
37
58
  }): {
38
59
  promoteTitleTool: boolean;
39
60
  generateTitleInParallel: boolean;
@@ -43,8 +64,9 @@ export function sessionTitleToolPlan(input: {
43
64
  const titleToolAvailable =
44
65
  input.shouldRequestTitle &&
45
66
  input.tools.some((tool) => tool.kind === "mcp" && tool.id === "opengeni");
46
- const generateTitleInParallel = titleToolAvailable && input.parallelGenerationAvailable;
47
- const promoteTitleTool = titleToolAvailable && !generateTitleInParallel;
67
+ const titleRequestAllowed = titleToolAvailable && input.routeAllowsTitleRequests;
68
+ const generateTitleInParallel = titleRequestAllowed && input.parallelGenerationAvailable;
69
+ const promoteTitleTool = titleRequestAllowed && !generateTitleInParallel;
48
70
  return {
49
71
  promoteTitleTool,
50
72
  generateTitleInParallel,
@@ -57,6 +79,54 @@ export function sessionTitleToolPlan(input: {
57
79
 
58
80
  export const PARALLEL_SESSION_TITLE_TIMEOUT_MS = 15_000;
59
81
 
82
+ /**
83
+ * The lowest reasoning effort the resolved model can run, for the auxiliary
84
+ * title request only. A title needs no deliberation, and a provider default
85
+ * effort can use most of the output budget before any visible text. Returns
86
+ * undefined when the model declares no runnable reasoning control, so the
87
+ * request carries no reasoning parameter.
88
+ */
89
+ export function sessionTitleReasoningEffort(
90
+ capabilities: Pick<ModelCapabilitiesV1, "reasoning"> | undefined,
91
+ ): ReasoningEffort | undefined {
92
+ const reasoning = capabilities?.reasoning;
93
+ if (!reasoning?.runnable) return undefined;
94
+ const order = ReasoningEffort.options;
95
+ let lowest: ReasoningEffort | undefined;
96
+ for (const effort of reasoning.efforts) {
97
+ if (!lowest || order.indexOf(effort) < order.indexOf(lowest)) lowest = effort;
98
+ }
99
+ return lowest;
100
+ }
101
+
102
+ /**
103
+ * Options for the parallel title request. It uses the turn's resolved
104
+ * provider and credential authority, but its own lowest runnable reasoning
105
+ * effort rather than the turn's effort.
106
+ */
107
+ export function sessionTitleGenerationOptions(input: {
108
+ resolvedModel: ReturnType<OpenGeniRuntime["resolveTurnModel"]>;
109
+ modelName: string;
110
+ serviceTier: GenerateSessionTitleOptions["serviceTier"] | null | undefined;
111
+ signal: AbortSignal;
112
+ }): GenerateSessionTitleOptions {
113
+ const { resolvedModel, serviceTier } = input;
114
+ const reasoningEffort = sessionTitleReasoningEffort(resolvedModel?.configured.capabilities);
115
+ return {
116
+ ...(resolvedModel
117
+ ? {
118
+ client: resolvedModel.client,
119
+ provider: resolvedModel.provider,
120
+ model: resolvedModel.model,
121
+ }
122
+ : {}),
123
+ modelName: input.modelName,
124
+ ...(serviceTier ? { serviceTier } : {}),
125
+ ...(reasoningEffort ? { reasoningEffort } : {}),
126
+ signal: input.signal,
127
+ };
128
+ }
129
+
60
130
  export type ParallelSessionTitleGeneration = {
61
131
  finish: () => Promise<GeneratedSessionTitle | null>;
62
132
  cancel: () => Promise<void>;