@opengeni/worker-bundle 2.0.3 → 2.1.0-canary.36239117573001

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/dist/activities/agent-turn/agent-build.d.ts +2 -0
  2. package/dist/activities/agent-turn/code-search.d.ts +60 -0
  3. package/dist/activities/agent-turn/codex-capacity.d.ts +13 -0
  4. package/dist/activities/agent-turn/errors.d.ts +12 -1
  5. package/dist/activities/agent-turn/governance-model.d.ts +2 -0
  6. package/dist/activities/agent-turn/session-title.d.ts +32 -2
  7. package/dist/activities/agent-turn/tool-environment.d.ts +4 -0
  8. package/dist/activities/context-compaction.d.ts +10 -2
  9. package/dist/activities/knowledge-indexing.d.ts +19 -0
  10. package/dist/{activities-control-IVTL723D.js → activities-control-I72CRMAE.js} +81 -14
  11. package/dist/activities-control-I72CRMAE.js.map +1 -0
  12. package/dist/{activities-turn-EZEDZTEZ.js → activities-turn-CSDKK4N3.js} +425 -46
  13. package/dist/activities-turn-CSDKK4N3.js.map +1 -0
  14. package/dist/artifact-outbox-entry.js +3 -0
  15. package/dist/artifact-outbox-entry.js.map +1 -1
  16. package/dist/{chunk-JMO5ZUW3.js → chunk-7M6G2SCF.js} +159 -11
  17. package/dist/chunk-7M6G2SCF.js.map +1 -0
  18. package/dist/{chunk-I7HKKMUJ.js → chunk-DSPP6CZL.js} +48 -3
  19. package/dist/chunk-DSPP6CZL.js.map +1 -0
  20. package/dist/index.js +4 -4
  21. package/dist/index.js.map +1 -1
  22. package/dist/observability-metrics.d.ts +11 -1
  23. package/dist/sandbox-resume.d.ts +46 -0
  24. package/dist/workflow-bundle.js +77 -2
  25. package/dist/workflows/session.d.ts +30 -0
  26. package/package.json +20 -19
  27. package/src/activities/agent-turn/agent-build.ts +4 -0
  28. package/src/activities/agent-turn/code-search.ts +275 -0
  29. package/src/activities/agent-turn/codex-capacity.ts +37 -2
  30. package/src/activities/agent-turn/compaction-prep.ts +52 -16
  31. package/src/activities/agent-turn/errors.ts +59 -1
  32. package/src/activities/agent-turn/governance-model.ts +17 -1
  33. package/src/activities/agent-turn/run.ts +4 -0
  34. package/src/activities/agent-turn/sandbox-establish.ts +12 -1
  35. package/src/activities/agent-turn/session-title.ts +73 -3
  36. package/src/activities/agent-turn/stream-attempt.ts +14 -13
  37. package/src/activities/agent-turn/tool-environment.ts +65 -0
  38. package/src/activities/context-compaction.ts +49 -14
  39. package/src/activities/knowledge-indexing.ts +107 -9
  40. package/src/activities/scheduled-tasks.ts +18 -2
  41. package/src/activity-services.ts +16 -4
  42. package/src/editable-artifact-hint-broker.ts +3 -0
  43. package/src/index.ts +2 -2
  44. package/src/observability-metrics.ts +70 -1
  45. package/src/personal-github-git-credentials.ts +2 -0
  46. package/src/sandbox-resume.ts +205 -5
  47. package/src/workflows/session.ts +102 -6
  48. package/dist/activities-control-IVTL723D.js.map +0 -1
  49. package/dist/activities-turn-EZEDZTEZ.js.map +0 -1
  50. package/dist/chunk-I7HKKMUJ.js.map +0 -1
  51. package/dist/chunk-JMO5ZUW3.js.map +0 -1
@@ -22,6 +22,11 @@ import {
22
22
  SandboxMaterializationVerificationError,
23
23
  materializationVerificationDiagnostic,
24
24
  type MaterializationVerificationDiagnostic,
25
+ PROVIDER_QUOTA_EXHAUSTED_CODE,
26
+ type ProviderQuotaExhaustion,
27
+ type ProviderQuotaScope,
28
+ classifyProviderQuotaError,
29
+ providerQuotaExhaustedMessage,
25
30
  SelfhostedWorkspaceRootChangedError,
26
31
  UNKNOWN_MODEL_FINISH_REASON_CODE,
27
32
  } from "@opengeni/runtime";
@@ -34,6 +39,7 @@ import { CODEX_USAGE_EXHAUSTED_PCT } from "../codex-rotation";
34
39
  import { RetainedAttachmentTransportLimitError } from "../run-input";
35
40
  import type { CodexAccountStatus } from "@opengeni/db";
36
41
  import {
42
+ CODEX_USAGE_LIMIT_ERROR_TYPE,
37
43
  CodexReloginRequired,
38
44
  classifyCodexEncryptedArtifactRejection,
39
45
  classifyCodexResponseTimeoutError,
@@ -614,6 +620,14 @@ export function compactionFailureReasonFromError(error: unknown): string {
614
620
  `the model provider rejected the compaction request (${describeCompactionProviderRejection(rejection)}). Active history was preserved. ${COMPACTION_PROVIDER_REJECTION_GUIDANCE}`,
615
621
  );
616
622
  }
623
+ // An exhausted provider quota is not retried (see agentRunFailurePayload),
624
+ // so name the refusal plainly instead of the raw diagnostic envelope.
625
+ const quota = classifyProviderQuotaExhaustionError(error);
626
+ if (quota) {
627
+ return compactionFailureReason(
628
+ `${providerQuotaExhaustedMessage(quota.scope)} Active history was preserved.`,
629
+ );
630
+ }
617
631
  if (
618
632
  error instanceof CompactionProviderResponseError ||
619
633
  error instanceof EmptyCompactionSummaryError
@@ -640,8 +654,12 @@ export function compactionFailureTurnEventPayload(
640
654
  recovery: "user_message";
641
655
  compacted: false;
642
656
  providerRejection?: CompactionProviderRejection;
657
+ quotaScope?: ProviderQuotaScope;
643
658
  } {
644
659
  const rejection = compactionProviderRejection(error);
660
+ // The same closed marker as a `provider_quota_exhausted` turn failure, so
661
+ // clients can name the exhausted limit and offer another model here too.
662
+ const quota = rejection ? null : classifyProviderQuotaExhaustionError(error);
645
663
  return {
646
664
  error: overrides.error ?? compactionFailureReasonFromError(error),
647
665
  code: "context_compaction_failed",
@@ -649,6 +667,7 @@ export function compactionFailureTurnEventPayload(
649
667
  recovery: "user_message",
650
668
  compacted: false,
651
669
  ...(rejection ? { providerRejection: rejection } : {}),
670
+ ...(quota ? { quotaScope: quota.scope } : {}),
652
671
  };
653
672
  }
654
673
 
@@ -846,6 +865,27 @@ export function isTransientProviderError(error: unknown): boolean {
846
865
  );
847
866
  }
848
867
 
868
+ /** An explicit `usage_limit_reached` type or code string anywhere on the error chain. */
869
+ function hasCodexUsageLimitType(error: unknown): boolean {
870
+ return collectErrorStrings(error).some((value) => value.includes(CODEX_USAGE_LIMIT_ERROR_TYPE));
871
+ }
872
+
873
+ /**
874
+ * Recognize an exhausted API-key provider quota (a daily or monthly allowance,
875
+ * a free-tier day cap, or an account out of credits) as distinct from an
876
+ * ordinary per-minute rate limit. Retrying within the bounded same-turn budget
877
+ * cannot succeed, so the turn fails promptly instead. Subscription transports
878
+ * own their quota semantics through credential rotation and durable capacity
879
+ * waits, so a Codex or SuperGrok transport error never classifies here.
880
+ */
881
+ export function classifyProviderQuotaExhaustionError(
882
+ error: unknown,
883
+ ): ProviderQuotaExhaustion | null {
884
+ if (isCodexTransportError(error) || isXaiSubscriptionTransportError(error)) return null;
885
+ // The same reader the OpenAI SDK retry veto uses, so the two never disagree.
886
+ return classifyProviderQuotaError(error);
887
+ }
888
+
849
889
  export type XaiCredentialFailure = {
850
890
  kind: "auth" | "forbidden" | "rate_limit";
851
891
  cooldownMs: number | null;
@@ -952,6 +992,7 @@ function baseAgentRunFailurePayload(
952
992
  historyPersistenceStage?: MandatoryHistoryPersistenceStage;
953
993
  mcpTransportDiagnostic?: McpTransportRequestFailureDiagnostic;
954
994
  materializationDiagnostic?: MaterializationVerificationDiagnostic;
995
+ quotaScope?: ProviderQuotaScope;
955
996
  } {
956
997
  if (error instanceof SandboxMaterializationVerificationError) {
957
998
  return {
@@ -1104,8 +1145,11 @@ function baseAgentRunFailurePayload(
1104
1145
  // `usage_limit_reached` shape must still outrank generic 429 retryability.
1105
1146
  // Credential quarantine/failover remains separately provenance-gated by
1106
1147
  // `isCodexTransportError`; this branch only chooses the truthful user payload.
1148
+ // The looser "429 ... usage limit" wording counts only on a Codex transport
1149
+ // error: an API-key provider's 429 that says "usage limit" is provider quota
1150
+ // evidence, not a ChatGPT/Codex subscription cap.
1107
1151
  const usageLimit = classifyCodexUsageLimitError(error);
1108
- if (usageLimit) {
1152
+ if (usageLimit && (isCodexTransportError(error) || hasCodexUsageLimitType(error))) {
1109
1153
  return codexUsageLimitFailurePayload(usageLimit, message);
1110
1154
  }
1111
1155
  const codexTimeout = classifyCodexResponseTimeoutError(error, {
@@ -1157,6 +1201,20 @@ function baseAgentRunFailurePayload(
1157
1201
  retryable: true,
1158
1202
  };
1159
1203
  }
1204
+ // An exhausted quota also arrives as HTTP 429 (or 402), but no retry within
1205
+ // the finite same-turn budget can succeed. Fail the turn promptly with a
1206
+ // distinct code so the client can offer another model; ordinary short rate
1207
+ // limits fall through to the retryable branch below.
1208
+ const quota = classifyProviderQuotaExhaustionError(error);
1209
+ if (quota) {
1210
+ return {
1211
+ error: providerQuotaExhaustedMessage(quota.scope),
1212
+ code: PROVIDER_QUOTA_EXHAUSTED_CODE,
1213
+ retryable: false,
1214
+ quotaScope: quota.scope,
1215
+ ...(message ? { detail: message } : {}),
1216
+ };
1217
+ }
1160
1218
  if (
1161
1219
  status === 429 ||
1162
1220
  code === "rate_limit_exceeded" ||
@@ -20,7 +20,11 @@ import {
20
20
  renderWorkspaceGovernanceContext,
21
21
  type OpenGeniRuntime,
22
22
  } from "@opengeni/runtime";
23
- import { settingsWithResolvedModelContext, type Settings } from "@opengeni/config";
23
+ import {
24
+ codeSearchDeploymentPolicy,
25
+ settingsWithResolvedModelContext,
26
+ type Settings,
27
+ } from "@opengeni/config";
24
28
  import { projectReasoningConfigurations, supportsReasoningConfiguration } from "@opengeni/codex";
25
29
  import { settingsWithSessionMcpServersForRun } from "../capabilities";
26
30
  import { resolveRigProviderImageForRun } from "@opengeni/core";
@@ -46,6 +50,7 @@ import {
46
50
  resolveWorkspaceAgentHumanInputEnabled,
47
51
  type MediaGenerationResult,
48
52
  } from "@opengeni/contracts";
53
+ import { codeSearchEnabledForTurn } from "@opengeni/contracts/code-search";
49
54
 
50
55
  import { assertWorkspaceHumanInputAllowed } from "./admission";
51
56
  import {
@@ -89,6 +94,8 @@ export type GovernanceModelOk = {
89
94
  | null;
90
95
  rigName: string | null;
91
96
  agentHumanInputEnabled: boolean;
97
+ /** Deployment and workspace allow the Jev-backed code_search tool. */
98
+ codeSearchEnabled: boolean;
92
99
  workspaceAgentInstructions: string | null | undefined;
93
100
  workspaceGovernance: ReturnType<typeof renderWorkspaceGovernanceContext>;
94
101
  structuredWorkspacePolicyActive: boolean;
@@ -230,6 +237,14 @@ export async function prepareGovernanceAndModel(
230
237
  workspaceRefs.rigVersionId = session.rigVersionId ?? "";
231
238
  if (!workspace) throw new Error(`Workspace not found: ${input.workspaceId}`);
232
239
  const agentHumanInputEnabled = resolveWorkspaceAgentHumanInputEnabled(workspace.settings);
240
+ // The session's decision was frozen when it was created, so only a
241
+ // deliberate switch-off (deployment or workspace Off), or undoing one,
242
+ // changes its tool list.
243
+ const codeSearchEnabled = codeSearchEnabledForTurn(
244
+ session.codeSearchEnabled,
245
+ workspace.settings,
246
+ codeSearchDeploymentPolicy(capabilitySettings),
247
+ );
233
248
  const contextSelection = await resolveCompanyBrainContextSelection(db, governanceClaims);
234
249
  const workspaceAgentInstructions = contextSelection.legacyWorkspaceInstructions;
235
250
  const memoryPromptMode = contextSelection.receipt.memoryPromptMode;
@@ -460,6 +475,7 @@ export async function prepareGovernanceAndModel(
460
475
  rigVersion,
461
476
  rigName,
462
477
  agentHumanInputEnabled,
478
+ codeSearchEnabled,
463
479
  workspaceAgentInstructions,
464
480
  workspaceGovernance,
465
481
  structuredWorkspacePolicyActive,
@@ -535,6 +535,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
535
535
  rigVersion,
536
536
  rigName,
537
537
  agentHumanInputEnabled,
538
+ codeSearchEnabled,
538
539
  workspaceAgentInstructions,
539
540
  workspaceGovernance,
540
541
  structuredWorkspacePolicyActive,
@@ -1278,6 +1279,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
1278
1279
  turnExecutionPolicy,
1279
1280
  trigger,
1280
1281
  runSettings,
1282
+ resolvedModel,
1281
1283
  lazyToolTransport,
1282
1284
  turnTools,
1283
1285
  connectionScope,
@@ -1289,6 +1291,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
1289
1291
  credentialSubjectId,
1290
1292
  interactionInterventionResume,
1291
1293
  runWorkspaceMutationForSandbox,
1294
+ codeSearchEnabled,
1292
1295
  throwIfWorkerShuttingDown,
1293
1296
  throwIfTurnCancelled,
1294
1297
  });
@@ -1352,6 +1355,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
1352
1355
  connectorActionPolicy,
1353
1356
  trigger,
1354
1357
  preparationIndependentToolNames,
1358
+ codeSearchAvailable: toolRuntime.codeSearchAvailable,
1355
1359
  videoGenerationAcceptancesByCallId,
1356
1360
  activeSandboxBackend,
1357
1361
  groupBoxBackend,
@@ -35,7 +35,11 @@ import {
35
35
  import { rigProviderImageSourceImage } from "../sandbox-images";
36
36
  import type { TurnActivityServices as ActivityServices, RunAgentTurnInput } from "../types";
37
37
  import type { currentActivityContext } from "../streaming";
38
- import { resumeBoxForTurn, type ResumedTurnSandbox } from "../../sandbox-resume";
38
+ import {
39
+ createFreshSandboxReadinessReplacementBudget,
40
+ resumeBoxForTurn,
41
+ type ResumedTurnSandbox,
42
+ } from "../../sandbox-resume";
39
43
  import {
40
44
  wrapTurnBoxWithRouting,
41
45
  wrapLazyTurnBoxWithRouting,
@@ -298,6 +302,9 @@ export async function resolveSandboxRoute(deps: SandboxRouteDeps): Promise<Sandb
298
302
  }
299
303
 
300
304
  export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Promise<void> {
305
+ // One fresh-box readiness replacement per turn attempt, shared by the eager
306
+ // establish and every lazy provisioner retry of this attempt.
307
+ const freshSandboxReadinessReplacementBudget = createFreshSandboxReadinessReplacementBudget();
301
308
  const {
302
309
  input,
303
310
  settings,
@@ -592,6 +599,8 @@ export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Prom
592
599
  logicalFallbackSettings: logicalSandboxSettings,
593
600
  cancellationSignal: sandboxResumeSignal,
594
601
  sandboxMetrics: runtimeMetricsHooksForObservability(observability),
602
+ observability,
603
+ freshSandboxReadinessReplacementBudget,
595
604
  onSandboxLost: publishSandboxLost,
596
605
  objectStorage,
597
606
  },
@@ -667,6 +676,8 @@ export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Prom
667
676
  logicalFallbackSettings: logicalSandboxSettings,
668
677
  cancellationSignal: sandboxResumeSignal,
669
678
  sandboxMetrics: runtimeMetricsHooksForObservability(observability),
679
+ observability,
680
+ freshSandboxReadinessReplacementBudget,
670
681
  onSandboxLost: publishSandboxLost,
671
682
  objectStorage,
672
683
  },
@@ -1,9 +1,15 @@
1
1
  import { hasPermission } from "@opengeni/core";
2
2
  import type { AttemptToolDefinition } from "@opengeni/codemode";
3
- import type { GeneratedSessionTitle } from "@opengeni/runtime";
3
+ import { isManagedOpenRouterFreeRoute, type ModelCapabilitiesV1 } from "@opengeni/config";
4
+ import type {
5
+ GeneratedSessionTitle,
6
+ GenerateSessionTitleOptions,
7
+ OpenGeniRuntime,
8
+ } from "@opengeni/runtime";
4
9
  import {
5
10
  AUTOMATIC_SESSION_TITLE_FALLBACK,
6
11
  DEFAULT_FIRST_PARTY_MCP_PERMISSIONS,
12
+ ReasoningEffort,
7
13
  type FirstPartyMcpToolName,
8
14
  type Permission,
9
15
  type ToolRef,
@@ -29,11 +35,26 @@ export function shouldRequestMissingSessionTitle(input: {
29
35
  return hasPermission([...permissions], "sessions:control");
30
36
  }
31
37
 
38
+ /**
39
+ * Whether the turn's route can afford a model request spent only on a title.
40
+ * The managed OpenRouter free route draws on one deployment-wide per-minute
41
+ * and per-day request quota that users' turns need, so an untitled session on
42
+ * it gets no title sidecar and no title tool (whose call would cost a
43
+ * follow-up request). Clients keep showing the prompt preview, and a later
44
+ * turn on another route titles the session.
45
+ */
46
+ export function routeAllowsSessionTitleRequests(
47
+ resolvedModel: ReturnType<OpenGeniRuntime["resolveTurnModel"]>,
48
+ ): boolean {
49
+ return !resolvedModel || !isManagedOpenRouterFreeRoute(resolvedModel);
50
+ }
51
+
32
52
  export function sessionTitleToolPlan(input: {
33
53
  tools: readonly ToolRef[];
34
54
  selectedFirstPartyMcpTools: readonly FirstPartyMcpToolName[];
35
55
  shouldRequestTitle: boolean;
36
56
  parallelGenerationAvailable: boolean;
57
+ routeAllowsTitleRequests: boolean;
37
58
  }): {
38
59
  promoteTitleTool: boolean;
39
60
  generateTitleInParallel: boolean;
@@ -43,8 +64,9 @@ export function sessionTitleToolPlan(input: {
43
64
  const titleToolAvailable =
44
65
  input.shouldRequestTitle &&
45
66
  input.tools.some((tool) => tool.kind === "mcp" && tool.id === "opengeni");
46
- const generateTitleInParallel = titleToolAvailable && input.parallelGenerationAvailable;
47
- const promoteTitleTool = titleToolAvailable && !generateTitleInParallel;
67
+ const titleRequestAllowed = titleToolAvailable && input.routeAllowsTitleRequests;
68
+ const generateTitleInParallel = titleRequestAllowed && input.parallelGenerationAvailable;
69
+ const promoteTitleTool = titleRequestAllowed && !generateTitleInParallel;
48
70
  return {
49
71
  promoteTitleTool,
50
72
  generateTitleInParallel,
@@ -57,6 +79,54 @@ export function sessionTitleToolPlan(input: {
57
79
 
58
80
  export const PARALLEL_SESSION_TITLE_TIMEOUT_MS = 15_000;
59
81
 
82
+ /**
83
+ * The lowest reasoning effort the resolved model can run, for the auxiliary
84
+ * title request only. A title needs no deliberation, and a provider default
85
+ * effort can use most of the output budget before any visible text. Returns
86
+ * undefined when the model declares no runnable reasoning control, so the
87
+ * request carries no reasoning parameter.
88
+ */
89
+ export function sessionTitleReasoningEffort(
90
+ capabilities: Pick<ModelCapabilitiesV1, "reasoning"> | undefined,
91
+ ): ReasoningEffort | undefined {
92
+ const reasoning = capabilities?.reasoning;
93
+ if (!reasoning?.runnable) return undefined;
94
+ const order = ReasoningEffort.options;
95
+ let lowest: ReasoningEffort | undefined;
96
+ for (const effort of reasoning.efforts) {
97
+ if (!lowest || order.indexOf(effort) < order.indexOf(lowest)) lowest = effort;
98
+ }
99
+ return lowest;
100
+ }
101
+
102
+ /**
103
+ * Options for the parallel title request. It uses the turn's resolved
104
+ * provider and credential authority, but its own lowest runnable reasoning
105
+ * effort rather than the turn's effort.
106
+ */
107
+ export function sessionTitleGenerationOptions(input: {
108
+ resolvedModel: ReturnType<OpenGeniRuntime["resolveTurnModel"]>;
109
+ modelName: string;
110
+ serviceTier: GenerateSessionTitleOptions["serviceTier"] | null | undefined;
111
+ signal: AbortSignal;
112
+ }): GenerateSessionTitleOptions {
113
+ const { resolvedModel, serviceTier } = input;
114
+ const reasoningEffort = sessionTitleReasoningEffort(resolvedModel?.configured.capabilities);
115
+ return {
116
+ ...(resolvedModel
117
+ ? {
118
+ client: resolvedModel.client,
119
+ provider: resolvedModel.provider,
120
+ model: resolvedModel.model,
121
+ }
122
+ : {}),
123
+ modelName: input.modelName,
124
+ ...(serviceTier ? { serviceTier } : {}),
125
+ ...(reasoningEffort ? { reasoningEffort } : {}),
126
+ signal: input.signal,
127
+ };
128
+ }
129
+
60
130
  export type ParallelSessionTitleGeneration = {
61
131
  finish: () => Promise<GeneratedSessionTitle | null>;
62
132
  cancel: () => Promise<void>;
@@ -126,7 +126,10 @@ import {
126
126
  recordModelUsageAndDebitCredits,
127
127
  recordAuthoritativeModelCallFact,
128
128
  } from "./model-usage";
129
- import { startParallelSessionTitleGeneration } from "./session-title";
129
+ import {
130
+ sessionTitleGenerationOptions,
131
+ startParallelSessionTitleGeneration,
132
+ } from "./session-title";
130
133
  import {
131
134
  assertAgentStreamNotCancelled,
132
135
  assertSuccessfulAgentStreamCompletion,
@@ -1786,18 +1789,16 @@ export async function runTurnStreamAttempt(
1786
1789
  signal: runtimeCancellationSignal,
1787
1790
  generate: async (signal) =>
1788
1791
  await withSessionTitleProviderRequestContext(() =>
1789
- runtime.generateSessionTitle!(runSettings, sessionTitlePrompt, {
1790
- ...(resolvedModel
1791
- ? {
1792
- client: resolvedModel.client,
1793
- provider: resolvedModel.provider,
1794
- model: resolvedModel.model,
1795
- }
1796
- : {}),
1797
- modelName: turnExecutionPolicy.upstreamModelId,
1798
- ...(serviceTier ? { serviceTier } : {}),
1799
- signal,
1800
- }),
1792
+ runtime.generateSessionTitle!(
1793
+ runSettings,
1794
+ sessionTitlePrompt,
1795
+ sessionTitleGenerationOptions({
1796
+ resolvedModel,
1797
+ modelName: turnExecutionPolicy.upstreamModelId,
1798
+ serviceTier,
1799
+ signal,
1800
+ }),
1801
+ ),
1801
1802
  ),
1802
1803
  onError: (error) => {
1803
1804
  observability.warn("parallel session title generation failed", {
@@ -15,6 +15,7 @@ import {
15
15
  organizationModelProviderConnectionActiveForWorkspace,
16
16
  persistAttemptToolCatalog,
17
17
  prepareConnectorActionApproval,
18
+ recordUsageEvent,
18
19
  previewConnectorActionApproval,
19
20
  namedSubjectHasLiveWorkspaceAuthority,
20
21
  updateSessionTitleWithEvent,
@@ -101,11 +102,13 @@ import type {
101
102
  } from "./turn-context";
102
103
  import {
103
104
  createSessionTitleAttemptToolDefinition,
105
+ routeAllowsSessionTitleRequests,
104
106
  sessionTitleToolPlan,
105
107
  shouldRequestMissingSessionTitle,
106
108
  } from "./session-title";
107
109
  import { resolveTurnSandboxAccess } from "./turn-sandbox-access";
108
110
  import { createListModelsAttemptToolDefinition } from "./list-models";
111
+ import { codeSearchToolDefinitions, codeSearchWorkspaceFromChannel } from "./code-search";
109
112
  import { createWorkspaceSkillTools } from "./skill-tools";
110
113
  import { loadConfiguredBundledSkills } from "./skill-selection";
111
114
  import { guardSkillFilesystem } from "./skill-transfer";
@@ -152,6 +155,7 @@ export type PrepareTurnToolRuntimeDeps = {
152
155
  turnExecutionPolicy: ClaimTurnOk["turnExecutionPolicy"];
153
156
  trigger: ClaimTurnOk["trigger"];
154
157
  runSettings: GovernanceModelOk["runSettings"];
158
+ resolvedModel: GovernanceModelOk["resolvedModel"];
155
159
  lazyToolTransport: GovernanceModelOk["lazyToolTransport"];
156
160
  turnTools: ReturnType<typeof withFirstPartyTools>;
157
161
  connectionScope: { accountId: string; workspaceId: string };
@@ -163,6 +167,8 @@ export type PrepareTurnToolRuntimeDeps = {
163
167
  credentialSubjectId: ClaimTurnOk["credentialSubjectId"];
164
168
  interactionInterventionResume: ClaimTurnOk["interactionInterventionResume"];
165
169
  runWorkspaceMutationForSandbox: SandboxTurnRuntime["runWorkspaceMutationForSandbox"];
170
+ /** Deployment and workspace allow the Jev-backed code_search tool. */
171
+ codeSearchEnabled: boolean;
166
172
  throwIfWorkerShuttingDown: () => void;
167
173
  throwIfTurnCancelled: () => void;
168
174
  };
@@ -371,6 +377,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
371
377
  turnExecutionPolicy,
372
378
  trigger,
373
379
  runSettings: canonicalRunSettings,
380
+ resolvedModel,
374
381
  lazyToolTransport,
375
382
  turnTools: canonicalTurnTools,
376
383
  sandboxArtifactRuntime,
@@ -381,6 +388,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
381
388
  credentialSubjectId,
382
389
  interactionInterventionResume,
383
390
  runWorkspaceMutationForSandbox,
391
+ codeSearchEnabled,
384
392
  throwIfWorkerShuttingDown,
385
393
  throwIfTurnCancelled,
386
394
  } = deps;
@@ -573,6 +581,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
573
581
  firstPartyMcpPermissions: effectiveFirstPartyPermissions,
574
582
  }),
575
583
  parallelGenerationAvailable: typeof runtime.generateSessionTitle === "function",
584
+ routeAllowsTitleRequests: routeAllowsSessionTitleRequests(resolvedModel),
576
585
  });
577
586
  const googleDrivePublicationAllowed =
578
587
  selectedFirstPartyMcpTools.includes("editable_artifact_export") &&
@@ -762,6 +771,60 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
762
771
  return ((await prepared.ready) ?? prepared).attemptToolEnvironment;
763
772
  },
764
773
  });
774
+ const codeSearchTools = codeSearchToolDefinitions({
775
+ enabled: codeSearchEnabled,
776
+ settings: runSettings,
777
+ backend: activeSandboxBackend ?? groupBoxBackend,
778
+ machineWorkspaceRoot: sandboxState.machinePrimarySession?.workspaceRoot ?? null,
779
+ observability,
780
+ // OpenGeni's Jev key pays for these calls whatever model billing the
781
+ // workspace uses; record them per workspace so the cost stays visible.
782
+ recordUsage: async (usage) => {
783
+ const shared = {
784
+ accountId: input.accountId,
785
+ workspaceId: input.workspaceId,
786
+ sourceResourceType: "code_search",
787
+ sourceResourceId: usage.operationId,
788
+ sessionId: input.sessionId,
789
+ turnId: turn.id,
790
+ turnAttemptId: input.attemptId,
791
+ };
792
+ await recordUsageEvent(db, {
793
+ ...shared,
794
+ eventType: "code_search.jev_input_tokens",
795
+ quantity: usage.jevInputTokens,
796
+ unit: "tokens",
797
+ idempotencyKey: `usage:code_search.jev_input_tokens:${input.attemptId}:${usage.operationId}`,
798
+ });
799
+ await recordUsageEvent(db, {
800
+ ...shared,
801
+ eventType: "code_search.jev_cost",
802
+ quantity: Math.round(usage.jevCostUsd * 1_000_000),
803
+ unit: "usd_micros",
804
+ idempotencyKey: `usage:code_search.jev_cost:${input.attemptId}:${usage.operationId}`,
805
+ });
806
+ },
807
+ workspace: async () => {
808
+ throwIfWorkerShuttingDown();
809
+ throwIfTurnCancelled();
810
+ const access = await resolveTurnSandboxAccess(
811
+ sandboxState,
812
+ media.sdkOwnedSandboxSession,
813
+ "code_search requires a sandbox or Connected Machine.",
814
+ );
815
+ const machineRoot = sandboxState.machinePrimarySession?.workspaceRoot;
816
+ const runAs = sandboxRunAs(runSettings);
817
+ return codeSearchWorkspaceFromChannel(
818
+ new SandboxChannelAService({
819
+ session: access.session,
820
+ workspaceRoot: machineRoot ?? "/workspace",
821
+ ...(machineRoot ? { providerPathMode: "workspace-relative" as const } : {}),
822
+ leaseEpoch: access.leaseEpoch,
823
+ ...(runAs ? { runAs } : {}),
824
+ }),
825
+ );
826
+ },
827
+ });
765
828
  const attemptToolDefinitions = [
766
829
  ...(operationReadStore
767
830
  ? [
@@ -903,6 +966,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
903
966
  ...(googleDrivePublicationTool && googleDrivePublicationAllowed
904
967
  ? [googleDrivePublicationTool]
905
968
  : []),
969
+ ...codeSearchTools,
906
970
  ];
907
971
  recordTurnStartupPhase(observability, {
908
972
  phase: "tool_context_preparation",
@@ -1124,6 +1188,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
1124
1188
  ...titleToolPlan.preparationIndependentToolNames,
1125
1189
  "skill_read",
1126
1190
  ],
1191
+ codeSearchAvailable: codeSearchTools.length > 0,
1127
1192
  skillCatalog,
1128
1193
  };
1129
1194
  }
@@ -9,7 +9,8 @@ import {
9
9
  import {
10
10
  EmptyCompactionSummaryError,
11
11
  REMOTE_COMPACTION_V2_IMPLEMENTATION,
12
- SUMMARY_BUFFER_TOKENS,
12
+ compactionSummaryOutputTokens,
13
+ buildSummaryItem,
13
14
  buildCompactionReplacementHistory,
14
15
  buildRemoteV2ReplacementHistory,
15
16
  compactionThresholdTokens,
@@ -58,7 +59,13 @@ export type MaybeCompactResult =
58
59
  * Codex, call Codex `/codex/responses` with `compaction_trigger` and persist the
59
60
  * opaque compaction item. Fail closed — never silently fall back to portable.
60
61
  */
61
- export type CompactionSummarizer = (settings: Settings, input: CompactionItem[]) => Promise<string>;
62
+ export type CompactionSummarizer = ((
63
+ settings: Settings,
64
+ input: CompactionItem[],
65
+ ) => Promise<string>) & {
66
+ /** Model-visible instructions and tool schemas outside the history estimate. */
67
+ estimatePrefixTokens?: () => number;
68
+ };
62
69
 
63
70
  /** Returns the opaque Codex remote compaction v2 item. */
64
71
  export type RemoteCompactionV2Requester = (
@@ -81,7 +88,7 @@ export async function maybeCompactContext(
81
88
  // Injectable for tests; defaults to the real provider-aware model call.
82
89
  summarize: CompactionSummarizer = (s, m) =>
83
90
  summarizeForCompaction(s, m, {
84
- maxOutputTokens: SUMMARY_BUFFER_TOKENS,
91
+ maxOutputTokens: compactionSummaryOutputTokens(s.contextWindowTokens),
85
92
  }),
86
93
  // Operator-forced (the /compact command): bypass the budget trigger and
87
94
  // compact now if there is anything to summarize. Structural guards still hold.
@@ -300,6 +307,7 @@ async function settleSkippedAfterStart(
300
307
  reason:
301
308
  | "no_history"
302
309
  | "replacement_not_smaller"
310
+ | "replacement_exceeds_model_budget"
303
311
  | "replacement_unchanged"
304
312
  | "summarization_failed",
305
313
  ): Promise<Extract<MaybeCompactResult, { compacted: false }>> {
@@ -455,12 +463,23 @@ async function compactContextPortable(
455
463
  const summarized = await summarizeWithCodexOverflowTrimming(summarize, settings, items);
456
464
  const summaryBody = summarized.summaryBody;
457
465
  const retainedTokens = await retentionTokenCounts(canonicalItems, projectForWire);
466
+ const outputReserve = compactionSummaryOutputTokens(settings.contextWindowTokens);
467
+ const structuralBudget = Math.min(
468
+ contextInputBudgetTokens(settings) || settings.contextWindowTokens - outputReserve,
469
+ settings.contextWindowTokens - outputReserve,
470
+ );
471
+ const prefixTokens = Math.max(0, Math.ceil(summarize.estimatePrefixTokens?.() ?? 0));
472
+ const summaryTokens = estimateTokens([buildSummaryItem(summaryBody)]);
458
473
  const replacementHistory = buildCompactionReplacementHistory(
459
474
  canonicalItems,
460
475
  summaryBody,
461
476
  (item) => retainedTokens.get(item) ?? estimateTokens([item]),
477
+ Math.min(outputReserve, Math.max(0, structuralBudget - prefixTokens - summaryTokens)),
462
478
  );
463
479
  const estimatedTokensAfter = estimateTokens(await projectForWire(replacementHistory));
480
+ if (estimatedTokensAfter + prefixTokens > structuralBudget) {
481
+ return await settleSkippedAfterStart(db, scope, options, "replacement_exceeds_model_budget");
482
+ }
464
483
  const replacementFingerprint = compactionReplacementFingerprint(replacementHistory);
465
484
  const previousReplacementFingerprint = latestCompactionReplacementFingerprint(canonicalItems);
466
485
  const summaryIndex =
@@ -520,7 +539,7 @@ async function compactContextPortable(
520
539
  };
521
540
  }
522
541
 
523
- async function summarizeWithCodexOverflowTrimming(
542
+ export async function summarizeWithCodexOverflowTrimming(
524
543
  summarize: CompactionSummarizer,
525
544
  settings: Settings,
526
545
  activeHistory: CompactionItem[],
@@ -531,17 +550,32 @@ async function summarizeWithCodexOverflowTrimming(
531
550
  }> {
532
551
  // Codex's estimator is intentionally coarse. Keep the explicit checkpoint
533
552
  // request below both the effective input window and the raw window minus the
534
- // requested summary, then leave 15% estimator headroom. This changes only the
535
- // temporary summarizer input; durable active history remains untouched until
536
- // applyContextCompaction succeeds under the attempt fence.
537
- const summaryAwareBudget = Math.max(0, settings.contextWindowTokens - SUMMARY_BUFFER_TOKENS);
553
+ // requested summary. Preserve the full portable history copy on the first
554
+ // call whenever it fits; only trim further after an actual provider overflow.
555
+ // Durable active history remains untouched until applyContextCompaction.
556
+ const summaryAwareBudget = Math.max(
557
+ 0,
558
+ settings.contextWindowTokens - compactionSummaryOutputTokens(settings.contextWindowTokens),
559
+ );
538
560
  const configuredInputBudget = contextInputBudgetTokens(settings);
539
561
  const structuralBudget = Math.min(
540
562
  configuredInputBudget > 0 ? configuredInputBudget : summaryAwareBudget,
541
563
  summaryAwareBudget,
542
564
  );
543
- const initialBudget = Math.floor(structuralBudget * 0.85);
565
+ const prefixTokens = Math.max(0, Math.ceil(summarize.estimatePrefixTokens?.() ?? 0));
566
+ const initialBudget = Math.max(0, structuralBudget - prefixTokens);
544
567
  let preparation = prepareCompactionPromptInput(activeHistory, initialBudget);
568
+ // A checkpoint prompt without source history cannot summarize that history.
569
+ // Do not let a plausible-sounding reply replace durable active context.
570
+ const requireHistory = () => {
571
+ if (activeHistory.length > 0 && preparation.input.length === 1) {
572
+ throw new EmptyCompactionSummaryError({
573
+ stage: "portable_input_budget",
574
+ reason: "no_history_fit",
575
+ });
576
+ }
577
+ };
578
+ requireHistory();
545
579
  try {
546
580
  return {
547
581
  summaryBody: await summarize(settings, preparation.input),
@@ -551,15 +585,16 @@ async function summarizeWithCodexOverflowTrimming(
551
585
  } catch (error) {
552
586
  if (!isContextWindowExceeded(error)) throw error;
553
587
  // The provider is more authoritative than the byte/4 estimate. Refit once
554
- // to 70% of both the configured target and the actual prepared estimate;
588
+ // to 40% of both the available target and the actual prepared estimate;
555
589
  // then fail terminally with prior history intact. Never issue one failing
556
- // request per oldest item. The production incident proved that the
557
- // provider can count slightly more than twice the byte/4 estimate, so a
558
- // half-size retry is the smallest honest bound for that observed skew.
590
+ // request per oldest item. The retry is bounded, not a guarantee: the
591
+ // provider can count slightly more than twice the byte/4 estimate. The
592
+ // prepared prefix has already been reserved from the available budget.
559
593
  const retryBudget = Math.floor(
560
- Math.min(initialBudget * 0.5, preparation.estimatedInputTokens * 0.5),
594
+ Math.min(initialBudget * 0.4, preparation.estimatedInputTokens * 0.4),
561
595
  );
562
596
  preparation = prepareCompactionPromptInput(activeHistory, retryBudget);
597
+ requireHistory();
563
598
  return {
564
599
  summaryBody: await summarize(settings, preparation.input),
565
600
  preparation,