@opengeni/worker-bundle 2.0.3 → 2.1.0-canary.36239117573001
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/activities/agent-turn/agent-build.d.ts +2 -0
- package/dist/activities/agent-turn/code-search.d.ts +60 -0
- package/dist/activities/agent-turn/codex-capacity.d.ts +13 -0
- package/dist/activities/agent-turn/errors.d.ts +12 -1
- package/dist/activities/agent-turn/governance-model.d.ts +2 -0
- package/dist/activities/agent-turn/session-title.d.ts +32 -2
- package/dist/activities/agent-turn/tool-environment.d.ts +4 -0
- package/dist/activities/context-compaction.d.ts +10 -2
- package/dist/activities/knowledge-indexing.d.ts +19 -0
- package/dist/{activities-control-IVTL723D.js → activities-control-I72CRMAE.js} +81 -14
- package/dist/activities-control-I72CRMAE.js.map +1 -0
- package/dist/{activities-turn-EZEDZTEZ.js → activities-turn-CSDKK4N3.js} +425 -46
- package/dist/activities-turn-CSDKK4N3.js.map +1 -0
- package/dist/artifact-outbox-entry.js +3 -0
- package/dist/artifact-outbox-entry.js.map +1 -1
- package/dist/{chunk-JMO5ZUW3.js → chunk-7M6G2SCF.js} +159 -11
- package/dist/chunk-7M6G2SCF.js.map +1 -0
- package/dist/{chunk-I7HKKMUJ.js → chunk-DSPP6CZL.js} +48 -3
- package/dist/chunk-DSPP6CZL.js.map +1 -0
- package/dist/index.js +4 -4
- package/dist/index.js.map +1 -1
- package/dist/observability-metrics.d.ts +11 -1
- package/dist/sandbox-resume.d.ts +46 -0
- package/dist/workflow-bundle.js +77 -2
- package/dist/workflows/session.d.ts +30 -0
- package/package.json +20 -19
- package/src/activities/agent-turn/agent-build.ts +4 -0
- package/src/activities/agent-turn/code-search.ts +275 -0
- package/src/activities/agent-turn/codex-capacity.ts +37 -2
- package/src/activities/agent-turn/compaction-prep.ts +52 -16
- package/src/activities/agent-turn/errors.ts +59 -1
- package/src/activities/agent-turn/governance-model.ts +17 -1
- package/src/activities/agent-turn/run.ts +4 -0
- package/src/activities/agent-turn/sandbox-establish.ts +12 -1
- package/src/activities/agent-turn/session-title.ts +73 -3
- package/src/activities/agent-turn/stream-attempt.ts +14 -13
- package/src/activities/agent-turn/tool-environment.ts +65 -0
- package/src/activities/context-compaction.ts +49 -14
- package/src/activities/knowledge-indexing.ts +107 -9
- package/src/activities/scheduled-tasks.ts +18 -2
- package/src/activity-services.ts +16 -4
- package/src/editable-artifact-hint-broker.ts +3 -0
- package/src/index.ts +2 -2
- package/src/observability-metrics.ts +70 -1
- package/src/personal-github-git-credentials.ts +2 -0
- package/src/sandbox-resume.ts +205 -5
- package/src/workflows/session.ts +102 -6
- package/dist/activities-control-IVTL723D.js.map +0 -1
- package/dist/activities-turn-EZEDZTEZ.js.map +0 -1
- package/dist/chunk-I7HKKMUJ.js.map +0 -1
- package/dist/chunk-JMO5ZUW3.js.map +0 -1
|
@@ -22,6 +22,11 @@ import {
|
|
|
22
22
|
SandboxMaterializationVerificationError,
|
|
23
23
|
materializationVerificationDiagnostic,
|
|
24
24
|
type MaterializationVerificationDiagnostic,
|
|
25
|
+
PROVIDER_QUOTA_EXHAUSTED_CODE,
|
|
26
|
+
type ProviderQuotaExhaustion,
|
|
27
|
+
type ProviderQuotaScope,
|
|
28
|
+
classifyProviderQuotaError,
|
|
29
|
+
providerQuotaExhaustedMessage,
|
|
25
30
|
SelfhostedWorkspaceRootChangedError,
|
|
26
31
|
UNKNOWN_MODEL_FINISH_REASON_CODE,
|
|
27
32
|
} from "@opengeni/runtime";
|
|
@@ -34,6 +39,7 @@ import { CODEX_USAGE_EXHAUSTED_PCT } from "../codex-rotation";
|
|
|
34
39
|
import { RetainedAttachmentTransportLimitError } from "../run-input";
|
|
35
40
|
import type { CodexAccountStatus } from "@opengeni/db";
|
|
36
41
|
import {
|
|
42
|
+
CODEX_USAGE_LIMIT_ERROR_TYPE,
|
|
37
43
|
CodexReloginRequired,
|
|
38
44
|
classifyCodexEncryptedArtifactRejection,
|
|
39
45
|
classifyCodexResponseTimeoutError,
|
|
@@ -614,6 +620,14 @@ export function compactionFailureReasonFromError(error: unknown): string {
|
|
|
614
620
|
`the model provider rejected the compaction request (${describeCompactionProviderRejection(rejection)}). Active history was preserved. ${COMPACTION_PROVIDER_REJECTION_GUIDANCE}`,
|
|
615
621
|
);
|
|
616
622
|
}
|
|
623
|
+
// An exhausted provider quota is not retried (see agentRunFailurePayload),
|
|
624
|
+
// so name the refusal plainly instead of the raw diagnostic envelope.
|
|
625
|
+
const quota = classifyProviderQuotaExhaustionError(error);
|
|
626
|
+
if (quota) {
|
|
627
|
+
return compactionFailureReason(
|
|
628
|
+
`${providerQuotaExhaustedMessage(quota.scope)} Active history was preserved.`,
|
|
629
|
+
);
|
|
630
|
+
}
|
|
617
631
|
if (
|
|
618
632
|
error instanceof CompactionProviderResponseError ||
|
|
619
633
|
error instanceof EmptyCompactionSummaryError
|
|
@@ -640,8 +654,12 @@ export function compactionFailureTurnEventPayload(
|
|
|
640
654
|
recovery: "user_message";
|
|
641
655
|
compacted: false;
|
|
642
656
|
providerRejection?: CompactionProviderRejection;
|
|
657
|
+
quotaScope?: ProviderQuotaScope;
|
|
643
658
|
} {
|
|
644
659
|
const rejection = compactionProviderRejection(error);
|
|
660
|
+
// The same closed marker as a `provider_quota_exhausted` turn failure, so
|
|
661
|
+
// clients can name the exhausted limit and offer another model here too.
|
|
662
|
+
const quota = rejection ? null : classifyProviderQuotaExhaustionError(error);
|
|
645
663
|
return {
|
|
646
664
|
error: overrides.error ?? compactionFailureReasonFromError(error),
|
|
647
665
|
code: "context_compaction_failed",
|
|
@@ -649,6 +667,7 @@ export function compactionFailureTurnEventPayload(
|
|
|
649
667
|
recovery: "user_message",
|
|
650
668
|
compacted: false,
|
|
651
669
|
...(rejection ? { providerRejection: rejection } : {}),
|
|
670
|
+
...(quota ? { quotaScope: quota.scope } : {}),
|
|
652
671
|
};
|
|
653
672
|
}
|
|
654
673
|
|
|
@@ -846,6 +865,27 @@ export function isTransientProviderError(error: unknown): boolean {
|
|
|
846
865
|
);
|
|
847
866
|
}
|
|
848
867
|
|
|
868
|
+
/** An explicit `usage_limit_reached` type or code string anywhere on the error chain. */
|
|
869
|
+
function hasCodexUsageLimitType(error: unknown): boolean {
|
|
870
|
+
return collectErrorStrings(error).some((value) => value.includes(CODEX_USAGE_LIMIT_ERROR_TYPE));
|
|
871
|
+
}
|
|
872
|
+
|
|
873
|
+
/**
|
|
874
|
+
* Recognize an exhausted API-key provider quota (a daily or monthly allowance,
|
|
875
|
+
* a free-tier day cap, or an account out of credits) as distinct from an
|
|
876
|
+
* ordinary per-minute rate limit. Retrying within the bounded same-turn budget
|
|
877
|
+
* cannot succeed, so the turn fails promptly instead. Subscription transports
|
|
878
|
+
* own their quota semantics through credential rotation and durable capacity
|
|
879
|
+
* waits, so a Codex or SuperGrok transport error never classifies here.
|
|
880
|
+
*/
|
|
881
|
+
export function classifyProviderQuotaExhaustionError(
|
|
882
|
+
error: unknown,
|
|
883
|
+
): ProviderQuotaExhaustion | null {
|
|
884
|
+
if (isCodexTransportError(error) || isXaiSubscriptionTransportError(error)) return null;
|
|
885
|
+
// The same reader the OpenAI SDK retry veto uses, so the two never disagree.
|
|
886
|
+
return classifyProviderQuotaError(error);
|
|
887
|
+
}
|
|
888
|
+
|
|
849
889
|
export type XaiCredentialFailure = {
|
|
850
890
|
kind: "auth" | "forbidden" | "rate_limit";
|
|
851
891
|
cooldownMs: number | null;
|
|
@@ -952,6 +992,7 @@ function baseAgentRunFailurePayload(
|
|
|
952
992
|
historyPersistenceStage?: MandatoryHistoryPersistenceStage;
|
|
953
993
|
mcpTransportDiagnostic?: McpTransportRequestFailureDiagnostic;
|
|
954
994
|
materializationDiagnostic?: MaterializationVerificationDiagnostic;
|
|
995
|
+
quotaScope?: ProviderQuotaScope;
|
|
955
996
|
} {
|
|
956
997
|
if (error instanceof SandboxMaterializationVerificationError) {
|
|
957
998
|
return {
|
|
@@ -1104,8 +1145,11 @@ function baseAgentRunFailurePayload(
|
|
|
1104
1145
|
// `usage_limit_reached` shape must still outrank generic 429 retryability.
|
|
1105
1146
|
// Credential quarantine/failover remains separately provenance-gated by
|
|
1106
1147
|
// `isCodexTransportError`; this branch only chooses the truthful user payload.
|
|
1148
|
+
// The looser "429 ... usage limit" wording counts only on a Codex transport
|
|
1149
|
+
// error: an API-key provider's 429 that says "usage limit" is provider quota
|
|
1150
|
+
// evidence, not a ChatGPT/Codex subscription cap.
|
|
1107
1151
|
const usageLimit = classifyCodexUsageLimitError(error);
|
|
1108
|
-
if (usageLimit) {
|
|
1152
|
+
if (usageLimit && (isCodexTransportError(error) || hasCodexUsageLimitType(error))) {
|
|
1109
1153
|
return codexUsageLimitFailurePayload(usageLimit, message);
|
|
1110
1154
|
}
|
|
1111
1155
|
const codexTimeout = classifyCodexResponseTimeoutError(error, {
|
|
@@ -1157,6 +1201,20 @@ function baseAgentRunFailurePayload(
|
|
|
1157
1201
|
retryable: true,
|
|
1158
1202
|
};
|
|
1159
1203
|
}
|
|
1204
|
+
// An exhausted quota also arrives as HTTP 429 (or 402), but no retry within
|
|
1205
|
+
// the finite same-turn budget can succeed. Fail the turn promptly with a
|
|
1206
|
+
// distinct code so the client can offer another model; ordinary short rate
|
|
1207
|
+
// limits fall through to the retryable branch below.
|
|
1208
|
+
const quota = classifyProviderQuotaExhaustionError(error);
|
|
1209
|
+
if (quota) {
|
|
1210
|
+
return {
|
|
1211
|
+
error: providerQuotaExhaustedMessage(quota.scope),
|
|
1212
|
+
code: PROVIDER_QUOTA_EXHAUSTED_CODE,
|
|
1213
|
+
retryable: false,
|
|
1214
|
+
quotaScope: quota.scope,
|
|
1215
|
+
...(message ? { detail: message } : {}),
|
|
1216
|
+
};
|
|
1217
|
+
}
|
|
1160
1218
|
if (
|
|
1161
1219
|
status === 429 ||
|
|
1162
1220
|
code === "rate_limit_exceeded" ||
|
|
@@ -20,7 +20,11 @@ import {
|
|
|
20
20
|
renderWorkspaceGovernanceContext,
|
|
21
21
|
type OpenGeniRuntime,
|
|
22
22
|
} from "@opengeni/runtime";
|
|
23
|
-
import {
|
|
23
|
+
import {
|
|
24
|
+
codeSearchDeploymentPolicy,
|
|
25
|
+
settingsWithResolvedModelContext,
|
|
26
|
+
type Settings,
|
|
27
|
+
} from "@opengeni/config";
|
|
24
28
|
import { projectReasoningConfigurations, supportsReasoningConfiguration } from "@opengeni/codex";
|
|
25
29
|
import { settingsWithSessionMcpServersForRun } from "../capabilities";
|
|
26
30
|
import { resolveRigProviderImageForRun } from "@opengeni/core";
|
|
@@ -46,6 +50,7 @@ import {
|
|
|
46
50
|
resolveWorkspaceAgentHumanInputEnabled,
|
|
47
51
|
type MediaGenerationResult,
|
|
48
52
|
} from "@opengeni/contracts";
|
|
53
|
+
import { codeSearchEnabledForTurn } from "@opengeni/contracts/code-search";
|
|
49
54
|
|
|
50
55
|
import { assertWorkspaceHumanInputAllowed } from "./admission";
|
|
51
56
|
import {
|
|
@@ -89,6 +94,8 @@ export type GovernanceModelOk = {
|
|
|
89
94
|
| null;
|
|
90
95
|
rigName: string | null;
|
|
91
96
|
agentHumanInputEnabled: boolean;
|
|
97
|
+
/** Deployment and workspace allow the Jev-backed code_search tool. */
|
|
98
|
+
codeSearchEnabled: boolean;
|
|
92
99
|
workspaceAgentInstructions: string | null | undefined;
|
|
93
100
|
workspaceGovernance: ReturnType<typeof renderWorkspaceGovernanceContext>;
|
|
94
101
|
structuredWorkspacePolicyActive: boolean;
|
|
@@ -230,6 +237,14 @@ export async function prepareGovernanceAndModel(
|
|
|
230
237
|
workspaceRefs.rigVersionId = session.rigVersionId ?? "";
|
|
231
238
|
if (!workspace) throw new Error(`Workspace not found: ${input.workspaceId}`);
|
|
232
239
|
const agentHumanInputEnabled = resolveWorkspaceAgentHumanInputEnabled(workspace.settings);
|
|
240
|
+
// The session's decision was frozen when it was created, so only a
|
|
241
|
+
// deliberate switch-off (deployment or workspace Off), or undoing one,
|
|
242
|
+
// changes its tool list.
|
|
243
|
+
const codeSearchEnabled = codeSearchEnabledForTurn(
|
|
244
|
+
session.codeSearchEnabled,
|
|
245
|
+
workspace.settings,
|
|
246
|
+
codeSearchDeploymentPolicy(capabilitySettings),
|
|
247
|
+
);
|
|
233
248
|
const contextSelection = await resolveCompanyBrainContextSelection(db, governanceClaims);
|
|
234
249
|
const workspaceAgentInstructions = contextSelection.legacyWorkspaceInstructions;
|
|
235
250
|
const memoryPromptMode = contextSelection.receipt.memoryPromptMode;
|
|
@@ -460,6 +475,7 @@ export async function prepareGovernanceAndModel(
|
|
|
460
475
|
rigVersion,
|
|
461
476
|
rigName,
|
|
462
477
|
agentHumanInputEnabled,
|
|
478
|
+
codeSearchEnabled,
|
|
463
479
|
workspaceAgentInstructions,
|
|
464
480
|
workspaceGovernance,
|
|
465
481
|
structuredWorkspacePolicyActive,
|
|
@@ -535,6 +535,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
|
|
|
535
535
|
rigVersion,
|
|
536
536
|
rigName,
|
|
537
537
|
agentHumanInputEnabled,
|
|
538
|
+
codeSearchEnabled,
|
|
538
539
|
workspaceAgentInstructions,
|
|
539
540
|
workspaceGovernance,
|
|
540
541
|
structuredWorkspacePolicyActive,
|
|
@@ -1278,6 +1279,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
|
|
|
1278
1279
|
turnExecutionPolicy,
|
|
1279
1280
|
trigger,
|
|
1280
1281
|
runSettings,
|
|
1282
|
+
resolvedModel,
|
|
1281
1283
|
lazyToolTransport,
|
|
1282
1284
|
turnTools,
|
|
1283
1285
|
connectionScope,
|
|
@@ -1289,6 +1291,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
|
|
|
1289
1291
|
credentialSubjectId,
|
|
1290
1292
|
interactionInterventionResume,
|
|
1291
1293
|
runWorkspaceMutationForSandbox,
|
|
1294
|
+
codeSearchEnabled,
|
|
1292
1295
|
throwIfWorkerShuttingDown,
|
|
1293
1296
|
throwIfTurnCancelled,
|
|
1294
1297
|
});
|
|
@@ -1352,6 +1355,7 @@ export function createRunAgentTurnActivity(services: () => Promise<ActivityServi
|
|
|
1352
1355
|
connectorActionPolicy,
|
|
1353
1356
|
trigger,
|
|
1354
1357
|
preparationIndependentToolNames,
|
|
1358
|
+
codeSearchAvailable: toolRuntime.codeSearchAvailable,
|
|
1355
1359
|
videoGenerationAcceptancesByCallId,
|
|
1356
1360
|
activeSandboxBackend,
|
|
1357
1361
|
groupBoxBackend,
|
|
@@ -35,7 +35,11 @@ import {
|
|
|
35
35
|
import { rigProviderImageSourceImage } from "../sandbox-images";
|
|
36
36
|
import type { TurnActivityServices as ActivityServices, RunAgentTurnInput } from "../types";
|
|
37
37
|
import type { currentActivityContext } from "../streaming";
|
|
38
|
-
import {
|
|
38
|
+
import {
|
|
39
|
+
createFreshSandboxReadinessReplacementBudget,
|
|
40
|
+
resumeBoxForTurn,
|
|
41
|
+
type ResumedTurnSandbox,
|
|
42
|
+
} from "../../sandbox-resume";
|
|
39
43
|
import {
|
|
40
44
|
wrapTurnBoxWithRouting,
|
|
41
45
|
wrapLazyTurnBoxWithRouting,
|
|
@@ -298,6 +302,9 @@ export async function resolveSandboxRoute(deps: SandboxRouteDeps): Promise<Sandb
|
|
|
298
302
|
}
|
|
299
303
|
|
|
300
304
|
export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Promise<void> {
|
|
305
|
+
// One fresh-box readiness replacement per turn attempt, shared by the eager
|
|
306
|
+
// establish and every lazy provisioner retry of this attempt.
|
|
307
|
+
const freshSandboxReadinessReplacementBudget = createFreshSandboxReadinessReplacementBudget();
|
|
301
308
|
const {
|
|
302
309
|
input,
|
|
303
310
|
settings,
|
|
@@ -592,6 +599,8 @@ export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Prom
|
|
|
592
599
|
logicalFallbackSettings: logicalSandboxSettings,
|
|
593
600
|
cancellationSignal: sandboxResumeSignal,
|
|
594
601
|
sandboxMetrics: runtimeMetricsHooksForObservability(observability),
|
|
602
|
+
observability,
|
|
603
|
+
freshSandboxReadinessReplacementBudget,
|
|
595
604
|
onSandboxLost: publishSandboxLost,
|
|
596
605
|
objectStorage,
|
|
597
606
|
},
|
|
@@ -667,6 +676,8 @@ export async function establishTurnSandbox(deps: EstablishTurnSandboxDeps): Prom
|
|
|
667
676
|
logicalFallbackSettings: logicalSandboxSettings,
|
|
668
677
|
cancellationSignal: sandboxResumeSignal,
|
|
669
678
|
sandboxMetrics: runtimeMetricsHooksForObservability(observability),
|
|
679
|
+
observability,
|
|
680
|
+
freshSandboxReadinessReplacementBudget,
|
|
670
681
|
onSandboxLost: publishSandboxLost,
|
|
671
682
|
objectStorage,
|
|
672
683
|
},
|
|
@@ -1,9 +1,15 @@
|
|
|
1
1
|
import { hasPermission } from "@opengeni/core";
|
|
2
2
|
import type { AttemptToolDefinition } from "@opengeni/codemode";
|
|
3
|
-
import type
|
|
3
|
+
import { isManagedOpenRouterFreeRoute, type ModelCapabilitiesV1 } from "@opengeni/config";
|
|
4
|
+
import type {
|
|
5
|
+
GeneratedSessionTitle,
|
|
6
|
+
GenerateSessionTitleOptions,
|
|
7
|
+
OpenGeniRuntime,
|
|
8
|
+
} from "@opengeni/runtime";
|
|
4
9
|
import {
|
|
5
10
|
AUTOMATIC_SESSION_TITLE_FALLBACK,
|
|
6
11
|
DEFAULT_FIRST_PARTY_MCP_PERMISSIONS,
|
|
12
|
+
ReasoningEffort,
|
|
7
13
|
type FirstPartyMcpToolName,
|
|
8
14
|
type Permission,
|
|
9
15
|
type ToolRef,
|
|
@@ -29,11 +35,26 @@ export function shouldRequestMissingSessionTitle(input: {
|
|
|
29
35
|
return hasPermission([...permissions], "sessions:control");
|
|
30
36
|
}
|
|
31
37
|
|
|
38
|
+
/**
|
|
39
|
+
* Whether the turn's route can afford a model request spent only on a title.
|
|
40
|
+
* The managed OpenRouter free route draws on one deployment-wide per-minute
|
|
41
|
+
* and per-day request quota that users' turns need, so an untitled session on
|
|
42
|
+
* it gets no title sidecar and no title tool (whose call would cost a
|
|
43
|
+
* follow-up request). Clients keep showing the prompt preview, and a later
|
|
44
|
+
* turn on another route titles the session.
|
|
45
|
+
*/
|
|
46
|
+
export function routeAllowsSessionTitleRequests(
|
|
47
|
+
resolvedModel: ReturnType<OpenGeniRuntime["resolveTurnModel"]>,
|
|
48
|
+
): boolean {
|
|
49
|
+
return !resolvedModel || !isManagedOpenRouterFreeRoute(resolvedModel);
|
|
50
|
+
}
|
|
51
|
+
|
|
32
52
|
export function sessionTitleToolPlan(input: {
|
|
33
53
|
tools: readonly ToolRef[];
|
|
34
54
|
selectedFirstPartyMcpTools: readonly FirstPartyMcpToolName[];
|
|
35
55
|
shouldRequestTitle: boolean;
|
|
36
56
|
parallelGenerationAvailable: boolean;
|
|
57
|
+
routeAllowsTitleRequests: boolean;
|
|
37
58
|
}): {
|
|
38
59
|
promoteTitleTool: boolean;
|
|
39
60
|
generateTitleInParallel: boolean;
|
|
@@ -43,8 +64,9 @@ export function sessionTitleToolPlan(input: {
|
|
|
43
64
|
const titleToolAvailable =
|
|
44
65
|
input.shouldRequestTitle &&
|
|
45
66
|
input.tools.some((tool) => tool.kind === "mcp" && tool.id === "opengeni");
|
|
46
|
-
const
|
|
47
|
-
const
|
|
67
|
+
const titleRequestAllowed = titleToolAvailable && input.routeAllowsTitleRequests;
|
|
68
|
+
const generateTitleInParallel = titleRequestAllowed && input.parallelGenerationAvailable;
|
|
69
|
+
const promoteTitleTool = titleRequestAllowed && !generateTitleInParallel;
|
|
48
70
|
return {
|
|
49
71
|
promoteTitleTool,
|
|
50
72
|
generateTitleInParallel,
|
|
@@ -57,6 +79,54 @@ export function sessionTitleToolPlan(input: {
|
|
|
57
79
|
|
|
58
80
|
export const PARALLEL_SESSION_TITLE_TIMEOUT_MS = 15_000;
|
|
59
81
|
|
|
82
|
+
/**
|
|
83
|
+
* The lowest reasoning effort the resolved model can run, for the auxiliary
|
|
84
|
+
* title request only. A title needs no deliberation, and a provider default
|
|
85
|
+
* effort can use most of the output budget before any visible text. Returns
|
|
86
|
+
* undefined when the model declares no runnable reasoning control, so the
|
|
87
|
+
* request carries no reasoning parameter.
|
|
88
|
+
*/
|
|
89
|
+
export function sessionTitleReasoningEffort(
|
|
90
|
+
capabilities: Pick<ModelCapabilitiesV1, "reasoning"> | undefined,
|
|
91
|
+
): ReasoningEffort | undefined {
|
|
92
|
+
const reasoning = capabilities?.reasoning;
|
|
93
|
+
if (!reasoning?.runnable) return undefined;
|
|
94
|
+
const order = ReasoningEffort.options;
|
|
95
|
+
let lowest: ReasoningEffort | undefined;
|
|
96
|
+
for (const effort of reasoning.efforts) {
|
|
97
|
+
if (!lowest || order.indexOf(effort) < order.indexOf(lowest)) lowest = effort;
|
|
98
|
+
}
|
|
99
|
+
return lowest;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Options for the parallel title request. It uses the turn's resolved
|
|
104
|
+
* provider and credential authority, but its own lowest runnable reasoning
|
|
105
|
+
* effort rather than the turn's effort.
|
|
106
|
+
*/
|
|
107
|
+
export function sessionTitleGenerationOptions(input: {
|
|
108
|
+
resolvedModel: ReturnType<OpenGeniRuntime["resolveTurnModel"]>;
|
|
109
|
+
modelName: string;
|
|
110
|
+
serviceTier: GenerateSessionTitleOptions["serviceTier"] | null | undefined;
|
|
111
|
+
signal: AbortSignal;
|
|
112
|
+
}): GenerateSessionTitleOptions {
|
|
113
|
+
const { resolvedModel, serviceTier } = input;
|
|
114
|
+
const reasoningEffort = sessionTitleReasoningEffort(resolvedModel?.configured.capabilities);
|
|
115
|
+
return {
|
|
116
|
+
...(resolvedModel
|
|
117
|
+
? {
|
|
118
|
+
client: resolvedModel.client,
|
|
119
|
+
provider: resolvedModel.provider,
|
|
120
|
+
model: resolvedModel.model,
|
|
121
|
+
}
|
|
122
|
+
: {}),
|
|
123
|
+
modelName: input.modelName,
|
|
124
|
+
...(serviceTier ? { serviceTier } : {}),
|
|
125
|
+
...(reasoningEffort ? { reasoningEffort } : {}),
|
|
126
|
+
signal: input.signal,
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
|
|
60
130
|
export type ParallelSessionTitleGeneration = {
|
|
61
131
|
finish: () => Promise<GeneratedSessionTitle | null>;
|
|
62
132
|
cancel: () => Promise<void>;
|
|
@@ -126,7 +126,10 @@ import {
|
|
|
126
126
|
recordModelUsageAndDebitCredits,
|
|
127
127
|
recordAuthoritativeModelCallFact,
|
|
128
128
|
} from "./model-usage";
|
|
129
|
-
import {
|
|
129
|
+
import {
|
|
130
|
+
sessionTitleGenerationOptions,
|
|
131
|
+
startParallelSessionTitleGeneration,
|
|
132
|
+
} from "./session-title";
|
|
130
133
|
import {
|
|
131
134
|
assertAgentStreamNotCancelled,
|
|
132
135
|
assertSuccessfulAgentStreamCompletion,
|
|
@@ -1786,18 +1789,16 @@ export async function runTurnStreamAttempt(
|
|
|
1786
1789
|
signal: runtimeCancellationSignal,
|
|
1787
1790
|
generate: async (signal) =>
|
|
1788
1791
|
await withSessionTitleProviderRequestContext(() =>
|
|
1789
|
-
runtime.generateSessionTitle!(
|
|
1790
|
-
|
|
1791
|
-
|
|
1792
|
-
|
|
1793
|
-
|
|
1794
|
-
|
|
1795
|
-
|
|
1796
|
-
|
|
1797
|
-
|
|
1798
|
-
|
|
1799
|
-
signal,
|
|
1800
|
-
}),
|
|
1792
|
+
runtime.generateSessionTitle!(
|
|
1793
|
+
runSettings,
|
|
1794
|
+
sessionTitlePrompt,
|
|
1795
|
+
sessionTitleGenerationOptions({
|
|
1796
|
+
resolvedModel,
|
|
1797
|
+
modelName: turnExecutionPolicy.upstreamModelId,
|
|
1798
|
+
serviceTier,
|
|
1799
|
+
signal,
|
|
1800
|
+
}),
|
|
1801
|
+
),
|
|
1801
1802
|
),
|
|
1802
1803
|
onError: (error) => {
|
|
1803
1804
|
observability.warn("parallel session title generation failed", {
|
|
@@ -15,6 +15,7 @@ import {
|
|
|
15
15
|
organizationModelProviderConnectionActiveForWorkspace,
|
|
16
16
|
persistAttemptToolCatalog,
|
|
17
17
|
prepareConnectorActionApproval,
|
|
18
|
+
recordUsageEvent,
|
|
18
19
|
previewConnectorActionApproval,
|
|
19
20
|
namedSubjectHasLiveWorkspaceAuthority,
|
|
20
21
|
updateSessionTitleWithEvent,
|
|
@@ -101,11 +102,13 @@ import type {
|
|
|
101
102
|
} from "./turn-context";
|
|
102
103
|
import {
|
|
103
104
|
createSessionTitleAttemptToolDefinition,
|
|
105
|
+
routeAllowsSessionTitleRequests,
|
|
104
106
|
sessionTitleToolPlan,
|
|
105
107
|
shouldRequestMissingSessionTitle,
|
|
106
108
|
} from "./session-title";
|
|
107
109
|
import { resolveTurnSandboxAccess } from "./turn-sandbox-access";
|
|
108
110
|
import { createListModelsAttemptToolDefinition } from "./list-models";
|
|
111
|
+
import { codeSearchToolDefinitions, codeSearchWorkspaceFromChannel } from "./code-search";
|
|
109
112
|
import { createWorkspaceSkillTools } from "./skill-tools";
|
|
110
113
|
import { loadConfiguredBundledSkills } from "./skill-selection";
|
|
111
114
|
import { guardSkillFilesystem } from "./skill-transfer";
|
|
@@ -152,6 +155,7 @@ export type PrepareTurnToolRuntimeDeps = {
|
|
|
152
155
|
turnExecutionPolicy: ClaimTurnOk["turnExecutionPolicy"];
|
|
153
156
|
trigger: ClaimTurnOk["trigger"];
|
|
154
157
|
runSettings: GovernanceModelOk["runSettings"];
|
|
158
|
+
resolvedModel: GovernanceModelOk["resolvedModel"];
|
|
155
159
|
lazyToolTransport: GovernanceModelOk["lazyToolTransport"];
|
|
156
160
|
turnTools: ReturnType<typeof withFirstPartyTools>;
|
|
157
161
|
connectionScope: { accountId: string; workspaceId: string };
|
|
@@ -163,6 +167,8 @@ export type PrepareTurnToolRuntimeDeps = {
|
|
|
163
167
|
credentialSubjectId: ClaimTurnOk["credentialSubjectId"];
|
|
164
168
|
interactionInterventionResume: ClaimTurnOk["interactionInterventionResume"];
|
|
165
169
|
runWorkspaceMutationForSandbox: SandboxTurnRuntime["runWorkspaceMutationForSandbox"];
|
|
170
|
+
/** Deployment and workspace allow the Jev-backed code_search tool. */
|
|
171
|
+
codeSearchEnabled: boolean;
|
|
166
172
|
throwIfWorkerShuttingDown: () => void;
|
|
167
173
|
throwIfTurnCancelled: () => void;
|
|
168
174
|
};
|
|
@@ -371,6 +377,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
|
|
|
371
377
|
turnExecutionPolicy,
|
|
372
378
|
trigger,
|
|
373
379
|
runSettings: canonicalRunSettings,
|
|
380
|
+
resolvedModel,
|
|
374
381
|
lazyToolTransport,
|
|
375
382
|
turnTools: canonicalTurnTools,
|
|
376
383
|
sandboxArtifactRuntime,
|
|
@@ -381,6 +388,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
|
|
|
381
388
|
credentialSubjectId,
|
|
382
389
|
interactionInterventionResume,
|
|
383
390
|
runWorkspaceMutationForSandbox,
|
|
391
|
+
codeSearchEnabled,
|
|
384
392
|
throwIfWorkerShuttingDown,
|
|
385
393
|
throwIfTurnCancelled,
|
|
386
394
|
} = deps;
|
|
@@ -573,6 +581,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
|
|
|
573
581
|
firstPartyMcpPermissions: effectiveFirstPartyPermissions,
|
|
574
582
|
}),
|
|
575
583
|
parallelGenerationAvailable: typeof runtime.generateSessionTitle === "function",
|
|
584
|
+
routeAllowsTitleRequests: routeAllowsSessionTitleRequests(resolvedModel),
|
|
576
585
|
});
|
|
577
586
|
const googleDrivePublicationAllowed =
|
|
578
587
|
selectedFirstPartyMcpTools.includes("editable_artifact_export") &&
|
|
@@ -762,6 +771,60 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
|
|
|
762
771
|
return ((await prepared.ready) ?? prepared).attemptToolEnvironment;
|
|
763
772
|
},
|
|
764
773
|
});
|
|
774
|
+
const codeSearchTools = codeSearchToolDefinitions({
|
|
775
|
+
enabled: codeSearchEnabled,
|
|
776
|
+
settings: runSettings,
|
|
777
|
+
backend: activeSandboxBackend ?? groupBoxBackend,
|
|
778
|
+
machineWorkspaceRoot: sandboxState.machinePrimarySession?.workspaceRoot ?? null,
|
|
779
|
+
observability,
|
|
780
|
+
// OpenGeni's Jev key pays for these calls whatever model billing the
|
|
781
|
+
// workspace uses; record them per workspace so the cost stays visible.
|
|
782
|
+
recordUsage: async (usage) => {
|
|
783
|
+
const shared = {
|
|
784
|
+
accountId: input.accountId,
|
|
785
|
+
workspaceId: input.workspaceId,
|
|
786
|
+
sourceResourceType: "code_search",
|
|
787
|
+
sourceResourceId: usage.operationId,
|
|
788
|
+
sessionId: input.sessionId,
|
|
789
|
+
turnId: turn.id,
|
|
790
|
+
turnAttemptId: input.attemptId,
|
|
791
|
+
};
|
|
792
|
+
await recordUsageEvent(db, {
|
|
793
|
+
...shared,
|
|
794
|
+
eventType: "code_search.jev_input_tokens",
|
|
795
|
+
quantity: usage.jevInputTokens,
|
|
796
|
+
unit: "tokens",
|
|
797
|
+
idempotencyKey: `usage:code_search.jev_input_tokens:${input.attemptId}:${usage.operationId}`,
|
|
798
|
+
});
|
|
799
|
+
await recordUsageEvent(db, {
|
|
800
|
+
...shared,
|
|
801
|
+
eventType: "code_search.jev_cost",
|
|
802
|
+
quantity: Math.round(usage.jevCostUsd * 1_000_000),
|
|
803
|
+
unit: "usd_micros",
|
|
804
|
+
idempotencyKey: `usage:code_search.jev_cost:${input.attemptId}:${usage.operationId}`,
|
|
805
|
+
});
|
|
806
|
+
},
|
|
807
|
+
workspace: async () => {
|
|
808
|
+
throwIfWorkerShuttingDown();
|
|
809
|
+
throwIfTurnCancelled();
|
|
810
|
+
const access = await resolveTurnSandboxAccess(
|
|
811
|
+
sandboxState,
|
|
812
|
+
media.sdkOwnedSandboxSession,
|
|
813
|
+
"code_search requires a sandbox or Connected Machine.",
|
|
814
|
+
);
|
|
815
|
+
const machineRoot = sandboxState.machinePrimarySession?.workspaceRoot;
|
|
816
|
+
const runAs = sandboxRunAs(runSettings);
|
|
817
|
+
return codeSearchWorkspaceFromChannel(
|
|
818
|
+
new SandboxChannelAService({
|
|
819
|
+
session: access.session,
|
|
820
|
+
workspaceRoot: machineRoot ?? "/workspace",
|
|
821
|
+
...(machineRoot ? { providerPathMode: "workspace-relative" as const } : {}),
|
|
822
|
+
leaseEpoch: access.leaseEpoch,
|
|
823
|
+
...(runAs ? { runAs } : {}),
|
|
824
|
+
}),
|
|
825
|
+
);
|
|
826
|
+
},
|
|
827
|
+
});
|
|
765
828
|
const attemptToolDefinitions = [
|
|
766
829
|
...(operationReadStore
|
|
767
830
|
? [
|
|
@@ -903,6 +966,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
|
|
|
903
966
|
...(googleDrivePublicationTool && googleDrivePublicationAllowed
|
|
904
967
|
? [googleDrivePublicationTool]
|
|
905
968
|
: []),
|
|
969
|
+
...codeSearchTools,
|
|
906
970
|
];
|
|
907
971
|
recordTurnStartupPhase(observability, {
|
|
908
972
|
phase: "tool_context_preparation",
|
|
@@ -1124,6 +1188,7 @@ export async function prepareTurnToolRuntime(deps: PrepareTurnToolRuntimeDeps) {
|
|
|
1124
1188
|
...titleToolPlan.preparationIndependentToolNames,
|
|
1125
1189
|
"skill_read",
|
|
1126
1190
|
],
|
|
1191
|
+
codeSearchAvailable: codeSearchTools.length > 0,
|
|
1127
1192
|
skillCatalog,
|
|
1128
1193
|
};
|
|
1129
1194
|
}
|
|
@@ -9,7 +9,8 @@ import {
|
|
|
9
9
|
import {
|
|
10
10
|
EmptyCompactionSummaryError,
|
|
11
11
|
REMOTE_COMPACTION_V2_IMPLEMENTATION,
|
|
12
|
-
|
|
12
|
+
compactionSummaryOutputTokens,
|
|
13
|
+
buildSummaryItem,
|
|
13
14
|
buildCompactionReplacementHistory,
|
|
14
15
|
buildRemoteV2ReplacementHistory,
|
|
15
16
|
compactionThresholdTokens,
|
|
@@ -58,7 +59,13 @@ export type MaybeCompactResult =
|
|
|
58
59
|
* Codex, call Codex `/codex/responses` with `compaction_trigger` and persist the
|
|
59
60
|
* opaque compaction item. Fail closed — never silently fall back to portable.
|
|
60
61
|
*/
|
|
61
|
-
export type CompactionSummarizer = (
|
|
62
|
+
export type CompactionSummarizer = ((
|
|
63
|
+
settings: Settings,
|
|
64
|
+
input: CompactionItem[],
|
|
65
|
+
) => Promise<string>) & {
|
|
66
|
+
/** Model-visible instructions and tool schemas outside the history estimate. */
|
|
67
|
+
estimatePrefixTokens?: () => number;
|
|
68
|
+
};
|
|
62
69
|
|
|
63
70
|
/** Returns the opaque Codex remote compaction v2 item. */
|
|
64
71
|
export type RemoteCompactionV2Requester = (
|
|
@@ -81,7 +88,7 @@ export async function maybeCompactContext(
|
|
|
81
88
|
// Injectable for tests; defaults to the real provider-aware model call.
|
|
82
89
|
summarize: CompactionSummarizer = (s, m) =>
|
|
83
90
|
summarizeForCompaction(s, m, {
|
|
84
|
-
maxOutputTokens:
|
|
91
|
+
maxOutputTokens: compactionSummaryOutputTokens(s.contextWindowTokens),
|
|
85
92
|
}),
|
|
86
93
|
// Operator-forced (the /compact command): bypass the budget trigger and
|
|
87
94
|
// compact now if there is anything to summarize. Structural guards still hold.
|
|
@@ -300,6 +307,7 @@ async function settleSkippedAfterStart(
|
|
|
300
307
|
reason:
|
|
301
308
|
| "no_history"
|
|
302
309
|
| "replacement_not_smaller"
|
|
310
|
+
| "replacement_exceeds_model_budget"
|
|
303
311
|
| "replacement_unchanged"
|
|
304
312
|
| "summarization_failed",
|
|
305
313
|
): Promise<Extract<MaybeCompactResult, { compacted: false }>> {
|
|
@@ -455,12 +463,23 @@ async function compactContextPortable(
|
|
|
455
463
|
const summarized = await summarizeWithCodexOverflowTrimming(summarize, settings, items);
|
|
456
464
|
const summaryBody = summarized.summaryBody;
|
|
457
465
|
const retainedTokens = await retentionTokenCounts(canonicalItems, projectForWire);
|
|
466
|
+
const outputReserve = compactionSummaryOutputTokens(settings.contextWindowTokens);
|
|
467
|
+
const structuralBudget = Math.min(
|
|
468
|
+
contextInputBudgetTokens(settings) || settings.contextWindowTokens - outputReserve,
|
|
469
|
+
settings.contextWindowTokens - outputReserve,
|
|
470
|
+
);
|
|
471
|
+
const prefixTokens = Math.max(0, Math.ceil(summarize.estimatePrefixTokens?.() ?? 0));
|
|
472
|
+
const summaryTokens = estimateTokens([buildSummaryItem(summaryBody)]);
|
|
458
473
|
const replacementHistory = buildCompactionReplacementHistory(
|
|
459
474
|
canonicalItems,
|
|
460
475
|
summaryBody,
|
|
461
476
|
(item) => retainedTokens.get(item) ?? estimateTokens([item]),
|
|
477
|
+
Math.min(outputReserve, Math.max(0, structuralBudget - prefixTokens - summaryTokens)),
|
|
462
478
|
);
|
|
463
479
|
const estimatedTokensAfter = estimateTokens(await projectForWire(replacementHistory));
|
|
480
|
+
if (estimatedTokensAfter + prefixTokens > structuralBudget) {
|
|
481
|
+
return await settleSkippedAfterStart(db, scope, options, "replacement_exceeds_model_budget");
|
|
482
|
+
}
|
|
464
483
|
const replacementFingerprint = compactionReplacementFingerprint(replacementHistory);
|
|
465
484
|
const previousReplacementFingerprint = latestCompactionReplacementFingerprint(canonicalItems);
|
|
466
485
|
const summaryIndex =
|
|
@@ -520,7 +539,7 @@ async function compactContextPortable(
|
|
|
520
539
|
};
|
|
521
540
|
}
|
|
522
541
|
|
|
523
|
-
async function summarizeWithCodexOverflowTrimming(
|
|
542
|
+
export async function summarizeWithCodexOverflowTrimming(
|
|
524
543
|
summarize: CompactionSummarizer,
|
|
525
544
|
settings: Settings,
|
|
526
545
|
activeHistory: CompactionItem[],
|
|
@@ -531,17 +550,32 @@ async function summarizeWithCodexOverflowTrimming(
|
|
|
531
550
|
}> {
|
|
532
551
|
// Codex's estimator is intentionally coarse. Keep the explicit checkpoint
|
|
533
552
|
// request below both the effective input window and the raw window minus the
|
|
534
|
-
// requested summary
|
|
535
|
-
//
|
|
536
|
-
//
|
|
537
|
-
const summaryAwareBudget = Math.max(
|
|
553
|
+
// requested summary. Preserve the full portable history copy on the first
|
|
554
|
+
// call whenever it fits; only trim further after an actual provider overflow.
|
|
555
|
+
// Durable active history remains untouched until applyContextCompaction.
|
|
556
|
+
const summaryAwareBudget = Math.max(
|
|
557
|
+
0,
|
|
558
|
+
settings.contextWindowTokens - compactionSummaryOutputTokens(settings.contextWindowTokens),
|
|
559
|
+
);
|
|
538
560
|
const configuredInputBudget = contextInputBudgetTokens(settings);
|
|
539
561
|
const structuralBudget = Math.min(
|
|
540
562
|
configuredInputBudget > 0 ? configuredInputBudget : summaryAwareBudget,
|
|
541
563
|
summaryAwareBudget,
|
|
542
564
|
);
|
|
543
|
-
const
|
|
565
|
+
const prefixTokens = Math.max(0, Math.ceil(summarize.estimatePrefixTokens?.() ?? 0));
|
|
566
|
+
const initialBudget = Math.max(0, structuralBudget - prefixTokens);
|
|
544
567
|
let preparation = prepareCompactionPromptInput(activeHistory, initialBudget);
|
|
568
|
+
// A checkpoint prompt without source history cannot summarize that history.
|
|
569
|
+
// Do not let a plausible-sounding reply replace durable active context.
|
|
570
|
+
const requireHistory = () => {
|
|
571
|
+
if (activeHistory.length > 0 && preparation.input.length === 1) {
|
|
572
|
+
throw new EmptyCompactionSummaryError({
|
|
573
|
+
stage: "portable_input_budget",
|
|
574
|
+
reason: "no_history_fit",
|
|
575
|
+
});
|
|
576
|
+
}
|
|
577
|
+
};
|
|
578
|
+
requireHistory();
|
|
545
579
|
try {
|
|
546
580
|
return {
|
|
547
581
|
summaryBody: await summarize(settings, preparation.input),
|
|
@@ -551,15 +585,16 @@ async function summarizeWithCodexOverflowTrimming(
|
|
|
551
585
|
} catch (error) {
|
|
552
586
|
if (!isContextWindowExceeded(error)) throw error;
|
|
553
587
|
// The provider is more authoritative than the byte/4 estimate. Refit once
|
|
554
|
-
// to
|
|
588
|
+
// to 40% of both the available target and the actual prepared estimate;
|
|
555
589
|
// then fail terminally with prior history intact. Never issue one failing
|
|
556
|
-
// request per oldest item. The
|
|
557
|
-
// provider can count slightly more than twice the byte/4 estimate
|
|
558
|
-
//
|
|
590
|
+
// request per oldest item. The retry is bounded, not a guarantee: the
|
|
591
|
+
// provider can count slightly more than twice the byte/4 estimate. The
|
|
592
|
+
// prepared prefix has already been reserved from the available budget.
|
|
559
593
|
const retryBudget = Math.floor(
|
|
560
|
-
Math.min(initialBudget * 0.
|
|
594
|
+
Math.min(initialBudget * 0.4, preparation.estimatedInputTokens * 0.4),
|
|
561
595
|
);
|
|
562
596
|
preparation = prepareCompactionPromptInput(activeHistory, retryBudget);
|
|
597
|
+
requireHistory();
|
|
563
598
|
return {
|
|
564
599
|
summaryBody: await summarize(settings, preparation.input),
|
|
565
600
|
preparation,
|