@opengeni/worker-bundle 2.0.1 → 2.0.3-canary.36199476632001
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/activities/agent-turn/agent-build.d.ts +2 -0
- package/dist/activities/agent-turn/code-search.d.ts +60 -0
- package/dist/activities/agent-turn/codex-capacity.d.ts +13 -0
- package/dist/activities/agent-turn/errors.d.ts +12 -1
- package/dist/activities/agent-turn/failure-settlement.d.ts +1 -1
- package/dist/activities/agent-turn/governance-model.d.ts +2 -0
- package/dist/activities/agent-turn/history.d.ts +2 -0
- package/dist/activities/agent-turn/media-artifacts.d.ts +5 -3
- package/dist/activities/agent-turn/session-title.d.ts +32 -2
- package/dist/activities/agent-turn/tool-environment.d.ts +4 -0
- package/dist/activities/context-compaction.d.ts +10 -2
- package/dist/activities/knowledge-indexing.d.ts +19 -0
- package/dist/activities/retained-screenshots.d.ts +7 -2
- package/dist/activities/sandbox-lease.d.ts +3 -4
- package/dist/{activities-control-QA6OLRQY.js → activities-control-I72CRMAE.js} +283 -35
- package/dist/activities-control-I72CRMAE.js.map +1 -0
- package/dist/{activities-turn-YVVKCGAK.js → activities-turn-CSDKK4N3.js} +531 -77
- package/dist/activities-turn-CSDKK4N3.js.map +1 -0
- package/dist/artifact-outbox-entry.js +3 -0
- package/dist/artifact-outbox-entry.js.map +1 -1
- package/dist/{chunk-M3NVMF2V.js → chunk-7M6G2SCF.js} +172 -17
- package/dist/chunk-7M6G2SCF.js.map +1 -0
- package/dist/{chunk-I7HKKMUJ.js → chunk-DSPP6CZL.js} +48 -3
- package/dist/chunk-DSPP6CZL.js.map +1 -0
- package/dist/index.js +10 -5
- package/dist/index.js.map +1 -1
- package/dist/observability-metrics.d.ts +11 -1
- package/dist/sandbox-resume.d.ts +46 -0
- package/dist/workflow-bundle.js +80 -2
- package/dist/workflows/session.d.ts +30 -0
- package/package.json +20 -19
- package/src/activities/agent-turn/agent-build.ts +4 -0
- package/src/activities/agent-turn/code-search.ts +275 -0
- package/src/activities/agent-turn/codex-capacity.ts +37 -2
- package/src/activities/agent-turn/compaction-prep.ts +52 -16
- package/src/activities/agent-turn/errors.ts +67 -4
- package/src/activities/agent-turn/failure-settlement.ts +40 -8
- package/src/activities/agent-turn/governance-model.ts +17 -1
- package/src/activities/agent-turn/history.ts +14 -1
- package/src/activities/agent-turn/media-artifacts.ts +2 -1
- package/src/activities/agent-turn/run.ts +4 -0
- package/src/activities/agent-turn/sandbox-establish.ts +12 -1
- package/src/activities/agent-turn/sandbox-runtime.ts +25 -9
- package/src/activities/agent-turn/session-title.ts +73 -3
- package/src/activities/agent-turn/stream-attempt.ts +27 -15
- package/src/activities/agent-turn/tool-environment.ts +65 -0
- package/src/activities/context-compaction.ts +49 -14
- package/src/activities/knowledge-indexing.ts +240 -15
- package/src/activities/retained-screenshots.ts +82 -11
- package/src/activities/sandbox-lease.ts +141 -21
- package/src/activities/scheduled-tasks.ts +18 -2
- package/src/activity-services.ts +16 -4
- package/src/editable-artifact-hint-broker.ts +3 -0
- package/src/index.ts +2 -2
- package/src/observability-metrics.ts +70 -1
- package/src/personal-github-git-credentials.ts +2 -0
- package/src/sandbox-resume.ts +211 -5
- package/src/workflows/activities.ts +14 -1
- package/src/workflows/session.ts +102 -6
- package/dist/activities-control-QA6OLRQY.js.map +0 -1
- package/dist/activities-turn-YVVKCGAK.js.map +0 -1
- package/dist/chunk-I7HKKMUJ.js.map +0 -1
- package/dist/chunk-M3NVMF2V.js.map +0 -1
|
@@ -1,5 +1,34 @@
|
|
|
1
1
|
import type * as activities from "../activities.js";
|
|
2
2
|
import { type EscapedMcpTimeoutRecoveryDetail, type PostClaimDatabaseRecoveryDetail, type PreClaimFailureDetail, type PreClaimFailureDisposition } from "../activities/types.js";
|
|
3
|
+
/**
|
|
4
|
+
* Provider capacity waits are shared: every waiter of one exhausted pool learns
|
|
5
|
+
* the same authoritative reset time, and one capacity mutation (for example the
|
|
6
|
+
* bounded refresh that verifies a quota reset) wakes every waiter at once, both
|
|
7
|
+
* as the typed capacity signal and as the generic durable workflow wake. Without
|
|
8
|
+
* spread, a reset resumes the whole backlog in the same few seconds and its
|
|
9
|
+
* turns stampede sandbox creation and the provider. Each workflow therefore
|
|
10
|
+
* delays its reconciliation by a bounded, replay-deterministic (Temporal-seeded
|
|
11
|
+
* `Math.random`) jitter: up to one minute past a scheduled reset timer and up to
|
|
12
|
+
* 30 seconds after a capacity or queue wake or an already-due waiter. A wake
|
|
13
|
+
* that lands inside a waiter's timer spread does not shorten it. The Postgres
|
|
14
|
+
* waiter stays authoritative; jitter only delays the same reconciliation
|
|
15
|
+
* activity and never creates queue rows, input, or inference. Interruptions are
|
|
16
|
+
* never delayed.
|
|
17
|
+
*
|
|
18
|
+
* The patch marker is not understood by workers built before it: rolling the
|
|
19
|
+
* worker image back past this change while a session has recorded the marker
|
|
20
|
+
* (it waited on capacity in its current run) fails that workflow's tasks as
|
|
21
|
+
* nondeterministic until a patched worker returns.
|
|
22
|
+
*/
|
|
23
|
+
export declare const CAPACITY_WAKE_JITTER_PATCH = "session-capacity-wake-jitter-v1";
|
|
24
|
+
export declare const CAPACITY_TIMER_WAKE_JITTER_MAX_MS = 60000;
|
|
25
|
+
export declare const CAPACITY_WAKE_JITTER_MAX_MS = 30000;
|
|
26
|
+
/**
|
|
27
|
+
* Pure + exported so the bound is unit-testable without a workflow environment.
|
|
28
|
+
* `overrideMaxMs` is the test-only SessionWorkflowInput ceiling; it can only
|
|
29
|
+
* narrow the production bound.
|
|
30
|
+
*/
|
|
31
|
+
export declare function capacityWakeJitterMs(kind: "timer" | "wake", sample: number, overrideMaxMs?: number): number;
|
|
3
32
|
/**
|
|
4
33
|
* How long the continuation loop must hold before re-admitting the next turn. 0 ⇒ no
|
|
5
34
|
* hold (re-dispatch immediately — a rotation candidate is ready, or no idle delay was
|
|
@@ -76,5 +105,6 @@ export type SessionWorkflowInput = {
|
|
|
76
105
|
initialEventId?: string;
|
|
77
106
|
maxTurnsPerRun?: number;
|
|
78
107
|
maxCapacityChecksPerRun?: number;
|
|
108
|
+
capacityWakeJitterMaxMs?: number;
|
|
79
109
|
};
|
|
80
110
|
export declare function sessionWorkflow(input: SessionWorkflowInput): Promise<void>;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@opengeni/worker-bundle",
|
|
3
|
-
"version": "2.0.
|
|
3
|
+
"version": "2.0.3-canary.36199476632001",
|
|
4
4
|
"description": "OpenGeni worker entry and reusable embedded lifecycle, shipped with a release-coherent pre-bundled Temporal workflow artifact.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"repository": {
|
|
@@ -54,24 +54,25 @@
|
|
|
54
54
|
},
|
|
55
55
|
"dependencies": {
|
|
56
56
|
"@llamaindex/liteparse": "2.14.2",
|
|
57
|
-
"@opengeni/agent-proto": "^0.6.0",
|
|
58
|
-
"@opengeni/capabilities": "^0.3.4",
|
|
59
|
-
"@opengeni/codemode": "^0.6.
|
|
60
|
-
"@opengeni/codex": "^0.2.
|
|
61
|
-
"@opengeni/config": "^2.1.
|
|
62
|
-
"@opengeni/contracts": "^5.1.
|
|
63
|
-
"@opengeni/core": "^4.0.
|
|
64
|
-
"@opengeni/db": "^6.0.
|
|
65
|
-
"@opengeni/documents": "^0.8.
|
|
66
|
-
"@opengeni/events": "^0.4.
|
|
67
|
-
"@opengeni/github": "^0.7.
|
|
68
|
-
"@opengeni/
|
|
69
|
-
"@opengeni/
|
|
70
|
-
"@opengeni/
|
|
71
|
-
"@opengeni/
|
|
72
|
-
"@opengeni/
|
|
73
|
-
"@opengeni/
|
|
74
|
-
"@opengeni/
|
|
57
|
+
"@opengeni/agent-proto": "^0.6.0-canary.36199476632001",
|
|
58
|
+
"@opengeni/capabilities": "^0.3.4-canary.36199476632001",
|
|
59
|
+
"@opengeni/codemode": "^0.6.2-canary.36199476632001",
|
|
60
|
+
"@opengeni/codex": "^0.2.26-canary.36199476632001",
|
|
61
|
+
"@opengeni/config": "^2.1.1-canary.36199476632001",
|
|
62
|
+
"@opengeni/contracts": "^5.1.1-canary.36199476632001",
|
|
63
|
+
"@opengeni/core": "^4.0.3-canary.36199476632001",
|
|
64
|
+
"@opengeni/db": "^6.0.3-canary.36199476632001",
|
|
65
|
+
"@opengeni/documents": "^0.8.34-canary.36199476632001",
|
|
66
|
+
"@opengeni/events": "^0.4.32-canary.36199476632001",
|
|
67
|
+
"@opengeni/github": "^0.7.16-canary.36199476632001",
|
|
68
|
+
"@opengeni/jev": "^0.1.0-canary.36199476632001",
|
|
69
|
+
"@opengeni/network": "^0.3.1-canary.36199476632001",
|
|
70
|
+
"@opengeni/observability": "^0.8.32-canary.36199476632001",
|
|
71
|
+
"@opengeni/runtime": "^4.0.3-canary.36199476632001",
|
|
72
|
+
"@opengeni/sdk": "^7.1.1-canary.36199476632001",
|
|
73
|
+
"@opengeni/storage": "^0.2.133-canary.36199476632001",
|
|
74
|
+
"@opengeni/tool-gateway": "^0.1.13-canary.36199476632001",
|
|
75
|
+
"@opengeni/xai-subscription": "^0.1.4-canary.36199476632001",
|
|
75
76
|
"@temporalio/activity": "^1.17.0",
|
|
76
77
|
"@temporalio/client": "^1.17.0",
|
|
77
78
|
"@temporalio/worker": "^1.17.0",
|
|
@@ -131,6 +131,8 @@ export type BuildTurnAgentDeps = {
|
|
|
131
131
|
connectorActionPolicy: ConnectorActionPolicyHooks;
|
|
132
132
|
trigger: ClaimTurnOk["trigger"];
|
|
133
133
|
preparationIndependentToolNames: readonly string[];
|
|
134
|
+
/** The attempt's tool catalog includes the Jev-backed code_search tool. */
|
|
135
|
+
codeSearchAvailable: boolean;
|
|
134
136
|
videoGenerationAcceptancesByCallId: Map<string, { operationId: string; requestDigest: string }>;
|
|
135
137
|
activeSandboxBackend: Settings["sandboxBackend"] | undefined;
|
|
136
138
|
groupBoxBackend: Settings["sandboxBackend"];
|
|
@@ -189,6 +191,7 @@ export async function buildTurnAgent(deps: BuildTurnAgentDeps) {
|
|
|
189
191
|
connectorActionPolicy,
|
|
190
192
|
trigger,
|
|
191
193
|
preparationIndependentToolNames,
|
|
194
|
+
codeSearchAvailable,
|
|
192
195
|
videoGenerationAcceptancesByCallId,
|
|
193
196
|
activeSandboxBackend,
|
|
194
197
|
groupBoxBackend,
|
|
@@ -676,6 +679,7 @@ export async function buildTurnAgent(deps: BuildTurnAgentDeps) {
|
|
|
676
679
|
? { gitTokenSeed: sandboxGitToken }
|
|
677
680
|
: {}),
|
|
678
681
|
...(sandboxCodemodeToken ? { codemodeAvailable: true } : {}),
|
|
682
|
+
...(codeSearchAvailable ? { codeSearchAvailable: true } : {}),
|
|
679
683
|
// Managed boxes receive the bearer through their protected per-session
|
|
680
684
|
// token file. Connected Machines use transient per-exec delivery above,
|
|
681
685
|
// so they must not run the file-seeding lifecycle hook.
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
import type { AttemptToolDefinition } from "@opengeni/codemode";
|
|
2
|
+
import { usableJevApiKey, type Settings } from "@opengeni/config";
|
|
3
|
+
import {
|
|
4
|
+
CODE_SEARCH_TOOL_DESCRIPTION,
|
|
5
|
+
CODE_SEARCH_TOOL_NAME,
|
|
6
|
+
CodeSearchArgumentError,
|
|
7
|
+
CodeSearchRipgrepMissingError,
|
|
8
|
+
CodeSearchWorkspaceError,
|
|
9
|
+
JevCircuitBreaker,
|
|
10
|
+
JevClient,
|
|
11
|
+
JevRequestError,
|
|
12
|
+
JevUnavailableError,
|
|
13
|
+
codeSearchInputSchema,
|
|
14
|
+
parseCodeSearchArguments,
|
|
15
|
+
renderCodeSearchError,
|
|
16
|
+
runCodeSearch,
|
|
17
|
+
type CodeSearchWorkspace,
|
|
18
|
+
type JevCircuitLease,
|
|
19
|
+
} from "@opengeni/jev";
|
|
20
|
+
import type { Observability } from "@opengeni/observability";
|
|
21
|
+
import {
|
|
22
|
+
ChannelANotFoundError,
|
|
23
|
+
ChannelAUnavailableError,
|
|
24
|
+
ChannelAUnsupportedError,
|
|
25
|
+
ChannelAValidationError,
|
|
26
|
+
isWindowsConnectedMachinePath,
|
|
27
|
+
type SandboxChannelAService,
|
|
28
|
+
} from "@opengeni/runtime/sandbox";
|
|
29
|
+
import { recordCodeSearchCall, type CodeSearchCallOutcome } from "../../observability-metrics";
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* One breaker per worker process. Repeated Jev outages (including an exhausted
|
|
33
|
+
* account) make `code_search` calls on this worker fail at once for a cooldown
|
|
34
|
+
* instead of each running a full retry cycle. The breaker never changes which
|
|
35
|
+
* tools a turn is offered: the tool list and instructions are the start of the
|
|
36
|
+
* model's cached prompt, and sessions move between workers whose breakers
|
|
37
|
+
* disagree.
|
|
38
|
+
*/
|
|
39
|
+
export const codeSearchCircuitBreaker = new JevCircuitBreaker();
|
|
40
|
+
|
|
41
|
+
/** Ripgrep output kept on the box per call before framing. */
|
|
42
|
+
const CODE_SEARCH_RIPGREP_MAX_BYTES = 32 * 1024 * 1024;
|
|
43
|
+
|
|
44
|
+
type CodeSearchChannel = Pick<
|
|
45
|
+
SandboxChannelAService,
|
|
46
|
+
"codeSearchRipgrep" | "codeSearchPathKinds" | "fsRead"
|
|
47
|
+
>;
|
|
48
|
+
|
|
49
|
+
/** Adapt the turn's sandbox or Connected Machine to the code search engine. */
|
|
50
|
+
export function codeSearchWorkspaceFromChannel(channel: CodeSearchChannel): CodeSearchWorkspace {
|
|
51
|
+
return {
|
|
52
|
+
ripgrep: async (args, options) => {
|
|
53
|
+
options.signal?.throwIfAborted();
|
|
54
|
+
const outcome = await channel.codeSearchRipgrep(args, {
|
|
55
|
+
timeoutMs: options.timeoutMs,
|
|
56
|
+
maxBytes: CODE_SEARCH_RIPGREP_MAX_BYTES,
|
|
57
|
+
});
|
|
58
|
+
if (!outcome.available) throw new CodeSearchRipgrepMissingError();
|
|
59
|
+
return {
|
|
60
|
+
stdout: outcome.stdout,
|
|
61
|
+
exitCode: outcome.exitCode,
|
|
62
|
+
truncated: outcome.truncated,
|
|
63
|
+
timedOut: outcome.timedOut,
|
|
64
|
+
};
|
|
65
|
+
},
|
|
66
|
+
readText: async (path, options) => {
|
|
67
|
+
options.signal?.throwIfAborted();
|
|
68
|
+
try {
|
|
69
|
+
const read = await channel.fsRead({ path, encoding: "utf8", maxBytes: options.maxBytes });
|
|
70
|
+
return { text: read.content, truncated: read.truncated, binary: read.isBinary };
|
|
71
|
+
} catch (error) {
|
|
72
|
+
if (error instanceof ChannelANotFoundError) return null;
|
|
73
|
+
throw error;
|
|
74
|
+
}
|
|
75
|
+
},
|
|
76
|
+
pathKinds: async (paths, options) => {
|
|
77
|
+
options.signal?.throwIfAborted();
|
|
78
|
+
return await channel.codeSearchPathKinds(paths);
|
|
79
|
+
},
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** Jev work done by one completed `code_search` call, for per-workspace usage records. */
|
|
84
|
+
export type CodeSearchUsage = {
|
|
85
|
+
operationId: string;
|
|
86
|
+
jevRequests: number;
|
|
87
|
+
jevInputTokens: number;
|
|
88
|
+
jevCostUsd: number;
|
|
89
|
+
};
|
|
90
|
+
|
|
91
|
+
function textResult(text: string, isError: boolean) {
|
|
92
|
+
return { isError, content: [{ type: "text" as const, text }] };
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* The model-facing `code_search` tool. Jev scores candidates inside the worker;
|
|
97
|
+
* the sandbox only runs read-only ripgrep and file reads, so the Jev key never
|
|
98
|
+
* leaves this process. A Jev failure is reported to the model instead of
|
|
99
|
+
* degrading to keyword-only ranking, which lowered answer quality in testing.
|
|
100
|
+
*/
|
|
101
|
+
export function createCodeSearchAttemptToolDefinition(input: {
|
|
102
|
+
settings: Pick<Settings, "jevApiKey" | "jevBaseUrl" | "jevModel" | "jevRequestTimeoutMs">;
|
|
103
|
+
apiKey: string;
|
|
104
|
+
workspace: () => Promise<CodeSearchWorkspace>;
|
|
105
|
+
observability: Observability;
|
|
106
|
+
/** Records Jev usage against the workspace. Failures are logged, never surfaced. */
|
|
107
|
+
recordUsage?: (usage: CodeSearchUsage) => Promise<void>;
|
|
108
|
+
breaker?: JevCircuitBreaker;
|
|
109
|
+
fetch?: typeof fetch;
|
|
110
|
+
}): AttemptToolDefinition {
|
|
111
|
+
const breaker = input.breaker ?? codeSearchCircuitBreaker;
|
|
112
|
+
const jev = new JevClient({
|
|
113
|
+
apiKey: input.apiKey,
|
|
114
|
+
baseUrl: input.settings.jevBaseUrl,
|
|
115
|
+
model: input.settings.jevModel,
|
|
116
|
+
timeoutMs: input.settings.jevRequestTimeoutMs,
|
|
117
|
+
...(input.fetch ? { fetch: input.fetch } : {}),
|
|
118
|
+
});
|
|
119
|
+
return {
|
|
120
|
+
identity: { serverId: "opengeni", toolName: CODE_SEARCH_TOOL_NAME },
|
|
121
|
+
modelName: CODE_SEARCH_TOOL_NAME,
|
|
122
|
+
codemodePath: ["opengeni", CODE_SEARCH_TOOL_NAME],
|
|
123
|
+
title: "Search code",
|
|
124
|
+
description: CODE_SEARCH_TOOL_DESCRIPTION,
|
|
125
|
+
inputSchema: codeSearchInputSchema,
|
|
126
|
+
annotations: {
|
|
127
|
+
title: "Search code",
|
|
128
|
+
readOnlyHint: true,
|
|
129
|
+
destructiveHint: false,
|
|
130
|
+
idempotentHint: true,
|
|
131
|
+
openWorldHint: false,
|
|
132
|
+
},
|
|
133
|
+
source: "opengeni",
|
|
134
|
+
approval: "none",
|
|
135
|
+
execute: async (args, context) => {
|
|
136
|
+
const startedAt = performance.now();
|
|
137
|
+
let outcome: CodeSearchCallOutcome = "failed";
|
|
138
|
+
let jevRequests = 0;
|
|
139
|
+
let jevCostUsd = 0;
|
|
140
|
+
// The breaker lease (the single trial while half-open) is settled exactly
|
|
141
|
+
// once: by what Jev did, or released when Jev was never judged. Settling
|
|
142
|
+
// takes it, so the finally block releases only a lease still held.
|
|
143
|
+
let lease: JevCircuitLease | null = null;
|
|
144
|
+
try {
|
|
145
|
+
const request = parseCodeSearchArguments(args);
|
|
146
|
+
lease = breaker.tryAcquire(Date.now());
|
|
147
|
+
if (!lease) {
|
|
148
|
+
outcome = "breaker_open";
|
|
149
|
+
return textResult(
|
|
150
|
+
renderCodeSearchError(
|
|
151
|
+
new JevUnavailableError("Jev is not responding; calls are paused for a few minutes"),
|
|
152
|
+
),
|
|
153
|
+
true,
|
|
154
|
+
);
|
|
155
|
+
}
|
|
156
|
+
const workspace = await input.workspace();
|
|
157
|
+
const result = await runCodeSearch({
|
|
158
|
+
...request,
|
|
159
|
+
workspace,
|
|
160
|
+
jev,
|
|
161
|
+
...(context.signal ? { signal: context.signal } : {}),
|
|
162
|
+
});
|
|
163
|
+
const held = lease;
|
|
164
|
+
lease = null;
|
|
165
|
+
// A pack whose final status check hit an outage still counts against Jev.
|
|
166
|
+
if (result.statusCheckError instanceof JevUnavailableError) {
|
|
167
|
+
breaker.recordFailure(result.statusCheckError, Date.now(), held);
|
|
168
|
+
} else if (result.stats.jev.requests > 0) {
|
|
169
|
+
breaker.recordSuccess(held);
|
|
170
|
+
} else {
|
|
171
|
+
breaker.release(held);
|
|
172
|
+
}
|
|
173
|
+
outcome = "completed";
|
|
174
|
+
jevRequests = result.stats.jev.requests;
|
|
175
|
+
jevCostUsd = result.stats.jev.costUsd;
|
|
176
|
+
if (input.recordUsage && jevRequests > 0) {
|
|
177
|
+
await input
|
|
178
|
+
.recordUsage({
|
|
179
|
+
operationId: context.operationId,
|
|
180
|
+
jevRequests,
|
|
181
|
+
jevInputTokens: result.stats.jev.inputTokens,
|
|
182
|
+
jevCostUsd,
|
|
183
|
+
})
|
|
184
|
+
.catch((error: unknown) => {
|
|
185
|
+
input.observability.warn("code_search usage record failed", {
|
|
186
|
+
error: error instanceof Error ? error.message : String(error),
|
|
187
|
+
});
|
|
188
|
+
});
|
|
189
|
+
}
|
|
190
|
+
return textResult(result.text, false);
|
|
191
|
+
} catch (error) {
|
|
192
|
+
if (context.signal?.aborted) {
|
|
193
|
+
outcome = "cancelled";
|
|
194
|
+
throw error;
|
|
195
|
+
}
|
|
196
|
+
if (error instanceof CodeSearchArgumentError) {
|
|
197
|
+
outcome = "invalid_arguments";
|
|
198
|
+
return textResult(error.message, true);
|
|
199
|
+
}
|
|
200
|
+
if (error instanceof JevUnavailableError) {
|
|
201
|
+
const held = lease;
|
|
202
|
+
lease = null;
|
|
203
|
+
breaker.recordFailure(error, Date.now(), held);
|
|
204
|
+
outcome = "jev_unavailable";
|
|
205
|
+
return textResult(renderCodeSearchError(error), true);
|
|
206
|
+
}
|
|
207
|
+
if (error instanceof JevRequestError) {
|
|
208
|
+
outcome = "jev_rejected";
|
|
209
|
+
return textResult(renderCodeSearchError(error), true);
|
|
210
|
+
}
|
|
211
|
+
if (error instanceof CodeSearchWorkspaceError) {
|
|
212
|
+
outcome = "workspace_unavailable";
|
|
213
|
+
return textResult(renderCodeSearchError(error), true);
|
|
214
|
+
}
|
|
215
|
+
if (
|
|
216
|
+
error instanceof ChannelAUnavailableError ||
|
|
217
|
+
error instanceof ChannelAUnsupportedError ||
|
|
218
|
+
error instanceof ChannelAValidationError
|
|
219
|
+
) {
|
|
220
|
+
outcome = "workspace_unavailable";
|
|
221
|
+
return textResult(
|
|
222
|
+
renderCodeSearchError(new CodeSearchWorkspaceError(error.message)),
|
|
223
|
+
true,
|
|
224
|
+
);
|
|
225
|
+
}
|
|
226
|
+
throw error;
|
|
227
|
+
} finally {
|
|
228
|
+
if (lease) breaker.release(lease);
|
|
229
|
+
recordCodeSearchCall(input.observability, {
|
|
230
|
+
outcome,
|
|
231
|
+
durationSeconds: (performance.now() - startedAt) / 1_000,
|
|
232
|
+
jevRequests,
|
|
233
|
+
jevCostUsd,
|
|
234
|
+
});
|
|
235
|
+
}
|
|
236
|
+
},
|
|
237
|
+
};
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* The `code_search` definition for one turn, or none. It is offered only when
|
|
242
|
+
* the session's frozen decision and the deployment enable it, a usable Jev key
|
|
243
|
+
* exists, and the turn has compute that can run its POSIX shell commands (not
|
|
244
|
+
* a Windows Connected Machine). Every input is durable, so the tool list stays
|
|
245
|
+
* the same from turn to turn and on every worker. Transient Jev health never
|
|
246
|
+
* hides the tool; the breaker only refuses calls.
|
|
247
|
+
*/
|
|
248
|
+
export function codeSearchToolDefinitions(input: {
|
|
249
|
+
enabled: boolean;
|
|
250
|
+
settings: Pick<Settings, "jevApiKey" | "jevBaseUrl" | "jevModel" | "jevRequestTimeoutMs">;
|
|
251
|
+
backend: Settings["sandboxBackend"];
|
|
252
|
+
/** The turn's Connected Machine workspace root, when a machine is primary. */
|
|
253
|
+
machineWorkspaceRoot?: string | null;
|
|
254
|
+
observability: Observability;
|
|
255
|
+
workspace: () => Promise<CodeSearchWorkspace>;
|
|
256
|
+
recordUsage?: (usage: CodeSearchUsage) => Promise<void>;
|
|
257
|
+
breaker?: JevCircuitBreaker;
|
|
258
|
+
}): AttemptToolDefinition[] {
|
|
259
|
+
const apiKey = usableJevApiKey(input.settings);
|
|
260
|
+
const breaker = input.breaker ?? codeSearchCircuitBreaker;
|
|
261
|
+
if (!input.enabled || !apiKey || input.backend === "none") return [];
|
|
262
|
+
if (input.machineWorkspaceRoot && isWindowsConnectedMachinePath(input.machineWorkspaceRoot)) {
|
|
263
|
+
return [];
|
|
264
|
+
}
|
|
265
|
+
return [
|
|
266
|
+
createCodeSearchAttemptToolDefinition({
|
|
267
|
+
settings: input.settings,
|
|
268
|
+
apiKey,
|
|
269
|
+
workspace: input.workspace,
|
|
270
|
+
observability: input.observability,
|
|
271
|
+
...(input.recordUsage ? { recordUsage: input.recordUsage } : {}),
|
|
272
|
+
breaker,
|
|
273
|
+
}),
|
|
274
|
+
];
|
|
275
|
+
}
|
|
@@ -43,6 +43,7 @@ import { recordTurnStartupPhase } from "../../observability-metrics";
|
|
|
43
43
|
import { createTurnCredentialLeases } from "./credential-leases";
|
|
44
44
|
import { deliverFailedChildTurnToParent } from "../parent-wake";
|
|
45
45
|
import { randomUUID } from "node:crypto";
|
|
46
|
+
import { createLogThrottle, type LogThrottle } from "@opengeni/observability";
|
|
46
47
|
|
|
47
48
|
import { refreshCappedCodexUsageRows } from "./codex";
|
|
48
49
|
import { codexUsageLimitFailurePayload, CODEX_USAGE_LIMIT_MAX_RESUME_MS } from "./errors";
|
|
@@ -56,6 +57,39 @@ import type {
|
|
|
56
57
|
TurnControlState,
|
|
57
58
|
} from "./turn-context";
|
|
58
59
|
|
|
60
|
+
/** The eligible-pool gauge and `opengeni_codex_pool_low_total` stay per turn;
|
|
61
|
+
* the warning line is the first observation per workspace pool depth, then at
|
|
62
|
+
* most one per interval with the count it hid. The public log projection drops
|
|
63
|
+
* the identifiers and counts, so the closed `reason` keeps the depth visible. */
|
|
64
|
+
export const CODEX_POOL_LOW_WARNING_INTERVAL_MS = 10 * 60_000;
|
|
65
|
+
const codexPoolLowWarningThrottle = createLogThrottle({
|
|
66
|
+
intervalMs: CODEX_POOL_LOW_WARNING_INTERVAL_MS,
|
|
67
|
+
maxKeys: 1_024,
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
export function warnCodexPoolLow(
|
|
71
|
+
observability: Pick<ActivityServices["observability"], "warn">,
|
|
72
|
+
input: {
|
|
73
|
+
workspaceKey: string;
|
|
74
|
+
workspaceId: string;
|
|
75
|
+
eligibleCount: number;
|
|
76
|
+
connectedCount: number;
|
|
77
|
+
depth: "zero" | "one";
|
|
78
|
+
},
|
|
79
|
+
throttle: LogThrottle = codexPoolLowWarningThrottle,
|
|
80
|
+
): void {
|
|
81
|
+
const admission = throttle.admit(`${input.workspaceKey}:${input.depth}`);
|
|
82
|
+
if (!admission) return;
|
|
83
|
+
observability.warn("Codex eligible credential pool is low", {
|
|
84
|
+
workspaceId: input.workspaceId,
|
|
85
|
+
eligibleCount: input.eligibleCount,
|
|
86
|
+
connectedCount: input.connectedCount,
|
|
87
|
+
depth: input.depth,
|
|
88
|
+
reason: input.depth === "zero" ? "eligible_pool_zero" : "eligible_pool_one",
|
|
89
|
+
...(admission.suppressedCount > 0 ? { suppressedCount: admission.suppressedCount } : {}),
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
|
|
59
93
|
export type CapacityPhaseDeps = {
|
|
60
94
|
input: RunAgentTurnInput;
|
|
61
95
|
settings: Settings;
|
|
@@ -402,11 +436,12 @@ export async function selectCodexTurnCapacity(
|
|
|
402
436
|
help: "Alert signal emitted when the eligible Codex pool is zero or one.",
|
|
403
437
|
labels: { workspace_key: codexWorkspaceKey, depth: poolDepth },
|
|
404
438
|
});
|
|
405
|
-
observability
|
|
439
|
+
warnCodexPoolLow(observability, {
|
|
440
|
+
workspaceKey: codexWorkspaceKey,
|
|
406
441
|
workspaceId: input.workspaceId,
|
|
407
442
|
eligibleCount,
|
|
408
443
|
connectedCount: leased.accounts.length,
|
|
409
|
-
depth: poolDepth,
|
|
444
|
+
depth: poolDepth === "zero" ? "zero" : "one",
|
|
410
445
|
});
|
|
411
446
|
}
|
|
412
447
|
|
|
@@ -11,7 +11,8 @@ import {
|
|
|
11
11
|
compactionThresholdTokens,
|
|
12
12
|
CompactionNeededError,
|
|
13
13
|
compactionProviderRejection,
|
|
14
|
-
|
|
14
|
+
estimateSerializedValueTokens,
|
|
15
|
+
compactionSummaryOutputTokens,
|
|
15
16
|
type ModelResponseUsage,
|
|
16
17
|
} from "@opengeni/runtime";
|
|
17
18
|
import { type Settings } from "@opengeni/config";
|
|
@@ -168,6 +169,13 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
|
|
|
168
169
|
const remotePrefix: RemoteCompactionPrefix = {
|
|
169
170
|
agent: null,
|
|
170
171
|
};
|
|
172
|
+
const portableResponsesNeedsAgentPrefix =
|
|
173
|
+
resolvedModel?.provider.api === "responses" &&
|
|
174
|
+
!(billingState.isCodexTurn && session.codexCompactionMode === "remote_v2");
|
|
175
|
+
const preparedPortableRequest = () => {
|
|
176
|
+
if (!remotePrefix.agent) throw new Error("Compaction agent is unavailable");
|
|
177
|
+
return preparedCompactionRequest(remotePrefix.agent);
|
|
178
|
+
};
|
|
171
179
|
|
|
172
180
|
const promptCacheKey = acceptsPromptCacheKeyForTurn(resolvedModel) ? input.sessionId : undefined;
|
|
173
181
|
const compactionUsageState = createCompactionModelUsageEventState(claimedModelUsageSourceKeys);
|
|
@@ -200,8 +208,8 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
|
|
|
200
208
|
contextContributions: eventing.companyBrainContextContributions,
|
|
201
209
|
});
|
|
202
210
|
};
|
|
203
|
-
const compactionSummarizerFor = (systemInstructions?: string) =>
|
|
204
|
-
resolvedModel
|
|
211
|
+
const compactionSummarizerFor = (systemInstructions?: string): CompactionSummarizer => {
|
|
212
|
+
const summarize: CompactionSummarizer = resolvedModel
|
|
205
213
|
? (s: Settings, m: Array<Record<string, unknown>>) =>
|
|
206
214
|
withProviderRequestContext(() =>
|
|
207
215
|
summarizeContextForCompaction(s, m, {
|
|
@@ -209,20 +217,37 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
|
|
|
209
217
|
provider: resolvedModel.provider,
|
|
210
218
|
api: resolvedModel.provider.api,
|
|
211
219
|
model: turnExecutionPolicy.upstreamModelId,
|
|
212
|
-
maxOutputTokens:
|
|
220
|
+
maxOutputTokens: compactionSummaryOutputTokens(s.contextWindowTokens),
|
|
221
|
+
...(cancellationSignal ? { signal: cancellationSignal } : {}),
|
|
213
222
|
onUsage: recordCompactionUsage,
|
|
214
223
|
...(systemInstructions ? { systemInstructions } : {}),
|
|
215
224
|
...(promptCacheKey ? { promptCacheKey } : {}),
|
|
225
|
+
...(portableResponsesNeedsAgentPrefix
|
|
226
|
+
? { preparedRequest: preparedPortableRequest() }
|
|
227
|
+
: {}),
|
|
216
228
|
}),
|
|
217
229
|
)
|
|
218
230
|
: (s: Settings, m: Array<Record<string, unknown>>) =>
|
|
219
231
|
summarizeContextForCompaction(s, m, {
|
|
220
232
|
model: turnExecutionPolicy.upstreamModelId,
|
|
221
|
-
maxOutputTokens:
|
|
233
|
+
maxOutputTokens: compactionSummaryOutputTokens(s.contextWindowTokens),
|
|
234
|
+
...(cancellationSignal ? { signal: cancellationSignal } : {}),
|
|
222
235
|
onUsage: recordCompactionUsage,
|
|
223
236
|
...(systemInstructions ? { systemInstructions } : {}),
|
|
224
237
|
...(promptCacheKey ? { promptCacheKey } : {}),
|
|
225
238
|
});
|
|
239
|
+
summarize.estimatePrefixTokens = () => {
|
|
240
|
+
if (resolvedModel?.provider.api === "chat") {
|
|
241
|
+
return estimateSerializedValueTokens(systemInstructions ?? "");
|
|
242
|
+
}
|
|
243
|
+
const prepared = portableResponsesNeedsAgentPrefix ? preparedPortableRequest() : null;
|
|
244
|
+
return (
|
|
245
|
+
estimateSerializedValueTokens(prepared?.systemInstructions ?? systemInstructions ?? "") +
|
|
246
|
+
(prepared ? estimateSerializedValueTokens(prepared.tools) : 0)
|
|
247
|
+
);
|
|
248
|
+
};
|
|
249
|
+
return summarize;
|
|
250
|
+
};
|
|
226
251
|
// Prompt-cache prefix for remote_v2 MUST match ordinary turns:
|
|
227
252
|
// tools → instructions → history. Filled after buildAgent for every
|
|
228
253
|
// compact path (including operator /compact, which now builds the agent
|
|
@@ -265,13 +290,9 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
|
|
|
265
290
|
...(remoteCompactionRequester ? { requestRemoteCompactionV2: remoteCompactionRequester } : {}),
|
|
266
291
|
} as const;
|
|
267
292
|
|
|
268
|
-
//
|
|
269
|
-
//
|
|
270
|
-
//
|
|
271
|
-
// - remote_v2: fall through to prepareTools/buildAgent so the compact
|
|
272
|
-
// request reuses the ordinary tools→instructions cache prefix, then
|
|
273
|
-
// settle without inference. (Requester is also wired for Codex portable
|
|
274
|
-
// turns but unused there — gate on the frozen session mode.)
|
|
293
|
+
// Responses compaction, portable or remote, prepares the ordinary agent
|
|
294
|
+
// request first so tool schemas and instructions match the warm cache prefix.
|
|
295
|
+
// Chat providers keep the standalone portable maintenance path.
|
|
275
296
|
const compactionOnlyTurn = turn.source === "compaction";
|
|
276
297
|
const remoteV2CompactionNeedsAgentPrefix =
|
|
277
298
|
Boolean(remoteCompactionRequester) && session.codexCompactionMode === "remote_v2";
|
|
@@ -308,7 +329,11 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
|
|
|
308
329
|
control.activityStatus = "idle";
|
|
309
330
|
return claimedResult({ status: "idle" });
|
|
310
331
|
};
|
|
311
|
-
if (
|
|
332
|
+
if (
|
|
333
|
+
compactionOnlyTurn &&
|
|
334
|
+
!remoteV2CompactionNeedsAgentPrefix &&
|
|
335
|
+
!portableResponsesNeedsAgentPrefix
|
|
336
|
+
) {
|
|
312
337
|
const compactionInstructions = appendWorkspaceMemory(
|
|
313
338
|
appendSessionInstructions(
|
|
314
339
|
appendWorkspaceGovernance(
|
|
@@ -466,6 +491,7 @@ export async function runPostAgentCompaction(
|
|
|
466
491
|
claimedResult,
|
|
467
492
|
turn,
|
|
468
493
|
session,
|
|
494
|
+
resolvedModel,
|
|
469
495
|
remotePrefix,
|
|
470
496
|
remoteCompactionRequester,
|
|
471
497
|
publishCompactionLiveEvents,
|
|
@@ -480,10 +506,13 @@ export async function runPostAgentCompaction(
|
|
|
480
506
|
} = deps;
|
|
481
507
|
|
|
482
508
|
const agentInstructions = typeof agent.instructions === "string" ? agent.instructions : "";
|
|
509
|
+
const preparedPortable =
|
|
510
|
+
resolvedModel?.provider.api === "responses" &&
|
|
511
|
+
!(remoteCompactionRequester && session.codexCompactionMode === "remote_v2");
|
|
483
512
|
const compactSummarizer = compactionSummarizerFor(
|
|
484
513
|
agentInstructions.trim() ? agentInstructions : undefined,
|
|
485
514
|
);
|
|
486
|
-
if (remoteCompactionRequester) {
|
|
515
|
+
if (remoteCompactionRequester || preparedPortable) {
|
|
487
516
|
// The prefix is captured only after this agent passes normal SDK preparation.
|
|
488
517
|
remotePrefix.agent = agent;
|
|
489
518
|
}
|
|
@@ -491,7 +520,11 @@ export async function runPostAgentCompaction(
|
|
|
491
520
|
if (compactionOnlyTurn) {
|
|
492
521
|
const requested = await isSessionCompactionRequested(db, input.workspaceId, input.sessionId);
|
|
493
522
|
let outcome: Awaited<ReturnType<typeof maybeCompactContext>> | null = null;
|
|
494
|
-
if (
|
|
523
|
+
if (
|
|
524
|
+
requested &&
|
|
525
|
+
(preparedPortable ||
|
|
526
|
+
(remoteCompactionRequester && session.codexCompactionMode === "remote_v2"))
|
|
527
|
+
) {
|
|
495
528
|
queuePreparedCompaction(
|
|
496
529
|
agent,
|
|
497
530
|
new CompactionNeededError({
|
|
@@ -610,7 +643,10 @@ export async function runPostAgentCompaction(
|
|
|
610
643
|
return { exit: claimedResult({ status: "idle" }) };
|
|
611
644
|
}
|
|
612
645
|
|
|
613
|
-
if (
|
|
646
|
+
if (
|
|
647
|
+
preparedPortable ||
|
|
648
|
+
(remoteCompactionRequester && session.codexCompactionMode === "remote_v2")
|
|
649
|
+
) {
|
|
614
650
|
const forced = await isSessionCompactionRequested(db, input.workspaceId, input.sessionId);
|
|
615
651
|
const thresholdTokens = compactionThresholdTokens(eventing.modelRunSettings);
|
|
616
652
|
if (
|