@opengeni/worker-bundle 1.0.2 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/activities/agent-run-admission.d.ts +1 -14
  2. package/dist/activities/agent-turn/admission.d.ts +3 -2
  3. package/dist/activities/agent-turn/compaction-prep.d.ts +1 -0
  4. package/dist/activities/agent-turn/errors.d.ts +3 -0
  5. package/dist/activities/agent-turn/finalization-monitor.d.ts +6 -1
  6. package/dist/activities/agent-turn/model-usage.d.ts +10 -0
  7. package/dist/activities/agent-turn/provider-recovery-metrics.d.ts +18 -0
  8. package/dist/activities/agent-turn/turn-context.d.ts +5 -0
  9. package/dist/activities/goals.d.ts +1 -17
  10. package/dist/activities/types.d.ts +1 -1
  11. package/dist/{activities-control-BNPTRTOE.js → activities-control-QSDPXLLD.js} +26 -10
  12. package/dist/activities-control-QSDPXLLD.js.map +1 -0
  13. package/dist/{activities-turn-5LGZOSVU.js → activities-turn-BMY3CDKL.js} +191 -35
  14. package/dist/activities-turn-BMY3CDKL.js.map +1 -0
  15. package/dist/{chunk-LJX7Z2KP.js → chunk-JKMG5ZHH.js} +55 -167
  16. package/dist/chunk-JKMG5ZHH.js.map +1 -0
  17. package/dist/{chunk-3ZTRWDMN.js → chunk-XONQEVAD.js} +1 -1
  18. package/dist/chunk-XONQEVAD.js.map +1 -0
  19. package/dist/index.js +3 -4
  20. package/dist/index.js.map +1 -1
  21. package/dist/observability-metrics.d.ts +1 -1
  22. package/dist/workflow-bundle.js +1 -1
  23. package/package.json +21 -21
  24. package/src/activities/agent-run-admission.ts +1 -98
  25. package/src/activities/agent-turn/admission.ts +11 -5
  26. package/src/activities/agent-turn/claim.ts +12 -0
  27. package/src/activities/agent-turn/compaction-prep.ts +35 -1
  28. package/src/activities/agent-turn/errors.ts +51 -19
  29. package/src/activities/agent-turn/failure-settlement.ts +31 -1
  30. package/src/activities/agent-turn/finalization-monitor.ts +14 -1
  31. package/src/activities/agent-turn/finalization.ts +5 -0
  32. package/src/activities/agent-turn/model-usage.ts +38 -0
  33. package/src/activities/agent-turn/provider-recovery-metrics.ts +64 -0
  34. package/src/activities/agent-turn/run.ts +1 -0
  35. package/src/activities/agent-turn/stream-attempt.ts +32 -16
  36. package/src/activities/agent-turn/tool-environment.ts +9 -2
  37. package/src/activities/agent-turn/turn-context.ts +4 -0
  38. package/src/activities/goals.ts +17 -113
  39. package/src/activities/knowledge-indexing.ts +2 -2
  40. package/src/activities/scheduled-tasks.ts +18 -3
  41. package/src/activities/types.ts +1 -0
  42. package/src/index.ts +0 -2
  43. package/src/observability-metrics.ts +1 -0
  44. package/dist/activities-control-BNPTRTOE.js.map +0 -1
  45. package/dist/activities-turn-5LGZOSVU.js.map +0 -1
  46. package/dist/chunk-3ZTRWDMN.js.map +0 -1
  47. package/dist/chunk-LJX7Z2KP.js.map +0 -1
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@opengeni/worker-bundle",
3
- "version": "1.0.2",
3
+ "version": "1.2.0",
4
4
  "description": "OpenGeni worker entry and reusable embedded lifecycle, shipped with a release-coherent pre-bundled Temporal workflow artifact.",
5
5
  "license": "Apache-2.0",
6
6
  "repository": {
@@ -54,26 +54,26 @@
54
54
  },
55
55
  "dependencies": {
56
56
  "@llamaindex/liteparse": "2.14.2",
57
- "@opengeni/agent-proto": "1.0.2",
58
- "@opengeni/capabilities": "1.0.2",
59
- "@opengeni/codemode": "1.0.2",
60
- "@opengeni/codex": "1.0.2",
61
- "@opengeni/config": "1.0.2",
62
- "@opengeni/contracts": "1.0.2",
63
- "@opengeni/core": "1.0.2",
64
- "@opengeni/db": "1.0.2",
65
- "@opengeni/documents": "1.0.2",
66
- "@opengeni/events": "1.0.2",
67
- "@opengeni/github": "1.0.2",
68
- "@opengeni/interaction": "1.0.2",
69
- "@opengeni/jev": "1.0.2",
70
- "@opengeni/network": "1.0.2",
71
- "@opengeni/observability": "1.0.2",
72
- "@opengeni/runtime": "1.0.2",
73
- "@opengeni/sdk": "1.0.2",
74
- "@opengeni/storage": "1.0.2",
75
- "@opengeni/tool-gateway": "1.0.2",
76
- "@opengeni/xai-subscription": "1.0.2",
57
+ "@opengeni/agent-proto": "1.2.0",
58
+ "@opengeni/capabilities": "1.2.0",
59
+ "@opengeni/codemode": "1.2.0",
60
+ "@opengeni/codex": "1.2.0",
61
+ "@opengeni/config": "1.2.0",
62
+ "@opengeni/contracts": "1.2.0",
63
+ "@opengeni/core": "1.2.0",
64
+ "@opengeni/db": "1.2.0",
65
+ "@opengeni/documents": "1.2.0",
66
+ "@opengeni/events": "1.2.0",
67
+ "@opengeni/github": "1.2.0",
68
+ "@opengeni/interaction": "1.2.0",
69
+ "@opengeni/jev": "1.2.0",
70
+ "@opengeni/network": "1.2.0",
71
+ "@opengeni/observability": "1.2.0",
72
+ "@opengeni/runtime": "1.2.0",
73
+ "@opengeni/sdk": "1.2.0",
74
+ "@opengeni/storage": "1.2.0",
75
+ "@opengeni/tool-gateway": "1.2.0",
76
+ "@opengeni/xai-subscription": "1.2.0",
77
77
  "@temporalio/activity": "^1.17.0",
78
78
  "@temporalio/client": "^1.17.0",
79
79
  "@temporalio/worker": "^1.17.0",
@@ -1,98 +1 @@
1
- import { configuredStaticUsageLimits, type Settings } from "@opengeni/config";
2
- import { modelFundingForAdmission } from "@opengeni/core";
3
- import {
4
- checkWorkspaceAllowance,
5
- getBillingBalance,
6
- isCodexBilledTurn,
7
- sumUsageQuantity,
8
- } from "@opengeni/db";
9
- import type { ControlActivityServices } from "./types";
10
-
11
- export type AgentRunAdmissionDenial =
12
- | "insufficient_credits"
13
- | "allowance_exhausted"
14
- | "monthly_model_cost_limit"
15
- | "monthly_agent_run_limit";
16
-
17
- /** One worker-side admission boundary for service-authored agent runs. */
18
- export async function agentRunAdmissionDenial(
19
- services: Pick<ControlActivityServices, "db" | "entitlements"> & { settings: Settings },
20
- input: {
21
- accountId: string;
22
- workspaceId: string;
23
- model: string;
24
- requestedAgentRuns: number;
25
- /** The accepted work's causal human, not the scheduler/service caller. */
26
- initiatingHumanSubjectId?: string | null;
27
- },
28
- ): Promise<AgentRunAdmissionDenial | null> {
29
- const codexBilled = await isCodexBilledTurn({
30
- db: services.db,
31
- settings: services.settings,
32
- workspaceId: input.workspaceId,
33
- model: input.model,
34
- });
35
- const externallyBilled = modelFundingForAdmission(
36
- services.settings,
37
- input.model,
38
- codexBilled,
39
- ).fundedWithoutCredits;
40
- if (
41
- !externallyBilled &&
42
- (services.settings.billingMode === "stripe" || services.settings.usageLimitsMode === "managed")
43
- ) {
44
- if (services.entitlements) {
45
- const decision = await services.entitlements.admitRun({
46
- accountId: input.accountId,
47
- workspaceId: input.workspaceId,
48
- action: "agent_run:create",
49
- quantity: input.requestedAgentRuns,
50
- });
51
- if (!decision.allowed) return "insufficient_credits";
52
- } else {
53
- const balance = await getBillingBalance(services.db, input.accountId);
54
- if (balance.balanceMicros <= 0) return "insufficient_credits";
55
- }
56
- }
57
- if (!externallyBilled) {
58
- const refusal = await checkWorkspaceAllowance(services.db, {
59
- accountId: input.accountId,
60
- workspaceId: input.workspaceId,
61
- subjectId: input.initiatingHumanSubjectId ?? null,
62
- });
63
- if (refusal) return refusal.code;
64
- }
65
- if (
66
- services.settings.usageLimitsMode !== "static" &&
67
- services.settings.usageLimitsMode !== "managed"
68
- ) {
69
- return null;
70
- }
71
- const limits = configuredStaticUsageLimits(services.settings);
72
- if (!externallyBilled && limits.maxMonthlyCostMicrosPerAccount) {
73
- const used = await sumUsageQuantity(services.db, {
74
- accountId: input.accountId,
75
- eventType: "model.cost",
76
- since: startOfUtcMonth(),
77
- });
78
- if (used >= limits.maxMonthlyCostMicrosPerAccount) {
79
- return "monthly_model_cost_limit";
80
- }
81
- }
82
- if (limits.maxMonthlyAgentRunsPerWorkspace) {
83
- const used = await sumUsageQuantity(services.db, {
84
- workspaceId: input.workspaceId,
85
- eventType: "agent_run.created",
86
- since: startOfUtcMonth(),
87
- });
88
- if (used + input.requestedAgentRuns > limits.maxMonthlyAgentRunsPerWorkspace) {
89
- return "monthly_agent_run_limit";
90
- }
91
- }
92
- return null;
93
- }
94
-
95
- function startOfUtcMonth(): Date {
96
- const now = new Date();
97
- return new Date(Date.UTC(now.getUTCFullYear(), now.getUTCMonth(), 1));
98
- }
1
+ export { agentRunAdmissionDenial, type AgentRunAdmissionDenial } from "@opengeni/core";
@@ -1,4 +1,4 @@
1
- import { checkWorkspaceAllowance, getBillingBalance, sumUsageQuantity } from "@opengeni/db";
1
+ import { checkWorkspaceAllowance, getSpendableCreditBalance, sumUsageQuantity } from "@opengeni/db";
2
2
  import {
3
3
  configuredStaticUsageLimits,
4
4
  resolveTurnExecutionPolicyV1,
@@ -274,10 +274,11 @@ export async function ensureRunAllowedBetweenModelCalls(input: {
274
274
  chargesOpenGeniCredits: boolean;
275
275
  countsTowardTokenCap: boolean;
276
276
  initiatingHumanSubjectId: string | null;
277
+ modelId?: string;
277
278
  serializedRunState?: () => string | null;
278
- }): Promise<void> {
279
+ }): Promise<number | undefined> {
279
280
  try {
280
- await ensureRunAllowed(
281
+ return await ensureRunAllowed(
281
282
  input.settings,
282
283
  input.db,
283
284
  input.accountId,
@@ -287,6 +288,7 @@ export async function ensureRunAllowedBetweenModelCalls(input: {
287
288
  input.chargesOpenGeniCredits,
288
289
  input.countsTowardTokenCap,
289
290
  input.initiatingHumanSubjectId,
291
+ input.modelId,
290
292
  );
291
293
  } catch (limitError) {
292
294
  let serializedRunState: string | null = null;
@@ -317,7 +319,9 @@ export async function ensureRunAllowed(
317
319
  chargesOpenGeniCredits = !isExternallyBilledTurn,
318
320
  countsTowardTokenCap = !isExternallyBilledTurn,
319
321
  initiatingHumanSubjectId: string | null = null,
320
- ): Promise<void> {
322
+ modelId?: string,
323
+ ): Promise<number | undefined> {
324
+ let creditPolicyRevision: number | undefined;
321
325
  // Upstream settlement and workspace-facing cost are independent. External
322
326
  // metering skips the token cap; free/subscription/workspace cost skips the
323
327
  // OpenGeni credit gate. The agent-run COUNT cap below is a volume/fairness
@@ -352,7 +356,8 @@ export async function ensureRunAllowed(
352
356
  chargesOpenGeniCredits &&
353
357
  (settings.billingMode === "stripe" || settings.usageLimitsMode === "managed")
354
358
  ) {
355
- const balance = await getBillingBalance(db, accountId);
359
+ const balance = await getSpendableCreditBalance(db, accountId, modelId);
360
+ creditPolicyRevision = balance.creditPolicyRevision;
356
361
  if (balance.balanceMicros <= 0) {
357
362
  throw new Error("insufficient Opengeni credits");
358
363
  }
@@ -393,4 +398,5 @@ export async function ensureRunAllowed(
393
398
  }
394
399
  }
395
400
  }
401
+ return creditPolicyRevision;
396
402
  }
@@ -60,6 +60,7 @@ import { createTurnCredentialLeases } from "./credential-leases";
60
60
  import { createTurnMediaArtifacts } from "./media-artifacts";
61
61
  import { readTurnExecutionPolicyV1 } from "@opengeni/contracts";
62
62
  import { turnCredentialRestriction } from "./credential-restriction";
63
+ import { readProviderRecoveryObservation } from "./provider-recovery-metrics";
63
64
 
64
65
  import {
65
66
  credentialSubjectIdForTurnInitiator,
@@ -237,6 +238,7 @@ export async function claimTurnAttempt(deps: ClaimTurnDeps): Promise<ClaimTurnOu
237
238
  attempt.dispatchId = dispatchId;
238
239
  attempt.executionGeneration = turn.executionGeneration;
239
240
  attempt.providerRecoveryCount = providerRecoveryCountFromMetadata(turn.metadata);
241
+ attempt.providerRecoveryObservation = readProviderRecoveryObservation(turn.metadata ?? {});
240
242
  const authRecovery = turn.metadata?.claudeAuthRecovery;
241
243
  attempt.claudeAuthRecovery =
242
244
  authRecovery &&
@@ -391,6 +393,11 @@ export async function claimTurnAttempt(deps: ClaimTurnDeps): Promise<ClaimTurnOu
391
393
  latencyMode: turn.latencyMode,
392
394
  },
393
395
  );
396
+ // The durable same-turn recovery lane owns provider retries. Hidden SDK
397
+ // retries multiply that budget and keep the UI looking active during backoff.
398
+ // Apply before configuring/resolving clients so main, compaction and title
399
+ // requests all share this policy; standalone runtime consumers keep theirs.
400
+ capabilitySettings = { ...capabilitySettings, openaiMaxRetries: 0 };
394
401
  runtime.configure(capabilitySettings);
395
402
  const verifiedExecutionPolicy = assertTurnExecutionPolicyMatchesConfigV1(
396
403
  capabilitySettings,
@@ -402,6 +409,10 @@ export async function claimTurnAttempt(deps: ClaimTurnDeps): Promise<ClaimTurnOu
402
409
  },
403
410
  );
404
411
  const turnExecutionPolicy = verifiedExecutionPolicy.policy;
412
+ attempt.modelMetricRoute = {
413
+ provider: turnExecutionPolicy.providerId,
414
+ model: turnExecutionPolicy.productModelId,
415
+ };
405
416
  assertSessionAllowsProductModel(session, turnExecutionPolicy.productModelId);
406
417
  const billingIdentity = turnExecutionPolicyBillingIdentity(turnExecutionPolicy);
407
418
  billingState.isExternallyBilledTurn = billingIdentity.externallyBilled;
@@ -462,6 +473,7 @@ export async function claimTurnAttempt(deps: ClaimTurnDeps): Promise<ClaimTurnOu
462
473
  billingState.chargesOpenGeniCredits,
463
474
  billingState.countsTowardTokenCap,
464
475
  turn.initiatingHumanSubjectId,
476
+ turnExecutionPolicy.productModelId,
465
477
  ),
466
478
  cancellationSignal,
467
479
  undefined,
@@ -1,3 +1,4 @@
1
+ import { ensureRunAllowedBetweenModelCalls } from "./admission";
1
2
  import { hasPendingSteerAfterContextCompaction, isSessionCompactionRequested } from "@opengeni/db";
2
3
  import { publishDurableSessionEvents } from "@opengeni/events";
3
4
  import {
@@ -46,6 +47,7 @@ import {
46
47
  processCompactionModelUsageEvent,
47
48
  } from "./model-usage";
48
49
  import { waitForTurnOperation } from "./sandbox-provision";
50
+ import { recordProviderRecoveryOutcome } from "./provider-recovery-metrics";
49
51
 
50
52
  import type { ClaimTurnOk } from "./claim";
51
53
  import type { GovernanceModelOk } from "./governance-model";
@@ -66,6 +68,7 @@ export type CompactionPrepDeps = {
66
68
  input: RunAgentTurnInput;
67
69
  settings: Settings;
68
70
  db: ActivityServices["db"];
71
+ entitlements?: ActivityServices["entitlements"];
69
72
  bus: ActivityServices["bus"];
70
73
  observability: ActivityServices["observability"];
71
74
  cancellationSignal: AbortSignal | undefined;
@@ -236,9 +239,11 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
236
239
 
237
240
  const promptCacheKey = acceptsPromptCacheKeyForTurn(resolvedModel) ? input.sessionId : undefined;
238
241
  const compactionUsageState = createCompactionModelUsageEventState(claimedModelUsageSourceKeys);
242
+ let compactionCreditPolicyRevision: number | undefined;
239
243
  const recordCompactionUsage = async (usage: ModelResponseUsage) => {
240
244
  await processCompactionModelUsageEvent({
241
245
  usage,
246
+ creditPolicyRevision: compactionCreditPolicyRevision,
242
247
  state: compactionUsageState,
243
248
  dispatchId: modelUsageDispatchId,
244
249
  settings,
@@ -266,7 +271,7 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
266
271
  });
267
272
  };
268
273
  const compactionSummarizerFor = (systemInstructions?: string): CompactionSummarizer => {
269
- const summarize: CompactionSummarizer = resolvedModel
274
+ const summarizeModel: CompactionSummarizer = resolvedModel
270
275
  ? (s: Settings, m: Array<Record<string, unknown>>) =>
271
276
  withProviderRequestContext(() =>
272
277
  summarizeContextForCompaction(s, m, {
@@ -293,6 +298,21 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
293
298
  ...(systemInstructions ? { systemInstructions } : {}),
294
299
  ...(promptCacheKey ? { promptCacheKey } : {}),
295
300
  });
301
+ const summarize: CompactionSummarizer = async (s, m) => {
302
+ compactionCreditPolicyRevision = await ensureRunAllowedBetweenModelCalls({
303
+ settings: s,
304
+ db,
305
+ accountId: input.accountId,
306
+ workspaceId: input.workspaceId,
307
+ modelId: resolvedModel?.configured.id ?? turn.model,
308
+ isExternallyBilledTurn: billingState.isExternallyBilledTurn,
309
+ chargesOpenGeniCredits: billingState.chargesOpenGeniCredits,
310
+ countsTowardTokenCap: billingState.countsTowardTokenCap,
311
+ initiatingHumanSubjectId: turn.initiatingHumanSubjectId,
312
+ entitlements: deps.entitlements,
313
+ });
314
+ return await summarizeModel(s, m);
315
+ };
296
316
  summarize.estimatePrefixTokens = () => {
297
317
  if (resolvedModel?.provider.api === "chat") {
298
318
  return estimateSerializedValueTokens(systemInstructions ?? "");
@@ -330,6 +350,20 @@ export async function prepareCompaction(deps: CompactionPrepDeps): Promise<Compa
330
350
  await publishDurableSessionEvents(bus, input.workspaceId, input.sessionId, events);
331
351
  };
332
352
  const publishCompactionOutcomeEvents = async (events: SessionEvent[]) => {
353
+ if (events.some((event) => event.type === "session.context.compacted")) {
354
+ // The summary and recovery reset already committed under the attempt
355
+ // fence. Skipped compaction does not prove successful model progress.
356
+ attempt.providerRecoveryCount = 0;
357
+ if (attempt.providerRecoveryObservation) {
358
+ recordProviderRecoveryOutcome(observability, {
359
+ route: attempt.modelMetricRoute,
360
+ cause: attempt.providerRecoveryObservation.cause,
361
+ outcome: "recovered",
362
+ elapsedMs: Date.now() - attempt.providerRecoveryObservation.startedAt,
363
+ });
364
+ attempt.providerRecoveryObservation = undefined;
365
+ }
366
+ }
333
367
  // `compaction.started` was already fanout via publishCompactionLiveEvents.
334
368
  await publishDurableSessionEvents(
335
369
  bus,
@@ -11,6 +11,7 @@ import {
11
11
  safeDatabaseErrorFacts,
12
12
  isRetryableDatabaseTransportFailure,
13
13
  isSessionEventPersistenceError,
14
+ DatabaseTransactionError,
14
15
  SandboxLeaseTransitionError,
15
16
  } from "@opengeni/db";
16
17
  import {
@@ -100,6 +101,11 @@ export const MAX_AUTOMATIC_PROVIDER_RECOVERIES = PROVIDER_CONNECTIVITY_BACKOFF_M
100
101
  * alone would spend every automatic recovery before the window resets.
101
102
  */
102
103
  export const PROVIDER_RATE_LIMIT_BACKOFF_MS = [10_000, 20_000, 40_000, 60_000, 120_000] as const;
104
+ /** Positive-only spread: never shorten the provider's minimum delay. */
105
+ export function providerRecoveryJitterMs(delayMs: number, sample: number): number {
106
+ const bounded = Number.isFinite(sample) ? Math.max(0, Math.min(sample, 1)) : 0;
107
+ return Math.floor(Math.min(5_000, delayMs * 0.2) * bounded);
108
+ }
103
109
  export const POST_COMPACTION_CONTINUATION_EMPTY_CODE = "post_compaction_continuation_empty";
104
110
 
105
111
  export class PostCompactionContinuationEmptyError extends Error {
@@ -126,6 +132,7 @@ export function providerRecoveryResult(input: {
126
132
  failureCode: string | undefined;
127
133
  attemptNumber: number;
128
134
  retryAfterMs?: number | null;
135
+ jitterSample?: number;
129
136
  }): ProviderRecoveryResult {
130
137
  if (input.attemptNumber > MAX_AUTOMATIC_PROVIDER_RECOVERIES) {
131
138
  return {
@@ -171,7 +178,13 @@ export function providerRecoveryResult(input: {
171
178
  : PROVIDER_BACKPRESSURE_DELAY_MS;
172
179
  return {
173
180
  status: "recovering",
174
- continueDelayMs,
181
+ continueDelayMs:
182
+ continueDelayMs +
183
+ (input.failureCode === "provider_rate_limited" ||
184
+ input.failureCode === "provider_unavailable" ||
185
+ input.failureCode === "upstream_connectivity_unavailable"
186
+ ? providerRecoveryJitterMs(continueDelayMs, input.jitterSample ?? 0)
187
+ : 0),
175
188
  };
176
189
  }
177
190
 
@@ -407,6 +420,12 @@ function retryableDatabaseFailureCode(
407
420
  const boundaries = new Set<object>();
408
421
  const ownDatabaseNodes = new Set<object>();
409
422
  const codes = new Set<PostClaimDatabaseRecoveryDetail["code"]>();
423
+ const ownChildren = (node: object): object[] =>
424
+ node instanceof DatabaseTransactionError
425
+ ? // Callback failures retained beside a failed rollback supply vetoes,
426
+ // never driver provenance for a provider's connection-looking error.
427
+ [...graph.get(node)!].filter((child) => child === node.cause)
428
+ : [...graph.get(node)!];
410
429
  // Ask the canonical transport predicate about ONLY this node's facts. Its
411
430
  // recursive search must not pair a DB sibling with an unrelated provider.
412
431
  for (const node of graph.keys()) {
@@ -429,7 +448,11 @@ function retryableDatabaseFailureCode(
429
448
  : isDatabaseConnectionSqlState(sqlState)
430
449
  )
431
450
  transports.add(node);
432
- if (node instanceof DrizzleQueryError || isSessionEventPersistenceError(node)) {
451
+ if (
452
+ node instanceof DrizzleQueryError ||
453
+ node instanceof DatabaseTransactionError ||
454
+ isSessionEventPersistenceError(node)
455
+ ) {
433
456
  // Only actual errors raised at our ORM/typed persistence boundary own
434
457
  // their driver subtree. A PostgresError name, SDK wrapper or provider
435
458
  // socket by itself is never own-client provenance for a running turn.
@@ -437,7 +460,7 @@ function retryableDatabaseFailureCode(
437
460
  for (const source of queue) {
438
461
  if (ownDatabaseNodes.has(source)) continue;
439
462
  ownDatabaseNodes.add(source);
440
- queue.push(...graph.get(source)!);
463
+ queue.push(...ownChildren(source));
441
464
  }
442
465
  }
443
466
  }
@@ -448,7 +471,7 @@ function retryableDatabaseFailureCode(
448
471
  if (seen.has(node)) continue;
449
472
  seen.add(node);
450
473
  if (transports.has(node)) return true;
451
- queue.push(...graph.get(node)!);
474
+ queue.push(...ownChildren(node));
452
475
  }
453
476
  return false;
454
477
  };
@@ -457,7 +480,7 @@ function retryableDatabaseFailureCode(
457
480
  // order; a deeper transport/reset cannot override it either.
458
481
  for (const node of graph.keys()) {
459
482
  const record = node as Record<string, unknown>;
460
- if (node instanceof DrizzleQueryError) {
483
+ if (node instanceof DrizzleQueryError || node instanceof DatabaseTransactionError) {
461
484
  boundaries.add(node);
462
485
  if (hasOwnTransport(node)) codes.add("db_failure");
463
486
  }
@@ -501,6 +524,12 @@ function retryableDatabaseFailureCode(
501
524
  }
502
525
 
503
526
  const RUNNING_TURN_DATABASE_TRANSPORT_CODES = new Set([
527
+ // postgres.js reports these for a physically lost connection, including
528
+ // transaction cleanup after its socket has already closed. Own-client
529
+ // provenance and the unknown-outcome veto remain mandatory above.
530
+ "CONNECTION_CLOSED",
531
+ "CONNECTION_DESTROYED",
532
+ "CONNECTION_ENDED",
504
533
  "ECONNREFUSED",
505
534
  "ECONNRESET",
506
535
  "CONNECT_TIMEOUT",
@@ -1086,19 +1115,28 @@ function isProviderSafetyRefusal(error: unknown): boolean {
1086
1115
  return providerSafetyRefusalDiagnostic(error) !== undefined;
1087
1116
  }
1088
1117
 
1118
+ /** Preserve the closest real HTTP status through SDK Error.cause wrappers. */
1119
+ function providerHttpStatus(error: unknown): number | undefined {
1120
+ let current = error;
1121
+ const seen = new Set<unknown>();
1122
+ for (let depth = 0; depth < 6 && current && typeof current === "object"; depth += 1) {
1123
+ if (seen.has(current)) break;
1124
+ seen.add(current);
1125
+ const value = current as { status?: unknown; statusCode?: unknown; cause?: unknown };
1126
+ const status = Number(value.status ?? value.statusCode);
1127
+ if (Number.isInteger(status) && status >= 100 && status < 600) return status;
1128
+ current = value.cause;
1129
+ }
1130
+ return undefined;
1131
+ }
1132
+
1089
1133
  export function isTransientProviderError(error: unknown): boolean {
1090
1134
  if (error instanceof ResponsesStreamingTerminalError) {
1091
1135
  return error.category === "unavailable";
1092
1136
  }
1093
1137
  // A semantic refusal can arrive inside a 5xx transport envelope.
1094
1138
  if (isProviderSafetyRefusal(error)) return false;
1095
- const status =
1096
- typeof error === "object" && error !== null
1097
- ? Number(
1098
- (error as { status?: unknown; statusCode?: unknown }).status ??
1099
- (error as { statusCode?: unknown }).statusCode,
1100
- )
1101
- : undefined;
1139
+ const status = providerHttpStatus(error);
1102
1140
  // A real HTTP status is AUTHORITATIVE: a 5xx is transient, and ANY other status
1103
1141
  // (4xx validation/auth/404, plus the 429 the earlier branches already handled) is
1104
1142
  // a request fault that must NOT auto-retry — even if its body happens to read like
@@ -1399,13 +1437,7 @@ function baseAgentRunFailurePayload(
1399
1437
  };
1400
1438
  }
1401
1439
  const message = error instanceof Error ? error.message : String(error);
1402
- const status =
1403
- typeof error === "object" && error !== null
1404
- ? Number(
1405
- (error as { status?: unknown; statusCode?: unknown }).status ??
1406
- (error as { statusCode?: unknown }).statusCode,
1407
- )
1408
- : undefined;
1440
+ const status = providerHttpStatus(error);
1409
1441
  const code =
1410
1442
  typeof error === "object" && error !== null && "code" in error
1411
1443
  ? String((error as { code?: unknown }).code)
@@ -100,6 +100,7 @@ import type {
100
100
  } from "./turn-context";
101
101
  import type { CodexCredentialPolicySnapshotV1 } from "@opengeni/contracts";
102
102
  import { armAndReconcileCodexCapacityWait } from "../codex-capacity";
103
+ import { providerRecoveryCause, recordProviderRecoveryOutcome } from "./provider-recovery-metrics";
103
104
 
104
105
  export type TurnFailureDeps = {
105
106
  error: unknown;
@@ -276,12 +277,20 @@ export async function settleTurnFailure(deps: TurnFailureDeps): Promise<RunAgent
276
277
  try {
277
278
  return await settleTurnFailureInAttempt(deps);
278
279
  } catch (error) {
280
+ if (
281
+ deps.control.activityStatus === "recovering" &&
282
+ error instanceof ApplicationFailure &&
283
+ error.type === "OpenGeniPostClaimDatabaseRecovery"
284
+ )
285
+ throw error;
279
286
  // Connectivity can disappear while settling an unrelated run error too.
280
287
  // Do not overwrite a possibly committed settlement; the control lane
281
288
  // re-reads exact ownership and becomes a stale no-op if it already closed.
282
289
  if (deps.attempt.turnId && deps.attempt.triggerEventId) {
283
290
  const recovery = postClaimDatabaseRecoveryFailure({
284
- error,
291
+ // A failed rollback/terminal write must not erase no-replay evidence
292
+ // from the failure we were settling (notably unknown tool effects).
293
+ error: new AggregateError([error, deps.error], "Turn failure settlement failed"),
285
294
  turnId: deps.attempt.turnId,
286
295
  triggerEventId: deps.attempt.triggerEventId,
287
296
  executionGeneration: deps.attempt.executionGeneration,
@@ -1995,6 +2004,7 @@ async function settleTurnFailureInAttempt(deps: TurnFailureDeps): Promise<RunAge
1995
2004
  failureCode: failure.code,
1996
2005
  attemptNumber: nextProviderRecoveryCount,
1997
2006
  retryAfterMs: providerRetryAfterMs(error),
2007
+ jitterSample: Math.random(),
1998
2008
  });
1999
2009
  const setupRecoveryExhausted =
2000
2010
  earlyCommandStartUnavailable &&
@@ -2064,9 +2074,29 @@ async function settleTurnFailureInAttempt(deps: TurnFailureDeps): Promise<RunAge
2064
2074
  control.turnMetricOutcome = "recovering";
2065
2075
  control.activityStatus = "recovering";
2066
2076
  control.activityError = error;
2077
+ const recoveryCause = providerRecoveryCause(failure.code);
2078
+ if (recoveryCause) {
2079
+ recordProviderRecoveryOutcome(observability, {
2080
+ route: attempt.modelMetricRoute,
2081
+ cause: recoveryCause,
2082
+ outcome: "scheduled",
2083
+ delayMs: recoveryResult.continueDelayMs,
2084
+ });
2085
+ }
2067
2086
  return claimedResult(recoveryResult);
2068
2087
  }
2069
2088
  failure = providerRecoveryExhaustedFailure(failure, recoveryResult);
2089
+ const recoveryCause = providerRecoveryCause(failure.code);
2090
+ if (recoveryCause) {
2091
+ recordProviderRecoveryOutcome(observability, {
2092
+ route: attempt.modelMetricRoute,
2093
+ cause: recoveryCause,
2094
+ outcome: "exhausted",
2095
+ ...(attempt.providerRecoveryObservation
2096
+ ? { elapsedMs: Date.now() - attempt.providerRecoveryObservation.startedAt }
2097
+ : {}),
2098
+ });
2099
+ }
2070
2100
  if (earlyRecoverableSetup) {
2071
2101
  // Setup has no eventing sink yet. Carry only the fixed, safe diagnostic
2072
2102
  // through Temporal into exact-attempt workflow failure settlement.
@@ -1,4 +1,4 @@
1
- import type { Observability } from "@opengeni/observability";
1
+ import { turnExecutionTelemetryKey, type Observability } from "@opengeni/observability";
2
2
  import type { TurnHeartbeatDetails } from "../../op-journal";
3
3
  import { armTurnQuiescenceWatchdog } from "./quiescence";
4
4
 
@@ -42,10 +42,21 @@ export function startTurnFinalizationMonitor(input: {
42
42
  details: TurnHeartbeatDetails;
43
43
  heartbeat: (details: TurnHeartbeatDetails) => void;
44
44
  requestWorkerDrain: () => void;
45
+ execution?: { workspaceId: string; sessionId: string; attemptId: string };
45
46
  timeoutMs?: number;
46
47
  slowAfterMs?: number;
47
48
  }) {
48
49
  const { observability, details } = input;
50
+ const correlation: { correlationId?: string } = {};
51
+ diagnose(() => {
52
+ if (input.execution) {
53
+ correlation.correlationId = turnExecutionTelemetryKey(
54
+ input.execution.workspaceId,
55
+ input.execution.sessionId,
56
+ input.execution.attemptId,
57
+ );
58
+ }
59
+ });
49
60
  if (!initialized.has(observability)) {
50
61
  for (const stage of TURN_FINALIZATION_STAGES) {
51
62
  diagnose(() => observability.incrementGauge({ ...INFLIGHT, labels: { stage }, amount: 0 }));
@@ -88,6 +99,7 @@ export function startTurnFinalizationMonitor(input: {
88
99
  surface: "turn_finalization",
89
100
  outcome: "containment",
90
101
  reason: next,
102
+ ...correlation,
91
103
  });
92
104
  } catch {
93
105
  // Diagnostic failure must never disable physical containment.
@@ -104,6 +116,7 @@ export function startTurnFinalizationMonitor(input: {
104
116
  surface: "turn_finalization",
105
117
  outcome: "slow",
106
118
  reason: next,
119
+ ...correlation,
107
120
  });
108
121
  }),
109
122
  input.slowAfterMs ?? 30_000,
@@ -128,6 +128,11 @@ export async function finalizeTurnAttempt(deps: TurnFinalizationDeps): Promise<v
128
128
  }
129
129
  },
130
130
  requestWorkerDrain: deps.requestWorkerDrain,
131
+ execution: {
132
+ workspaceId: deps.input.workspaceId,
133
+ sessionId: deps.input.sessionId,
134
+ attemptId: deps.input.attemptId,
135
+ },
131
136
  });
132
137
  try {
133
138
  monitor.enter("tool_writers");