@oh-my-pi/pi-agent-core 18.3.5 → 18.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +1 -1
- package/dist/types/compaction/branch-summarization.d.ts +1 -1
- package/dist/types/compaction/compaction.d.ts +2 -2
- package/dist/types/compaction/openai.d.ts +11 -0
- package/dist/types/output-budget.d.ts +12 -1
- package/dist/types/telemetry.d.ts +103 -60
- package/package.json +8 -8
- package/src/agent-loop.ts +9 -2
- package/src/compaction/branch-summarization.ts +1 -1
- package/src/compaction/compaction-v2-streaming.ts +11 -2
- package/src/compaction/compaction.ts +4 -2
- package/src/compaction/openai.ts +26 -1
- package/src/output-budget.ts +14 -2
- package/src/telemetry.ts +240 -117
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,20 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.4.1] - 2026-09-28
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Fixed fitted output caps overshooting the context window by a few tokens on strict Chat Completions hosts (e.g. llama.cpp), causing 400s.
|
|
10
|
+
- Fixed native remote compaction sending requests already estimated past the model's context window (e.g. after re-expanding history behind another provider's native boundary); it now fails fast so the next configured compaction method runs ([#13502](https://github.com/can1357/oh-my-pi/issues/13502))
|
|
11
|
+
- Fixed V2 remote compaction retrying a standalone stream `error` event three times and reporting it as `stream closed before response.completed`; the upstream status, code, and message (e.g. `context_too_large`) are now surfaced ([#13502](https://github.com/can1357/oh-my-pi/issues/13502))
|
|
12
|
+
|
|
13
|
+
## [18.4.0] - 2026-09-28
|
|
14
|
+
|
|
15
|
+
### Changed
|
|
16
|
+
|
|
17
|
+
- Updated telemetry attribute names from the `pi.*` namespace to the `omp.*` namespace.
|
|
18
|
+
|
|
5
19
|
## [18.3.3] - 2026-09-27
|
|
6
20
|
|
|
7
21
|
### Added
|
package/README.md
CHANGED
|
@@ -462,7 +462,7 @@ const runCoverage = aggregateAgentRunCoverage(coverages);
|
|
|
462
462
|
|
|
463
463
|
### Tool status reporting
|
|
464
464
|
|
|
465
|
-
`execute_tool` spans carry `
|
|
465
|
+
`execute_tool` spans carry `omp.gen_ai.tool.status` ∈
|
|
466
466
|
`"ok" | "error" | "skipped" | "blocked" | "timeout" | "aborted"`.
|
|
467
467
|
`beforeToolCall` blocks throw a distinguishable `ToolCallBlockedError`
|
|
468
468
|
internally; the catch path reports `status: "blocked"` instead of conflating
|
|
@@ -55,7 +55,7 @@ export interface GenerateBranchSummaryOptions {
|
|
|
55
55
|
convertToLlm?: ConvertToLlm;
|
|
56
56
|
/**
|
|
57
57
|
* Optional telemetry handle. When provided, the branch summary LLM call is
|
|
58
|
-
* wrapped in an OTEL chat span tagged with `
|
|
58
|
+
* wrapped in an OTEL chat span tagged with `omp.gen_ai.oneshot.kind = "branch_summary"`.
|
|
59
59
|
*/
|
|
60
60
|
telemetry?: AgentTelemetry;
|
|
61
61
|
/**
|
|
@@ -180,7 +180,7 @@ export interface SummaryOptions {
|
|
|
180
180
|
/**
|
|
181
181
|
* Optional telemetry handle. When provided, every LLM call emitted during
|
|
182
182
|
* compaction is wrapped in an OTEL chat span tagged with
|
|
183
|
-
* `
|
|
183
|
+
* `omp.gen_ai.oneshot.kind` (`compaction_summary`, `compaction_short_summary`,
|
|
184
184
|
* or `compaction_turn_prefix`). `undefined` keeps the call paths zero-cost.
|
|
185
185
|
*/
|
|
186
186
|
telemetry?: AgentTelemetry;
|
|
@@ -243,7 +243,7 @@ export interface HandoffOptions {
|
|
|
243
243
|
metadata?: Record<string, unknown>;
|
|
244
244
|
/**
|
|
245
245
|
* Optional telemetry handle. When provided, the handoff LLM call is
|
|
246
|
-
* wrapped in an OTEL chat span tagged with `
|
|
246
|
+
* wrapped in an OTEL chat span tagged with `omp.gen_ai.oneshot.kind = "handoff"`.
|
|
247
247
|
*/
|
|
248
248
|
telemetry?: AgentTelemetry;
|
|
249
249
|
/**
|
|
@@ -33,6 +33,8 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
33
33
|
rewrittenOutputs: number;
|
|
34
34
|
estimatedTokensBefore: number;
|
|
35
35
|
estimatedTokensAfter: number;
|
|
36
|
+
/** Whether `input` fits the model window; false means it must not be sent. */
|
|
37
|
+
fits: boolean;
|
|
36
38
|
}
|
|
37
39
|
/**
|
|
38
40
|
* Preserve the full native transcript unless trailing tool outputs alone push a
|
|
@@ -41,6 +43,15 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
41
43
|
* matching Codex's recovery path for oversized tool turns.
|
|
42
44
|
*/
|
|
43
45
|
export declare function trimRemoteCompactionInputToContextWindow(input: Array<Record<string, unknown>>, tokenizer: Tokenizer, contextWindow: number | null | undefined, instructions: string, tools?: unknown[]): TrimRemoteCompactionInputResult;
|
|
46
|
+
/**
|
|
47
|
+
* Refuse a native compaction request whose prepared input cannot fit the model
|
|
48
|
+
* window, before any network I/O. Re-expanded history behind an unreadable
|
|
49
|
+
* native boundary can exceed the window even when live context does not.
|
|
50
|
+
*
|
|
51
|
+
* @throws Error flagged `ContextOverflow` when `trimmed.fits` is false, so
|
|
52
|
+
* compaction callers skip retries and advance to the next method.
|
|
53
|
+
*/
|
|
54
|
+
export declare function assertRemoteCompactionInputFits(trimmed: TrimRemoteCompactionInputResult, model: Model): void;
|
|
44
55
|
export type OpenAiRemoteCompactionItem = {
|
|
45
56
|
type: "compaction" | "compaction_summary";
|
|
46
57
|
encrypted_content?: string;
|
|
@@ -2,6 +2,16 @@ import type { Context, Model } from "@oh-my-pi/pi-ai";
|
|
|
2
2
|
import type { Tokenizer } from "./tokenizer.js";
|
|
3
3
|
/** Smallest output cap {@link fitOutputTokensToContextWindow} will request. */
|
|
4
4
|
export declare const MIN_FITTED_OUTPUT_TOKENS = 1024;
|
|
5
|
+
/**
|
|
6
|
+
* Absolute headway subtracted from the remaining room (unconditionally, on anchored and
|
|
7
|
+
* fully-local counts alike). Proportional padding covers
|
|
8
|
+
* tokenizer drift that scales with the prompt, but a host can still count a few tokens more
|
|
9
|
+
* than any local estimate can see (chat-template framing, reasoning wrappers). Measured
|
|
10
|
+
* against a 262,144-token host: fitted caps landed +11 to +42 tokens over and were rejected
|
|
11
|
+
* with 400s. 64 tokens covers the observed drift with margin to spare; the
|
|
12
|
+
* {@link MIN_FITTED_OUTPUT_TOKENS} floor still applies.
|
|
13
|
+
*/
|
|
14
|
+
export declare const OUTPUT_FIT_HEADWAY_TOKENS = 64;
|
|
5
15
|
/**
|
|
6
16
|
* Output cap for a request, so prompt plus output stays inside the model's
|
|
7
17
|
* context window.
|
|
@@ -26,7 +36,8 @@ export declare const MIN_FITTED_OUTPUT_TOKENS = 1024;
|
|
|
26
36
|
* OpenRouter-hosted model with no caller cap: the transport omits the catalog
|
|
27
37
|
* default there so each upstream self-caps, and a fitted value would turn into
|
|
28
38
|
* an explicit cap that filters upstreams). Otherwise returns
|
|
29
|
-
* the remaining room
|
|
39
|
+
* the remaining room minus {@link OUTPUT_FIT_HEADWAY_TOKENS} (never below
|
|
40
|
+
* {@link MIN_FITTED_OUTPUT_TOKENS}); a
|
|
30
41
|
* prompt that fills the whole window still overflows and is left to the
|
|
31
42
|
* caller's compaction. Near a full window the floor means a turn can stop on
|
|
32
43
|
* `length` instead of failing with a 400.
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* OpenTelemetry instrumentation for the agent loop.
|
|
3
3
|
*
|
|
4
4
|
* Implements the OpenTelemetry GenAI semantic conventions
|
|
5
|
-
* (https://opentelemetry.io/docs/specs/semconv/gen-ai/) plus `
|
|
5
|
+
* (https://opentelemetry.io/docs/specs/semconv/gen-ai/) plus `omp.gen_ai.*`
|
|
6
6
|
* extension attributes for run summaries, dashboard summaries, and cost hints
|
|
7
7
|
* that are useful to downstream observability UIs.
|
|
8
8
|
*
|
|
@@ -78,35 +78,41 @@ export declare const enum OpenAIAttr {
|
|
|
78
78
|
ResponseServiceTier = "openai.response.service_tier"
|
|
79
79
|
}
|
|
80
80
|
/** Project extension attributes. Kept out of the reserved `gen_ai.*` namespace. */
|
|
81
|
-
export declare const enum
|
|
82
|
-
AgentStepNumber = "
|
|
83
|
-
AgentStepCount = "
|
|
84
|
-
RequestReasoningEffort = "
|
|
85
|
-
RequestToolChoice = "
|
|
86
|
-
RequestAvailableTools = "
|
|
87
|
-
RequestMessages = "
|
|
88
|
-
ResponseText = "
|
|
89
|
-
ResponseToolCalls = "
|
|
90
|
-
ResponseUpstreamProvider = "
|
|
91
|
-
UsageTotalTokens = "
|
|
92
|
-
UsageServerSideTools = "
|
|
93
|
-
CostEstimatedUsd = "
|
|
94
|
-
CostInputUsd = "
|
|
95
|
-
CostOutputUsd = "
|
|
96
|
-
CostUnavailableReason = "
|
|
97
|
-
ToolStatus = "
|
|
98
|
-
ToolCallIntent = "
|
|
99
|
-
HandoffFromAgentName = "
|
|
100
|
-
HandoffFromAgentId = "
|
|
101
|
-
HandoffToAgentName = "
|
|
102
|
-
HandoffToAgentId = "
|
|
103
|
-
OneshotKind = "
|
|
104
|
-
GatewayName = "
|
|
105
|
-
GatewayEndpoint = "
|
|
106
|
-
GatewayCallId = "
|
|
107
|
-
GatewayRoutedTo = "
|
|
81
|
+
export declare const enum OmpGenAIAttr {
|
|
82
|
+
AgentStepNumber = "omp.gen_ai.agent.step.number",
|
|
83
|
+
AgentStepCount = "omp.gen_ai.agent.step.count",
|
|
84
|
+
RequestReasoningEffort = "omp.gen_ai.request.reasoning.effort",
|
|
85
|
+
RequestToolChoice = "omp.gen_ai.request.tool.choice",
|
|
86
|
+
RequestAvailableTools = "omp.gen_ai.request.available_tools",
|
|
87
|
+
RequestMessages = "omp.gen_ai.request.messages",
|
|
88
|
+
ResponseText = "omp.gen_ai.response.text",
|
|
89
|
+
ResponseToolCalls = "omp.gen_ai.response.tool_calls",
|
|
90
|
+
ResponseUpstreamProvider = "omp.gen_ai.response.upstream_provider",
|
|
91
|
+
UsageTotalTokens = "omp.gen_ai.usage.total_tokens",
|
|
92
|
+
UsageServerSideTools = "omp.gen_ai.usage.server_tool_requests",
|
|
93
|
+
CostEstimatedUsd = "omp.gen_ai.cost.estimated_usd",
|
|
94
|
+
CostInputUsd = "omp.gen_ai.cost.input_usd",
|
|
95
|
+
CostOutputUsd = "omp.gen_ai.cost.output_usd",
|
|
96
|
+
CostUnavailableReason = "omp.gen_ai.cost.unavailable_reason",
|
|
97
|
+
ToolStatus = "omp.gen_ai.tool.status",
|
|
98
|
+
ToolCallIntent = "omp.gen_ai.tool.call.intent",
|
|
99
|
+
HandoffFromAgentName = "omp.gen_ai.handoff.from_agent.name",
|
|
100
|
+
HandoffFromAgentId = "omp.gen_ai.handoff.from_agent.id",
|
|
101
|
+
HandoffToAgentName = "omp.gen_ai.handoff.to_agent.name",
|
|
102
|
+
HandoffToAgentId = "omp.gen_ai.handoff.to_agent.id",
|
|
103
|
+
OneshotKind = "omp.gen_ai.oneshot.kind",
|
|
104
|
+
GatewayName = "omp.gen_ai.gateway.name",
|
|
105
|
+
GatewayEndpoint = "omp.gen_ai.gateway.endpoint",
|
|
106
|
+
GatewayCallId = "omp.gen_ai.gateway.call_id",
|
|
107
|
+
GatewayRoutedTo = "omp.gen_ai.gateway.routed_to",
|
|
108
108
|
/** Cloudflare AI Gateway response-cache status (`cf-aig-cache-status`), never prompt-cache. */
|
|
109
|
-
GatewayResponseCacheStatus = "
|
|
109
|
+
GatewayResponseCacheStatus = "omp.gen_ai.gateway.response_cache.status",
|
|
110
|
+
/** Caller-level reason a judgment ran (`find`, `ttsr`, `judge_batch`, …). */
|
|
111
|
+
JudgmentPurpose = "omp.gen_ai.judgment.purpose",
|
|
112
|
+
/** Questions the caller asked in one judgment request. */
|
|
113
|
+
JudgmentQuestions = "omp.gen_ai.judgment.questions",
|
|
114
|
+
/** Questions answered from the local judgment cache instead of the provider. */
|
|
115
|
+
JudgmentCachedQuestions = "omp.gen_ai.judgment.cached_questions"
|
|
110
116
|
}
|
|
111
117
|
/** GenAI operation names — values for {@link GenAIAttr.OperationName}. */
|
|
112
118
|
export declare const GenAIOperation: {
|
|
@@ -114,6 +120,8 @@ export declare const GenAIOperation: {
|
|
|
114
120
|
readonly ExecuteTool: "execute_tool";
|
|
115
121
|
readonly InvokeAgent: "invoke_agent";
|
|
116
122
|
readonly Handoff: "handoff";
|
|
123
|
+
/** Typed judgment over a state (System One or a prompted judge model); a project extension value. */
|
|
124
|
+
readonly Judgment: "judgment";
|
|
117
125
|
readonly GenerateContent: "generate_content";
|
|
118
126
|
readonly TextCompletion: "text_completion";
|
|
119
127
|
readonly CreateAgent: "create_agent";
|
|
@@ -121,7 +129,7 @@ export declare const GenAIOperation: {
|
|
|
121
129
|
};
|
|
122
130
|
export type GenAIOperationName = (typeof GenAIOperation)[keyof typeof GenAIOperation];
|
|
123
131
|
/** Identifies which agent span a callback is reporting on. */
|
|
124
|
-
export type TelemetrySpanKind = "invoke_agent" | "chat" | "execute_tool" | "handoff";
|
|
132
|
+
export type TelemetrySpanKind = "invoke_agent" | "chat" | "execute_tool" | "handoff" | "judgment";
|
|
125
133
|
/**
|
|
126
134
|
* Aggregated usage + cost surface passed to {@link AgentTelemetryConfig.costEstimator}.
|
|
127
135
|
* Mirrors the bucketed shape we already emit as span attributes so the
|
|
@@ -144,9 +152,9 @@ export interface CostEstimatorContext {
|
|
|
144
152
|
}
|
|
145
153
|
/**
|
|
146
154
|
* Cost estimator result.
|
|
147
|
-
* { usd: number } — cost is known; emitted as
|
|
155
|
+
* { usd: number } — cost is known; emitted as omp.gen_ai.cost.estimated_usd
|
|
148
156
|
* { unavailable: string } — cost is intentionally unknown; emitted as
|
|
149
|
-
*
|
|
157
|
+
* omp.gen_ai.cost.unavailable_reason
|
|
150
158
|
* undefined — no opinion; nothing emitted
|
|
151
159
|
*/
|
|
152
160
|
export type CostEstimate = {
|
|
@@ -177,6 +185,8 @@ export interface CostDelta {
|
|
|
177
185
|
*/
|
|
178
186
|
export interface ChatUsageEvent {
|
|
179
187
|
readonly span: Span;
|
|
188
|
+
/** Operation that produced the usage: `chat` for model turns, `judgment` for typed judgments. */
|
|
189
|
+
readonly operation: GenAIOperationName;
|
|
180
190
|
readonly agent: AgentIdentity | undefined;
|
|
181
191
|
readonly conversationId: string | undefined;
|
|
182
192
|
readonly stepNumber: number | undefined;
|
|
@@ -195,7 +205,7 @@ export interface ChatUsageEvent {
|
|
|
195
205
|
*
|
|
196
206
|
* Use this to reconcile gateway-issued ids (e.g. `x-litellm-call-id`) with
|
|
197
207
|
* downstream billing / spend dashboards. Known gateway patterns are also
|
|
198
|
-
* auto-stamped on the chat span as `
|
|
208
|
+
* auto-stamped on the chat span as `omp.gen_ai.gateway.*` attributes.
|
|
199
209
|
*/
|
|
200
210
|
readonly headers: Readonly<Record<string, string>> | undefined;
|
|
201
211
|
}
|
|
@@ -417,7 +427,7 @@ export interface GatewayHeaderDetection {
|
|
|
417
427
|
export declare function detectGatewayFromHeaders(headers: Readonly<Record<string, string>> | undefined): GatewayHeaderDetection | undefined;
|
|
418
428
|
/**
|
|
419
429
|
* Bounded Cloudflare AI Gateway response-cache statuses emitted on
|
|
420
|
-
* {@link
|
|
430
|
+
* {@link OmpGenAIAttr.GatewayResponseCacheStatus}. Distinct from provider
|
|
421
431
|
* prompt-cache token counters (`gen_ai.usage.cache_*`).
|
|
422
432
|
*
|
|
423
433
|
* Cloudflare documents `HIT` / `MISS` on `cf-aig-cache-status`; `bypass` covers
|
|
@@ -453,6 +463,39 @@ export interface ManualChatTelemetryOptions {
|
|
|
453
463
|
readonly endSpan?: boolean;
|
|
454
464
|
}
|
|
455
465
|
export declare function recordManualChatTelemetry(telemetry: AgentTelemetry | undefined, options: ManualChatTelemetryOptions): Promise<Span | undefined>;
|
|
466
|
+
/**
|
|
467
|
+
* One judgment attempt reported to {@link recordJudgmentTelemetry}: a billed
|
|
468
|
+
* System One request, one prompted-model attempt, or a request answered
|
|
469
|
+
* entirely from the local judgment cache.
|
|
470
|
+
*/
|
|
471
|
+
export interface JudgmentTelemetryOptions {
|
|
472
|
+
readonly provider: string;
|
|
473
|
+
/** Requested model id. */
|
|
474
|
+
readonly model: string;
|
|
475
|
+
/** Model that answered, when the provider resolves an alias (`jev-latest` → `jev-1.13.0`). */
|
|
476
|
+
readonly responseModel?: string;
|
|
477
|
+
/** Caller-level reason the judgment ran; stamped as {@link OmpGenAIAttr.JudgmentPurpose}. */
|
|
478
|
+
readonly purpose: string;
|
|
479
|
+
/** Priced usage of this attempt — the same amount the session ledger bills. */
|
|
480
|
+
readonly usage: Usage;
|
|
481
|
+
readonly stopReason: StopReason;
|
|
482
|
+
readonly errorMessage?: string;
|
|
483
|
+
/** Epoch ms the attempt started. */
|
|
484
|
+
readonly startTime: number;
|
|
485
|
+
/** Questions in the request; omitted for prompted backends, which report per completion attempt. */
|
|
486
|
+
readonly questions?: number;
|
|
487
|
+
/** Questions answered from the local cache rather than the provider. */
|
|
488
|
+
readonly cachedQuestions?: number;
|
|
489
|
+
}
|
|
490
|
+
/**
|
|
491
|
+
* Emit one `judgment` span per judgment attempt and forward any billed usage
|
|
492
|
+
* to `onCostDelta` / `onChatUsage` with `operation: "judgment"`. Cost is the
|
|
493
|
+
* attempt's own priced usage rather than {@link AgentTelemetryConfig.costEstimator},
|
|
494
|
+
* so spans, metrics, and the session ledger agree on every judgment dollar.
|
|
495
|
+
* Attempts that billed nothing (cache hits, pre-response failures) keep their
|
|
496
|
+
* span but fire no usage hooks. No-op when telemetry is disabled.
|
|
497
|
+
*/
|
|
498
|
+
export declare function recordJudgmentTelemetry(telemetry: AgentTelemetry | undefined, options: JudgmentTelemetryOptions): Promise<void>;
|
|
456
499
|
/**
|
|
457
500
|
* Options accepted by {@link instrumentedCompleteSimple}. Mirrors the
|
|
458
501
|
* `streamAssistantResponse` chat-span lifecycle for oneshot LLM calls
|
|
@@ -465,7 +508,7 @@ export interface InstrumentedChatSpanOptions {
|
|
|
465
508
|
/** Step index recorded on the span; defaults to `-1` for non-loop calls. */
|
|
466
509
|
readonly stepNumber?: number;
|
|
467
510
|
/**
|
|
468
|
-
* Tag stamped onto `
|
|
511
|
+
* Tag stamped onto `omp.gen_ai.oneshot.kind`. Values used by the agent:
|
|
469
512
|
* `compaction_summary`, `compaction_short_summary`, `compaction_turn_prefix`,
|
|
470
513
|
* `handoff`, `branch_summary`, `image_question`. Free-form to allow callers
|
|
471
514
|
* outside this package to add new kinds without bumping the helper.
|
|
@@ -535,7 +578,7 @@ export declare function finishExecuteToolSpan(telemetry: AgentTelemetry | undefi
|
|
|
535
578
|
readonly toolName: string;
|
|
536
579
|
}): void;
|
|
537
580
|
/** Span attribute carrying the terminal {@link ToolStatus}. */
|
|
538
|
-
export declare const EXECUTE_TOOL_STATUS_ATTR =
|
|
581
|
+
export declare const EXECUTE_TOOL_STATUS_ATTR = OmpGenAIAttr.ToolStatus;
|
|
539
582
|
/**
|
|
540
583
|
* Record a tool that bypassed the span lifecycle entirely (pre-run
|
|
541
584
|
* interrupt, post-execution tail sweep for calls that never produced a
|
|
@@ -570,30 +613,30 @@ export declare function finishInvokeAgentSpan(telemetry: AgentTelemetry | undefi
|
|
|
570
613
|
* calling this helper.
|
|
571
614
|
*/
|
|
572
615
|
export declare function fireOnRunEnd(telemetry: AgentTelemetry, summary: AgentRunSummary, coverage: AgentRunCoverage): void;
|
|
573
|
-
/** Aggregate `
|
|
574
|
-
export declare const enum
|
|
575
|
-
ChatsCount = "
|
|
576
|
-
ChatsTotalLatencyMs = "
|
|
577
|
-
ChatsStopReasonPrefix = "
|
|
578
|
-
ToolsCount = "
|
|
579
|
-
ToolsOkCount = "
|
|
580
|
-
ToolsErrorCount = "
|
|
581
|
-
ToolsSkippedCount = "
|
|
582
|
-
ToolsBlockedCount = "
|
|
583
|
-
ToolsTimeoutCount = "
|
|
584
|
-
ToolsAbortedCount = "
|
|
585
|
-
ToolsTotalLatencyMs = "
|
|
586
|
-
ToolsInvoked = "
|
|
587
|
-
ToolsAvailable = "
|
|
588
|
-
ToolsUnused = "
|
|
589
|
-
UsageInputTokensTotal = "
|
|
590
|
-
UsageOutputTokensTotal = "
|
|
591
|
-
UsageCacheReadInputTokensTotal = "
|
|
592
|
-
UsageCacheCreationInputTokensTotal = "
|
|
593
|
-
UsageReasoningOutputTokensTotal = "
|
|
594
|
-
UsageTotalTokensTotal = "
|
|
595
|
-
CostEstimatedUsdTotal = "
|
|
596
|
-
ErrorsCount = "
|
|
616
|
+
/** Aggregate `omp.gen_ai.agent.*` attributes stamped on the `invoke_agent` span. */
|
|
617
|
+
export declare const enum OmpGenAIAggregateAttr {
|
|
618
|
+
ChatsCount = "omp.gen_ai.agent.chats.count",
|
|
619
|
+
ChatsTotalLatencyMs = "omp.gen_ai.agent.chats.total_latency_ms",
|
|
620
|
+
ChatsStopReasonPrefix = "omp.gen_ai.agent.chats.stop_reason.",
|
|
621
|
+
ToolsCount = "omp.gen_ai.agent.tools.count",
|
|
622
|
+
ToolsOkCount = "omp.gen_ai.agent.tools.ok.count",
|
|
623
|
+
ToolsErrorCount = "omp.gen_ai.agent.tools.error.count",
|
|
624
|
+
ToolsSkippedCount = "omp.gen_ai.agent.tools.skipped.count",
|
|
625
|
+
ToolsBlockedCount = "omp.gen_ai.agent.tools.blocked.count",
|
|
626
|
+
ToolsTimeoutCount = "omp.gen_ai.agent.tools.timeout.count",
|
|
627
|
+
ToolsAbortedCount = "omp.gen_ai.agent.tools.aborted.count",
|
|
628
|
+
ToolsTotalLatencyMs = "omp.gen_ai.agent.tools.total_latency_ms",
|
|
629
|
+
ToolsInvoked = "omp.gen_ai.agent.tools.invoked",
|
|
630
|
+
ToolsAvailable = "omp.gen_ai.agent.tools.available",
|
|
631
|
+
ToolsUnused = "omp.gen_ai.agent.tools.unused",
|
|
632
|
+
UsageInputTokensTotal = "omp.gen_ai.agent.usage.input_tokens.total",
|
|
633
|
+
UsageOutputTokensTotal = "omp.gen_ai.agent.usage.output_tokens.total",
|
|
634
|
+
UsageCacheReadInputTokensTotal = "omp.gen_ai.agent.usage.cache_read.input_tokens.total",
|
|
635
|
+
UsageCacheCreationInputTokensTotal = "omp.gen_ai.agent.usage.cache_creation.input_tokens.total",
|
|
636
|
+
UsageReasoningOutputTokensTotal = "omp.gen_ai.agent.usage.reasoning.output_tokens.total",
|
|
637
|
+
UsageTotalTokensTotal = "omp.gen_ai.agent.usage.total_tokens.total",
|
|
638
|
+
CostEstimatedUsdTotal = "omp.gen_ai.agent.cost.estimated_usd.total",
|
|
639
|
+
ErrorsCount = "omp.gen_ai.agent.errors.count"
|
|
597
640
|
}
|
|
598
641
|
/**
|
|
599
642
|
* Run `fn` with `span` activated on the OTEL context. Spans created
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-agent-core",
|
|
4
|
-
"version": "18.
|
|
4
|
+
"version": "18.4.1",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": {
|
|
@@ -38,16 +38,16 @@
|
|
|
38
38
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
39
39
|
},
|
|
40
40
|
"dependencies": {
|
|
41
|
-
"@oh-my-pi/pi-ai": "18.
|
|
42
|
-
"@oh-my-pi/pi-catalog": "18.
|
|
43
|
-
"@oh-my-pi/pi-natives": "18.
|
|
44
|
-
"@oh-my-pi/pi-utils": "18.
|
|
45
|
-
"@oh-my-pi/pi-wire": "18.
|
|
46
|
-
"@oh-my-pi/snapcompact": "18.
|
|
41
|
+
"@oh-my-pi/pi-ai": "18.4.1",
|
|
42
|
+
"@oh-my-pi/pi-catalog": "18.4.1",
|
|
43
|
+
"@oh-my-pi/pi-natives": "18.4.1",
|
|
44
|
+
"@oh-my-pi/pi-utils": "18.4.1",
|
|
45
|
+
"@oh-my-pi/pi-wire": "18.4.1",
|
|
46
|
+
"@oh-my-pi/snapcompact": "18.4.1",
|
|
47
47
|
"@opentelemetry/api": "^1.9.1"
|
|
48
48
|
},
|
|
49
49
|
"devDependencies": {
|
|
50
|
-
"@oh-my-pi/omptype": "18.
|
|
50
|
+
"@oh-my-pi/omptype": "18.4.1",
|
|
51
51
|
"@opentelemetry/context-async-hooks": "^2.9.0",
|
|
52
52
|
"@opentelemetry/sdk-trace-base": "^2.9.0",
|
|
53
53
|
"@types/bun": "^1.3.14"
|
package/src/agent-loop.ts
CHANGED
|
@@ -64,7 +64,7 @@ import {
|
|
|
64
64
|
finishExecuteToolSpan,
|
|
65
65
|
finishInvokeAgentSpan,
|
|
66
66
|
fireOnRunEnd,
|
|
67
|
-
|
|
67
|
+
OmpGenAIAttr,
|
|
68
68
|
recordSkippedTool,
|
|
69
69
|
resolveTelemetry,
|
|
70
70
|
runInActiveSpan,
|
|
@@ -2859,6 +2859,13 @@ async function prepareToolCallDispatch(
|
|
|
2859
2859
|
if (toolCall.type !== "toolCall") continue;
|
|
2860
2860
|
if ((toolCall as CursorExecResolvedCarrier)[kCursorExecResolved] === true) continue;
|
|
2861
2861
|
const tool = resolveToolForCall(context.tools, toolCall, resolveFallbackTool);
|
|
2862
|
+
// A host fallback accepts aliases (`xd://recall`, a mis-separated MCP
|
|
2863
|
+
// name) that providers reject when replayed as a function-call name.
|
|
2864
|
+
// Record the call under the resolved tool's canonical name so history,
|
|
2865
|
+
// persistence, and replay agree; custom-wire calls keep their wire name.
|
|
2866
|
+
if (tool && toolCall.name !== tool.name && toolCall.name !== tool.customWireName) {
|
|
2867
|
+
toolCall.name = tool.name;
|
|
2868
|
+
}
|
|
2862
2869
|
const entry: PreparedToolCall = { tool, args: toolCall.arguments as Record<string, unknown> };
|
|
2863
2870
|
prepared.set(toolCall.id, entry);
|
|
2864
2871
|
let argsForExecution = toolCall.arguments as Record<string, unknown>;
|
|
@@ -3316,7 +3323,7 @@ async function executeToolCalls(
|
|
|
3316
3323
|
parent: invokeAgentSpan,
|
|
3317
3324
|
});
|
|
3318
3325
|
if (toolSpan && toolCall.intent) {
|
|
3319
|
-
toolSpan.setAttribute(
|
|
3326
|
+
toolSpan.setAttribute(OmpGenAIAttr.ToolCallIntent, toolCall.intent);
|
|
3320
3327
|
}
|
|
3321
3328
|
|
|
3322
3329
|
let result: AgentToolResult<any> = { content: [], details: {} };
|
|
@@ -86,7 +86,7 @@ export interface GenerateBranchSummaryOptions {
|
|
|
86
86
|
convertToLlm?: ConvertToLlm;
|
|
87
87
|
/**
|
|
88
88
|
* Optional telemetry handle. When provided, the branch summary LLM call is
|
|
89
|
-
* wrapped in an OTEL chat span tagged with `
|
|
89
|
+
* wrapped in an OTEL chat span tagged with `omp.gen_ai.oneshot.kind = "branch_summary"`.
|
|
90
90
|
*/
|
|
91
91
|
telemetry?: AgentTelemetry;
|
|
92
92
|
/**
|
|
@@ -597,6 +597,14 @@ function handleCompactionV2Event(
|
|
|
597
597
|
if (type === "response.failed" || type === "response.incomplete") {
|
|
598
598
|
throw new Error(formatCompactionV2Failure(event, type));
|
|
599
599
|
}
|
|
600
|
+
|
|
601
|
+
// A standalone `error` event terminates the stream. Keep its status so a
|
|
602
|
+
// deterministic 4xx (e.g. context_too_large) is not retried as a dropped stream.
|
|
603
|
+
if (type === "error") {
|
|
604
|
+
const message = formatCompactionV2Failure(event, type);
|
|
605
|
+
const status = numberField(event, "status");
|
|
606
|
+
throw status === undefined ? new Error(message) : new AIError.ProviderHttpError(message, status);
|
|
607
|
+
}
|
|
600
608
|
}
|
|
601
609
|
|
|
602
610
|
function parseCompactionV2Usage(event: Record<string, unknown>): CompactionV2Usage | undefined {
|
|
@@ -629,8 +637,9 @@ function formatCompactionV2Failure(event: Record<string, unknown>, type: string)
|
|
|
629
637
|
: response && isRecord(response.error)
|
|
630
638
|
? response.error
|
|
631
639
|
: undefined;
|
|
632
|
-
|
|
633
|
-
const
|
|
640
|
+
// Responses `error` events carry code/message at the top level.
|
|
641
|
+
const message = stringField(error ?? event, "message");
|
|
642
|
+
const code = error ? (stringField(error, "code") ?? stringField(error, "type")) : stringField(event, "code");
|
|
634
643
|
return `V2 compaction stream ${type}${code ? ` (${code})` : ""}${message ? `: ${message}` : ""}`;
|
|
635
644
|
}
|
|
636
645
|
|
|
@@ -69,6 +69,7 @@ import {
|
|
|
69
69
|
defaultConvertToLlm,
|
|
70
70
|
} from "./messages";
|
|
71
71
|
import {
|
|
72
|
+
assertRemoteCompactionInputFits,
|
|
72
73
|
buildOpenAiNativeHistory,
|
|
73
74
|
getPreservedOpenAiRemoteCompactionData,
|
|
74
75
|
isOpenAiRemoteCompactionApi,
|
|
@@ -678,7 +679,7 @@ export interface SummaryOptions {
|
|
|
678
679
|
/**
|
|
679
680
|
* Optional telemetry handle. When provided, every LLM call emitted during
|
|
680
681
|
* compaction is wrapped in an OTEL chat span tagged with
|
|
681
|
-
* `
|
|
682
|
+
* `omp.gen_ai.oneshot.kind` (`compaction_summary`, `compaction_short_summary`,
|
|
682
683
|
* or `compaction_turn_prefix`). `undefined` keeps the call paths zero-cost.
|
|
683
684
|
*/
|
|
684
685
|
telemetry?: AgentTelemetry;
|
|
@@ -1041,7 +1042,7 @@ export interface HandoffOptions {
|
|
|
1041
1042
|
metadata?: Record<string, unknown>;
|
|
1042
1043
|
/**
|
|
1043
1044
|
* Optional telemetry handle. When provided, the handoff LLM call is
|
|
1044
|
-
* wrapped in an OTEL chat span tagged with `
|
|
1045
|
+
* wrapped in an OTEL chat span tagged with `omp.gen_ai.oneshot.kind = "handoff"`.
|
|
1045
1046
|
*/
|
|
1046
1047
|
telemetry?: AgentTelemetry;
|
|
1047
1048
|
/**
|
|
@@ -1748,6 +1749,7 @@ export async function compact(
|
|
|
1748
1749
|
contextWindow: model.contextWindow,
|
|
1749
1750
|
});
|
|
1750
1751
|
}
|
|
1752
|
+
assertRemoteCompactionInputFits(trimmed, model);
|
|
1751
1753
|
const requestOptions = {
|
|
1752
1754
|
sessionId: summaryOptions.sessionId,
|
|
1753
1755
|
promptCacheKey: summaryOptions.promptCacheKey,
|
package/src/compaction/openai.ts
CHANGED
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* with `{ summary, shortSummary? }`.
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
|
-
import { ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
|
18
|
+
import { attach, create, Flag, ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
|
19
19
|
import { getCodexAttestationHeader } from "@oh-my-pi/pi-ai/providers/openai-codex-attestation";
|
|
20
20
|
import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-compaction";
|
|
21
21
|
import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer";
|
|
@@ -122,6 +122,8 @@ export interface TrimRemoteCompactionInputResult {
|
|
|
122
122
|
rewrittenOutputs: number;
|
|
123
123
|
estimatedTokensBefore: number;
|
|
124
124
|
estimatedTokensAfter: number;
|
|
125
|
+
/** Whether `input` fits the model window; false means it must not be sent. */
|
|
126
|
+
fits: boolean;
|
|
125
127
|
}
|
|
126
128
|
|
|
127
129
|
/** Verdict for one remote-compaction request measured against the model window. */
|
|
@@ -199,6 +201,7 @@ export function trimRemoteCompactionInputToContextWindow(
|
|
|
199
201
|
rewrittenOutputs: 0,
|
|
200
202
|
estimatedTokensBefore: before.tokens,
|
|
201
203
|
estimatedTokensAfter: before.tokens,
|
|
204
|
+
fits: true,
|
|
202
205
|
};
|
|
203
206
|
}
|
|
204
207
|
|
|
@@ -222,6 +225,7 @@ export function trimRemoteCompactionInputToContextWindow(
|
|
|
222
225
|
rewrittenOutputs: 0,
|
|
223
226
|
estimatedTokensBefore: before.tokens,
|
|
224
227
|
estimatedTokensAfter: before.tokens,
|
|
228
|
+
fits: false,
|
|
225
229
|
};
|
|
226
230
|
}
|
|
227
231
|
|
|
@@ -230,9 +234,29 @@ export function trimRemoteCompactionInputToContextWindow(
|
|
|
230
234
|
rewrittenOutputs,
|
|
231
235
|
estimatedTokensBefore: before.tokens,
|
|
232
236
|
estimatedTokensAfter: after.tokens,
|
|
237
|
+
fits: true,
|
|
233
238
|
};
|
|
234
239
|
}
|
|
235
240
|
|
|
241
|
+
/**
|
|
242
|
+
* Refuse a native compaction request whose prepared input cannot fit the model
|
|
243
|
+
* window, before any network I/O. Re-expanded history behind an unreadable
|
|
244
|
+
* native boundary can exceed the window even when live context does not.
|
|
245
|
+
*
|
|
246
|
+
* @throws Error flagged `ContextOverflow` when `trimmed.fits` is false, so
|
|
247
|
+
* compaction callers skip retries and advance to the next method.
|
|
248
|
+
*/
|
|
249
|
+
export function assertRemoteCompactionInputFits(trimmed: TrimRemoteCompactionInputResult, model: Model): void {
|
|
250
|
+
if (trimmed.fits) return;
|
|
251
|
+
throw attach(
|
|
252
|
+
new Error(
|
|
253
|
+
`Remote compaction input exceeds the context window of ${model.provider}/${model.id}: ` +
|
|
254
|
+
`estimated ${trimmed.estimatedTokensAfter} tokens > ${model.contextWindow}`,
|
|
255
|
+
),
|
|
256
|
+
create(Flag.ContextOverflow),
|
|
257
|
+
);
|
|
258
|
+
}
|
|
259
|
+
|
|
236
260
|
/** Race the caller's signal against the request timeout; `timeoutMs <= 0` disables the watchdog. */
|
|
237
261
|
function withRequestTimeout(signal: AbortSignal | undefined, timeoutMs: number): AbortSignal | undefined {
|
|
238
262
|
if (timeoutMs <= 0) return signal;
|
|
@@ -794,6 +818,7 @@ export async function requestOpenAiRemoteCompaction(
|
|
|
794
818
|
contextWindow: model.contextWindow,
|
|
795
819
|
});
|
|
796
820
|
}
|
|
821
|
+
assertRemoteCompactionInputFits(trimmed, model);
|
|
797
822
|
const request: OpenAiRemoteCompactionRequest = {
|
|
798
823
|
model: requestModel,
|
|
799
824
|
// Preserve the native transcript. Only oversized trailing tool outputs are
|
package/src/output-budget.ts
CHANGED
|
@@ -17,6 +17,17 @@ export const MIN_FITTED_OUTPUT_TOKENS = 1024;
|
|
|
17
17
|
*/
|
|
18
18
|
const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
|
|
19
19
|
|
|
20
|
+
/**
|
|
21
|
+
* Absolute headway subtracted from the remaining room (unconditionally, on anchored and
|
|
22
|
+
* fully-local counts alike). Proportional padding covers
|
|
23
|
+
* tokenizer drift that scales with the prompt, but a host can still count a few tokens more
|
|
24
|
+
* than any local estimate can see (chat-template framing, reasoning wrappers). Measured
|
|
25
|
+
* against a 262,144-token host: fitted caps landed +11 to +42 tokens over and were rejected
|
|
26
|
+
* with 400s. 64 tokens covers the observed drift with margin to spare; the
|
|
27
|
+
* {@link MIN_FITTED_OUTPUT_TOKENS} floor still applies.
|
|
28
|
+
*/
|
|
29
|
+
export const OUTPUT_FIT_HEADWAY_TOKENS = 64;
|
|
30
|
+
|
|
20
31
|
/**
|
|
21
32
|
* Output cap for a request, so prompt plus output stays inside the model's
|
|
22
33
|
* context window.
|
|
@@ -41,7 +52,8 @@ const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
|
|
|
41
52
|
* OpenRouter-hosted model with no caller cap: the transport omits the catalog
|
|
42
53
|
* default there so each upstream self-caps, and a fitted value would turn into
|
|
43
54
|
* an explicit cap that filters upstreams). Otherwise returns
|
|
44
|
-
* the remaining room
|
|
55
|
+
* the remaining room minus {@link OUTPUT_FIT_HEADWAY_TOKENS} (never below
|
|
56
|
+
* {@link MIN_FITTED_OUTPUT_TOKENS}); a
|
|
45
57
|
* prompt that fills the whole window still overflows and is left to the
|
|
46
58
|
* caller's compaction. Near a full window the floor means a turn can stop on
|
|
47
59
|
* `length` instead of failing with a 400.
|
|
@@ -67,7 +79,7 @@ export function fitOutputTokensToContextWindow(
|
|
|
67
79
|
if (!requested || !contextWindow || contextWindow <= 0) return maxTokens;
|
|
68
80
|
if (stopsOutputAtContextWindow(model)) return maxTokens;
|
|
69
81
|
|
|
70
|
-
const room = contextWindow - countPromptTokens(context, tokenizer);
|
|
82
|
+
const room = contextWindow - countPromptTokens(context, tokenizer) - OUTPUT_FIT_HEADWAY_TOKENS;
|
|
71
83
|
if (room >= requested) return maxTokens;
|
|
72
84
|
return Math.max(MIN_FITTED_OUTPUT_TOKENS, room);
|
|
73
85
|
}
|
package/src/telemetry.ts
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* OpenTelemetry instrumentation for the agent loop.
|
|
3
3
|
*
|
|
4
4
|
* Implements the OpenTelemetry GenAI semantic conventions
|
|
5
|
-
* (https://opentelemetry.io/docs/specs/semconv/gen-ai/) plus `
|
|
5
|
+
* (https://opentelemetry.io/docs/specs/semconv/gen-ai/) plus `omp.gen_ai.*`
|
|
6
6
|
* extension attributes for run summaries, dashboard summaries, and cost hints
|
|
7
7
|
* that are useful to downstream observability UIs.
|
|
8
8
|
*
|
|
@@ -125,42 +125,48 @@ export const enum OpenAIAttr {
|
|
|
125
125
|
}
|
|
126
126
|
|
|
127
127
|
/** Project extension attributes. Kept out of the reserved `gen_ai.*` namespace. */
|
|
128
|
-
export const enum
|
|
129
|
-
AgentStepNumber = "
|
|
130
|
-
AgentStepCount = "
|
|
131
|
-
RequestReasoningEffort = "
|
|
132
|
-
RequestToolChoice = "
|
|
133
|
-
RequestAvailableTools = "
|
|
134
|
-
RequestMessages = "
|
|
135
|
-
ResponseText = "
|
|
136
|
-
ResponseToolCalls = "
|
|
137
|
-
ResponseUpstreamProvider = "
|
|
138
|
-
UsageTotalTokens = "
|
|
139
|
-
UsageServerSideTools = "
|
|
140
|
-
CostEstimatedUsd = "
|
|
141
|
-
CostInputUsd = "
|
|
142
|
-
CostOutputUsd = "
|
|
143
|
-
CostUnavailableReason = "
|
|
144
|
-
ToolStatus = "
|
|
145
|
-
ToolCallIntent = "
|
|
146
|
-
HandoffFromAgentName = "
|
|
147
|
-
HandoffFromAgentId = "
|
|
148
|
-
HandoffToAgentName = "
|
|
149
|
-
HandoffToAgentId = "
|
|
128
|
+
export const enum OmpGenAIAttr {
|
|
129
|
+
AgentStepNumber = "omp.gen_ai.agent.step.number",
|
|
130
|
+
AgentStepCount = "omp.gen_ai.agent.step.count",
|
|
131
|
+
RequestReasoningEffort = "omp.gen_ai.request.reasoning.effort",
|
|
132
|
+
RequestToolChoice = "omp.gen_ai.request.tool.choice",
|
|
133
|
+
RequestAvailableTools = "omp.gen_ai.request.available_tools",
|
|
134
|
+
RequestMessages = "omp.gen_ai.request.messages",
|
|
135
|
+
ResponseText = "omp.gen_ai.response.text",
|
|
136
|
+
ResponseToolCalls = "omp.gen_ai.response.tool_calls",
|
|
137
|
+
ResponseUpstreamProvider = "omp.gen_ai.response.upstream_provider",
|
|
138
|
+
UsageTotalTokens = "omp.gen_ai.usage.total_tokens",
|
|
139
|
+
UsageServerSideTools = "omp.gen_ai.usage.server_tool_requests",
|
|
140
|
+
CostEstimatedUsd = "omp.gen_ai.cost.estimated_usd",
|
|
141
|
+
CostInputUsd = "omp.gen_ai.cost.input_usd",
|
|
142
|
+
CostOutputUsd = "omp.gen_ai.cost.output_usd",
|
|
143
|
+
CostUnavailableReason = "omp.gen_ai.cost.unavailable_reason",
|
|
144
|
+
ToolStatus = "omp.gen_ai.tool.status",
|
|
145
|
+
ToolCallIntent = "omp.gen_ai.tool.call.intent",
|
|
146
|
+
HandoffFromAgentName = "omp.gen_ai.handoff.from_agent.name",
|
|
147
|
+
HandoffFromAgentId = "omp.gen_ai.handoff.from_agent.id",
|
|
148
|
+
HandoffToAgentName = "omp.gen_ai.handoff.to_agent.name",
|
|
149
|
+
HandoffToAgentId = "omp.gen_ai.handoff.to_agent.id",
|
|
150
150
|
// Marks chat spans emitted outside the agent loop (compaction, handoff, branch
|
|
151
151
|
// summary, image inspection, …). Lets dashboards split oneshot cost / latency
|
|
152
152
|
// from main-turn cost without overloading the semconv `gen_ai.operation.name`.
|
|
153
|
-
OneshotKind = "
|
|
153
|
+
OneshotKind = "omp.gen_ai.oneshot.kind",
|
|
154
154
|
// Gateway / proxy (LiteLLM, Helicone, Portkey, …) — populated when a known
|
|
155
155
|
// gateway header pattern is detected on the upstream response. The base
|
|
156
156
|
// `gen_ai.provider.name` continues to track the *upstream* provider (e.g.
|
|
157
157
|
// `anthropic`) that the gateway routed to.
|
|
158
|
-
GatewayName = "
|
|
159
|
-
GatewayEndpoint = "
|
|
160
|
-
GatewayCallId = "
|
|
161
|
-
GatewayRoutedTo = "
|
|
158
|
+
GatewayName = "omp.gen_ai.gateway.name",
|
|
159
|
+
GatewayEndpoint = "omp.gen_ai.gateway.endpoint",
|
|
160
|
+
GatewayCallId = "omp.gen_ai.gateway.call_id",
|
|
161
|
+
GatewayRoutedTo = "omp.gen_ai.gateway.routed_to",
|
|
162
162
|
/** Cloudflare AI Gateway response-cache status (`cf-aig-cache-status`), never prompt-cache. */
|
|
163
|
-
GatewayResponseCacheStatus = "
|
|
163
|
+
GatewayResponseCacheStatus = "omp.gen_ai.gateway.response_cache.status",
|
|
164
|
+
/** Caller-level reason a judgment ran (`find`, `ttsr`, `judge_batch`, …). */
|
|
165
|
+
JudgmentPurpose = "omp.gen_ai.judgment.purpose",
|
|
166
|
+
/** Questions the caller asked in one judgment request. */
|
|
167
|
+
JudgmentQuestions = "omp.gen_ai.judgment.questions",
|
|
168
|
+
/** Questions answered from the local judgment cache instead of the provider. */
|
|
169
|
+
JudgmentCachedQuestions = "omp.gen_ai.judgment.cached_questions",
|
|
164
170
|
}
|
|
165
171
|
|
|
166
172
|
/** GenAI operation names — values for {@link GenAIAttr.OperationName}. */
|
|
@@ -169,6 +175,8 @@ export const GenAIOperation = {
|
|
|
169
175
|
ExecuteTool: "execute_tool",
|
|
170
176
|
InvokeAgent: "invoke_agent",
|
|
171
177
|
Handoff: "handoff",
|
|
178
|
+
/** Typed judgment over a state (System One or a prompted judge model); a project extension value. */
|
|
179
|
+
Judgment: "judgment",
|
|
172
180
|
GenerateContent: "generate_content",
|
|
173
181
|
TextCompletion: "text_completion",
|
|
174
182
|
CreateAgent: "create_agent",
|
|
@@ -178,7 +186,7 @@ export const GenAIOperation = {
|
|
|
178
186
|
export type GenAIOperationName = (typeof GenAIOperation)[keyof typeof GenAIOperation];
|
|
179
187
|
|
|
180
188
|
/** Identifies which agent span a callback is reporting on. */
|
|
181
|
-
export type TelemetrySpanKind = "invoke_agent" | "chat" | "execute_tool" | "handoff";
|
|
189
|
+
export type TelemetrySpanKind = "invoke_agent" | "chat" | "execute_tool" | "handoff" | "judgment";
|
|
182
190
|
|
|
183
191
|
/**
|
|
184
192
|
* Aggregated usage + cost surface passed to {@link AgentTelemetryConfig.costEstimator}.
|
|
@@ -204,9 +212,9 @@ export interface CostEstimatorContext {
|
|
|
204
212
|
|
|
205
213
|
/**
|
|
206
214
|
* Cost estimator result.
|
|
207
|
-
* { usd: number } — cost is known; emitted as
|
|
215
|
+
* { usd: number } — cost is known; emitted as omp.gen_ai.cost.estimated_usd
|
|
208
216
|
* { unavailable: string } — cost is intentionally unknown; emitted as
|
|
209
|
-
*
|
|
217
|
+
* omp.gen_ai.cost.unavailable_reason
|
|
210
218
|
* undefined — no opinion; nothing emitted
|
|
211
219
|
*/
|
|
212
220
|
export type CostEstimate =
|
|
@@ -235,6 +243,8 @@ export interface CostDelta {
|
|
|
235
243
|
*/
|
|
236
244
|
export interface ChatUsageEvent {
|
|
237
245
|
readonly span: Span;
|
|
246
|
+
/** Operation that produced the usage: `chat` for model turns, `judgment` for typed judgments. */
|
|
247
|
+
readonly operation: GenAIOperationName;
|
|
238
248
|
readonly agent: AgentIdentity | undefined;
|
|
239
249
|
readonly conversationId: string | undefined;
|
|
240
250
|
readonly stepNumber: number | undefined;
|
|
@@ -253,7 +263,7 @@ export interface ChatUsageEvent {
|
|
|
253
263
|
*
|
|
254
264
|
* Use this to reconcile gateway-issued ids (e.g. `x-litellm-call-id`) with
|
|
255
265
|
* downstream billing / spend dashboards. Known gateway patterns are also
|
|
256
|
-
* auto-stamped on the chat span as `
|
|
266
|
+
* auto-stamped on the chat span as `omp.gen_ai.gateway.*` attributes.
|
|
257
267
|
*/
|
|
258
268
|
readonly headers: Readonly<Record<string, string>> | undefined;
|
|
259
269
|
}
|
|
@@ -475,6 +485,8 @@ function startSpan(
|
|
|
475
485
|
readonly stepNumber?: number;
|
|
476
486
|
readonly toolCallId?: string;
|
|
477
487
|
readonly toolName?: string;
|
|
488
|
+
/** Epoch ms; defaults to now. */
|
|
489
|
+
readonly startTime?: number;
|
|
478
490
|
},
|
|
479
491
|
): Span | undefined {
|
|
480
492
|
if (!telemetry) return undefined;
|
|
@@ -497,7 +509,11 @@ function startSpan(
|
|
|
497
509
|
if (options.attributes) Object.assign(attrs, options.attributes);
|
|
498
510
|
|
|
499
511
|
const ctx = options.parent ? trace.setSpan(context.active(), options.parent) : context.active();
|
|
500
|
-
const span = telemetry.tracer.startSpan(
|
|
512
|
+
const span = telemetry.tracer.startSpan(
|
|
513
|
+
name,
|
|
514
|
+
{ kind: options.spanKind, attributes: attrs, startTime: options.startTime },
|
|
515
|
+
ctx,
|
|
516
|
+
);
|
|
501
517
|
safeOnSpanStart(telemetry, { ...attrCtx, span });
|
|
502
518
|
return span;
|
|
503
519
|
}
|
|
@@ -548,6 +564,8 @@ function kindToOperation(kind: TelemetrySpanKind): GenAIOperationName | undefine
|
|
|
548
564
|
return GenAIOperation.ExecuteTool;
|
|
549
565
|
case "handoff":
|
|
550
566
|
return GenAIOperation.Handoff;
|
|
567
|
+
case "judgment":
|
|
568
|
+
return GenAIOperation.Judgment;
|
|
551
569
|
}
|
|
552
570
|
}
|
|
553
571
|
|
|
@@ -683,7 +701,7 @@ export function startInvokeAgentSpan(telemetry: AgentTelemetry | undefined, mode
|
|
|
683
701
|
/** Stamp the final step count on the `invoke_agent` span. */
|
|
684
702
|
export function applyInvokeAgentFinish(span: Span | undefined, stepCount: number): void {
|
|
685
703
|
if (!span) return;
|
|
686
|
-
span.setAttribute(
|
|
704
|
+
span.setAttribute(OmpGenAIAttr.AgentStepCount, stepCount);
|
|
687
705
|
}
|
|
688
706
|
|
|
689
707
|
/**
|
|
@@ -740,7 +758,7 @@ export interface ChatRequestSnapshot {
|
|
|
740
758
|
|
|
741
759
|
function buildChatRequestAttributes(stepNumber: number, request: ChatRequestSnapshot, provider: string): Attributes {
|
|
742
760
|
const attrs: Attributes = {
|
|
743
|
-
[
|
|
761
|
+
[OmpGenAIAttr.AgentStepNumber]: stepNumber,
|
|
744
762
|
[GenAIAttr.OutputType]: "text",
|
|
745
763
|
[GenAIAttr.RequestStream]: true,
|
|
746
764
|
};
|
|
@@ -757,11 +775,11 @@ function buildChatRequestAttributes(stepNumber: number, request: ChatRequestSnap
|
|
|
757
775
|
if (request.serviceTier && shouldSendServiceTier(request.serviceTier, provider)) {
|
|
758
776
|
attrs[OpenAIAttr.RequestServiceTier] = request.serviceTier;
|
|
759
777
|
}
|
|
760
|
-
if (request.reasoningEffort) attrs[
|
|
778
|
+
if (request.reasoningEffort) attrs[OmpGenAIAttr.RequestReasoningEffort] = request.reasoningEffort;
|
|
761
779
|
const toolChoice = serializeToolChoice(request.toolChoice);
|
|
762
|
-
if (toolChoice) attrs[
|
|
780
|
+
if (toolChoice) attrs[OmpGenAIAttr.RequestToolChoice] = toolChoice;
|
|
763
781
|
if (request.tools && request.tools.length > 0) {
|
|
764
|
-
attrs[
|
|
782
|
+
attrs[OmpGenAIAttr.RequestAvailableTools] = request.tools.map(tool => tool.name);
|
|
765
783
|
}
|
|
766
784
|
return attrs;
|
|
767
785
|
}
|
|
@@ -779,7 +797,7 @@ function serializeToolChoice(toolChoice: ToolChoice | undefined): string | undef
|
|
|
779
797
|
|
|
780
798
|
function applyContentCaptureForRequest(telemetry: AgentTelemetry, span: Span, request: ChatRequestSnapshot): void {
|
|
781
799
|
const requestMessages = serializeRequestMessagesForTelemetry(telemetry, request);
|
|
782
|
-
if (requestMessages) span.setAttribute(
|
|
800
|
+
if (requestMessages) span.setAttribute(OmpGenAIAttr.RequestMessages, requestMessages);
|
|
783
801
|
if (telemetry.contentCapture !== "full") return;
|
|
784
802
|
const systemInstructions = serializeFullSystemInstructionsForTelemetry(request);
|
|
785
803
|
if (systemInstructions) span.setAttribute(GenAIAttr.SystemInstructions, systemInstructions);
|
|
@@ -789,9 +807,9 @@ function applyContentCaptureForRequest(telemetry: AgentTelemetry, span: Span, re
|
|
|
789
807
|
|
|
790
808
|
function applyContentCaptureForResponse(telemetry: AgentTelemetry, span: Span, message: AssistantMessage): void {
|
|
791
809
|
const responseText = serializeResponseTextForTelemetry(telemetry, message);
|
|
792
|
-
if (responseText) span.setAttribute(
|
|
810
|
+
if (responseText) span.setAttribute(OmpGenAIAttr.ResponseText, responseText);
|
|
793
811
|
const responseToolCalls = serializeResponseToolCallsForTelemetry(telemetry, message);
|
|
794
|
-
if (responseToolCalls) span.setAttribute(
|
|
812
|
+
if (responseToolCalls) span.setAttribute(OmpGenAIAttr.ResponseToolCalls, responseToolCalls);
|
|
795
813
|
if (telemetry.contentCapture === "full") {
|
|
796
814
|
const outputMessages = serializeFullOutputMessagesForTelemetry(message);
|
|
797
815
|
if (outputMessages) span.setAttribute(GenAIAttr.OutputMessages, outputMessages);
|
|
@@ -1131,6 +1149,7 @@ export async function finishChatSpan(
|
|
|
1131
1149
|
const cost = applyCostEstimate(telemetry, span, message, options.serviceTier, options.stepNumber);
|
|
1132
1150
|
if (telemetry) {
|
|
1133
1151
|
await emitChatUsage(telemetry, span, {
|
|
1152
|
+
operation: GenAIOperation.Chat,
|
|
1134
1153
|
model: message.model,
|
|
1135
1154
|
provider: message.provider,
|
|
1136
1155
|
serviceTier: options.serviceTier,
|
|
@@ -1199,7 +1218,7 @@ function applyChatResponseAttributes(span: Span, message: AssistantMessage): voi
|
|
|
1199
1218
|
span.setAttribute(GenAIAttr.ResponseModel, message.model);
|
|
1200
1219
|
if (message.responseId) span.setAttribute(GenAIAttr.ResponseId, message.responseId);
|
|
1201
1220
|
if (message.upstreamProvider) {
|
|
1202
|
-
span.setAttribute(
|
|
1221
|
+
span.setAttribute(OmpGenAIAttr.ResponseUpstreamProvider, message.upstreamProvider);
|
|
1203
1222
|
}
|
|
1204
1223
|
if (message.ttft != null) span.setAttribute(GenAIAttr.ResponseTimeToFirstChunk, message.ttft / 1000);
|
|
1205
1224
|
const finishReason = mapStopReason(message.stopReason);
|
|
@@ -1215,7 +1234,7 @@ function applyUsageAttributes(span: Span, usage: Usage | undefined): void {
|
|
|
1215
1234
|
span.setAttribute(GenAIAttr.UsageInputTokens, inputTokens);
|
|
1216
1235
|
span.setAttribute(GenAIAttr.UsageOutputTokens, outputTokens);
|
|
1217
1236
|
const total = usage.totalTokens ?? inputTokens + outputTokens;
|
|
1218
|
-
span.setAttribute(
|
|
1237
|
+
span.setAttribute(OmpGenAIAttr.UsageTotalTokens, total);
|
|
1219
1238
|
if (usage.cacheRead != null) span.setAttribute(GenAIAttr.UsageCacheReadInputTokens, usage.cacheRead);
|
|
1220
1239
|
if (usage.cacheWrite != null) span.setAttribute(GenAIAttr.UsageCacheCreationInputTokens, usage.cacheWrite);
|
|
1221
1240
|
if (usage.reasoningTokens != null) {
|
|
@@ -1223,7 +1242,7 @@ function applyUsageAttributes(span: Span, usage: Usage | undefined): void {
|
|
|
1223
1242
|
}
|
|
1224
1243
|
if (usage.server) {
|
|
1225
1244
|
const sums = (usage.server.webSearch ?? 0) + (usage.server.webFetch ?? 0);
|
|
1226
|
-
if (sums > 0) span.setAttribute(
|
|
1245
|
+
if (sums > 0) span.setAttribute(OmpGenAIAttr.UsageServerSideTools, sums);
|
|
1227
1246
|
}
|
|
1228
1247
|
}
|
|
1229
1248
|
|
|
@@ -1284,7 +1303,7 @@ export function detectGatewayFromHeaders(
|
|
|
1284
1303
|
|
|
1285
1304
|
/**
|
|
1286
1305
|
* Bounded Cloudflare AI Gateway response-cache statuses emitted on
|
|
1287
|
-
* {@link
|
|
1306
|
+
* {@link OmpGenAIAttr.GatewayResponseCacheStatus}. Distinct from provider
|
|
1288
1307
|
* prompt-cache token counters (`gen_ai.usage.cache_*`).
|
|
1289
1308
|
*
|
|
1290
1309
|
* Cloudflare documents `HIT` / `MISS` on `cf-aig-cache-status`; `bypass` covers
|
|
@@ -1323,14 +1342,14 @@ function applyGatewayAttributes(
|
|
|
1323
1342
|
): void {
|
|
1324
1343
|
const gateway = detectGatewayFromHeaders(headers);
|
|
1325
1344
|
if (gateway) {
|
|
1326
|
-
span.setAttribute(
|
|
1327
|
-
if (baseUrl) span.setAttribute(
|
|
1328
|
-
if (gateway.callId) span.setAttribute(
|
|
1329
|
-
if (gateway.routedTo) span.setAttribute(
|
|
1345
|
+
span.setAttribute(OmpGenAIAttr.GatewayName, gateway.name);
|
|
1346
|
+
if (baseUrl) span.setAttribute(OmpGenAIAttr.GatewayEndpoint, baseUrl);
|
|
1347
|
+
if (gateway.callId) span.setAttribute(OmpGenAIAttr.GatewayCallId, gateway.callId);
|
|
1348
|
+
if (gateway.routedTo) span.setAttribute(OmpGenAIAttr.GatewayRoutedTo, gateway.routedTo);
|
|
1330
1349
|
}
|
|
1331
1350
|
const responseCacheStatus = classifyGatewayResponseCacheStatus(headers);
|
|
1332
1351
|
if (responseCacheStatus) {
|
|
1333
|
-
span.setAttribute(
|
|
1352
|
+
span.setAttribute(OmpGenAIAttr.GatewayResponseCacheStatus, responseCacheStatus);
|
|
1334
1353
|
}
|
|
1335
1354
|
}
|
|
1336
1355
|
|
|
@@ -1392,7 +1411,7 @@ function applyCostEstimateForUsage(
|
|
|
1392
1411
|
}
|
|
1393
1412
|
if (!result) return EMPTY_COST;
|
|
1394
1413
|
if ("unavailable" in result) {
|
|
1395
|
-
span.setAttribute(
|
|
1414
|
+
span.setAttribute(OmpGenAIAttr.CostUnavailableReason, result.unavailable);
|
|
1396
1415
|
const cost: AppliedCostEstimate = {
|
|
1397
1416
|
costUsd: undefined,
|
|
1398
1417
|
inputUsd: undefined,
|
|
@@ -1414,9 +1433,9 @@ function applyCostEstimateForUsage(
|
|
|
1414
1433
|
});
|
|
1415
1434
|
return cost;
|
|
1416
1435
|
}
|
|
1417
|
-
span.setAttribute(
|
|
1418
|
-
if (result.inputUsd != null) span.setAttribute(
|
|
1419
|
-
if (result.outputUsd != null) span.setAttribute(
|
|
1436
|
+
span.setAttribute(OmpGenAIAttr.CostEstimatedUsd, result.usd);
|
|
1437
|
+
if (result.inputUsd != null) span.setAttribute(OmpGenAIAttr.CostInputUsd, result.inputUsd);
|
|
1438
|
+
if (result.outputUsd != null) span.setAttribute(OmpGenAIAttr.CostOutputUsd, result.outputUsd);
|
|
1420
1439
|
const cost: AppliedCostEstimate = {
|
|
1421
1440
|
costUsd: result.usd,
|
|
1422
1441
|
inputUsd: result.inputUsd,
|
|
@@ -1470,6 +1489,7 @@ async function emitChatUsage(
|
|
|
1470
1489
|
telemetry: AgentTelemetry,
|
|
1471
1490
|
span: Span,
|
|
1472
1491
|
input: {
|
|
1492
|
+
readonly operation: GenAIOperationName;
|
|
1473
1493
|
readonly model: string;
|
|
1474
1494
|
readonly provider: string | undefined;
|
|
1475
1495
|
readonly serviceTier: ServiceTier | undefined;
|
|
@@ -1483,6 +1503,7 @@ async function emitChatUsage(
|
|
|
1483
1503
|
if (!hook || !input.usage) return;
|
|
1484
1504
|
const event: ChatUsageEvent = {
|
|
1485
1505
|
span,
|
|
1506
|
+
operation: input.operation,
|
|
1486
1507
|
agent: normalizedTelemetryAgent(telemetry),
|
|
1487
1508
|
conversationId: telemetry.conversationId,
|
|
1488
1509
|
stepNumber: input.stepNumber,
|
|
@@ -1493,7 +1514,9 @@ async function emitChatUsage(
|
|
|
1493
1514
|
cost: costEstimateFromApplied(input.applied),
|
|
1494
1515
|
attributes: resolveDynamicAttributes(
|
|
1495
1516
|
telemetry,
|
|
1496
|
-
buildTelemetryAttributeContext(telemetry, "chat", {
|
|
1517
|
+
buildTelemetryAttributeContext(telemetry, input.operation === GenAIOperation.Judgment ? "judgment" : "chat", {
|
|
1518
|
+
stepNumber: input.stepNumber,
|
|
1519
|
+
}),
|
|
1497
1520
|
),
|
|
1498
1521
|
headers: input.headers,
|
|
1499
1522
|
};
|
|
@@ -1586,7 +1609,7 @@ export async function recordManualChatTelemetry(
|
|
|
1586
1609
|
});
|
|
1587
1610
|
if (!span) return undefined;
|
|
1588
1611
|
if (options.span && options.attributes) span.setAttributes(options.attributes);
|
|
1589
|
-
if (options.stepNumber != null) span.setAttribute(
|
|
1612
|
+
if (options.stepNumber != null) span.setAttribute(OmpGenAIAttr.AgentStepNumber, options.stepNumber);
|
|
1590
1613
|
span.setAttribute(GenAIAttr.ResponseModel, options.responseModel ?? options.model.name);
|
|
1591
1614
|
if (options.responseId) span.setAttribute(GenAIAttr.ResponseId, options.responseId);
|
|
1592
1615
|
const finishReason = mapStopReason(options.finishReason);
|
|
@@ -1602,6 +1625,7 @@ export async function recordManualChatTelemetry(
|
|
|
1602
1625
|
usage: options.usage,
|
|
1603
1626
|
});
|
|
1604
1627
|
await emitChatUsage(telemetry, span, {
|
|
1628
|
+
operation: GenAIOperation.Chat,
|
|
1605
1629
|
model: options.responseModel ?? options.model.id,
|
|
1606
1630
|
provider: options.model.provider,
|
|
1607
1631
|
serviceTier: options.serviceTier,
|
|
@@ -1619,7 +1643,7 @@ export async function recordManualChatTelemetry(
|
|
|
1619
1643
|
}
|
|
1620
1644
|
if (options.responseText) {
|
|
1621
1645
|
const responseText = stringifyJsonAttribute(summarizeTelemetryTexts([options.responseText]));
|
|
1622
|
-
if (responseText) span.setAttribute(
|
|
1646
|
+
if (responseText) span.setAttribute(OmpGenAIAttr.ResponseText, responseText);
|
|
1623
1647
|
}
|
|
1624
1648
|
if (options.responseToolCalls && options.responseToolCalls.length > 0) {
|
|
1625
1649
|
const calls = options.responseToolCalls.map(call => ({
|
|
@@ -1628,13 +1652,112 @@ export async function recordManualChatTelemetry(
|
|
|
1628
1652
|
input: summarizeTelemetryValue(call.input),
|
|
1629
1653
|
}));
|
|
1630
1654
|
const responseToolCalls = stringifyJsonAttribute(limitTelemetryToolCalls(calls));
|
|
1631
|
-
if (responseToolCalls) span.setAttribute(
|
|
1655
|
+
if (responseToolCalls) span.setAttribute(OmpGenAIAttr.ResponseToolCalls, responseToolCalls);
|
|
1632
1656
|
}
|
|
1633
1657
|
applyTerminalStatus(span, options.finishReason, undefined);
|
|
1634
1658
|
if (options.endSpan ?? options.span === undefined) span.end();
|
|
1635
1659
|
return span;
|
|
1636
1660
|
}
|
|
1637
1661
|
|
|
1662
|
+
/**
|
|
1663
|
+
* One judgment attempt reported to {@link recordJudgmentTelemetry}: a billed
|
|
1664
|
+
* System One request, one prompted-model attempt, or a request answered
|
|
1665
|
+
* entirely from the local judgment cache.
|
|
1666
|
+
*/
|
|
1667
|
+
export interface JudgmentTelemetryOptions {
|
|
1668
|
+
readonly provider: string;
|
|
1669
|
+
/** Requested model id. */
|
|
1670
|
+
readonly model: string;
|
|
1671
|
+
/** Model that answered, when the provider resolves an alias (`jev-latest` → `jev-1.13.0`). */
|
|
1672
|
+
readonly responseModel?: string;
|
|
1673
|
+
/** Caller-level reason the judgment ran; stamped as {@link OmpGenAIAttr.JudgmentPurpose}. */
|
|
1674
|
+
readonly purpose: string;
|
|
1675
|
+
/** Priced usage of this attempt — the same amount the session ledger bills. */
|
|
1676
|
+
readonly usage: Usage;
|
|
1677
|
+
readonly stopReason: StopReason;
|
|
1678
|
+
readonly errorMessage?: string;
|
|
1679
|
+
/** Epoch ms the attempt started. */
|
|
1680
|
+
readonly startTime: number;
|
|
1681
|
+
/** Questions in the request; omitted for prompted backends, which report per completion attempt. */
|
|
1682
|
+
readonly questions?: number;
|
|
1683
|
+
/** Questions answered from the local cache rather than the provider. */
|
|
1684
|
+
readonly cachedQuestions?: number;
|
|
1685
|
+
}
|
|
1686
|
+
|
|
1687
|
+
/**
|
|
1688
|
+
* Emit one `judgment` span per judgment attempt and forward any billed usage
|
|
1689
|
+
* to `onCostDelta` / `onChatUsage` with `operation: "judgment"`. Cost is the
|
|
1690
|
+
* attempt's own priced usage rather than {@link AgentTelemetryConfig.costEstimator},
|
|
1691
|
+
* so spans, metrics, and the session ledger agree on every judgment dollar.
|
|
1692
|
+
* Attempts that billed nothing (cache hits, pre-response failures) keep their
|
|
1693
|
+
* span but fire no usage hooks. No-op when telemetry is disabled.
|
|
1694
|
+
*/
|
|
1695
|
+
export async function recordJudgmentTelemetry(
|
|
1696
|
+
telemetry: AgentTelemetry | undefined,
|
|
1697
|
+
options: JudgmentTelemetryOptions,
|
|
1698
|
+
): Promise<void> {
|
|
1699
|
+
if (!telemetry) return;
|
|
1700
|
+
const attributes: Attributes = {
|
|
1701
|
+
[GenAIAttr.RequestModel]: options.model,
|
|
1702
|
+
[OmpGenAIAttr.JudgmentPurpose]: options.purpose,
|
|
1703
|
+
};
|
|
1704
|
+
const provider = normalizeProviderName(telemetry, options.provider);
|
|
1705
|
+
if (provider) attributes[GenAIAttr.ProviderName] = provider;
|
|
1706
|
+
if (options.questions !== undefined) attributes[OmpGenAIAttr.JudgmentQuestions] = options.questions;
|
|
1707
|
+
if (options.cachedQuestions !== undefined) {
|
|
1708
|
+
attributes[OmpGenAIAttr.JudgmentCachedQuestions] = options.cachedQuestions;
|
|
1709
|
+
}
|
|
1710
|
+
const span = startSpan(telemetry, "judgment", `judgment ${options.model}`, {
|
|
1711
|
+
spanKind: SpanKind.CLIENT,
|
|
1712
|
+
attributes,
|
|
1713
|
+
startTime: options.startTime,
|
|
1714
|
+
});
|
|
1715
|
+
if (!span) return;
|
|
1716
|
+
const model = options.responseModel ?? options.model;
|
|
1717
|
+
span.setAttribute(GenAIAttr.ResponseModel, model);
|
|
1718
|
+
const finishReason = mapStopReason(options.stopReason);
|
|
1719
|
+
if (finishReason) span.setAttribute(GenAIAttr.ResponseFinishReasons, [finishReason]);
|
|
1720
|
+
applyUsageAttributes(span, options.usage);
|
|
1721
|
+
const { cost } = options.usage;
|
|
1722
|
+
span.setAttribute(OmpGenAIAttr.CostEstimatedUsd, cost.total);
|
|
1723
|
+
span.setAttribute(OmpGenAIAttr.CostInputUsd, cost.input);
|
|
1724
|
+
span.setAttribute(OmpGenAIAttr.CostOutputUsd, cost.output);
|
|
1725
|
+
if (options.usage.totalTokens > 0 || cost.total > 0) {
|
|
1726
|
+
const applied: AppliedCostEstimate = {
|
|
1727
|
+
costUsd: cost.total,
|
|
1728
|
+
inputUsd: cost.input,
|
|
1729
|
+
outputUsd: cost.output,
|
|
1730
|
+
costUnavailableReason: undefined,
|
|
1731
|
+
};
|
|
1732
|
+
emitCostDelta(telemetry, {
|
|
1733
|
+
agent: normalizedTelemetryAgent(telemetry),
|
|
1734
|
+
conversationId: telemetry.conversationId,
|
|
1735
|
+
stepNumber: undefined,
|
|
1736
|
+
provider: provider ?? options.provider,
|
|
1737
|
+
model,
|
|
1738
|
+
serviceTier: undefined,
|
|
1739
|
+
usage: buildUsageSnapshot(options.usage),
|
|
1740
|
+
costUsd: applied.costUsd,
|
|
1741
|
+
inputUsd: applied.inputUsd,
|
|
1742
|
+
outputUsd: applied.outputUsd,
|
|
1743
|
+
costUnavailableReason: undefined,
|
|
1744
|
+
});
|
|
1745
|
+
await emitChatUsage(telemetry, span, {
|
|
1746
|
+
operation: GenAIOperation.Judgment,
|
|
1747
|
+
model,
|
|
1748
|
+
provider: options.provider,
|
|
1749
|
+
serviceTier: undefined,
|
|
1750
|
+
stepNumber: undefined,
|
|
1751
|
+
usage: options.usage,
|
|
1752
|
+
applied,
|
|
1753
|
+
headers: undefined,
|
|
1754
|
+
});
|
|
1755
|
+
}
|
|
1756
|
+
applyTerminalStatus(span, options.stopReason, options.errorMessage);
|
|
1757
|
+
safeOnSpanEnd(telemetry, { ...buildTelemetryAttributeContext(telemetry, "judgment", {}), span });
|
|
1758
|
+
span.end();
|
|
1759
|
+
}
|
|
1760
|
+
|
|
1638
1761
|
/**
|
|
1639
1762
|
* Options accepted by {@link instrumentedCompleteSimple}. Mirrors the
|
|
1640
1763
|
* `streamAssistantResponse` chat-span lifecycle for oneshot LLM calls
|
|
@@ -1647,7 +1770,7 @@ export interface InstrumentedChatSpanOptions {
|
|
|
1647
1770
|
/** Step index recorded on the span; defaults to `-1` for non-loop calls. */
|
|
1648
1771
|
readonly stepNumber?: number;
|
|
1649
1772
|
/**
|
|
1650
|
-
* Tag stamped onto `
|
|
1773
|
+
* Tag stamped onto `omp.gen_ai.oneshot.kind`. Values used by the agent:
|
|
1651
1774
|
* `compaction_summary`, `compaction_short_summary`, `compaction_turn_prefix`,
|
|
1652
1775
|
* `handoff`, `branch_summary`, `image_question`. Free-form to allow callers
|
|
1653
1776
|
* outside this package to add new kinds without bumping the helper.
|
|
@@ -1726,7 +1849,7 @@ export async function instrumentedCompleteSimple<TApi extends Api>(
|
|
|
1726
1849
|
},
|
|
1727
1850
|
});
|
|
1728
1851
|
if (chatSpan) {
|
|
1729
|
-
if (oneshotKind) chatSpan.setAttribute(
|
|
1852
|
+
if (oneshotKind) chatSpan.setAttribute(OmpGenAIAttr.OneshotKind, oneshotKind);
|
|
1730
1853
|
if (span.attributes) chatSpan.setAttributes(span.attributes);
|
|
1731
1854
|
}
|
|
1732
1855
|
|
|
@@ -1881,7 +2004,7 @@ export function finishExecuteToolSpan(
|
|
|
1881
2004
|
}
|
|
1882
2005
|
|
|
1883
2006
|
/** Span attribute carrying the terminal {@link ToolStatus}. */
|
|
1884
|
-
export const EXECUTE_TOOL_STATUS_ATTR =
|
|
2007
|
+
export const EXECUTE_TOOL_STATUS_ATTR = OmpGenAIAttr.ToolStatus;
|
|
1885
2008
|
|
|
1886
2009
|
/**
|
|
1887
2010
|
* Mapping from non-ok {@link ToolStatus} values to the `error.type` attribute
|
|
@@ -1974,66 +2097,66 @@ export function fireOnRunEnd(telemetry: AgentTelemetry, summary: AgentRunSummary
|
|
|
1974
2097
|
}
|
|
1975
2098
|
}
|
|
1976
2099
|
|
|
1977
|
-
/** Aggregate `
|
|
1978
|
-
export const enum
|
|
1979
|
-
ChatsCount = "
|
|
1980
|
-
ChatsTotalLatencyMs = "
|
|
1981
|
-
ChatsStopReasonPrefix = "
|
|
1982
|
-
ToolsCount = "
|
|
1983
|
-
ToolsOkCount = "
|
|
1984
|
-
ToolsErrorCount = "
|
|
1985
|
-
ToolsSkippedCount = "
|
|
1986
|
-
ToolsBlockedCount = "
|
|
1987
|
-
ToolsTimeoutCount = "
|
|
1988
|
-
ToolsAbortedCount = "
|
|
1989
|
-
ToolsTotalLatencyMs = "
|
|
1990
|
-
ToolsInvoked = "
|
|
1991
|
-
ToolsAvailable = "
|
|
1992
|
-
ToolsUnused = "
|
|
1993
|
-
UsageInputTokensTotal = "
|
|
1994
|
-
UsageOutputTokensTotal = "
|
|
1995
|
-
UsageCacheReadInputTokensTotal = "
|
|
1996
|
-
UsageCacheCreationInputTokensTotal = "
|
|
1997
|
-
UsageReasoningOutputTokensTotal = "
|
|
1998
|
-
UsageTotalTokensTotal = "
|
|
1999
|
-
CostEstimatedUsdTotal = "
|
|
2000
|
-
ErrorsCount = "
|
|
2001
|
-
}
|
|
2002
|
-
|
|
2003
|
-
/** Stamp the aggregate `
|
|
2100
|
+
/** Aggregate `omp.gen_ai.agent.*` attributes stamped on the `invoke_agent` span. */
|
|
2101
|
+
export const enum OmpGenAIAggregateAttr {
|
|
2102
|
+
ChatsCount = "omp.gen_ai.agent.chats.count",
|
|
2103
|
+
ChatsTotalLatencyMs = "omp.gen_ai.agent.chats.total_latency_ms",
|
|
2104
|
+
ChatsStopReasonPrefix = "omp.gen_ai.agent.chats.stop_reason.",
|
|
2105
|
+
ToolsCount = "omp.gen_ai.agent.tools.count",
|
|
2106
|
+
ToolsOkCount = "omp.gen_ai.agent.tools.ok.count",
|
|
2107
|
+
ToolsErrorCount = "omp.gen_ai.agent.tools.error.count",
|
|
2108
|
+
ToolsSkippedCount = "omp.gen_ai.agent.tools.skipped.count",
|
|
2109
|
+
ToolsBlockedCount = "omp.gen_ai.agent.tools.blocked.count",
|
|
2110
|
+
ToolsTimeoutCount = "omp.gen_ai.agent.tools.timeout.count",
|
|
2111
|
+
ToolsAbortedCount = "omp.gen_ai.agent.tools.aborted.count",
|
|
2112
|
+
ToolsTotalLatencyMs = "omp.gen_ai.agent.tools.total_latency_ms",
|
|
2113
|
+
ToolsInvoked = "omp.gen_ai.agent.tools.invoked",
|
|
2114
|
+
ToolsAvailable = "omp.gen_ai.agent.tools.available",
|
|
2115
|
+
ToolsUnused = "omp.gen_ai.agent.tools.unused",
|
|
2116
|
+
UsageInputTokensTotal = "omp.gen_ai.agent.usage.input_tokens.total",
|
|
2117
|
+
UsageOutputTokensTotal = "omp.gen_ai.agent.usage.output_tokens.total",
|
|
2118
|
+
UsageCacheReadInputTokensTotal = "omp.gen_ai.agent.usage.cache_read.input_tokens.total",
|
|
2119
|
+
UsageCacheCreationInputTokensTotal = "omp.gen_ai.agent.usage.cache_creation.input_tokens.total",
|
|
2120
|
+
UsageReasoningOutputTokensTotal = "omp.gen_ai.agent.usage.reasoning.output_tokens.total",
|
|
2121
|
+
UsageTotalTokensTotal = "omp.gen_ai.agent.usage.total_tokens.total",
|
|
2122
|
+
CostEstimatedUsdTotal = "omp.gen_ai.agent.cost.estimated_usd.total",
|
|
2123
|
+
ErrorsCount = "omp.gen_ai.agent.errors.count",
|
|
2124
|
+
}
|
|
2125
|
+
|
|
2126
|
+
/** Stamp the aggregate `omp.gen_ai.agent.*` attributes on the given span. */
|
|
2004
2127
|
function applyAggregateAttributes(span: Span, summary: AgentRunSummary, coverage: AgentRunCoverage): void {
|
|
2005
|
-
span.setAttribute(
|
|
2006
|
-
span.setAttribute(
|
|
2128
|
+
span.setAttribute(OmpGenAIAggregateAttr.ChatsCount, summary.chats.total);
|
|
2129
|
+
span.setAttribute(OmpGenAIAggregateAttr.ChatsTotalLatencyMs, summary.chats.totalLatencyMs);
|
|
2007
2130
|
for (const [reason, count] of Object.entries(summary.chats.byStopReason)) {
|
|
2008
|
-
span.setAttribute(`${
|
|
2009
|
-
}
|
|
2010
|
-
span.setAttribute(
|
|
2011
|
-
span.setAttribute(
|
|
2012
|
-
span.setAttribute(
|
|
2013
|
-
span.setAttribute(
|
|
2014
|
-
span.setAttribute(
|
|
2015
|
-
span.setAttribute(
|
|
2016
|
-
span.setAttribute(
|
|
2017
|
-
span.setAttribute(
|
|
2131
|
+
span.setAttribute(`${OmpGenAIAggregateAttr.ChatsStopReasonPrefix}${reason}.count`, count);
|
|
2132
|
+
}
|
|
2133
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsCount, summary.tools.total);
|
|
2134
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsOkCount, summary.tools.ok);
|
|
2135
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsErrorCount, summary.tools.error);
|
|
2136
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsSkippedCount, summary.tools.skipped);
|
|
2137
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsBlockedCount, summary.tools.blocked);
|
|
2138
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsTimeoutCount, summary.tools.timeout);
|
|
2139
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsAbortedCount, summary.tools.aborted);
|
|
2140
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsTotalLatencyMs, summary.tools.totalLatencyMs);
|
|
2018
2141
|
if (coverage.toolsInvoked.length > 0) {
|
|
2019
|
-
span.setAttribute(
|
|
2142
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsInvoked, [...coverage.toolsInvoked]);
|
|
2020
2143
|
}
|
|
2021
2144
|
if (coverage.toolsAvailable.length > 0) {
|
|
2022
|
-
span.setAttribute(
|
|
2145
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsAvailable, [...coverage.toolsAvailable]);
|
|
2023
2146
|
}
|
|
2024
2147
|
if (coverage.toolsUnused.length > 0) {
|
|
2025
|
-
span.setAttribute(
|
|
2026
|
-
}
|
|
2027
|
-
span.setAttribute(
|
|
2028
|
-
span.setAttribute(
|
|
2029
|
-
span.setAttribute(
|
|
2030
|
-
span.setAttribute(
|
|
2031
|
-
span.setAttribute(
|
|
2032
|
-
span.setAttribute(
|
|
2148
|
+
span.setAttribute(OmpGenAIAggregateAttr.ToolsUnused, [...coverage.toolsUnused]);
|
|
2149
|
+
}
|
|
2150
|
+
span.setAttribute(OmpGenAIAggregateAttr.UsageInputTokensTotal, summary.usage.inputTokens);
|
|
2151
|
+
span.setAttribute(OmpGenAIAggregateAttr.UsageOutputTokensTotal, summary.usage.outputTokens);
|
|
2152
|
+
span.setAttribute(OmpGenAIAggregateAttr.UsageCacheReadInputTokensTotal, summary.usage.cachedInputTokens);
|
|
2153
|
+
span.setAttribute(OmpGenAIAggregateAttr.UsageCacheCreationInputTokensTotal, summary.usage.cacheWriteTokens);
|
|
2154
|
+
span.setAttribute(OmpGenAIAggregateAttr.UsageReasoningOutputTokensTotal, summary.usage.reasoningOutputTokens);
|
|
2155
|
+
span.setAttribute(OmpGenAIAggregateAttr.UsageTotalTokensTotal, summary.usage.totalTokens);
|
|
2033
2156
|
if (summary.cost.estimatedUsd > 0) {
|
|
2034
|
-
span.setAttribute(
|
|
2157
|
+
span.setAttribute(OmpGenAIAggregateAttr.CostEstimatedUsdTotal, summary.cost.estimatedUsd);
|
|
2035
2158
|
}
|
|
2036
|
-
span.setAttribute(
|
|
2159
|
+
span.setAttribute(OmpGenAIAggregateAttr.ErrorsCount, summary.errors.total);
|
|
2037
2160
|
}
|
|
2038
2161
|
|
|
2039
2162
|
/**
|
|
@@ -2068,10 +2191,10 @@ export function recordHandoff(
|
|
|
2068
2191
|
const attrs: Attributes = {};
|
|
2069
2192
|
const fromAgent = options.fromAgent ? normalizeAgentIdentity(telemetry, options.fromAgent) : undefined;
|
|
2070
2193
|
const toAgent = normalizeAgentIdentity(telemetry, options.toAgent);
|
|
2071
|
-
if (fromAgent?.name) attrs[
|
|
2072
|
-
if (fromAgent?.id) attrs[
|
|
2073
|
-
if (toAgent.name) attrs[
|
|
2074
|
-
if (toAgent.id) attrs[
|
|
2194
|
+
if (fromAgent?.name) attrs[OmpGenAIAttr.HandoffFromAgentName] = fromAgent.name;
|
|
2195
|
+
if (fromAgent?.id) attrs[OmpGenAIAttr.HandoffFromAgentId] = fromAgent.id;
|
|
2196
|
+
if (toAgent.name) attrs[OmpGenAIAttr.HandoffToAgentName] = toAgent.name;
|
|
2197
|
+
if (toAgent.id) attrs[OmpGenAIAttr.HandoffToAgentId] = toAgent.id;
|
|
2075
2198
|
const name = toAgent.name
|
|
2076
2199
|
? fromAgent?.name
|
|
2077
2200
|
? `handoff ${fromAgent.name} → ${toAgent.name}`
|