sediment-capture 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sediment_capture/__init__.py +48 -0
- sediment_capture/gateway.py +551 -0
- sediment_capture/github.py +646 -0
- sediment_capture/otlp.py +1394 -0
- sediment_capture/session_identity.py +281 -0
- sediment_capture-0.1.0.dist-info/METADATA +12 -0
- sediment_capture-0.1.0.dist-info/RECORD +9 -0
- sediment_capture-0.1.0.dist-info/WHEEL +4 -0
- sediment_capture-0.1.0.dist-info/licenses/LICENSE +661 -0
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
"""Sediment capture: translators that turn provider payloads into facts."""
|
|
3
|
+
|
|
4
|
+
from .gateway import ADAPTERS, LiteLLMAdapter
|
|
5
|
+
from .github import (
|
|
6
|
+
PullRequestRevisionSkipReason,
|
|
7
|
+
RepositoryCaptureSkipReason,
|
|
8
|
+
parse_repository_rename,
|
|
9
|
+
parse_pull_request_merge,
|
|
10
|
+
parse_pull_request_revision,
|
|
11
|
+
parse_push,
|
|
12
|
+
parse_workflow_run,
|
|
13
|
+
sign_payload,
|
|
14
|
+
verify_signature,
|
|
15
|
+
)
|
|
16
|
+
from .otlp import (
|
|
17
|
+
OTLPCaptureResult,
|
|
18
|
+
parse_otlp_logs,
|
|
19
|
+
parse_otlp_decisions,
|
|
20
|
+
parse_otlp_edit_observations,
|
|
21
|
+
parse_otlp_rejected_edits,
|
|
22
|
+
parse_otlp_retry_linkages,
|
|
23
|
+
RetryLinkageSkipReason,
|
|
24
|
+
)
|
|
25
|
+
from .session_identity import SessionIdentity, resolve_identity
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"ADAPTERS",
|
|
29
|
+
"LiteLLMAdapter",
|
|
30
|
+
"SessionIdentity",
|
|
31
|
+
"PullRequestRevisionSkipReason",
|
|
32
|
+
"RepositoryCaptureSkipReason",
|
|
33
|
+
"parse_repository_rename",
|
|
34
|
+
"OTLPCaptureResult",
|
|
35
|
+
"parse_otlp_logs",
|
|
36
|
+
"parse_otlp_decisions",
|
|
37
|
+
"parse_otlp_edit_observations",
|
|
38
|
+
"parse_otlp_rejected_edits",
|
|
39
|
+
"parse_otlp_retry_linkages",
|
|
40
|
+
"RetryLinkageSkipReason",
|
|
41
|
+
"parse_push",
|
|
42
|
+
"parse_pull_request_merge",
|
|
43
|
+
"parse_pull_request_revision",
|
|
44
|
+
"parse_workflow_run",
|
|
45
|
+
"resolve_identity",
|
|
46
|
+
"sign_payload",
|
|
47
|
+
"verify_signature",
|
|
48
|
+
]
|
|
@@ -0,0 +1,551 @@
|
|
|
1
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
"""
|
|
3
|
+
Gateway adapters: provider payloads into structured inference-call facts.
|
|
4
|
+
|
|
5
|
+
Pure payload→fact translation — no I/O, no storage, no attribution at
|
|
6
|
+
ingest (ADR 0001). Identity (session/user/org) arrives as parameters: the
|
|
7
|
+
ingest route binds ``org_id`` to the credential and the callback
|
|
8
|
+
supplies real session/user ids (ADR 0002 — no placeholders here). The
|
|
9
|
+
``ADAPTERS`` registry lives in the same module as the adapters it registers.
|
|
10
|
+
|
|
11
|
+
Fail-soft posture throughout (the same contract as ``github.py``): a
|
|
12
|
+
malformed payload must degrade, never raise — the ingest route would turn
|
|
13
|
+
an exception into a 500 and the inference call would be lost. The original
|
|
14
|
+
payload always survives on ``raw``.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
import logging
|
|
21
|
+
import math
|
|
22
|
+
from datetime import datetime
|
|
23
|
+
from typing import Any
|
|
24
|
+
from uuid import UUID, uuid5
|
|
25
|
+
|
|
26
|
+
from sediment_core import (
|
|
27
|
+
GatewayProvider,
|
|
28
|
+
InferenceCall,
|
|
29
|
+
InferenceMessage,
|
|
30
|
+
ReasoningPart,
|
|
31
|
+
TextPart,
|
|
32
|
+
ToolCallPart,
|
|
33
|
+
ToolCallResponsePart,
|
|
34
|
+
normalize_org_id,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
logger = logging.getLogger("sediment.capture.gateway")
|
|
38
|
+
|
|
39
|
+
# PostgreSQL BIGINT is a signed 64-bit integer. Larger values fail at INSERT,
|
|
40
|
+
# and int(NaN)/int(inf) raise here in the adapter — bound once so
|
|
41
|
+
# a crafted numeric degrades instead of crashing (same guard as github.py's
|
|
42
|
+
# pr_number).
|
|
43
|
+
_INT64_MAX = 2**63 - 1
|
|
44
|
+
|
|
45
|
+
# Dedicated namespace for callback-prepared capture identities (ADR 0017).
|
|
46
|
+
_CAPTURE_NAMESPACE = UUID("f8bfd6e0-f182-4bcd-a1f8-85a987a8c6e2")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _as_dict(value: Any) -> dict[str, Any]:
|
|
50
|
+
"""Gateway payloads are untrusted — coerce a missing/non-dict field to {}."""
|
|
51
|
+
return value if isinstance(value, dict) else {}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _as_list(value: Any) -> list[Any]:
|
|
55
|
+
"""Untrusted sibling of ``_as_dict`` for list-shaped fields."""
|
|
56
|
+
return value if isinstance(value, list) else []
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _int_in_bounds(value: Any) -> int | None:
|
|
60
|
+
"""Coerce a numeric to the fact store's int64 range; fail soft otherwise.
|
|
61
|
+
|
|
62
|
+
Negatives are out of bounds too: every consumer here is a count or a
|
|
63
|
+
latency, ge=0 at the schema — a crafted negative must degrade to the
|
|
64
|
+
fallback, not raise at InferenceCall construction.
|
|
65
|
+
"""
|
|
66
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
67
|
+
return None
|
|
68
|
+
if isinstance(value, float) and not math.isfinite(value):
|
|
69
|
+
return None
|
|
70
|
+
n = int(value)
|
|
71
|
+
if not 0 <= n <= _INT64_MAX:
|
|
72
|
+
return None
|
|
73
|
+
return n
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _first_int(*values: Any) -> int | None:
|
|
77
|
+
"""Return the first storable int value, else ``None``."""
|
|
78
|
+
for value in values:
|
|
79
|
+
n = _int_in_bounds(value)
|
|
80
|
+
if n is not None:
|
|
81
|
+
return n
|
|
82
|
+
return None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _text(value: Any) -> str:
|
|
86
|
+
"""Best-effort text for a message field. Strings pass
|
|
87
|
+
through; non-string shapes (e.g. multimodal content parts) degrade to
|
|
88
|
+
their JSON form rather than raising during message-part validation."""
|
|
89
|
+
if isinstance(value, str):
|
|
90
|
+
return value
|
|
91
|
+
if value is None:
|
|
92
|
+
return ""
|
|
93
|
+
try:
|
|
94
|
+
return json.dumps(value)
|
|
95
|
+
except (TypeError, ValueError):
|
|
96
|
+
return str(value)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _tool_call_input(arguments: Any) -> dict[str, Any] | None:
|
|
100
|
+
"""The parsed argument object, or None when malformed.
|
|
101
|
+
|
|
102
|
+
OpenAI-shape ``arguments`` is a JSON string; Anthropic-shape ``input`` is
|
|
103
|
+
already an object. Anything that fails to parse as a JSON *object* —
|
|
104
|
+
invalid JSON, or valid JSON that isn't a dict — is malformed: the caller
|
|
105
|
+
keeps the tool-call part with ``arguments={}`` (the id is join-critical
|
|
106
|
+
and must survive a cosmetic defect) and counts the occurrence. Never
|
|
107
|
+
wrap the raw string into input — the unparsed original is already on
|
|
108
|
+
``InferenceCall.raw`` for storage-seam Basic redaction; normalized fields
|
|
109
|
+
carry normalized data or nothing.
|
|
110
|
+
"""
|
|
111
|
+
if isinstance(arguments, str):
|
|
112
|
+
try:
|
|
113
|
+
arguments = json.loads(arguments)
|
|
114
|
+
except (TypeError, ValueError):
|
|
115
|
+
return None
|
|
116
|
+
return arguments if isinstance(arguments, dict) else None
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _usable_id(value: Any) -> bool:
|
|
120
|
+
"""A tool-call id that can join something: a non-blank string."""
|
|
121
|
+
return isinstance(value, str) and bool(value.strip())
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _optional_text(value: Any) -> str | None:
|
|
125
|
+
if not isinstance(value, str):
|
|
126
|
+
return None
|
|
127
|
+
value = value.strip()
|
|
128
|
+
return value or None
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _decoded_json(value: Any) -> Any:
|
|
132
|
+
if not isinstance(value, str):
|
|
133
|
+
return value
|
|
134
|
+
try:
|
|
135
|
+
return json.loads(value)
|
|
136
|
+
except (TypeError, ValueError):
|
|
137
|
+
return value
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _tool_part(
|
|
141
|
+
call_id: Any,
|
|
142
|
+
name: Any,
|
|
143
|
+
arguments: Any,
|
|
144
|
+
degraded: dict[str, int],
|
|
145
|
+
) -> ToolCallPart | None:
|
|
146
|
+
if not _usable_id(call_id):
|
|
147
|
+
degraded["missing_id"] += 1
|
|
148
|
+
return None
|
|
149
|
+
parsed = _tool_call_input(arguments)
|
|
150
|
+
if parsed is None:
|
|
151
|
+
degraded["malformed_arguments"] += 1
|
|
152
|
+
parsed = {}
|
|
153
|
+
return ToolCallPart(id=call_id, name=_text(name), arguments=parsed)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _reasoning_part(content: Any, degraded: dict[str, int]) -> ReasoningPart | None:
|
|
157
|
+
if isinstance(content, str) and content.strip():
|
|
158
|
+
return ReasoningPart(content=content)
|
|
159
|
+
degraded["opaque_reasoning"] += 1
|
|
160
|
+
return None
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _discriminator_declined(value: Any, *, position: int, field: str) -> None:
|
|
164
|
+
logger.warning(
|
|
165
|
+
"gateway_discriminator_declined",
|
|
166
|
+
extra={
|
|
167
|
+
"source": "litellm",
|
|
168
|
+
"record_position": position,
|
|
169
|
+
"field": field,
|
|
170
|
+
"reason": "unsupported_discriminator"
|
|
171
|
+
if isinstance(value, str)
|
|
172
|
+
else "malformed_discriminator",
|
|
173
|
+
},
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _message_part(
|
|
178
|
+
block: Any,
|
|
179
|
+
degraded: dict[str, int],
|
|
180
|
+
*,
|
|
181
|
+
position: int,
|
|
182
|
+
):
|
|
183
|
+
if block is None:
|
|
184
|
+
return None
|
|
185
|
+
if isinstance(block, str):
|
|
186
|
+
return TextPart(content=block)
|
|
187
|
+
if not isinstance(block, dict):
|
|
188
|
+
degraded["unsupported_content"] += 1
|
|
189
|
+
return None
|
|
190
|
+
part_type = block.get("type")
|
|
191
|
+
if not isinstance(part_type, str):
|
|
192
|
+
_discriminator_declined(part_type, position=position, field="content.type")
|
|
193
|
+
degraded["unsupported_content"] += 1
|
|
194
|
+
return None
|
|
195
|
+
if part_type in {"thinking", "reasoning"}:
|
|
196
|
+
content = block.get("thinking", block.get("text", block.get("content")))
|
|
197
|
+
return _reasoning_part(content, degraded)
|
|
198
|
+
if part_type == "redacted_thinking":
|
|
199
|
+
degraded["opaque_reasoning"] += 1
|
|
200
|
+
return None
|
|
201
|
+
if part_type in {"text", "input_text", "output_text"}:
|
|
202
|
+
text = block.get("text", block.get("content"))
|
|
203
|
+
if isinstance(text, str):
|
|
204
|
+
return TextPart(content=text)
|
|
205
|
+
degraded["unsupported_content"] += 1
|
|
206
|
+
return None
|
|
207
|
+
elif part_type in {"tool_use", "tool_call"}:
|
|
208
|
+
return _tool_part(
|
|
209
|
+
block.get("id"),
|
|
210
|
+
block.get("name"),
|
|
211
|
+
block.get("input", block.get("arguments")),
|
|
212
|
+
degraded,
|
|
213
|
+
)
|
|
214
|
+
elif part_type in {"tool_result", "tool_call_response"}:
|
|
215
|
+
call_id = block.get("tool_use_id", block.get("id"))
|
|
216
|
+
if not _usable_id(call_id):
|
|
217
|
+
degraded["missing_id"] += 1
|
|
218
|
+
return None
|
|
219
|
+
result = block.get("result", block.get("content"))
|
|
220
|
+
return ToolCallResponsePart(id=call_id, result=_decoded_json(result))
|
|
221
|
+
_discriminator_declined(part_type, position=position, field="content.type")
|
|
222
|
+
degraded["unsupported_content"] += 1
|
|
223
|
+
return None
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _message_reasoning_parts(
|
|
227
|
+
item: dict[str, Any], degraded: dict[str, int], *, position: int
|
|
228
|
+
) -> list[ReasoningPart]:
|
|
229
|
+
thinking_blocks = item.get("thinking_blocks")
|
|
230
|
+
if thinking_blocks is not None and not isinstance(thinking_blocks, list):
|
|
231
|
+
degraded["unsupported_content"] += 1
|
|
232
|
+
thinking_blocks = None
|
|
233
|
+
if thinking_blocks is None:
|
|
234
|
+
provider_thinking_blocks = _as_dict(item.get("provider_specific_fields")).get(
|
|
235
|
+
"thinking_blocks"
|
|
236
|
+
)
|
|
237
|
+
if provider_thinking_blocks is not None and not isinstance(
|
|
238
|
+
provider_thinking_blocks, list
|
|
239
|
+
):
|
|
240
|
+
degraded["unsupported_content"] += 1
|
|
241
|
+
else:
|
|
242
|
+
thinking_blocks = provider_thinking_blocks
|
|
243
|
+
|
|
244
|
+
parts = []
|
|
245
|
+
if isinstance(thinking_blocks, list):
|
|
246
|
+
for block in thinking_blocks:
|
|
247
|
+
if not isinstance(block, dict):
|
|
248
|
+
degraded["unsupported_content"] += 1
|
|
249
|
+
continue
|
|
250
|
+
part_type = block.get("type")
|
|
251
|
+
if not isinstance(part_type, str) or part_type not in {
|
|
252
|
+
"thinking",
|
|
253
|
+
"reasoning",
|
|
254
|
+
"redacted_thinking",
|
|
255
|
+
}:
|
|
256
|
+
_discriminator_declined(
|
|
257
|
+
part_type, position=position, field="thinking_blocks.type"
|
|
258
|
+
)
|
|
259
|
+
degraded["unsupported_content"] += 1
|
|
260
|
+
continue
|
|
261
|
+
part = _message_part(block, degraded, position=position)
|
|
262
|
+
if isinstance(part, ReasoningPart):
|
|
263
|
+
parts.append(part)
|
|
264
|
+
if parts:
|
|
265
|
+
return parts
|
|
266
|
+
|
|
267
|
+
reasoning_content = item.get("reasoning_content")
|
|
268
|
+
if reasoning_content is not None:
|
|
269
|
+
part = _reasoning_part(reasoning_content, degraded)
|
|
270
|
+
return [part] if part is not None else []
|
|
271
|
+
return []
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _inference_message(
|
|
275
|
+
item: dict[str, Any],
|
|
276
|
+
degraded: dict[str, int],
|
|
277
|
+
*,
|
|
278
|
+
finish_reason: Any = None,
|
|
279
|
+
position: int,
|
|
280
|
+
) -> InferenceMessage | None:
|
|
281
|
+
item_type = item.get("type")
|
|
282
|
+
if item_type is not None and (
|
|
283
|
+
not isinstance(item_type, str)
|
|
284
|
+
or item_type not in {"message", "function_call", "function_call_output"}
|
|
285
|
+
):
|
|
286
|
+
_discriminator_declined(item_type, position=position, field="type")
|
|
287
|
+
degraded["unsupported_content"] += 1
|
|
288
|
+
return None
|
|
289
|
+
if item_type == "function_call":
|
|
290
|
+
call_id = item.get("call_id")
|
|
291
|
+
if not _usable_id(call_id):
|
|
292
|
+
call_id = item.get("id")
|
|
293
|
+
part = _tool_part(call_id, item.get("name"), item.get("arguments"), degraded)
|
|
294
|
+
return InferenceMessage(role="assistant", parts=[part]) if part else None
|
|
295
|
+
if item_type == "function_call_output":
|
|
296
|
+
call_id = item.get("call_id")
|
|
297
|
+
if not _usable_id(call_id):
|
|
298
|
+
degraded["missing_id"] += 1
|
|
299
|
+
return None
|
|
300
|
+
return InferenceMessage(
|
|
301
|
+
role="tool",
|
|
302
|
+
parts=[
|
|
303
|
+
ToolCallResponsePart(
|
|
304
|
+
id=call_id, result=_decoded_json(item.get("output"))
|
|
305
|
+
)
|
|
306
|
+
],
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
role = _optional_text(item.get("role"))
|
|
310
|
+
if role is None:
|
|
311
|
+
degraded["unsupported_content"] += 1
|
|
312
|
+
return None
|
|
313
|
+
|
|
314
|
+
content = item.get("content")
|
|
315
|
+
if role == "tool" and _usable_id(item.get("tool_call_id")):
|
|
316
|
+
parts = [
|
|
317
|
+
ToolCallResponsePart(id=item["tool_call_id"], result=_decoded_json(content))
|
|
318
|
+
]
|
|
319
|
+
else:
|
|
320
|
+
blocks = content if isinstance(content, list) else [content]
|
|
321
|
+
content_parts = [
|
|
322
|
+
part
|
|
323
|
+
for block in blocks
|
|
324
|
+
if (part := _message_part(block, degraded, position=position)) is not None
|
|
325
|
+
]
|
|
326
|
+
sibling_reasoning = (
|
|
327
|
+
_message_reasoning_parts(item, degraded, position=position)
|
|
328
|
+
if role == "assistant"
|
|
329
|
+
else []
|
|
330
|
+
)
|
|
331
|
+
parts = (
|
|
332
|
+
sibling_reasoning
|
|
333
|
+
if not any(isinstance(part, ReasoningPart) for part in content_parts)
|
|
334
|
+
else []
|
|
335
|
+
)
|
|
336
|
+
parts.extend(content_parts)
|
|
337
|
+
for call in _as_list(item.get("tool_calls")):
|
|
338
|
+
if not isinstance(call, dict):
|
|
339
|
+
degraded["unsupported_content"] += 1
|
|
340
|
+
continue
|
|
341
|
+
call_type = call.get("type")
|
|
342
|
+
if call_type is not None and (
|
|
343
|
+
not isinstance(call_type, str) or call_type != "function"
|
|
344
|
+
):
|
|
345
|
+
_discriminator_declined(
|
|
346
|
+
call_type, position=position, field="tool_calls.type"
|
|
347
|
+
)
|
|
348
|
+
degraded["unsupported_content"] += 1
|
|
349
|
+
continue
|
|
350
|
+
function = _as_dict(call.get("function"))
|
|
351
|
+
part = _tool_part(
|
|
352
|
+
call.get("id"),
|
|
353
|
+
function.get("name"),
|
|
354
|
+
function.get("arguments"),
|
|
355
|
+
degraded,
|
|
356
|
+
)
|
|
357
|
+
if part is not None:
|
|
358
|
+
parts.append(part)
|
|
359
|
+
|
|
360
|
+
finish = _optional_text(finish_reason)
|
|
361
|
+
return InferenceMessage(role=role, parts=parts, finish_reason=finish)
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _messages(items: list[Any], *, finish_reason: Any = None) -> list[InferenceMessage]:
|
|
365
|
+
degraded = {
|
|
366
|
+
"malformed_arguments": 0,
|
|
367
|
+
"missing_id": 0,
|
|
368
|
+
"opaque_reasoning": 0,
|
|
369
|
+
"unsupported_content": 0,
|
|
370
|
+
}
|
|
371
|
+
messages = []
|
|
372
|
+
for position, item in enumerate(items):
|
|
373
|
+
if not isinstance(item, dict):
|
|
374
|
+
degraded["unsupported_content"] += 1
|
|
375
|
+
continue
|
|
376
|
+
message = _inference_message(
|
|
377
|
+
item, degraded, finish_reason=finish_reason, position=position
|
|
378
|
+
)
|
|
379
|
+
if message is not None:
|
|
380
|
+
messages.append(message)
|
|
381
|
+
if any(degraded.values()):
|
|
382
|
+
level = (
|
|
383
|
+
logging.WARNING
|
|
384
|
+
if any(
|
|
385
|
+
value for key, value in degraded.items() if key != "opaque_reasoning"
|
|
386
|
+
)
|
|
387
|
+
else logging.INFO
|
|
388
|
+
)
|
|
389
|
+
logger.log(
|
|
390
|
+
level,
|
|
391
|
+
"inference_messages_degraded malformed_arguments=%d missing_id=%d "
|
|
392
|
+
"unsupported_content=%d opaque_reasoning=%d kept=%d",
|
|
393
|
+
degraded["malformed_arguments"],
|
|
394
|
+
degraded["missing_id"],
|
|
395
|
+
degraded["unsupported_content"],
|
|
396
|
+
degraded["opaque_reasoning"],
|
|
397
|
+
len(messages),
|
|
398
|
+
)
|
|
399
|
+
return messages
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
class LiteLLMAdapter:
|
|
403
|
+
"""Normalizes LiteLLM inference-call payloads.
|
|
404
|
+
|
|
405
|
+
Handles the real proxy ``StandardLoggingPayload`` (what the custom
|
|
406
|
+
callback forwards in production — see ``litellm/sediment_callback.py``)
|
|
407
|
+
as well as the shaped payload the same callback's ``_fallback_payload``
|
|
408
|
+
builds when no SLO is present (also the shape of the hand-written unit
|
|
409
|
+
fixture). The two differ in where tokens and timing live:
|
|
410
|
+
|
|
411
|
+
- tokens: top-level ``usage`` (shaped) vs ``response.usage`` / top-level
|
|
412
|
+
``prompt_tokens``/``completion_tokens`` (StandardLoggingPayload)
|
|
413
|
+
- timing: ``response_time_ms`` (shaped) vs ``response_time`` seconds or
|
|
414
|
+
``endTime`` - ``startTime`` unix seconds (StandardLoggingPayload)
|
|
415
|
+
|
|
416
|
+
https://docs.litellm.ai/docs/proxy/logging
|
|
417
|
+
"""
|
|
418
|
+
|
|
419
|
+
def normalize(
|
|
420
|
+
self,
|
|
421
|
+
payload: dict[str, Any],
|
|
422
|
+
*,
|
|
423
|
+
session_id: str,
|
|
424
|
+
user_id: str | None,
|
|
425
|
+
org_id: str,
|
|
426
|
+
capture_id: UUID | None = None,
|
|
427
|
+
observed_at: datetime | None = None,
|
|
428
|
+
) -> InferenceCall:
|
|
429
|
+
if (capture_id is None) != (observed_at is None):
|
|
430
|
+
raise ValueError("capture_id and observed_at must be supplied together")
|
|
431
|
+
capture_fields: dict[str, Any] = {}
|
|
432
|
+
if capture_id is not None:
|
|
433
|
+
capture_fields = {
|
|
434
|
+
"inference_call_id": str(
|
|
435
|
+
uuid5(
|
|
436
|
+
_CAPTURE_NAMESPACE,
|
|
437
|
+
json.dumps(
|
|
438
|
+
[normalize_org_id(org_id), str(UUID(str(capture_id)))],
|
|
439
|
+
separators=(",", ":"),
|
|
440
|
+
),
|
|
441
|
+
)
|
|
442
|
+
),
|
|
443
|
+
"observed_at": observed_at,
|
|
444
|
+
}
|
|
445
|
+
raw_response = payload.get("response")
|
|
446
|
+
response_text = raw_response if isinstance(raw_response, str) else ""
|
|
447
|
+
response = _as_dict(raw_response)
|
|
448
|
+
choices = _as_list(response.get("choices"))
|
|
449
|
+
first_choice = _as_dict(choices[0]) if choices else {}
|
|
450
|
+
message = _as_dict(first_choice.get("message"))
|
|
451
|
+
|
|
452
|
+
input_messages = _messages(_as_list(payload.get("messages")))
|
|
453
|
+
if message:
|
|
454
|
+
output_messages = _messages(
|
|
455
|
+
[message], finish_reason=first_choice.get("finish_reason")
|
|
456
|
+
)
|
|
457
|
+
elif isinstance(first_choice.get("text"), str):
|
|
458
|
+
output_messages = [
|
|
459
|
+
InferenceMessage(
|
|
460
|
+
role="assistant",
|
|
461
|
+
parts=[TextPart(content=first_choice["text"])],
|
|
462
|
+
finish_reason=_optional_text(first_choice.get("finish_reason")),
|
|
463
|
+
)
|
|
464
|
+
]
|
|
465
|
+
elif _as_list(response.get("output")):
|
|
466
|
+
output_messages = _messages(_as_list(response.get("output")))
|
|
467
|
+
elif response_text:
|
|
468
|
+
output_messages = [
|
|
469
|
+
InferenceMessage(
|
|
470
|
+
role="assistant", parts=[TextPart(content=response_text)]
|
|
471
|
+
)
|
|
472
|
+
]
|
|
473
|
+
else:
|
|
474
|
+
output_messages = []
|
|
475
|
+
|
|
476
|
+
# Tokens may be under a top-level ``usage`` (shaped) or under
|
|
477
|
+
# ``response.usage`` / top-level keys (StandardLoggingPayload). A
|
|
478
|
+
# malformed or empty top-level usage falls through to response.usage.
|
|
479
|
+
usage = _as_dict(payload.get("usage")) or _as_dict(response.get("usage"))
|
|
480
|
+
|
|
481
|
+
# The proxy-assigned call id — the idempotency identity for a
|
|
482
|
+
# redelivered/retried callback POST. Fall back to the response's own
|
|
483
|
+
# completion id; a payload with neither stays keyless — model_call_id
|
|
484
|
+
# None means no dedup key under uq_inference_calls_model_call.
|
|
485
|
+
# _usable_id, not truthiness: model_call_id is the dedup key, and a
|
|
486
|
+
# truthy non-string (True → "True") or a blank string would be a
|
|
487
|
+
# shared key that collapses every later call as a redelivery.
|
|
488
|
+
model_call_id = payload.get("litellm_call_id")
|
|
489
|
+
if not _usable_id(model_call_id):
|
|
490
|
+
model_call_id = response.get("id")
|
|
491
|
+
if not _usable_id(model_call_id):
|
|
492
|
+
model_call_id = None
|
|
493
|
+
|
|
494
|
+
model = _optional_text(payload.get("model")) or _optional_text(
|
|
495
|
+
response.get("model")
|
|
496
|
+
)
|
|
497
|
+
return InferenceCall(
|
|
498
|
+
session_id=session_id,
|
|
499
|
+
user_id=user_id,
|
|
500
|
+
org_id=org_id,
|
|
501
|
+
gateway_provider=GatewayProvider.LITELLM,
|
|
502
|
+
model_provider=_optional_text(payload.get("custom_llm_provider")),
|
|
503
|
+
model=model,
|
|
504
|
+
input_messages=input_messages,
|
|
505
|
+
output_messages=output_messages,
|
|
506
|
+
model_call_id=model_call_id,
|
|
507
|
+
input_tokens=_first_int(
|
|
508
|
+
usage.get("prompt_tokens"), payload.get("prompt_tokens")
|
|
509
|
+
),
|
|
510
|
+
output_tokens=_first_int(
|
|
511
|
+
usage.get("completion_tokens"), payload.get("completion_tokens")
|
|
512
|
+
),
|
|
513
|
+
duration_ms=self._latency_ms(payload),
|
|
514
|
+
raw=payload,
|
|
515
|
+
**capture_fields,
|
|
516
|
+
)
|
|
517
|
+
|
|
518
|
+
@staticmethod
|
|
519
|
+
def _latency_ms(payload: dict[str, Any]) -> int | None:
|
|
520
|
+
"""Resolve latency in ms across the shapes LiteLLM emits. Each
|
|
521
|
+
branch is taken only when its value is a storable numeric, so a
|
|
522
|
+
malformed field (non-numeric, NaN/inf, beyond int64) falls through
|
|
523
|
+
to the next shape instead of zeroing the latency or raising."""
|
|
524
|
+
ms = _int_in_bounds(payload.get("response_time_ms"))
|
|
525
|
+
if ms is not None:
|
|
526
|
+
return ms
|
|
527
|
+
seconds = payload.get("response_time")
|
|
528
|
+
if isinstance(seconds, (int, float)) and not isinstance(seconds, bool):
|
|
529
|
+
ms = _int_in_bounds(seconds * 1000)
|
|
530
|
+
if ms is not None:
|
|
531
|
+
return ms
|
|
532
|
+
start, end = payload.get("startTime"), payload.get("endTime") # unix seconds
|
|
533
|
+
if (
|
|
534
|
+
isinstance(start, (int, float))
|
|
535
|
+
and isinstance(end, (int, float))
|
|
536
|
+
and not isinstance(start, bool)
|
|
537
|
+
and not isinstance(end, bool)
|
|
538
|
+
):
|
|
539
|
+
ms = _int_in_bounds((end - start) * 1000)
|
|
540
|
+
if ms is not None:
|
|
541
|
+
return ms
|
|
542
|
+
return None
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
# Only fully-implemented adapters are registered: an enum provider with no
|
|
546
|
+
# entry here (portkey, helicone, unknown) gets a clean 400 from the ingest
|
|
547
|
+
# route. A new provider is one adapter class + an entry; the contract is the
|
|
548
|
+
# route's call shape — normalize(payload, *, session_id, user_id, org_id).
|
|
549
|
+
ADAPTERS = {
|
|
550
|
+
GatewayProvider.LITELLM: LiteLLMAdapter(),
|
|
551
|
+
}
|