sediment-capture 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,48 @@
1
+ # SPDX-License-Identifier: AGPL-3.0-or-later
2
+ """Sediment capture: translators that turn provider payloads into facts."""
3
+
4
+ from .gateway import ADAPTERS, LiteLLMAdapter
5
+ from .github import (
6
+ PullRequestRevisionSkipReason,
7
+ RepositoryCaptureSkipReason,
8
+ parse_repository_rename,
9
+ parse_pull_request_merge,
10
+ parse_pull_request_revision,
11
+ parse_push,
12
+ parse_workflow_run,
13
+ sign_payload,
14
+ verify_signature,
15
+ )
16
+ from .otlp import (
17
+ OTLPCaptureResult,
18
+ parse_otlp_logs,
19
+ parse_otlp_decisions,
20
+ parse_otlp_edit_observations,
21
+ parse_otlp_rejected_edits,
22
+ parse_otlp_retry_linkages,
23
+ RetryLinkageSkipReason,
24
+ )
25
+ from .session_identity import SessionIdentity, resolve_identity
26
+
27
+ __all__ = [
28
+ "ADAPTERS",
29
+ "LiteLLMAdapter",
30
+ "SessionIdentity",
31
+ "PullRequestRevisionSkipReason",
32
+ "RepositoryCaptureSkipReason",
33
+ "parse_repository_rename",
34
+ "OTLPCaptureResult",
35
+ "parse_otlp_logs",
36
+ "parse_otlp_decisions",
37
+ "parse_otlp_edit_observations",
38
+ "parse_otlp_rejected_edits",
39
+ "parse_otlp_retry_linkages",
40
+ "RetryLinkageSkipReason",
41
+ "parse_push",
42
+ "parse_pull_request_merge",
43
+ "parse_pull_request_revision",
44
+ "parse_workflow_run",
45
+ "resolve_identity",
46
+ "sign_payload",
47
+ "verify_signature",
48
+ ]
@@ -0,0 +1,551 @@
1
+ # SPDX-License-Identifier: AGPL-3.0-or-later
2
+ """
3
+ Gateway adapters: provider payloads into structured inference-call facts.
4
+
5
+ Pure payload→fact translation — no I/O, no storage, no attribution at
6
+ ingest (ADR 0001). Identity (session/user/org) arrives as parameters: the
7
+ ingest route binds ``org_id`` to the credential and the callback
8
+ supplies real session/user ids (ADR 0002 — no placeholders here). The
9
+ ``ADAPTERS`` registry lives in the same module as the adapters it registers.
10
+
11
+ Fail-soft posture throughout (the same contract as ``github.py``): a
12
+ malformed payload must degrade, never raise — the ingest route would turn
13
+ an exception into a 500 and the inference call would be lost. The original
14
+ payload always survives on ``raw``.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ import logging
21
+ import math
22
+ from datetime import datetime
23
+ from typing import Any
24
+ from uuid import UUID, uuid5
25
+
26
+ from sediment_core import (
27
+ GatewayProvider,
28
+ InferenceCall,
29
+ InferenceMessage,
30
+ ReasoningPart,
31
+ TextPart,
32
+ ToolCallPart,
33
+ ToolCallResponsePart,
34
+ normalize_org_id,
35
+ )
36
+
37
+ logger = logging.getLogger("sediment.capture.gateway")
38
+
39
+ # PostgreSQL BIGINT is a signed 64-bit integer. Larger values fail at INSERT,
40
+ # and int(NaN)/int(inf) raise here in the adapter — bound once so
41
+ # a crafted numeric degrades instead of crashing (same guard as github.py's
42
+ # pr_number).
43
+ _INT64_MAX = 2**63 - 1
44
+
45
+ # Dedicated namespace for callback-prepared capture identities (ADR 0017).
46
+ _CAPTURE_NAMESPACE = UUID("f8bfd6e0-f182-4bcd-a1f8-85a987a8c6e2")
47
+
48
+
49
+ def _as_dict(value: Any) -> dict[str, Any]:
50
+ """Gateway payloads are untrusted — coerce a missing/non-dict field to {}."""
51
+ return value if isinstance(value, dict) else {}
52
+
53
+
54
+ def _as_list(value: Any) -> list[Any]:
55
+ """Untrusted sibling of ``_as_dict`` for list-shaped fields."""
56
+ return value if isinstance(value, list) else []
57
+
58
+
59
+ def _int_in_bounds(value: Any) -> int | None:
60
+ """Coerce a numeric to the fact store's int64 range; fail soft otherwise.
61
+
62
+ Negatives are out of bounds too: every consumer here is a count or a
63
+ latency, ge=0 at the schema — a crafted negative must degrade to the
64
+ fallback, not raise at InferenceCall construction.
65
+ """
66
+ if isinstance(value, bool) or not isinstance(value, (int, float)):
67
+ return None
68
+ if isinstance(value, float) and not math.isfinite(value):
69
+ return None
70
+ n = int(value)
71
+ if not 0 <= n <= _INT64_MAX:
72
+ return None
73
+ return n
74
+
75
+
76
+ def _first_int(*values: Any) -> int | None:
77
+ """Return the first storable int value, else ``None``."""
78
+ for value in values:
79
+ n = _int_in_bounds(value)
80
+ if n is not None:
81
+ return n
82
+ return None
83
+
84
+
85
+ def _text(value: Any) -> str:
86
+ """Best-effort text for a message field. Strings pass
87
+ through; non-string shapes (e.g. multimodal content parts) degrade to
88
+ their JSON form rather than raising during message-part validation."""
89
+ if isinstance(value, str):
90
+ return value
91
+ if value is None:
92
+ return ""
93
+ try:
94
+ return json.dumps(value)
95
+ except (TypeError, ValueError):
96
+ return str(value)
97
+
98
+
99
+ def _tool_call_input(arguments: Any) -> dict[str, Any] | None:
100
+ """The parsed argument object, or None when malformed.
101
+
102
+ OpenAI-shape ``arguments`` is a JSON string; Anthropic-shape ``input`` is
103
+ already an object. Anything that fails to parse as a JSON *object* —
104
+ invalid JSON, or valid JSON that isn't a dict — is malformed: the caller
105
+ keeps the tool-call part with ``arguments={}`` (the id is join-critical
106
+ and must survive a cosmetic defect) and counts the occurrence. Never
107
+ wrap the raw string into input — the unparsed original is already on
108
+ ``InferenceCall.raw`` for storage-seam Basic redaction; normalized fields
109
+ carry normalized data or nothing.
110
+ """
111
+ if isinstance(arguments, str):
112
+ try:
113
+ arguments = json.loads(arguments)
114
+ except (TypeError, ValueError):
115
+ return None
116
+ return arguments if isinstance(arguments, dict) else None
117
+
118
+
119
+ def _usable_id(value: Any) -> bool:
120
+ """A tool-call id that can join something: a non-blank string."""
121
+ return isinstance(value, str) and bool(value.strip())
122
+
123
+
124
+ def _optional_text(value: Any) -> str | None:
125
+ if not isinstance(value, str):
126
+ return None
127
+ value = value.strip()
128
+ return value or None
129
+
130
+
131
+ def _decoded_json(value: Any) -> Any:
132
+ if not isinstance(value, str):
133
+ return value
134
+ try:
135
+ return json.loads(value)
136
+ except (TypeError, ValueError):
137
+ return value
138
+
139
+
140
+ def _tool_part(
141
+ call_id: Any,
142
+ name: Any,
143
+ arguments: Any,
144
+ degraded: dict[str, int],
145
+ ) -> ToolCallPart | None:
146
+ if not _usable_id(call_id):
147
+ degraded["missing_id"] += 1
148
+ return None
149
+ parsed = _tool_call_input(arguments)
150
+ if parsed is None:
151
+ degraded["malformed_arguments"] += 1
152
+ parsed = {}
153
+ return ToolCallPart(id=call_id, name=_text(name), arguments=parsed)
154
+
155
+
156
+ def _reasoning_part(content: Any, degraded: dict[str, int]) -> ReasoningPart | None:
157
+ if isinstance(content, str) and content.strip():
158
+ return ReasoningPart(content=content)
159
+ degraded["opaque_reasoning"] += 1
160
+ return None
161
+
162
+
163
+ def _discriminator_declined(value: Any, *, position: int, field: str) -> None:
164
+ logger.warning(
165
+ "gateway_discriminator_declined",
166
+ extra={
167
+ "source": "litellm",
168
+ "record_position": position,
169
+ "field": field,
170
+ "reason": "unsupported_discriminator"
171
+ if isinstance(value, str)
172
+ else "malformed_discriminator",
173
+ },
174
+ )
175
+
176
+
177
+ def _message_part(
178
+ block: Any,
179
+ degraded: dict[str, int],
180
+ *,
181
+ position: int,
182
+ ):
183
+ if block is None:
184
+ return None
185
+ if isinstance(block, str):
186
+ return TextPart(content=block)
187
+ if not isinstance(block, dict):
188
+ degraded["unsupported_content"] += 1
189
+ return None
190
+ part_type = block.get("type")
191
+ if not isinstance(part_type, str):
192
+ _discriminator_declined(part_type, position=position, field="content.type")
193
+ degraded["unsupported_content"] += 1
194
+ return None
195
+ if part_type in {"thinking", "reasoning"}:
196
+ content = block.get("thinking", block.get("text", block.get("content")))
197
+ return _reasoning_part(content, degraded)
198
+ if part_type == "redacted_thinking":
199
+ degraded["opaque_reasoning"] += 1
200
+ return None
201
+ if part_type in {"text", "input_text", "output_text"}:
202
+ text = block.get("text", block.get("content"))
203
+ if isinstance(text, str):
204
+ return TextPart(content=text)
205
+ degraded["unsupported_content"] += 1
206
+ return None
207
+ elif part_type in {"tool_use", "tool_call"}:
208
+ return _tool_part(
209
+ block.get("id"),
210
+ block.get("name"),
211
+ block.get("input", block.get("arguments")),
212
+ degraded,
213
+ )
214
+ elif part_type in {"tool_result", "tool_call_response"}:
215
+ call_id = block.get("tool_use_id", block.get("id"))
216
+ if not _usable_id(call_id):
217
+ degraded["missing_id"] += 1
218
+ return None
219
+ result = block.get("result", block.get("content"))
220
+ return ToolCallResponsePart(id=call_id, result=_decoded_json(result))
221
+ _discriminator_declined(part_type, position=position, field="content.type")
222
+ degraded["unsupported_content"] += 1
223
+ return None
224
+
225
+
226
+ def _message_reasoning_parts(
227
+ item: dict[str, Any], degraded: dict[str, int], *, position: int
228
+ ) -> list[ReasoningPart]:
229
+ thinking_blocks = item.get("thinking_blocks")
230
+ if thinking_blocks is not None and not isinstance(thinking_blocks, list):
231
+ degraded["unsupported_content"] += 1
232
+ thinking_blocks = None
233
+ if thinking_blocks is None:
234
+ provider_thinking_blocks = _as_dict(item.get("provider_specific_fields")).get(
235
+ "thinking_blocks"
236
+ )
237
+ if provider_thinking_blocks is not None and not isinstance(
238
+ provider_thinking_blocks, list
239
+ ):
240
+ degraded["unsupported_content"] += 1
241
+ else:
242
+ thinking_blocks = provider_thinking_blocks
243
+
244
+ parts = []
245
+ if isinstance(thinking_blocks, list):
246
+ for block in thinking_blocks:
247
+ if not isinstance(block, dict):
248
+ degraded["unsupported_content"] += 1
249
+ continue
250
+ part_type = block.get("type")
251
+ if not isinstance(part_type, str) or part_type not in {
252
+ "thinking",
253
+ "reasoning",
254
+ "redacted_thinking",
255
+ }:
256
+ _discriminator_declined(
257
+ part_type, position=position, field="thinking_blocks.type"
258
+ )
259
+ degraded["unsupported_content"] += 1
260
+ continue
261
+ part = _message_part(block, degraded, position=position)
262
+ if isinstance(part, ReasoningPart):
263
+ parts.append(part)
264
+ if parts:
265
+ return parts
266
+
267
+ reasoning_content = item.get("reasoning_content")
268
+ if reasoning_content is not None:
269
+ part = _reasoning_part(reasoning_content, degraded)
270
+ return [part] if part is not None else []
271
+ return []
272
+
273
+
274
+ def _inference_message(
275
+ item: dict[str, Any],
276
+ degraded: dict[str, int],
277
+ *,
278
+ finish_reason: Any = None,
279
+ position: int,
280
+ ) -> InferenceMessage | None:
281
+ item_type = item.get("type")
282
+ if item_type is not None and (
283
+ not isinstance(item_type, str)
284
+ or item_type not in {"message", "function_call", "function_call_output"}
285
+ ):
286
+ _discriminator_declined(item_type, position=position, field="type")
287
+ degraded["unsupported_content"] += 1
288
+ return None
289
+ if item_type == "function_call":
290
+ call_id = item.get("call_id")
291
+ if not _usable_id(call_id):
292
+ call_id = item.get("id")
293
+ part = _tool_part(call_id, item.get("name"), item.get("arguments"), degraded)
294
+ return InferenceMessage(role="assistant", parts=[part]) if part else None
295
+ if item_type == "function_call_output":
296
+ call_id = item.get("call_id")
297
+ if not _usable_id(call_id):
298
+ degraded["missing_id"] += 1
299
+ return None
300
+ return InferenceMessage(
301
+ role="tool",
302
+ parts=[
303
+ ToolCallResponsePart(
304
+ id=call_id, result=_decoded_json(item.get("output"))
305
+ )
306
+ ],
307
+ )
308
+
309
+ role = _optional_text(item.get("role"))
310
+ if role is None:
311
+ degraded["unsupported_content"] += 1
312
+ return None
313
+
314
+ content = item.get("content")
315
+ if role == "tool" and _usable_id(item.get("tool_call_id")):
316
+ parts = [
317
+ ToolCallResponsePart(id=item["tool_call_id"], result=_decoded_json(content))
318
+ ]
319
+ else:
320
+ blocks = content if isinstance(content, list) else [content]
321
+ content_parts = [
322
+ part
323
+ for block in blocks
324
+ if (part := _message_part(block, degraded, position=position)) is not None
325
+ ]
326
+ sibling_reasoning = (
327
+ _message_reasoning_parts(item, degraded, position=position)
328
+ if role == "assistant"
329
+ else []
330
+ )
331
+ parts = (
332
+ sibling_reasoning
333
+ if not any(isinstance(part, ReasoningPart) for part in content_parts)
334
+ else []
335
+ )
336
+ parts.extend(content_parts)
337
+ for call in _as_list(item.get("tool_calls")):
338
+ if not isinstance(call, dict):
339
+ degraded["unsupported_content"] += 1
340
+ continue
341
+ call_type = call.get("type")
342
+ if call_type is not None and (
343
+ not isinstance(call_type, str) or call_type != "function"
344
+ ):
345
+ _discriminator_declined(
346
+ call_type, position=position, field="tool_calls.type"
347
+ )
348
+ degraded["unsupported_content"] += 1
349
+ continue
350
+ function = _as_dict(call.get("function"))
351
+ part = _tool_part(
352
+ call.get("id"),
353
+ function.get("name"),
354
+ function.get("arguments"),
355
+ degraded,
356
+ )
357
+ if part is not None:
358
+ parts.append(part)
359
+
360
+ finish = _optional_text(finish_reason)
361
+ return InferenceMessage(role=role, parts=parts, finish_reason=finish)
362
+
363
+
364
+ def _messages(items: list[Any], *, finish_reason: Any = None) -> list[InferenceMessage]:
365
+ degraded = {
366
+ "malformed_arguments": 0,
367
+ "missing_id": 0,
368
+ "opaque_reasoning": 0,
369
+ "unsupported_content": 0,
370
+ }
371
+ messages = []
372
+ for position, item in enumerate(items):
373
+ if not isinstance(item, dict):
374
+ degraded["unsupported_content"] += 1
375
+ continue
376
+ message = _inference_message(
377
+ item, degraded, finish_reason=finish_reason, position=position
378
+ )
379
+ if message is not None:
380
+ messages.append(message)
381
+ if any(degraded.values()):
382
+ level = (
383
+ logging.WARNING
384
+ if any(
385
+ value for key, value in degraded.items() if key != "opaque_reasoning"
386
+ )
387
+ else logging.INFO
388
+ )
389
+ logger.log(
390
+ level,
391
+ "inference_messages_degraded malformed_arguments=%d missing_id=%d "
392
+ "unsupported_content=%d opaque_reasoning=%d kept=%d",
393
+ degraded["malformed_arguments"],
394
+ degraded["missing_id"],
395
+ degraded["unsupported_content"],
396
+ degraded["opaque_reasoning"],
397
+ len(messages),
398
+ )
399
+ return messages
400
+
401
+
402
+ class LiteLLMAdapter:
403
+ """Normalizes LiteLLM inference-call payloads.
404
+
405
+ Handles the real proxy ``StandardLoggingPayload`` (what the custom
406
+ callback forwards in production — see ``litellm/sediment_callback.py``)
407
+ as well as the shaped payload the same callback's ``_fallback_payload``
408
+ builds when no SLO is present (also the shape of the hand-written unit
409
+ fixture). The two differ in where tokens and timing live:
410
+
411
+ - tokens: top-level ``usage`` (shaped) vs ``response.usage`` / top-level
412
+ ``prompt_tokens``/``completion_tokens`` (StandardLoggingPayload)
413
+ - timing: ``response_time_ms`` (shaped) vs ``response_time`` seconds or
414
+ ``endTime`` - ``startTime`` unix seconds (StandardLoggingPayload)
415
+
416
+ https://docs.litellm.ai/docs/proxy/logging
417
+ """
418
+
419
+ def normalize(
420
+ self,
421
+ payload: dict[str, Any],
422
+ *,
423
+ session_id: str,
424
+ user_id: str | None,
425
+ org_id: str,
426
+ capture_id: UUID | None = None,
427
+ observed_at: datetime | None = None,
428
+ ) -> InferenceCall:
429
+ if (capture_id is None) != (observed_at is None):
430
+ raise ValueError("capture_id and observed_at must be supplied together")
431
+ capture_fields: dict[str, Any] = {}
432
+ if capture_id is not None:
433
+ capture_fields = {
434
+ "inference_call_id": str(
435
+ uuid5(
436
+ _CAPTURE_NAMESPACE,
437
+ json.dumps(
438
+ [normalize_org_id(org_id), str(UUID(str(capture_id)))],
439
+ separators=(",", ":"),
440
+ ),
441
+ )
442
+ ),
443
+ "observed_at": observed_at,
444
+ }
445
+ raw_response = payload.get("response")
446
+ response_text = raw_response if isinstance(raw_response, str) else ""
447
+ response = _as_dict(raw_response)
448
+ choices = _as_list(response.get("choices"))
449
+ first_choice = _as_dict(choices[0]) if choices else {}
450
+ message = _as_dict(first_choice.get("message"))
451
+
452
+ input_messages = _messages(_as_list(payload.get("messages")))
453
+ if message:
454
+ output_messages = _messages(
455
+ [message], finish_reason=first_choice.get("finish_reason")
456
+ )
457
+ elif isinstance(first_choice.get("text"), str):
458
+ output_messages = [
459
+ InferenceMessage(
460
+ role="assistant",
461
+ parts=[TextPart(content=first_choice["text"])],
462
+ finish_reason=_optional_text(first_choice.get("finish_reason")),
463
+ )
464
+ ]
465
+ elif _as_list(response.get("output")):
466
+ output_messages = _messages(_as_list(response.get("output")))
467
+ elif response_text:
468
+ output_messages = [
469
+ InferenceMessage(
470
+ role="assistant", parts=[TextPart(content=response_text)]
471
+ )
472
+ ]
473
+ else:
474
+ output_messages = []
475
+
476
+ # Tokens may be under a top-level ``usage`` (shaped) or under
477
+ # ``response.usage`` / top-level keys (StandardLoggingPayload). A
478
+ # malformed or empty top-level usage falls through to response.usage.
479
+ usage = _as_dict(payload.get("usage")) or _as_dict(response.get("usage"))
480
+
481
+ # The proxy-assigned call id — the idempotency identity for a
482
+ # redelivered/retried callback POST. Fall back to the response's own
483
+ # completion id; a payload with neither stays keyless — model_call_id
484
+ # None means no dedup key under uq_inference_calls_model_call.
485
+ # _usable_id, not truthiness: model_call_id is the dedup key, and a
486
+ # truthy non-string (True → "True") or a blank string would be a
487
+ # shared key that collapses every later call as a redelivery.
488
+ model_call_id = payload.get("litellm_call_id")
489
+ if not _usable_id(model_call_id):
490
+ model_call_id = response.get("id")
491
+ if not _usable_id(model_call_id):
492
+ model_call_id = None
493
+
494
+ model = _optional_text(payload.get("model")) or _optional_text(
495
+ response.get("model")
496
+ )
497
+ return InferenceCall(
498
+ session_id=session_id,
499
+ user_id=user_id,
500
+ org_id=org_id,
501
+ gateway_provider=GatewayProvider.LITELLM,
502
+ model_provider=_optional_text(payload.get("custom_llm_provider")),
503
+ model=model,
504
+ input_messages=input_messages,
505
+ output_messages=output_messages,
506
+ model_call_id=model_call_id,
507
+ input_tokens=_first_int(
508
+ usage.get("prompt_tokens"), payload.get("prompt_tokens")
509
+ ),
510
+ output_tokens=_first_int(
511
+ usage.get("completion_tokens"), payload.get("completion_tokens")
512
+ ),
513
+ duration_ms=self._latency_ms(payload),
514
+ raw=payload,
515
+ **capture_fields,
516
+ )
517
+
518
+ @staticmethod
519
+ def _latency_ms(payload: dict[str, Any]) -> int | None:
520
+ """Resolve latency in ms across the shapes LiteLLM emits. Each
521
+ branch is taken only when its value is a storable numeric, so a
522
+ malformed field (non-numeric, NaN/inf, beyond int64) falls through
523
+ to the next shape instead of zeroing the latency or raising."""
524
+ ms = _int_in_bounds(payload.get("response_time_ms"))
525
+ if ms is not None:
526
+ return ms
527
+ seconds = payload.get("response_time")
528
+ if isinstance(seconds, (int, float)) and not isinstance(seconds, bool):
529
+ ms = _int_in_bounds(seconds * 1000)
530
+ if ms is not None:
531
+ return ms
532
+ start, end = payload.get("startTime"), payload.get("endTime") # unix seconds
533
+ if (
534
+ isinstance(start, (int, float))
535
+ and isinstance(end, (int, float))
536
+ and not isinstance(start, bool)
537
+ and not isinstance(end, bool)
538
+ ):
539
+ ms = _int_in_bounds((end - start) * 1000)
540
+ if ms is not None:
541
+ return ms
542
+ return None
543
+
544
+
545
+ # Only fully-implemented adapters are registered: an enum provider with no
546
+ # entry here (portkey, helicone, unknown) gets a clean 400 from the ingest
547
+ # route. A new provider is one adapter class + an entry; the contract is the
548
+ # route's call shape — normalize(payload, *, session_id, user_id, org_id).
549
+ ADAPTERS = {
550
+ GatewayProvider.LITELLM: LiteLLMAdapter(),
551
+ }