audr 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
audr/__init__.py ADDED
@@ -0,0 +1,90 @@
1
+ """The audr public API: build, validate, and deliver AUDR usage records."""
2
+
3
+ from audr._version import __version__
4
+ from audr.client import Client
5
+ from audr.errors import (
6
+ AudrError,
7
+ ConfigurationError,
8
+ LifecycleError,
9
+ ValidationError,
10
+ ValidationIssue,
11
+ )
12
+ from audr.record import (
13
+ AUDR,
14
+ SPEC_VERSION,
15
+ Attribution,
16
+ Cost,
17
+ Emitter,
18
+ EmitterComponent,
19
+ Environment,
20
+ ErrorCode,
21
+ LlmCost,
22
+ LlmUsage,
23
+ Modality,
24
+ Operation,
25
+ Resource,
26
+ ResourceType,
27
+ Run,
28
+ RunOutcome,
29
+ RunType,
30
+ Timing,
31
+ ToolCost,
32
+ ToolUsage,
33
+ Usage,
34
+ )
35
+ from audr.results import (
36
+ DeliveredCallback,
37
+ DeliveryStats,
38
+ Disposition,
39
+ FailedRecord,
40
+ FailureCallback,
41
+ FailureReason,
42
+ SubmitOutcome,
43
+ SubmitResult,
44
+ )
45
+ from audr.sinks import BatchOutcome, BatchResult, FileFormat, FileSink, RejectedRecord, Sink
46
+
47
+ __all__ = [
48
+ "AUDR",
49
+ "SPEC_VERSION",
50
+ "Attribution",
51
+ "AudrError",
52
+ "BatchOutcome",
53
+ "BatchResult",
54
+ "Client",
55
+ "ConfigurationError",
56
+ "Cost",
57
+ "DeliveredCallback",
58
+ "DeliveryStats",
59
+ "Disposition",
60
+ "Emitter",
61
+ "EmitterComponent",
62
+ "Environment",
63
+ "ErrorCode",
64
+ "FailedRecord",
65
+ "FailureCallback",
66
+ "FailureReason",
67
+ "FileFormat",
68
+ "FileSink",
69
+ "LifecycleError",
70
+ "LlmCost",
71
+ "LlmUsage",
72
+ "Modality",
73
+ "Operation",
74
+ "RejectedRecord",
75
+ "Resource",
76
+ "ResourceType",
77
+ "Run",
78
+ "RunOutcome",
79
+ "RunType",
80
+ "Sink",
81
+ "SubmitOutcome",
82
+ "SubmitResult",
83
+ "Timing",
84
+ "ToolCost",
85
+ "ToolUsage",
86
+ "Usage",
87
+ "ValidationError",
88
+ "ValidationIssue",
89
+ "__version__",
90
+ ]
audr/_pipeline.py ADDED
@@ -0,0 +1,462 @@
1
+ """Asynchronous, bounded batching pipeline for AUDR record delivery."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import logging
7
+ from dataclasses import dataclass
8
+ from enum import Enum, auto
9
+ from typing import assert_never
10
+
11
+ from audr.record import AUDR
12
+ from audr.results import (
13
+ DeliveredCallback,
14
+ DeliveryStats,
15
+ Disposition,
16
+ FailedRecord,
17
+ FailureCallback,
18
+ FailureReason,
19
+ SubmitOutcome,
20
+ SubmitResult,
21
+ )
22
+ from audr.sinks import BatchOutcome, BatchResult, Sink
23
+
24
+ _LOGGER = logging.getLogger("audr.pipeline")
25
+
26
+ _DEFAULT_DRAIN_TIMEOUT_SECONDS = 30.0
27
+
28
+
29
+ class _RecordState(Enum):
30
+ """Terminal accounting state for a record removed from the queue."""
31
+
32
+ IN_FLIGHT = auto()
33
+ SENT = auto()
34
+ DROPPED = auto()
35
+ UNKNOWN = auto()
36
+
37
+
38
+ @dataclass(slots=True)
39
+ class _QueuedRecord:
40
+ """In-flight wrapper with one monotonic terminal accounting transition."""
41
+
42
+ record: AUDR
43
+ state: _RecordState = _RecordState.IN_FLIGHT
44
+
45
+
46
+ class Pipeline:
47
+ """Queue single records and deliver them in background micro-batches."""
48
+
49
+ def __init__(
50
+ self,
51
+ sink: Sink,
52
+ *,
53
+ max_queue_size: int,
54
+ batch_max_size: int,
55
+ linger_seconds: float,
56
+ owns_sink: bool,
57
+ on_failure: FailureCallback | None,
58
+ on_delivered: DeliveredCallback | None = None,
59
+ ) -> None:
60
+ self._sink = sink
61
+ self._max_queue_size = max_queue_size
62
+ self._batch_max_size = batch_max_size
63
+ self._linger_seconds = linger_seconds
64
+ self._owns_sink = owns_sink
65
+ self._on_failure = on_failure
66
+ self._on_delivered = on_delivered
67
+
68
+ self._queue: asyncio.Queue[AUDR] = asyncio.Queue()
69
+ self._sink_lock = asyncio.Lock()
70
+ self._worker_task: asyncio.Task[None] | None = None
71
+ self._in_flight_records: dict[int, _QueuedRecord] = {}
72
+ self._in_flight = 0
73
+ self._running = False
74
+ self._draining = asyncio.Event()
75
+ self._stopped = False
76
+ self._submitted = 0
77
+ self._sent = 0
78
+ self._dropped = 0
79
+ self._unknown = 0
80
+ self._batches = 0
81
+
82
+ @property
83
+ def stats(self) -> DeliveryStats:
84
+ """Return an immutable snapshot of delivery counters."""
85
+ return DeliveryStats(
86
+ submitted=self._submitted,
87
+ sent=self._sent,
88
+ dropped=self._dropped,
89
+ unknown=self._unknown,
90
+ batches=self._batches,
91
+ queue_depth=self._current_depth(),
92
+ queue_capacity=self._max_queue_size,
93
+ )
94
+
95
+ @property
96
+ def running(self) -> bool:
97
+ """Return whether the worker is accepting submissions."""
98
+ return self._running
99
+
100
+ def start(self) -> None:
101
+ """Start the delivery worker from inside the host application's event loop."""
102
+ if self._running or self._stopped:
103
+ return
104
+ loop = asyncio.get_running_loop()
105
+ self._running = True
106
+ self._draining.clear()
107
+ self._worker_task = loop.create_task(self._worker(), name="audr-delivery")
108
+
109
+ def submit(self, record: AUDR) -> SubmitResult:
110
+ """Queue one record and return immediately; the worker performs all I/O."""
111
+ self._submitted += 1
112
+ if not self._running:
113
+ self._dropped += 1
114
+ self._notify_failure(
115
+ record,
116
+ disposition=Disposition.DROPPED,
117
+ reason=FailureReason.NOT_RUNNING,
118
+ retryable=True,
119
+ )
120
+ return SubmitResult(SubmitOutcome.DROPPED_NOT_RUNNING)
121
+
122
+ depth = self._current_depth()
123
+ if depth >= self._max_queue_size:
124
+ self._dropped += 1
125
+ _LOGGER.warning("record dropped: queue full (depth=%s)", depth)
126
+ self._notify_failure(
127
+ record,
128
+ disposition=Disposition.DROPPED,
129
+ reason=FailureReason.QUEUE_FULL,
130
+ retryable=True,
131
+ )
132
+ return SubmitResult(SubmitOutcome.DROPPED_QUEUE_FULL)
133
+
134
+ self._queue.put_nowait(record)
135
+ return SubmitResult(SubmitOutcome.QUEUED)
136
+
137
+ async def flush(self, timeout: float | None = None) -> bool:
138
+ """Deliver queued and in-flight work now, waiting up to `timeout` seconds.
139
+
140
+ Returns True when everything queued at the time of the call has reached a
141
+ terminal state. Returns False when the bound expired first; the remaining
142
+ records stay queued and continue to be delivered in the background.
143
+ """
144
+ if not self._running:
145
+ return True
146
+ return await self._drain(
147
+ _DEFAULT_DRAIN_TIMEOUT_SECONDS if timeout is None else timeout,
148
+ abandon_on_timeout=False,
149
+ )
150
+
151
+ async def stop(self, timeout: float | None = None) -> None:
152
+ """Stop accepting records, drain bounded work, then close the sink."""
153
+ if self._stopped:
154
+ return
155
+ self._running = False
156
+ self._stopped = True
157
+ try:
158
+ await self._drain(
159
+ _DEFAULT_DRAIN_TIMEOUT_SECONDS if timeout is None else timeout,
160
+ abandon_on_timeout=True,
161
+ )
162
+ finally:
163
+ await self._shutdown()
164
+
165
+ async def _worker(self) -> None:
166
+ while True:
167
+ batch = await self._collect_batch()
168
+ try:
169
+ await self._deliver(batch)
170
+ finally:
171
+ for _ in batch:
172
+ self._queue.task_done()
173
+
174
+ async def _collect_batch(self) -> list[_QueuedRecord]:
175
+ """Block for one record, then fill until count, linger, or drain."""
176
+ first_record = await self._queue.get()
177
+ first = self._begin_in_flight(first_record)
178
+ batch = [first]
179
+ deadline = asyncio.get_running_loop().time() + self._linger_seconds
180
+ while len(batch) < self._batch_max_size:
181
+ next_record: AUDR | None
182
+ if self._draining.is_set():
183
+ try:
184
+ next_record = self._queue.get_nowait()
185
+ except asyncio.QueueEmpty:
186
+ break
187
+ else:
188
+ next_record = await self._next_within_linger(deadline)
189
+ if next_record is None:
190
+ break
191
+ batch.append(self._begin_in_flight(next_record))
192
+ return batch
193
+
194
+ async def _next_within_linger(self, deadline: float) -> AUDR | None:
195
+ remaining = deadline - asyncio.get_running_loop().time()
196
+ if remaining <= 0:
197
+ return None
198
+ get_task: asyncio.Task[AUDR] = asyncio.ensure_future(self._queue.get())
199
+ drain_task: asyncio.Task[bool] = asyncio.ensure_future(self._draining.wait())
200
+ try:
201
+ done, _ = await asyncio.wait(
202
+ {get_task, drain_task},
203
+ timeout=remaining,
204
+ return_when=asyncio.FIRST_COMPLETED,
205
+ )
206
+ if get_task in done:
207
+ return get_task.result()
208
+ return None
209
+ finally:
210
+ for task in (get_task, drain_task):
211
+ if not task.done():
212
+ task.cancel()
213
+ await asyncio.gather(get_task, drain_task, return_exceptions=True)
214
+
215
+ async def _deliver(self, batch: list[_QueuedRecord]) -> None:
216
+ try:
217
+ records = tuple(queued.record for queued in batch)
218
+ try:
219
+ async with self._sink_lock:
220
+ result = await self._sink.deliver(records)
221
+ except asyncio.CancelledError:
222
+ for queued in batch:
223
+ self._mark_unknown(
224
+ queued,
225
+ "record outcome unknown: delivery cancelled",
226
+ reason=FailureReason.INCONCLUSIVE,
227
+ )
228
+ raise
229
+ except Exception:
230
+ _LOGGER.exception(
231
+ "batch outcome unknown after unexpected sink exception (size=%s)",
232
+ len(batch),
233
+ )
234
+ for queued in batch:
235
+ self._mark_unknown(
236
+ queued,
237
+ "record outcome unknown: sink exception",
238
+ reason=FailureReason.INCONCLUSIVE,
239
+ )
240
+ return
241
+
242
+ self._batches += 1
243
+ match result.outcome:
244
+ case BatchOutcome.ACCEPTED:
245
+ self._apply_accepted_result(batch, result)
246
+ self._notify_delivered(batch)
247
+ return
248
+ case BatchOutcome.RETRYABLE_FAILURE:
249
+ _LOGGER.warning(
250
+ "batch dropped after retryable sink failure (size=%s)", len(batch)
251
+ )
252
+ drop_reason = FailureReason.SINK_FAILURE
253
+ retryable = True
254
+ case BatchOutcome.PERMANENT_FAILURE:
255
+ _LOGGER.warning(
256
+ "batch dropped after permanent sink failure (size=%s)", len(batch)
257
+ )
258
+ drop_reason = FailureReason.SINK_FAILURE
259
+ retryable = False
260
+ case BatchOutcome.CLOSED:
261
+ _LOGGER.error("batch dropped because the sink is closed (size=%s)", len(batch))
262
+ drop_reason = FailureReason.SINK_CLOSED
263
+ retryable = True
264
+ case unreachable:
265
+ assert_never(unreachable)
266
+ for queued in batch:
267
+ self._drop(queued, reason=drop_reason, retryable=retryable)
268
+ finally:
269
+ for queued in batch:
270
+ if queued.state is _RecordState.IN_FLIGHT:
271
+ self._mark_unknown(
272
+ queued,
273
+ "record outcome unknown: incomplete delivery result",
274
+ reason=FailureReason.INCONCLUSIVE,
275
+ )
276
+ self._release_in_flight(queued)
277
+
278
+ def _apply_accepted_result(
279
+ self,
280
+ batch: list[_QueuedRecord],
281
+ result: BatchResult,
282
+ ) -> None:
283
+ rejected_by_id = {rejection.record_id: rejection for rejection in result.rejected}
284
+ unknown_ids = set(result.unknown)
285
+ for queued in batch:
286
+ identifier = queued.record.record_id
287
+ if identifier in rejected_by_id:
288
+ rejection = rejected_by_id[identifier]
289
+ self._drop(
290
+ queued,
291
+ reason=FailureReason.REJECTED,
292
+ retryable=False,
293
+ detail=rejection.detail,
294
+ )
295
+ elif identifier in unknown_ids:
296
+ self._mark_unknown(
297
+ queued,
298
+ "record outcome unknown: destination response was inconclusive",
299
+ reason=FailureReason.INCONCLUSIVE,
300
+ )
301
+ else:
302
+ self._mark_sent(queued)
303
+
304
+ async def _drain(self, timeout: float, *, abandon_on_timeout: bool) -> bool:
305
+ self._draining.set()
306
+ try:
307
+ await asyncio.wait_for(self._queue.join(), timeout=timeout)
308
+ except TimeoutError:
309
+ if abandon_on_timeout:
310
+ self._abandon_outstanding()
311
+ return False
312
+ finally:
313
+ if not self._stopped:
314
+ self._draining.clear()
315
+ return True
316
+
317
+ async def _shutdown(self) -> None:
318
+ if self._worker_task is not None:
319
+ self._worker_task.cancel()
320
+ await asyncio.gather(self._worker_task, return_exceptions=True)
321
+ self._worker_task = None
322
+ self._abandon_outstanding()
323
+ if self._owns_sink:
324
+ try:
325
+ await self._sink.close()
326
+ except Exception:
327
+ _LOGGER.exception("failed to close sink during pipeline shutdown")
328
+ _LOGGER.info(
329
+ "delivery stopped: submitted=%s sent=%s dropped=%s unknown=%s batches=%s",
330
+ self._submitted,
331
+ self._sent,
332
+ self._dropped,
333
+ self._unknown,
334
+ self._batches,
335
+ )
336
+
337
+ def _abandon_outstanding(self) -> None:
338
+ while True:
339
+ try:
340
+ record = self._queue.get_nowait()
341
+ except asyncio.QueueEmpty:
342
+ break
343
+ self._queue.task_done()
344
+ self._drop_queued(record)
345
+ for queued in list(self._in_flight_records.values()):
346
+ self._mark_unknown(
347
+ queued,
348
+ "record outcome unknown: shutdown",
349
+ reason=FailureReason.SHUTDOWN,
350
+ )
351
+
352
+ def _drop_queued(self, record: AUDR) -> None:
353
+ self._dropped += 1
354
+ _LOGGER.warning("record dropped: shutdown")
355
+ self._notify_failure(
356
+ record,
357
+ disposition=Disposition.DROPPED,
358
+ reason=FailureReason.SHUTDOWN,
359
+ retryable=True,
360
+ )
361
+
362
+ def _begin_in_flight(self, record: AUDR) -> _QueuedRecord:
363
+ queued = _QueuedRecord(record=record)
364
+ self._in_flight += 1
365
+ self._in_flight_records[id(queued)] = queued
366
+ return queued
367
+
368
+ def _finalize(self, queued: _QueuedRecord, state: _RecordState) -> bool:
369
+ if queued.state is not _RecordState.IN_FLIGHT:
370
+ return False
371
+ queued.state = state
372
+ return True
373
+
374
+ def _release_in_flight(self, queued: _QueuedRecord) -> None:
375
+ if self._in_flight_records.pop(id(queued), None) is not None:
376
+ self._in_flight -= 1
377
+
378
+ def _mark_sent(self, queued: _QueuedRecord) -> None:
379
+ if self._finalize(queued, _RecordState.SENT):
380
+ self._sent += 1
381
+
382
+ def _drop(
383
+ self,
384
+ queued: _QueuedRecord,
385
+ *,
386
+ reason: FailureReason,
387
+ retryable: bool,
388
+ detail: str | None = None,
389
+ ) -> None:
390
+ if self._finalize(queued, _RecordState.DROPPED):
391
+ self._dropped += 1
392
+ _LOGGER.warning("record dropped: reason=%s", reason.value)
393
+ self._notify_failure(
394
+ queued.record,
395
+ disposition=Disposition.DROPPED,
396
+ reason=reason,
397
+ retryable=retryable,
398
+ detail=detail,
399
+ )
400
+
401
+ def _mark_unknown(
402
+ self,
403
+ queued: _QueuedRecord,
404
+ message: str,
405
+ *,
406
+ reason: FailureReason,
407
+ ) -> None:
408
+ if self._finalize(queued, _RecordState.UNKNOWN):
409
+ self._unknown += 1
410
+ _LOGGER.warning("%s", message)
411
+ self._notify_failure(
412
+ queued.record,
413
+ disposition=Disposition.UNKNOWN,
414
+ reason=reason,
415
+ retryable=True,
416
+ )
417
+
418
+ def _notify_failure(
419
+ self,
420
+ record: AUDR,
421
+ *,
422
+ disposition: Disposition,
423
+ reason: FailureReason,
424
+ retryable: bool,
425
+ detail: str | None = None,
426
+ ) -> None:
427
+ if self._on_failure is None:
428
+ return
429
+ failure = FailedRecord(
430
+ record=record,
431
+ disposition=disposition,
432
+ reason=reason,
433
+ retryable=retryable,
434
+ detail=detail,
435
+ )
436
+ try:
437
+ self._on_failure(failure)
438
+ except Exception as error:
439
+ _LOGGER.warning(
440
+ "failure callback raised: %s",
441
+ type(error).__name__,
442
+ )
443
+
444
+ def _notify_delivered(self, batch: list[_QueuedRecord]) -> None:
445
+ if self._on_delivered is None:
446
+ return
447
+ sent_records = tuple(queued.record for queued in batch if queued.state is _RecordState.SENT)
448
+ if not sent_records:
449
+ return
450
+ try:
451
+ self._on_delivered(sent_records)
452
+ except Exception as error:
453
+ _LOGGER.warning(
454
+ "delivered callback raised: %s",
455
+ type(error).__name__,
456
+ )
457
+
458
+ def _current_depth(self) -> int:
459
+ return self._queue.qsize() + self._in_flight
460
+
461
+
462
+ __all__ = ["Pipeline"]
audr/_time.py ADDED
@@ -0,0 +1,20 @@
1
+ from __future__ import annotations
2
+
3
+ from datetime import UTC, datetime
4
+
5
+
6
+ def now_utc() -> datetime:
7
+ """Current time, tz-aware UTC, truncated to millisecond precision."""
8
+ now = datetime.now(UTC)
9
+ return now.replace(microsecond=now.microsecond - now.microsecond % 1000)
10
+
11
+
12
+ def to_rfc3339_ms(value: datetime) -> str:
13
+ if value.tzinfo is None:
14
+ raise ValueError("datetime must be timezone-aware")
15
+ utc = value.astimezone(UTC)
16
+ return utc.strftime("%Y-%m-%dT%H:%M:%S.") + f"{utc.microsecond // 1000:03d}Z"
17
+
18
+
19
+ def parse_rfc3339(text: str) -> datetime:
20
+ return datetime.fromisoformat(text.replace("Z", "+00:00"))
audr/_version.py ADDED
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"