dlogify 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dlogify/__init__.py +160 -0
- dlogify/__main__.py +3 -0
- dlogify/_capture_logging.py +56 -0
- dlogify/_client.py +239 -0
- dlogify/_config.py +194 -0
- dlogify/_diagnostics.py +54 -0
- dlogify/_encode.py +100 -0
- dlogify/_exporter.py +407 -0
- dlogify/_guard.py +29 -0
- dlogify/_hooks.py +95 -0
- dlogify/_http.py +52 -0
- dlogify/_levels.py +88 -0
- dlogify/_parse/__init__.py +0 -0
- dlogify/_parse/compact_json.py +55 -0
- dlogify/_parse/json_line.py +178 -0
- dlogify/_parse/lines.py +64 -0
- dlogify/_parse/stream_parser.py +106 -0
- dlogify/_parse/text_event.py +150 -0
- dlogify/_record.py +149 -0
- dlogify/_singleton.py +12 -0
- dlogify/_version.py +4 -0
- dlogify/handler.py +70 -0
- dlogify/py.typed +0 -0
- dlogify/run/__init__.py +0 -0
- dlogify/run/child.py +215 -0
- dlogify/run/cli.py +22 -0
- dlogify-0.1.0.dist-info/METADATA +111 -0
- dlogify-0.1.0.dist-info/RECORD +31 -0
- dlogify-0.1.0.dist-info/WHEEL +4 -0
- dlogify-0.1.0.dist-info/entry_points.txt +3 -0
- dlogify-0.1.0.dist-info/licenses/LICENSE +21 -0
dlogify/_diagnostics.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""The client's own messages go straight to the stderr stream, never through
|
|
2
|
+
logging or print, so capture never sees them (docs/12 §4, guarantee 3)."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import sys
|
|
7
|
+
import threading
|
|
8
|
+
from typing import Optional
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class Diagnostics:
|
|
12
|
+
def __init__(self, debug_enabled: bool) -> None:
|
|
13
|
+
self.debug_enabled = debug_enabled
|
|
14
|
+
self._warned: set[str] = set()
|
|
15
|
+
self._lock = threading.Lock()
|
|
16
|
+
|
|
17
|
+
def warn(self, message: str) -> None:
|
|
18
|
+
_write(f"logify: warning: {message}\n")
|
|
19
|
+
|
|
20
|
+
def warn_once(self, key: str, message: str) -> None:
|
|
21
|
+
with self._lock:
|
|
22
|
+
if key in self._warned:
|
|
23
|
+
return
|
|
24
|
+
self._warned.add(key)
|
|
25
|
+
self.warn(message)
|
|
26
|
+
|
|
27
|
+
def debug(self, message: str) -> None:
|
|
28
|
+
if self.debug_enabled:
|
|
29
|
+
_write(f"logify: {message}\n")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _write(text: str) -> None:
|
|
33
|
+
stream = sys.__stderr__
|
|
34
|
+
if stream is None:
|
|
35
|
+
return
|
|
36
|
+
try:
|
|
37
|
+
stream.write(text)
|
|
38
|
+
stream.flush()
|
|
39
|
+
except Exception:
|
|
40
|
+
# stderr is closed; there is nowhere left to report to.
|
|
41
|
+
pass
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def is_flag(value: Optional[str]) -> bool:
|
|
45
|
+
return value is not None and value.strip().lower() in ("1", "true")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def describe_error(exc: BaseException) -> str:
|
|
49
|
+
name = type(exc).__name__
|
|
50
|
+
try:
|
|
51
|
+
text = str(exc)
|
|
52
|
+
except Exception:
|
|
53
|
+
text = "<unprintable>"
|
|
54
|
+
return f"{name}: {text}" if text else name
|
dlogify/_encode.py
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""OTLP/JSON encoding (docs/12 §2–3). Tested by clients/conformance/otlp."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import socket
|
|
7
|
+
from collections.abc import Mapping, Sequence
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from typing import Any, Optional
|
|
10
|
+
|
|
11
|
+
from ._levels import severity
|
|
12
|
+
from ._record import LogRecord, normalize_value
|
|
13
|
+
from ._version import SDK_LANGUAGE, SDK_NAME, VERSION
|
|
14
|
+
|
|
15
|
+
_INT64_MIN = -(2**63)
|
|
16
|
+
_INT64_MAX = 2**63 - 1
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class EncodedRecord:
|
|
21
|
+
scope: str
|
|
22
|
+
record: dict[str, Any]
|
|
23
|
+
bytes: int
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def to_json_bytes(value: Any) -> bytes:
|
|
27
|
+
return json.dumps(value, separators=(",", ":"), ensure_ascii=False).encode("utf-8", "replace")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def any_value(raw: object) -> Optional[dict[str, Any]]:
|
|
31
|
+
value = normalize_value(raw)
|
|
32
|
+
if value is None:
|
|
33
|
+
return None
|
|
34
|
+
if isinstance(value, bool):
|
|
35
|
+
return {"boolValue": value}
|
|
36
|
+
if isinstance(value, int):
|
|
37
|
+
number = int(value)
|
|
38
|
+
# OTLP intValue is an int64; a larger Python int keeps its digits as a string.
|
|
39
|
+
return {"intValue": str(number)} if _INT64_MIN <= number <= _INT64_MAX else {"stringValue": str(number)}
|
|
40
|
+
if isinstance(value, float):
|
|
41
|
+
return {"doubleValue": value}
|
|
42
|
+
return {"stringValue": value}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def key_values(attributes: Mapping[str, object]) -> list[dict[str, Any]]:
|
|
46
|
+
out = []
|
|
47
|
+
for key, raw in attributes.items():
|
|
48
|
+
value = any_value(raw)
|
|
49
|
+
if value is not None:
|
|
50
|
+
out.append({"key": key, "value": value})
|
|
51
|
+
return out
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def host_name() -> str:
|
|
55
|
+
try:
|
|
56
|
+
return socket.gethostname()
|
|
57
|
+
except Exception:
|
|
58
|
+
return "unknown"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def resource_attributes(*, service: str, environment: str, release: str, attributes: Mapping[str, object]) -> list[dict[str, Any]]:
|
|
62
|
+
"""Fixed attributes first; init() attributes never replace them."""
|
|
63
|
+
fixed: dict[str, object] = {"service.name": service, "deployment.environment.name": environment}
|
|
64
|
+
if release:
|
|
65
|
+
fixed["service.version"] = release
|
|
66
|
+
fixed["host.name"] = host_name()
|
|
67
|
+
fixed["telemetry.sdk.name"] = SDK_NAME
|
|
68
|
+
fixed["telemetry.sdk.language"] = SDK_LANGUAGE
|
|
69
|
+
fixed["telemetry.sdk.version"] = VERSION
|
|
70
|
+
extra = {key: value for key, value in attributes.items() if key not in fixed}
|
|
71
|
+
return key_values(fixed) + key_values(extra)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def encode_record(record: LogRecord) -> EncodedRecord:
|
|
75
|
+
encoded: dict[str, Any] = {
|
|
76
|
+
"timeUnixNano": record.time_unix_nano,
|
|
77
|
+
"observedTimeUnixNano": record.observed_time_unix_nano,
|
|
78
|
+
}
|
|
79
|
+
if record.level is not None:
|
|
80
|
+
number, text = severity(record.level)
|
|
81
|
+
encoded["severityNumber"] = number
|
|
82
|
+
encoded["severityText"] = text
|
|
83
|
+
encoded["body"] = {"stringValue": record.body}
|
|
84
|
+
encoded["attributes"] = key_values(record.attributes)
|
|
85
|
+
return EncodedRecord(record.scope, encoded, len(to_json_bytes(encoded)))
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def encode_request(resource: list[dict[str, Any]], records: Sequence[EncodedRecord]) -> dict[str, Any]:
|
|
89
|
+
"""One ResourceLogs; one ScopeLogs per scope, in order of first appearance."""
|
|
90
|
+
scopes: dict[str, list[dict[str, Any]]] = {}
|
|
91
|
+
for item in records:
|
|
92
|
+
scopes.setdefault(item.scope, []).append(item.record)
|
|
93
|
+
return {
|
|
94
|
+
"resourceLogs": [
|
|
95
|
+
{
|
|
96
|
+
"resource": {"attributes": resource},
|
|
97
|
+
"scopeLogs": [{"scope": {"name": name}, "logRecords": logs} for name, logs in scopes.items()],
|
|
98
|
+
}
|
|
99
|
+
]
|
|
100
|
+
}
|
dlogify/_exporter.py
ADDED
|
@@ -0,0 +1,407 @@
|
|
|
1
|
+
"""Queue, batching and the export thread (docs/12 §4). Producers take a lock
|
|
2
|
+
for O(1) work; one daemon thread sends one request at a time; failures never
|
|
3
|
+
reach the caller."""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import collections
|
|
8
|
+
import gzip
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
import random as _random
|
|
12
|
+
import sys
|
|
13
|
+
import threading
|
|
14
|
+
import time
|
|
15
|
+
import urllib.error
|
|
16
|
+
import weakref
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from datetime import timezone
|
|
19
|
+
from email.utils import parsedate_to_datetime
|
|
20
|
+
from typing import Any, Callable, Optional
|
|
21
|
+
|
|
22
|
+
from . import _http
|
|
23
|
+
from ._diagnostics import Diagnostics, describe_error
|
|
24
|
+
from ._encode import EncodedRecord, encode_record, encode_request, to_json_bytes
|
|
25
|
+
from ._guard import guard_thread
|
|
26
|
+
from ._record import LogRecord
|
|
27
|
+
|
|
28
|
+
EXPORT_INTERVAL = 5.0
|
|
29
|
+
TRIGGER_RECORDS = 500
|
|
30
|
+
TRIGGER_BYTES = 512 * 1024
|
|
31
|
+
BATCH_MAX_RECORDS = 1000
|
|
32
|
+
BATCH_MAX_BYTES = 1_000_000
|
|
33
|
+
REQUEST_TIMEOUT = 10.0
|
|
34
|
+
RETRY_WINDOW = 300.0
|
|
35
|
+
_BACKOFF_BASE = 1.0
|
|
36
|
+
_BACKOFF_CAP = 30.0
|
|
37
|
+
# Repeated `Retry-After: 0` replies must not turn into a tight retry loop.
|
|
38
|
+
_RETRY_AFTER_FLOOR = 0.1
|
|
39
|
+
# Room for the scope wrappers around the records of one request.
|
|
40
|
+
_ENVELOPE_RESERVE_BYTES = 1024
|
|
41
|
+
|
|
42
|
+
Post = Callable[[str, bytes, dict[str, str], float], _http.Response]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True)
|
|
46
|
+
class Outcome:
|
|
47
|
+
kind: str # "sent", "retry", "split", "drop" or "stop"
|
|
48
|
+
reason: str = ""
|
|
49
|
+
retry_after: Optional[float] = None
|
|
50
|
+
status: int = 0
|
|
51
|
+
body: str = ""
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def parse_retry_after(value: Optional[str], now: float) -> Optional[float]:
|
|
55
|
+
"""Seconds to wait, from delay-seconds or an HTTP date; `now` is the wall clock."""
|
|
56
|
+
if not value:
|
|
57
|
+
return None
|
|
58
|
+
text = value.strip()
|
|
59
|
+
if text.isascii() and text.isdigit():
|
|
60
|
+
return float(int(text))
|
|
61
|
+
try:
|
|
62
|
+
when = parsedate_to_datetime(text)
|
|
63
|
+
except Exception:
|
|
64
|
+
return None
|
|
65
|
+
if when is None:
|
|
66
|
+
return None
|
|
67
|
+
if when.tzinfo is None:
|
|
68
|
+
when = when.replace(tzinfo=timezone.utc)
|
|
69
|
+
return max(0.0, when.timestamp() - now)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def backoff_delay(attempt: int, rand: float) -> float:
|
|
73
|
+
"""Exponential backoff with full jitter: 1 s, 2 s, 4 s … capped at 30 s."""
|
|
74
|
+
return rand * min(_BACKOFF_CAP, _BACKOFF_BASE * 2 ** min(attempt, 16))
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def classify(status: int, retry_after: Optional[str], body: str, now: float) -> Outcome:
|
|
78
|
+
if 200 <= status < 300:
|
|
79
|
+
return Outcome("sent", body=body)
|
|
80
|
+
if status in (401, 403):
|
|
81
|
+
return Outcome("stop", status=status)
|
|
82
|
+
if status == 413:
|
|
83
|
+
return Outcome("split")
|
|
84
|
+
if status in (429, 502, 503, 504):
|
|
85
|
+
return Outcome("retry", reason=f"HTTP {status}", retry_after=parse_retry_after(retry_after, now))
|
|
86
|
+
detail = f": {body[:500]}" if body else ""
|
|
87
|
+
return Outcome("drop", reason=f"HTTP {status}{detail}")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def is_certificate_error(err: BaseException) -> bool:
|
|
91
|
+
"""CERTIFICATE_VERIFY_FAILED, raised as is or wrapped by urllib."""
|
|
92
|
+
try:
|
|
93
|
+
import ssl
|
|
94
|
+
except ImportError: # a Python built without ssl cannot reach https at all
|
|
95
|
+
return False
|
|
96
|
+
if isinstance(err, urllib.error.URLError):
|
|
97
|
+
err = err.reason if isinstance(err.reason, BaseException) else err
|
|
98
|
+
return isinstance(err, ssl.SSLCertVerificationError)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# multiprocessing children end with os._exit(), which skips atexit: they flush
|
|
102
|
+
# through multiprocessing's own exit finalizers instead.
|
|
103
|
+
_CHILD_EXIT_FLUSH = 2.0
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
# One pair of at-fork callbacks for every exporter: callbacks cannot be
|
|
107
|
+
# unregistered, so one per exporter would pile up with many clients.
|
|
108
|
+
_live: "weakref.WeakSet[Exporter]" = weakref.WeakSet()
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _reset_in_child() -> None:
|
|
112
|
+
for exporter in list(_live):
|
|
113
|
+
exporter._reset_state()
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _before_fork() -> None:
|
|
117
|
+
for exporter in list(_live):
|
|
118
|
+
exporter._flush_multiprocessing_children()
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
if hasattr(os, "register_at_fork"):
|
|
122
|
+
os.register_at_fork(before=_before_fork, after_in_child=_reset_in_child)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _flush_at_child_exit(exporter: "Exporter") -> None:
|
|
126
|
+
# Runs in a multiprocessing child, after it cleared the finalizers it inherited.
|
|
127
|
+
util = sys.modules["multiprocessing.util"]
|
|
128
|
+
util.Finalize(None, exporter.flush, args=(_CHILD_EXIT_FLUSH,), exitpriority=10)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class Exporter:
|
|
132
|
+
def __init__(
|
|
133
|
+
self,
|
|
134
|
+
*,
|
|
135
|
+
url: str,
|
|
136
|
+
api_key: str,
|
|
137
|
+
user_agent: str,
|
|
138
|
+
resource: list[dict[str, Any]],
|
|
139
|
+
max_queue_size: int,
|
|
140
|
+
diagnostics: Diagnostics,
|
|
141
|
+
post: Optional[Post] = None,
|
|
142
|
+
clock: Callable[[], float] = time.monotonic,
|
|
143
|
+
sleep: Optional[Callable[[float], object]] = None,
|
|
144
|
+
random: Callable[[], float] = _random.random,
|
|
145
|
+
request_timeout: float = REQUEST_TIMEOUT,
|
|
146
|
+
interval: float = EXPORT_INTERVAL,
|
|
147
|
+
) -> None:
|
|
148
|
+
self._url = url
|
|
149
|
+
self._headers = {
|
|
150
|
+
"Content-Type": "application/json",
|
|
151
|
+
"Content-Encoding": "gzip",
|
|
152
|
+
"Authorization": f"Bearer {api_key}",
|
|
153
|
+
"User-Agent": user_agent,
|
|
154
|
+
}
|
|
155
|
+
self._resource = resource
|
|
156
|
+
self._max_queue_size = max_queue_size
|
|
157
|
+
self._diagnostics = diagnostics
|
|
158
|
+
self._post_fn = post or _http.post
|
|
159
|
+
self._clock = clock
|
|
160
|
+
self._sleep = sleep or self._wait
|
|
161
|
+
self._random = random
|
|
162
|
+
self._request_timeout = request_timeout
|
|
163
|
+
self._interval = interval
|
|
164
|
+
envelope = len(to_json_bytes(encode_request(resource, [])))
|
|
165
|
+
self._batch_budget = BATCH_MAX_BYTES - envelope - _ENVELOPE_RESERVE_BYTES
|
|
166
|
+
self._stopped = False
|
|
167
|
+
self._closed = False
|
|
168
|
+
self._multiprocessing_registered = False
|
|
169
|
+
self._reset_state()
|
|
170
|
+
_live.add(self)
|
|
171
|
+
|
|
172
|
+
def _reset_state(self) -> None:
|
|
173
|
+
# Also run in a forked child: the export thread does not exist there and
|
|
174
|
+
# may have held the lock. The parent still sends what was queued.
|
|
175
|
+
# Reentrant: a signal handler or a finalizer may call the SDK while this
|
|
176
|
+
# thread is inside enqueue() or flush().
|
|
177
|
+
self._cond = threading.Condition(threading.RLock())
|
|
178
|
+
self._wake = threading.Event()
|
|
179
|
+
self._queue: collections.deque[EncodedRecord] = collections.deque()
|
|
180
|
+
self._queued_bytes = 0
|
|
181
|
+
self._in_flight = 0
|
|
182
|
+
self._flush_waiters = 0
|
|
183
|
+
self._dropped = 0
|
|
184
|
+
self._reported_dropped = 0
|
|
185
|
+
self._thread: Optional[threading.Thread] = None
|
|
186
|
+
|
|
187
|
+
def _flush_multiprocessing_children(self) -> None:
|
|
188
|
+
"""Before a fork: when multiprocessing is in use, its children flush as they end."""
|
|
189
|
+
util = sys.modules.get("multiprocessing.util")
|
|
190
|
+
if util is None or self._multiprocessing_registered:
|
|
191
|
+
return
|
|
192
|
+
self._multiprocessing_registered = True
|
|
193
|
+
util.register_after_fork(self, _flush_at_child_exit)
|
|
194
|
+
|
|
195
|
+
def enqueue(self, record: LogRecord) -> None:
|
|
196
|
+
if self._stopped or self._closed:
|
|
197
|
+
return
|
|
198
|
+
encoded = encode_record(record)
|
|
199
|
+
with self._cond:
|
|
200
|
+
if self._stopped or self._closed:
|
|
201
|
+
return
|
|
202
|
+
if len(self._queue) >= self._max_queue_size:
|
|
203
|
+
oldest = self._queue.popleft()
|
|
204
|
+
self._queued_bytes -= oldest.bytes
|
|
205
|
+
self._dropped += 1
|
|
206
|
+
self._queue.append(encoded)
|
|
207
|
+
self._queued_bytes += encoded.bytes
|
|
208
|
+
self._ensure_thread()
|
|
209
|
+
if len(self._queue) >= TRIGGER_RECORDS or self._queued_bytes >= TRIGGER_BYTES:
|
|
210
|
+
self._cond.notify_all()
|
|
211
|
+
|
|
212
|
+
def flush(self, timeout: float) -> None:
|
|
213
|
+
"""Sends everything queued; returns when done or when the timeout expires."""
|
|
214
|
+
if threading.current_thread() is self._thread:
|
|
215
|
+
return
|
|
216
|
+
deadline = time.monotonic() + timeout
|
|
217
|
+
with self._cond:
|
|
218
|
+
if self._queue:
|
|
219
|
+
self._ensure_thread()
|
|
220
|
+
self._flush_waiters += 1
|
|
221
|
+
self._cond.notify_all()
|
|
222
|
+
try:
|
|
223
|
+
while (self._queue or self._in_flight) and not (self._stopped or self._closed) and self._thread is not None:
|
|
224
|
+
remaining = deadline - time.monotonic()
|
|
225
|
+
if remaining <= 0:
|
|
226
|
+
return
|
|
227
|
+
self._cond.wait(remaining)
|
|
228
|
+
finally:
|
|
229
|
+
self._flush_waiters -= 1
|
|
230
|
+
|
|
231
|
+
def close(self) -> None:
|
|
232
|
+
_live.discard(self)
|
|
233
|
+
with self._cond:
|
|
234
|
+
self._closed = True
|
|
235
|
+
self._queue.clear()
|
|
236
|
+
self._queued_bytes = 0
|
|
237
|
+
self._cond.notify_all()
|
|
238
|
+
self._wake.set()
|
|
239
|
+
|
|
240
|
+
def _wait(self, seconds: float) -> None:
|
|
241
|
+
self._wake.wait(seconds)
|
|
242
|
+
|
|
243
|
+
def _ensure_thread(self) -> None:
|
|
244
|
+
"""Starts the export thread on the first record (lock held)."""
|
|
245
|
+
if self._thread is not None:
|
|
246
|
+
return
|
|
247
|
+
thread = threading.Thread(target=self._run, name="dlogify-exporter", daemon=True)
|
|
248
|
+
try:
|
|
249
|
+
thread.start()
|
|
250
|
+
except RuntimeError as err: # the interpreter is shutting down
|
|
251
|
+
self._diagnostics.debug(f"could not start the export thread: {describe_error(err)}")
|
|
252
|
+
return
|
|
253
|
+
self._thread = thread
|
|
254
|
+
|
|
255
|
+
def _due(self, deadline: float) -> bool:
|
|
256
|
+
if self._stopped or self._closed or not self._queue:
|
|
257
|
+
return False
|
|
258
|
+
return (
|
|
259
|
+
self._flush_waiters > 0
|
|
260
|
+
or len(self._queue) >= TRIGGER_RECORDS
|
|
261
|
+
or self._queued_bytes >= TRIGGER_BYTES
|
|
262
|
+
or time.monotonic() >= deadline
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
def _run(self) -> None:
|
|
266
|
+
guard_thread()
|
|
267
|
+
deadline = time.monotonic() + self._interval
|
|
268
|
+
while True:
|
|
269
|
+
with self._cond:
|
|
270
|
+
while not self._due(deadline):
|
|
271
|
+
if self._stopped or self._closed:
|
|
272
|
+
return
|
|
273
|
+
now = time.monotonic()
|
|
274
|
+
if now >= deadline: # a tick with nothing queued
|
|
275
|
+
deadline = now + self._interval
|
|
276
|
+
self._cond.wait(deadline - now)
|
|
277
|
+
self._drain()
|
|
278
|
+
deadline = time.monotonic() + self._interval
|
|
279
|
+
|
|
280
|
+
def _drain(self) -> None:
|
|
281
|
+
while True:
|
|
282
|
+
batch: Optional[list[EncodedRecord]] = None
|
|
283
|
+
with self._cond:
|
|
284
|
+
note = self._dropped_note()
|
|
285
|
+
if self._queue and not (self._stopped or self._closed):
|
|
286
|
+
batch = self._take_batch()
|
|
287
|
+
self._in_flight = len(batch)
|
|
288
|
+
if note:
|
|
289
|
+
self._diagnostics.debug(note)
|
|
290
|
+
if batch is None:
|
|
291
|
+
return
|
|
292
|
+
try:
|
|
293
|
+
self._send(batch)
|
|
294
|
+
except Exception as err:
|
|
295
|
+
self._diagnostics.debug(f"export failed: {describe_error(err)}")
|
|
296
|
+
finally:
|
|
297
|
+
with self._cond:
|
|
298
|
+
self._in_flight = 0
|
|
299
|
+
self._cond.notify_all()
|
|
300
|
+
|
|
301
|
+
def _take_batch(self) -> list[EncodedRecord]:
|
|
302
|
+
"""Lock held: at most BATCH_MAX_RECORDS records and the byte budget."""
|
|
303
|
+
batch: list[EncodedRecord] = []
|
|
304
|
+
total = 0
|
|
305
|
+
while self._queue and len(batch) < BATCH_MAX_RECORDS:
|
|
306
|
+
size = self._queue[0].bytes + 1
|
|
307
|
+
if batch and total + size > self._batch_budget:
|
|
308
|
+
break
|
|
309
|
+
record = self._queue.popleft()
|
|
310
|
+
self._queued_bytes -= record.bytes
|
|
311
|
+
batch.append(record)
|
|
312
|
+
total += size
|
|
313
|
+
return batch
|
|
314
|
+
|
|
315
|
+
def _dropped_note(self) -> Optional[str]:
|
|
316
|
+
if self._dropped == self._reported_dropped:
|
|
317
|
+
return None
|
|
318
|
+
note = f"queue full: dropped {self._dropped - self._reported_dropped} records ({self._dropped} in total)"
|
|
319
|
+
self._reported_dropped = self._dropped
|
|
320
|
+
return note
|
|
321
|
+
|
|
322
|
+
def _body(self, batch: list[EncodedRecord]) -> bytes:
|
|
323
|
+
return gzip.compress(to_json_bytes(encode_request(self._resource, batch)))
|
|
324
|
+
|
|
325
|
+
def _send(self, batch: list[EncodedRecord]) -> None:
|
|
326
|
+
started = self._clock()
|
|
327
|
+
attempt = 0
|
|
328
|
+
while not (self._stopped or self._closed):
|
|
329
|
+
outcome = self._post(batch)
|
|
330
|
+
if outcome.kind == "sent":
|
|
331
|
+
self._report_partial(outcome.body)
|
|
332
|
+
return
|
|
333
|
+
if outcome.kind == "stop":
|
|
334
|
+
self._stop(outcome.status)
|
|
335
|
+
return
|
|
336
|
+
if outcome.kind == "drop":
|
|
337
|
+
self._diagnostics.debug(f"dropped {len(batch)} records: {outcome.reason}")
|
|
338
|
+
return
|
|
339
|
+
if outcome.kind == "split":
|
|
340
|
+
if len(batch) == 1:
|
|
341
|
+
self._diagnostics.debug("dropped a record that is too large (HTTP 413)")
|
|
342
|
+
return
|
|
343
|
+
half = (len(batch) + 1) // 2
|
|
344
|
+
self._send(batch[:half])
|
|
345
|
+
self._send(batch[half:])
|
|
346
|
+
return
|
|
347
|
+
remaining = RETRY_WINDOW - (self._clock() - started)
|
|
348
|
+
if remaining <= 0:
|
|
349
|
+
self._diagnostics.debug(f"dropped {len(batch)} records after retrying for 5 minutes: {outcome.reason}")
|
|
350
|
+
return
|
|
351
|
+
if outcome.retry_after is None:
|
|
352
|
+
wait = backoff_delay(attempt, self._random())
|
|
353
|
+
else:
|
|
354
|
+
wait = max(_RETRY_AFTER_FLOOR, outcome.retry_after)
|
|
355
|
+
delay = min(wait, remaining)
|
|
356
|
+
self._diagnostics.debug(f"export failed ({outcome.reason}); retrying in {round(delay * 1000)} ms")
|
|
357
|
+
self._sleep(delay)
|
|
358
|
+
attempt += 1
|
|
359
|
+
|
|
360
|
+
def _post(self, batch: list[EncodedRecord]) -> Outcome:
|
|
361
|
+
try:
|
|
362
|
+
body = self._body(batch)
|
|
363
|
+
except Exception as err:
|
|
364
|
+
return Outcome("drop", reason=f"encoding failed: {describe_error(err)}")
|
|
365
|
+
try:
|
|
366
|
+
response = self._post_fn(self._url, body, self._headers, self._request_timeout)
|
|
367
|
+
except Exception as err:
|
|
368
|
+
if is_certificate_error(err):
|
|
369
|
+
self._diagnostics.warn_once(
|
|
370
|
+
"certificate",
|
|
371
|
+
f"could not verify the TLS certificate of {self._url} ({describe_error(err)}). "
|
|
372
|
+
'With Python from python.org on macOS, run "Install Certificates.command" in its Applications folder; '
|
|
373
|
+
"behind a proxy that inspects TLS, point SSL_CERT_FILE to its CA bundle. Each batch is retried for 5 minutes, then dropped",
|
|
374
|
+
)
|
|
375
|
+
return Outcome("retry", reason=describe_error(err))
|
|
376
|
+
if 300 <= response.status < 400:
|
|
377
|
+
location = response.headers.get("location") or "an unknown location"
|
|
378
|
+
self._diagnostics.warn_once(
|
|
379
|
+
"redirect",
|
|
380
|
+
f"the endpoint {self._url} redirected to {location}; set LOGIFY_ENDPOINT to the final URL. Logs are dropped until then",
|
|
381
|
+
)
|
|
382
|
+
return Outcome("drop", reason=f"HTTP {response.status} redirect to {location}")
|
|
383
|
+
return classify(response.status, response.headers.get("retry-after"), response.body, time.time())
|
|
384
|
+
|
|
385
|
+
def _report_partial(self, body: str) -> None:
|
|
386
|
+
try:
|
|
387
|
+
parsed = json.loads(body)
|
|
388
|
+
partial = parsed.get("partialSuccess") or parsed.get("partial_success")
|
|
389
|
+
if not partial:
|
|
390
|
+
return
|
|
391
|
+
rejected = int(partial.get("rejectedLogRecords") or partial.get("rejected_log_records") or 0)
|
|
392
|
+
message = partial.get("errorMessage") or partial.get("error_message") or ""
|
|
393
|
+
if rejected > 0 or message:
|
|
394
|
+
self._diagnostics.debug(f"partial success: {rejected} records rejected: {message}")
|
|
395
|
+
except Exception:
|
|
396
|
+
pass # not JSON; nothing to report
|
|
397
|
+
|
|
398
|
+
def _stop(self, status: int) -> None:
|
|
399
|
+
"""401/403: the key is invalid, revoked or lacks the scope."""
|
|
400
|
+
with self._cond:
|
|
401
|
+
self._stopped = True
|
|
402
|
+
self._queue.clear()
|
|
403
|
+
self._queued_bytes = 0
|
|
404
|
+
self._cond.notify_all()
|
|
405
|
+
self._diagnostics.warn_once(
|
|
406
|
+
"auth", f"the API key was rejected (HTTP {status}); Logify stops sending logs for this process"
|
|
407
|
+
)
|
dlogify/_guard.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Set while Logify builds, queues or exports a record, so logging done on the
|
|
2
|
+
way (by Logify or by code it calls, such as a __str__ that logs) is not
|
|
3
|
+
captured again (docs/12 §4, guarantee 3)."""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import threading
|
|
8
|
+
from collections.abc import Iterator
|
|
9
|
+
from contextlib import contextmanager
|
|
10
|
+
|
|
11
|
+
_state = threading.local()
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def is_guarded() -> bool:
|
|
15
|
+
return getattr(_state, "depth", 0) > 0
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@contextmanager
|
|
19
|
+
def guarded() -> Iterator[None]:
|
|
20
|
+
_state.depth = getattr(_state, "depth", 0) + 1
|
|
21
|
+
try:
|
|
22
|
+
yield
|
|
23
|
+
finally:
|
|
24
|
+
_state.depth -= 1
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def guard_thread() -> None:
|
|
28
|
+
"""Marks the current thread (the export thread) as guarded for good."""
|
|
29
|
+
_state.depth = 1
|
dlogify/_hooks.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""Uncaught exceptions and the exit flush (docs/12 §4, guarantee 7; spec §3)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import atexit
|
|
6
|
+
import sys
|
|
7
|
+
import threading
|
|
8
|
+
import time
|
|
9
|
+
from types import TracebackType
|
|
10
|
+
from typing import TYPE_CHECKING, Any, Callable, Optional
|
|
11
|
+
|
|
12
|
+
from ._record import exception_attributes, exception_title
|
|
13
|
+
|
|
14
|
+
if TYPE_CHECKING:
|
|
15
|
+
from ._client import Client
|
|
16
|
+
|
|
17
|
+
EXIT_FLUSH_TIMEOUT = 2.0
|
|
18
|
+
_IGNORED = (KeyboardInterrupt, SystemExit)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class ExitBudget:
|
|
22
|
+
"""The 2 s an exit may spend flushing, shared by the excepthook and the
|
|
23
|
+
atexit flush that follows it."""
|
|
24
|
+
|
|
25
|
+
def __init__(self) -> None:
|
|
26
|
+
self.spent = 0.0
|
|
27
|
+
|
|
28
|
+
def start_over(self) -> None:
|
|
29
|
+
# In a REPL the excepthook runs for every error and the process goes on.
|
|
30
|
+
self.spent = 0.0
|
|
31
|
+
|
|
32
|
+
def flush(self, client: "Client") -> None:
|
|
33
|
+
remaining = EXIT_FLUSH_TIMEOUT - self.spent
|
|
34
|
+
if remaining <= 0:
|
|
35
|
+
return
|
|
36
|
+
started = time.monotonic()
|
|
37
|
+
try:
|
|
38
|
+
client.flush(remaining)
|
|
39
|
+
finally:
|
|
40
|
+
self.spent += time.monotonic() - started
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _record(client: "Client", exc: BaseException, tb: Optional[TracebackType], extra: dict[str, object]) -> None:
|
|
44
|
+
exception = exception_attributes(exc, tb)
|
|
45
|
+
body = exception_title(exception["exception.type"], exception["exception.message"])
|
|
46
|
+
client._emit("uncaught", "fatal", body, {**exception, **extra})
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def install_uncaught_capture(client: "Client", budget: ExitBudget) -> Callable[[], None]:
|
|
50
|
+
previous_hook = sys.excepthook
|
|
51
|
+
previous_thread_hook = threading.excepthook
|
|
52
|
+
|
|
53
|
+
def excepthook(exc_type: type, exc: BaseException, tb: Optional[TracebackType]) -> None:
|
|
54
|
+
try:
|
|
55
|
+
if isinstance(exc, BaseException) and not isinstance(exc, _IGNORED):
|
|
56
|
+
_record(client, exc, tb, {"exception.origin": "main"})
|
|
57
|
+
budget.start_over()
|
|
58
|
+
budget.flush(client)
|
|
59
|
+
except Exception:
|
|
60
|
+
pass # the application's error matters more than ours
|
|
61
|
+
previous_hook(exc_type, exc, tb)
|
|
62
|
+
|
|
63
|
+
def thread_excepthook(args: Any) -> None:
|
|
64
|
+
try:
|
|
65
|
+
exc = args.exc_value
|
|
66
|
+
if isinstance(exc, BaseException) and not isinstance(exc, _IGNORED):
|
|
67
|
+
extra: dict[str, object] = {"exception.origin": "thread"}
|
|
68
|
+
if args.thread is not None:
|
|
69
|
+
extra["thread.name"] = args.thread.name
|
|
70
|
+
_record(client, exc, args.exc_traceback, extra)
|
|
71
|
+
except Exception:
|
|
72
|
+
pass
|
|
73
|
+
previous_thread_hook(args)
|
|
74
|
+
|
|
75
|
+
sys.excepthook = excepthook
|
|
76
|
+
threading.excepthook = thread_excepthook
|
|
77
|
+
|
|
78
|
+
def undo() -> None:
|
|
79
|
+
if sys.excepthook is excepthook:
|
|
80
|
+
sys.excepthook = previous_hook
|
|
81
|
+
if threading.excepthook is thread_excepthook:
|
|
82
|
+
threading.excepthook = previous_thread_hook
|
|
83
|
+
|
|
84
|
+
return undo
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def install_exit_flush(client: "Client", budget: ExitBudget) -> Callable[[], None]:
|
|
88
|
+
"""atexit handlers run last-in first-out: those the application registers
|
|
89
|
+
after init() (and may log from) run before this flush."""
|
|
90
|
+
|
|
91
|
+
def flush_at_exit() -> None:
|
|
92
|
+
budget.flush(client)
|
|
93
|
+
|
|
94
|
+
atexit.register(flush_at_exit)
|
|
95
|
+
return lambda: atexit.unregister(flush_at_exit)
|