dlogify 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dlogify/__init__.py +160 -0
- dlogify/__main__.py +3 -0
- dlogify/_capture_logging.py +56 -0
- dlogify/_client.py +239 -0
- dlogify/_config.py +194 -0
- dlogify/_diagnostics.py +54 -0
- dlogify/_encode.py +100 -0
- dlogify/_exporter.py +407 -0
- dlogify/_guard.py +29 -0
- dlogify/_hooks.py +95 -0
- dlogify/_http.py +52 -0
- dlogify/_levels.py +88 -0
- dlogify/_parse/__init__.py +0 -0
- dlogify/_parse/compact_json.py +55 -0
- dlogify/_parse/json_line.py +178 -0
- dlogify/_parse/lines.py +64 -0
- dlogify/_parse/stream_parser.py +106 -0
- dlogify/_parse/text_event.py +150 -0
- dlogify/_record.py +149 -0
- dlogify/_singleton.py +12 -0
- dlogify/_version.py +4 -0
- dlogify/handler.py +70 -0
- dlogify/py.typed +0 -0
- dlogify/run/__init__.py +0 -0
- dlogify/run/child.py +215 -0
- dlogify/run/cli.py +22 -0
- dlogify-0.1.0.dist-info/METADATA +111 -0
- dlogify-0.1.0.dist-info/RECORD +31 -0
- dlogify-0.1.0.dist-info/WHEEL +4 -0
- dlogify-0.1.0.dist-info/entry_points.txt +3 -0
- dlogify-0.1.0.dist-info/licenses/LICENSE +21 -0
dlogify/_http.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""The exporter's transport: one POST with urllib.request."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import urllib.error
|
|
6
|
+
import urllib.request
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True)
|
|
12
|
+
class Response:
|
|
13
|
+
status: int
|
|
14
|
+
headers: dict[str, str]
|
|
15
|
+
body: str
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class _NoRedirect(urllib.request.HTTPRedirectHandler):
|
|
19
|
+
# A redirect would turn the POST into a GET or drop the body; the exporter
|
|
20
|
+
# reports it instead (urllib then raises HTTPError with the 3xx status).
|
|
21
|
+
def redirect_request(self, req: Any, fp: Any, code: int, msg: str, headers: Any, newurl: str) -> None:
|
|
22
|
+
return None
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
_opener = urllib.request.build_opener(_NoRedirect)
|
|
26
|
+
_MAX_BODY = 64 * 1024
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def post(url: str, body: bytes, headers: dict[str, str], timeout: float) -> Response:
|
|
30
|
+
"""Raises on network errors and timeouts; any HTTP status is a Response."""
|
|
31
|
+
request = urllib.request.Request(url, data=body, headers=headers, method="POST")
|
|
32
|
+
try:
|
|
33
|
+
with _opener.open(request, timeout=timeout) as response:
|
|
34
|
+
return Response(response.status, _headers(response.headers), _read(response))
|
|
35
|
+
except urllib.error.HTTPError as err:
|
|
36
|
+
try:
|
|
37
|
+
return Response(err.code, _headers(err.headers), _read(err))
|
|
38
|
+
finally:
|
|
39
|
+
err.close()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _headers(message: Any) -> dict[str, str]:
|
|
43
|
+
if message is None:
|
|
44
|
+
return {}
|
|
45
|
+
return {key.lower(): value for key, value in message.items()}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _read(response: Any) -> str:
|
|
49
|
+
try:
|
|
50
|
+
return response.read(_MAX_BODY).decode("utf-8", "replace")
|
|
51
|
+
except Exception:
|
|
52
|
+
return ""
|
dlogify/_levels.py
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Client levels and how sources map onto them (docs/12 §3, §5.2, §5.3)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
from typing import Literal, Optional
|
|
7
|
+
|
|
8
|
+
Level = Literal["trace", "debug", "info", "warn", "error", "fatal"]
|
|
9
|
+
LEVELS: tuple[Level, ...] = ("trace", "debug", "info", "warn", "error", "fatal")
|
|
10
|
+
|
|
11
|
+
_SEVERITY: dict[str, tuple[int, str]] = {
|
|
12
|
+
"trace": (1, "TRACE"),
|
|
13
|
+
"debug": (5, "DEBUG"),
|
|
14
|
+
"info": (9, "INFO"),
|
|
15
|
+
"warn": (13, "WARN"),
|
|
16
|
+
"error": (17, "ERROR"),
|
|
17
|
+
"fatal": (21, "FATAL"),
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
# The logging level a LogifyHandler gets for each min_level.
|
|
21
|
+
LEVELNO: dict[Level, int] = {"trace": 1, "debug": 10, "info": 20, "warn": 30, "error": 40, "fatal": 50}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def severity(level: Level) -> tuple[int, str]:
|
|
25
|
+
return _SEVERITY[level]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def is_level(value: object) -> bool:
|
|
29
|
+
return isinstance(value, str) and value in _SEVERITY
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def level_rank(level: Optional[Level]) -> int:
|
|
33
|
+
"""Unknown levels (only `logify run` produces them) rank as info."""
|
|
34
|
+
return LEVELS.index(level or "info")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
# Longer words first so a regex alternation prefers WARNING over WARN.
|
|
38
|
+
LEVEL_WORD_PATTERN = "WARNING|WARN|ERROR|ERR|TRACE|DEBUG|INFO|NOTICE|FATAL|CRITICAL|PANIC"
|
|
39
|
+
|
|
40
|
+
_WORDS: dict[str, Level] = {
|
|
41
|
+
"TRACE": "trace",
|
|
42
|
+
"DEBUG": "debug",
|
|
43
|
+
"INFO": "info",
|
|
44
|
+
"NOTICE": "info",
|
|
45
|
+
"WARN": "warn",
|
|
46
|
+
"WARNING": "warn",
|
|
47
|
+
"ERROR": "error",
|
|
48
|
+
"ERR": "error",
|
|
49
|
+
"FATAL": "fatal",
|
|
50
|
+
"CRITICAL": "fatal",
|
|
51
|
+
"PANIC": "fatal",
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def level_from_word(word: str) -> Optional[Level]:
|
|
56
|
+
return _WORDS.get(word.strip().upper())
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def level_from_number(number: float) -> Optional[Level]:
|
|
60
|
+
"""pino/bunyan scale; custom numbers map to the nearest lower standard level."""
|
|
61
|
+
if not math.isfinite(number) or number < 10:
|
|
62
|
+
return None
|
|
63
|
+
if number >= 60:
|
|
64
|
+
return "fatal"
|
|
65
|
+
if number >= 50:
|
|
66
|
+
return "error"
|
|
67
|
+
if number >= 40:
|
|
68
|
+
return "warn"
|
|
69
|
+
if number >= 30:
|
|
70
|
+
return "info"
|
|
71
|
+
if number >= 20:
|
|
72
|
+
return "debug"
|
|
73
|
+
return "trace"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def level_from_levelno(levelno: int) -> Level:
|
|
77
|
+
"""Python logging levels; custom levels map to the nearest lower standard level."""
|
|
78
|
+
if levelno >= 50:
|
|
79
|
+
return "fatal"
|
|
80
|
+
if levelno >= 40:
|
|
81
|
+
return "error"
|
|
82
|
+
if levelno >= 30:
|
|
83
|
+
return "warn"
|
|
84
|
+
if levelno >= 20:
|
|
85
|
+
return "info"
|
|
86
|
+
if levelno >= 10:
|
|
87
|
+
return "debug"
|
|
88
|
+
return "trace"
|
|
File without changes
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Compact JSON as JavaScript's JSON.stringify writes it, so nested values in
|
|
2
|
+
JSON lines render the same in both SDKs (docs/12 §5.2)."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
import math
|
|
8
|
+
from typing import Any, Union
|
|
9
|
+
|
|
10
|
+
MAX_SAFE_INT = 2**53
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def js_number(value: Union[int, float]) -> str:
|
|
14
|
+
if isinstance(value, int) and abs(value) <= MAX_SAFE_INT:
|
|
15
|
+
return str(value)
|
|
16
|
+
try:
|
|
17
|
+
number = float(value)
|
|
18
|
+
except OverflowError:
|
|
19
|
+
return "null"
|
|
20
|
+
if not math.isfinite(number):
|
|
21
|
+
return "null"
|
|
22
|
+
if number == 0:
|
|
23
|
+
return "0"
|
|
24
|
+
# repr() gives the shortest round-trip digits, like JavaScript; only the
|
|
25
|
+
# placement of the exponent differs.
|
|
26
|
+
mantissa, _, exponent = repr(number).partition("e")
|
|
27
|
+
sign = "-" if mantissa.startswith("-") else ""
|
|
28
|
+
mantissa = mantissa.lstrip("-")
|
|
29
|
+
if not exponent:
|
|
30
|
+
return sign + (mantissa[:-2] if mantissa.endswith(".0") else mantissa)
|
|
31
|
+
power = int(exponent)
|
|
32
|
+
if power >= 21 or power <= -7:
|
|
33
|
+
return f"{sign}{mantissa}e{'+' if power > 0 else '-'}{abs(power)}"
|
|
34
|
+
digits = mantissa.replace(".", "")
|
|
35
|
+
if power < 0:
|
|
36
|
+
return f"{sign}0.{'0' * (-power - 1)}{digits}"
|
|
37
|
+
return f"{sign}{digits}{'0' * (power + 1 - len(digits))}"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def compact_json(value: Any) -> str:
|
|
41
|
+
if value is None:
|
|
42
|
+
return "null"
|
|
43
|
+
if value is True:
|
|
44
|
+
return "true"
|
|
45
|
+
if value is False:
|
|
46
|
+
return "false"
|
|
47
|
+
if isinstance(value, str):
|
|
48
|
+
return json.dumps(value, ensure_ascii=False)
|
|
49
|
+
if isinstance(value, (int, float)):
|
|
50
|
+
return js_number(value)
|
|
51
|
+
if isinstance(value, list):
|
|
52
|
+
return "[" + ",".join(compact_json(item) for item in value) + "]"
|
|
53
|
+
if isinstance(value, dict):
|
|
54
|
+
return "{" + ",".join(f"{json.dumps(key, ensure_ascii=False)}:{compact_json(item)}" for key, item in value.items()) + "}"
|
|
55
|
+
return json.dumps(str(value), ensure_ascii=False)
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""docs/12 §5.2: a line that is a JSON object is one complete record."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import math
|
|
7
|
+
import re
|
|
8
|
+
from datetime import datetime, timedelta, timezone
|
|
9
|
+
from typing import Any, Callable, Optional
|
|
10
|
+
|
|
11
|
+
from .._levels import Level, level_from_number, level_from_word
|
|
12
|
+
from .._record import AttributeValue, ms_to_unix_nano
|
|
13
|
+
from .compact_json import MAX_SAFE_INT, compact_json
|
|
14
|
+
from .text_event import ParsedRecord, extract_exception
|
|
15
|
+
|
|
16
|
+
_BODY_KEYS = ("msg", "message", "@message", "event")
|
|
17
|
+
_LEVEL_KEYS = ("level", "severity", "levelname", "@level", "log.level")
|
|
18
|
+
_TIME_KEYS = ("time", "timestamp", "@timestamp", "ts")
|
|
19
|
+
_ERROR_KEYS = ("err", "error", "exception")
|
|
20
|
+
_ISO_TIME = re.compile(
|
|
21
|
+
r"^([0-9]{4})-([0-9]{2})-([0-9]{2})[T ]([0-9]{2}):([0-9]{2})(?::([0-9]{2}))?(?:[.,]([0-9]+))?(Z|[+-][0-9]{2}(?::?[0-9]{2})?)?$",
|
|
22
|
+
re.IGNORECASE,
|
|
23
|
+
)
|
|
24
|
+
_EPOCH = datetime(1970, 1, 1, tzinfo=timezone.utc)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _reject_constant(name: str) -> None:
|
|
28
|
+
raise ValueError(f"{name} is not JSON") # NaN and Infinity, which JavaScript rejects too
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def parse_json_line(line: str) -> Optional[ParsedRecord]:
|
|
32
|
+
if line.lstrip()[:1] != "{":
|
|
33
|
+
return None
|
|
34
|
+
try:
|
|
35
|
+
fields = json.loads(line, parse_constant=_reject_constant)
|
|
36
|
+
except (ValueError, RecursionError):
|
|
37
|
+
return None
|
|
38
|
+
if not isinstance(fields, dict):
|
|
39
|
+
return None
|
|
40
|
+
try:
|
|
41
|
+
return _record(fields)
|
|
42
|
+
except RecursionError:
|
|
43
|
+
return None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _record(fields: dict[str, Any]) -> ParsedRecord:
|
|
47
|
+
def first(keys: tuple[str, ...], accept: Callable[[Any], bool] = lambda value: True) -> Optional[tuple[str, Any]]:
|
|
48
|
+
for key in keys:
|
|
49
|
+
value = fields.get(key)
|
|
50
|
+
if value is not None and accept(value):
|
|
51
|
+
return key, value
|
|
52
|
+
return None
|
|
53
|
+
|
|
54
|
+
body_entry = first(_BODY_KEYS, lambda value: isinstance(value, str))
|
|
55
|
+
level_entry = first(_LEVEL_KEYS)
|
|
56
|
+
time_entry = first(_TIME_KEYS)
|
|
57
|
+
error_entry = first(_ERROR_KEYS)
|
|
58
|
+
exception = _error_attributes(error_entry[1]) if error_entry else None
|
|
59
|
+
used = {entry[0] for entry in (body_entry, level_entry, time_entry) if entry}
|
|
60
|
+
if error_entry and exception:
|
|
61
|
+
used.add(error_entry[0])
|
|
62
|
+
attributes: dict[str, AttributeValue] = dict(exception or {})
|
|
63
|
+
for key, value in fields.items():
|
|
64
|
+
if key in used or value is None:
|
|
65
|
+
continue
|
|
66
|
+
attributes[key] = _attribute(value)
|
|
67
|
+
return ParsedRecord(
|
|
68
|
+
level=_json_level(level_entry[1]) if level_entry else None,
|
|
69
|
+
body=body_entry[1] if body_entry else "",
|
|
70
|
+
attributes=attributes,
|
|
71
|
+
time_unix_nano=_json_time(time_entry[1]) if time_entry else None,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _to_float(value: Any) -> float:
|
|
76
|
+
try:
|
|
77
|
+
return float(value)
|
|
78
|
+
except OverflowError:
|
|
79
|
+
return math.inf if value > 0 else -math.inf
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _attribute(value: Any) -> AttributeValue:
|
|
83
|
+
if isinstance(value, (str, bool)):
|
|
84
|
+
return value
|
|
85
|
+
if isinstance(value, int):
|
|
86
|
+
return value if abs(value) <= MAX_SAFE_INT else _to_float(value)
|
|
87
|
+
if isinstance(value, float):
|
|
88
|
+
return int(value) if value.is_integer() and abs(value) <= MAX_SAFE_INT else value
|
|
89
|
+
return compact_json(value)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _json_level(value: Any) -> Optional[Level]:
|
|
93
|
+
if isinstance(value, bool):
|
|
94
|
+
return None
|
|
95
|
+
if isinstance(value, str):
|
|
96
|
+
return level_from_word(value)
|
|
97
|
+
if isinstance(value, (int, float)):
|
|
98
|
+
return level_from_number(_to_float(value))
|
|
99
|
+
return None
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _json_time(value: Any) -> Optional[str]:
|
|
103
|
+
if isinstance(value, bool):
|
|
104
|
+
return None
|
|
105
|
+
if isinstance(value, (int, float)):
|
|
106
|
+
number = _to_float(value)
|
|
107
|
+
if not math.isfinite(number) or number < 0:
|
|
108
|
+
return None
|
|
109
|
+
ms = number if number > 1e11 else number * 1000
|
|
110
|
+
return ms_to_unix_nano(ms) if math.isfinite(ms) else None
|
|
111
|
+
if isinstance(value, str):
|
|
112
|
+
return parse_iso_time(value)
|
|
113
|
+
return None
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def parse_iso_time(text: str) -> Optional[str]:
|
|
117
|
+
"""ISO 8601 date and time; without an offset it is local time, as in JavaScript's Date.parse."""
|
|
118
|
+
match = _ISO_TIME.match(text.strip())
|
|
119
|
+
if not match:
|
|
120
|
+
return None
|
|
121
|
+
year, month, day, hour, minute, second, fraction, zone = match.groups()
|
|
122
|
+
try:
|
|
123
|
+
tz = _zone(zone)
|
|
124
|
+
moment = datetime(int(year), int(month), int(day), int(hour), int(minute), int(second or 0), tzinfo=tz)
|
|
125
|
+
if tz is None:
|
|
126
|
+
moment = moment.astimezone()
|
|
127
|
+
delta = moment - _EPOCH
|
|
128
|
+
except (ValueError, OverflowError, OSError):
|
|
129
|
+
return None
|
|
130
|
+
seconds = delta.days * 86400 + delta.seconds
|
|
131
|
+
nanos = int(((fraction or "") + "000000000")[:9])
|
|
132
|
+
return str(seconds * 1_000_000_000 + nanos)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _zone(zone: Optional[str]) -> Optional[timezone]:
|
|
136
|
+
if not zone:
|
|
137
|
+
return None
|
|
138
|
+
if zone.upper() == "Z":
|
|
139
|
+
return timezone.utc
|
|
140
|
+
sign = -1 if zone[0] == "-" else 1
|
|
141
|
+
digits = zone[1:].replace(":", "")
|
|
142
|
+
hours = int(digits[:2])
|
|
143
|
+
minutes = int(digits[2:4] or 0)
|
|
144
|
+
return timezone(sign * timedelta(hours=hours, minutes=minutes))
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _error_attributes(value: Any) -> Optional[dict[str, str]]:
|
|
148
|
+
if isinstance(value, str):
|
|
149
|
+
exception = extract_exception(re.split(r"\r?\n", value))
|
|
150
|
+
if exception is None:
|
|
151
|
+
return None
|
|
152
|
+
return {
|
|
153
|
+
"exception.type": exception.type,
|
|
154
|
+
"exception.message": exception.message,
|
|
155
|
+
"exception.stacktrace": exception.stacktrace,
|
|
156
|
+
}
|
|
157
|
+
if not isinstance(value, dict):
|
|
158
|
+
return None
|
|
159
|
+
|
|
160
|
+
def text(key: str) -> Optional[str]:
|
|
161
|
+
item = value.get(key)
|
|
162
|
+
return item if isinstance(item, str) else None
|
|
163
|
+
|
|
164
|
+
out: dict[str, str] = {}
|
|
165
|
+
type_ = text("type")
|
|
166
|
+
if type_ is None:
|
|
167
|
+
type_ = text("name")
|
|
168
|
+
message = text("message")
|
|
169
|
+
stack = text("stack")
|
|
170
|
+
if stack is None:
|
|
171
|
+
stack = text("stacktrace")
|
|
172
|
+
if type_ is not None:
|
|
173
|
+
out["exception.type"] = type_
|
|
174
|
+
if message is not None:
|
|
175
|
+
out["exception.message"] = message
|
|
176
|
+
if stack is not None:
|
|
177
|
+
out["exception.stacktrace"] = stack
|
|
178
|
+
return out or None
|
dlogify/_parse/lines.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""docs/12 §5.1: lines end at \\n (a trailing \\r is removed), are cut to 64 KB
|
|
2
|
+
and decoded as UTF-8 with replacement characters."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
from .._record import utf8_cut_index
|
|
10
|
+
|
|
11
|
+
MAX_LINE_BYTES = 64 * 1024
|
|
12
|
+
# Kept beyond the limit so the cut can see the next byte (UTF-8 boundary).
|
|
13
|
+
_KEEP_BYTES = MAX_LINE_BYTES + 4
|
|
14
|
+
_ANSI = re.compile(r"\x1b\[[0-9;?]*[ -/]*[@-~]")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def strip_ansi(text: str) -> str:
|
|
18
|
+
return _ANSI.sub("", text)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class Line:
|
|
23
|
+
text: str
|
|
24
|
+
truncated: bool
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class LineSplitter:
|
|
28
|
+
def __init__(self) -> None:
|
|
29
|
+
self._buffer = bytearray()
|
|
30
|
+
self._total = 0
|
|
31
|
+
|
|
32
|
+
def push(self, chunk: bytes) -> list[Line]:
|
|
33
|
+
lines = []
|
|
34
|
+
start = 0
|
|
35
|
+
while True:
|
|
36
|
+
newline = chunk.find(b"\n", start)
|
|
37
|
+
if newline == -1:
|
|
38
|
+
break
|
|
39
|
+
self._append(chunk[start:newline])
|
|
40
|
+
lines.append(self._take())
|
|
41
|
+
start = newline + 1
|
|
42
|
+
if start < len(chunk):
|
|
43
|
+
self._append(chunk[start:])
|
|
44
|
+
return lines
|
|
45
|
+
|
|
46
|
+
def end(self) -> list[Line]:
|
|
47
|
+
return [self._take()] if self._total > 0 else []
|
|
48
|
+
|
|
49
|
+
def _append(self, data: bytes) -> None:
|
|
50
|
+
self._total += len(data)
|
|
51
|
+
room = _KEEP_BYTES - len(self._buffer)
|
|
52
|
+
if room > 0 and data:
|
|
53
|
+
self._buffer += data[:room]
|
|
54
|
+
|
|
55
|
+
def _take(self) -> Line:
|
|
56
|
+
data = bytes(self._buffer)
|
|
57
|
+
truncated = self._total > MAX_LINE_BYTES
|
|
58
|
+
self._buffer = bytearray()
|
|
59
|
+
self._total = 0
|
|
60
|
+
if truncated:
|
|
61
|
+
data = data[: utf8_cut_index(data, MAX_LINE_BYTES)]
|
|
62
|
+
elif data.endswith(b"\r"):
|
|
63
|
+
data = data[:-1]
|
|
64
|
+
return Line(data.decode("utf-8", "replace"), truncated)
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""docs/12 §5: one parser per stream, each with its own open event. The clock
|
|
2
|
+
(milliseconds) is injected so tests replay pauses without sleeping."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from typing import Callable, Optional
|
|
8
|
+
|
|
9
|
+
from .._levels import Level
|
|
10
|
+
from .._record import AttributeValue, ms_to_unix_nano
|
|
11
|
+
from .json_line import parse_json_line
|
|
12
|
+
from .lines import Line, LineSplitter, strip_ansi
|
|
13
|
+
from .text_event import ParsedRecord, build_text_record, continues_event, is_blank
|
|
14
|
+
|
|
15
|
+
IDLE_CLOSE_MS = 500
|
|
16
|
+
MAX_EVENT_LINES = 500
|
|
17
|
+
MAX_EVENT_BYTES = 40 * 1024
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class StreamRecord:
|
|
22
|
+
level: Optional[Level]
|
|
23
|
+
body: str
|
|
24
|
+
time_unix_nano: str
|
|
25
|
+
attributes: dict[str, AttributeValue]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class _OpenEvent:
|
|
30
|
+
started_at: float
|
|
31
|
+
lines: list[str] = field(default_factory=list)
|
|
32
|
+
bytes: int = 0
|
|
33
|
+
truncated: bool = False
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class StreamParser:
|
|
37
|
+
def __init__(self, stream: str, now: Callable[[], float], emit: Callable[[StreamRecord], None]) -> None:
|
|
38
|
+
self.stream = stream
|
|
39
|
+
self._now = now
|
|
40
|
+
self._emit_record = emit
|
|
41
|
+
self._splitter = LineSplitter()
|
|
42
|
+
self._event: Optional[_OpenEvent] = None
|
|
43
|
+
self._last_line_at = 0.0
|
|
44
|
+
|
|
45
|
+
def push(self, chunk: bytes) -> None:
|
|
46
|
+
for line in self._splitter.push(chunk):
|
|
47
|
+
self._line(line)
|
|
48
|
+
|
|
49
|
+
def tick(self) -> None:
|
|
50
|
+
"""Called by a ticker: closes the open event after 500 ms without a line."""
|
|
51
|
+
if self._event is not None and self._now() - self._last_line_at >= IDLE_CLOSE_MS:
|
|
52
|
+
self._close()
|
|
53
|
+
|
|
54
|
+
def end(self) -> None:
|
|
55
|
+
for line in self._splitter.end():
|
|
56
|
+
self._line(line)
|
|
57
|
+
self._close()
|
|
58
|
+
|
|
59
|
+
def _line(self, line: Line) -> None:
|
|
60
|
+
now = self._now()
|
|
61
|
+
self._last_line_at = now
|
|
62
|
+
text = strip_ansi(line.text)
|
|
63
|
+
parsed = parse_json_line(text)
|
|
64
|
+
if parsed is not None:
|
|
65
|
+
self._close()
|
|
66
|
+
if line.truncated:
|
|
67
|
+
parsed.attributes["logify.truncated"] = True
|
|
68
|
+
self._emit(parsed, now)
|
|
69
|
+
return
|
|
70
|
+
if self._event is not None and continues_event(self._event.lines, text):
|
|
71
|
+
self._append(text, line.truncated)
|
|
72
|
+
return
|
|
73
|
+
self._close()
|
|
74
|
+
if is_blank(text):
|
|
75
|
+
return
|
|
76
|
+
self._event = _OpenEvent(started_at=now)
|
|
77
|
+
self._append(text, line.truncated)
|
|
78
|
+
|
|
79
|
+
def _append(self, text: str, truncated: bool) -> None:
|
|
80
|
+
event = self._event
|
|
81
|
+
assert event is not None
|
|
82
|
+
event.lines.append(text)
|
|
83
|
+
event.bytes += len(text.encode("utf-8", "replace")) + 1
|
|
84
|
+
event.truncated = event.truncated or truncated
|
|
85
|
+
if len(event.lines) >= MAX_EVENT_LINES or event.bytes >= MAX_EVENT_BYTES:
|
|
86
|
+
event.truncated = True
|
|
87
|
+
self._close()
|
|
88
|
+
|
|
89
|
+
def _close(self) -> None:
|
|
90
|
+
event = self._event
|
|
91
|
+
if event is None:
|
|
92
|
+
return
|
|
93
|
+
self._event = None
|
|
94
|
+
record = build_text_record(event.lines, event.truncated)
|
|
95
|
+
if record is not None:
|
|
96
|
+
self._emit(record, event.started_at)
|
|
97
|
+
|
|
98
|
+
def _emit(self, record: ParsedRecord, arrived_at: float) -> None:
|
|
99
|
+
self._emit_record(
|
|
100
|
+
StreamRecord(
|
|
101
|
+
level=record.level,
|
|
102
|
+
body=record.body,
|
|
103
|
+
time_unix_nano=record.time_unix_nano or ms_to_unix_nano(arrived_at),
|
|
104
|
+
attributes={"log.iostream": self.stream, **record.attributes},
|
|
105
|
+
)
|
|
106
|
+
)
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""docs/12 §5.3–5.6. Shared by `logify run` and (for strings in JSON) the
|
|
2
|
+
JSON-line parser. A port of clients/typescript/src/parse/text-event.ts."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import re
|
|
7
|
+
from collections.abc import Sequence
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from typing import Optional
|
|
10
|
+
|
|
11
|
+
from .._levels import LEVEL_WORD_PATTERN, Level, level_from_word
|
|
12
|
+
from .._record import AttributeValue, exception_title
|
|
13
|
+
|
|
14
|
+
TRACEBACK = "Traceback (most recent call last):"
|
|
15
|
+
_CHAIN_LINES = frozenset(
|
|
16
|
+
{
|
|
17
|
+
"During handling of the above exception, another exception occurred:",
|
|
18
|
+
"The above exception was the direct cause of the following exception:",
|
|
19
|
+
}
|
|
20
|
+
)
|
|
21
|
+
_PREFIX_LEVEL = re.compile(rf"^[\[(]?({LEVEL_WORD_PATTERN})(?=[\]):|\s]|$)", re.IGNORECASE | re.ASCII)
|
|
22
|
+
_UPPER_LEVEL = re.compile(rf"(?<!\w)({LEVEL_WORD_PATTERN})(?!\w)", re.ASCII)
|
|
23
|
+
_UPPER_WINDOW = 64
|
|
24
|
+
_LEADING_TOKEN = re.compile(r"^\s*(\S+)")
|
|
25
|
+
_TIMESTAMP_TOKEN = re.compile(r"^[0-9\-:.,/+TZ]+$")
|
|
26
|
+
_DIGIT = re.compile(r"[0-9]")
|
|
27
|
+
_PY_EXCEPTION_LINE = re.compile(r"^([A-Za-z_][\w.]*)(?:: (.*))?$", re.ASCII)
|
|
28
|
+
_HEADER = re.compile(
|
|
29
|
+
r'^(?:Uncaught )?(?:Exception in thread "[^"]*" )?((?:[A-Za-z_$][\w$.]*)?(?:Error|Exception))(?: \[[\w-]+\])?: (.*)$', re.ASCII
|
|
30
|
+
)
|
|
31
|
+
_HEADER_TYPE_ONLY = re.compile(r"^((?:[A-Za-z_$][\w$.]*)?(?:Error|Exception))$", re.ASCII)
|
|
32
|
+
_JAVA_MORE = re.compile(r"^\.\.\. [0-9]+ (more|common frames omitted)$")
|
|
33
|
+
_NODE_VERSION = re.compile(r"^Node\.js v[0-9]+")
|
|
34
|
+
_GO_PANIC = re.compile(r"^panic: (.*?)(?: \[recovered\])?$")
|
|
35
|
+
_GO_FATAL = re.compile(r"^fatal error: (.*)$")
|
|
36
|
+
_STACK_LINES = (re.compile(r"^\s+at "), re.compile(r'^\s+File "'), re.compile(r"\.go:[0-9]+"))
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class ParsedRecord:
|
|
41
|
+
level: Optional[Level]
|
|
42
|
+
body: str
|
|
43
|
+
attributes: dict[str, AttributeValue] = field(default_factory=dict)
|
|
44
|
+
time_unix_nano: Optional[str] = None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True)
|
|
48
|
+
class ExtractedException:
|
|
49
|
+
type: str
|
|
50
|
+
message: str
|
|
51
|
+
stacktrace: str
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def is_blank(line: str) -> bool:
|
|
55
|
+
return line.strip() == ""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def detect_level(line: str) -> Optional[Level]:
|
|
59
|
+
rest = line
|
|
60
|
+
for _ in range(2):
|
|
61
|
+
token = _LEADING_TOKEN.match(rest)
|
|
62
|
+
if not token or not _TIMESTAMP_TOKEN.match(token.group(1)) or not _DIGIT.search(token.group(1)):
|
|
63
|
+
break
|
|
64
|
+
rest = rest[token.end() :]
|
|
65
|
+
prefix = _PREFIX_LEVEL.match(rest.lstrip())
|
|
66
|
+
if prefix:
|
|
67
|
+
return level_from_word(prefix.group(1))
|
|
68
|
+
upper = _UPPER_LEVEL.search(line)
|
|
69
|
+
if upper and upper.end() <= _UPPER_WINDOW:
|
|
70
|
+
return level_from_word(upper.group(1))
|
|
71
|
+
return None
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def exception_header(line: str) -> Optional[tuple[str, str]]:
|
|
75
|
+
full = _HEADER.match(line)
|
|
76
|
+
if full:
|
|
77
|
+
return full.group(1), full.group(2)
|
|
78
|
+
type_only = _HEADER_TYPE_ONLY.match(line)
|
|
79
|
+
return (type_only.group(1), "") if type_only else None
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def extract_exception(lines: Sequence[str]) -> Optional[ExtractedException]:
|
|
83
|
+
if TRACEBACK in lines:
|
|
84
|
+
start = list(lines).index(TRACEBACK)
|
|
85
|
+
for i in range(len(lines) - 1, start, -1):
|
|
86
|
+
match = _PY_EXCEPTION_LINE.match(lines[i])
|
|
87
|
+
if match:
|
|
88
|
+
return ExtractedException(match.group(1), match.group(2) or "", "\n".join(lines[start:]))
|
|
89
|
+
first = lines[0] if lines else ""
|
|
90
|
+
panic = _GO_PANIC.match(first)
|
|
91
|
+
if panic:
|
|
92
|
+
return ExtractedException("panic", panic.group(1), "\n".join(lines))
|
|
93
|
+
fatal = _GO_FATAL.match(first)
|
|
94
|
+
if fatal:
|
|
95
|
+
return ExtractedException("fatal error", fatal.group(1), "\n".join(lines))
|
|
96
|
+
for i, line in enumerate(lines):
|
|
97
|
+
header = exception_header(line)
|
|
98
|
+
if header:
|
|
99
|
+
return ExtractedException(header[0], header[1], "\n".join(lines[i:]))
|
|
100
|
+
return None
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def continues_event(event: Sequence[str], line: str) -> bool:
|
|
104
|
+
first = event[0] if event else ""
|
|
105
|
+
last = event[-1] if event else ""
|
|
106
|
+
# 1. Indented, non-blank.
|
|
107
|
+
if not is_blank(line) and line[:1] in (" ", "\t"):
|
|
108
|
+
return True
|
|
109
|
+
# 2. Java causes and omitted frames, Python traceback starts.
|
|
110
|
+
if line.startswith("Caused by: ") or _JAVA_MORE.match(line) or line == TRACEBACK:
|
|
111
|
+
return True
|
|
112
|
+
# 3. Inside a Python traceback.
|
|
113
|
+
if TRACEBACK in event:
|
|
114
|
+
if line in _CHAIN_LINES:
|
|
115
|
+
return True
|
|
116
|
+
if _PY_EXCEPTION_LINE.match(line):
|
|
117
|
+
last_non_blank = next((item for item in reversed(event) if not is_blank(item)), "")
|
|
118
|
+
if last_non_blank[:1].isspace() or last_non_blank == TRACEBACK:
|
|
119
|
+
return True
|
|
120
|
+
# 4. Go panics take every following text line.
|
|
121
|
+
if first.startswith("panic: ") or first.startswith("fatal error: "):
|
|
122
|
+
return True
|
|
123
|
+
# 5. After a blank line: an exception header or Node's version line.
|
|
124
|
+
if is_blank(last) and (exception_header(line) is not None or _NODE_VERSION.match(line)):
|
|
125
|
+
return True
|
|
126
|
+
# 6. A blank line inside a multi-line event or after a header (kept tentatively).
|
|
127
|
+
if is_blank(line) and (len(event) > 1 or exception_header(first) is not None):
|
|
128
|
+
return True
|
|
129
|
+
return False
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def build_text_record(lines: Sequence[str], truncated: bool) -> Optional[ParsedRecord]:
|
|
133
|
+
kept = list(lines)
|
|
134
|
+
while kept and is_blank(kept[-1]):
|
|
135
|
+
kept.pop()
|
|
136
|
+
if not kept:
|
|
137
|
+
return None
|
|
138
|
+
detected = detect_level(kept[0])
|
|
139
|
+
exception = extract_exception(kept)
|
|
140
|
+
has_stack = exception is not None or any(pattern.search(line) for line in kept for pattern in _STACK_LINES)
|
|
141
|
+
level: Optional[Level] = detected or ("error" if has_stack else None)
|
|
142
|
+
body = (exception_title(exception.type, exception.message) if exception and not detected else kept[0]).rstrip()
|
|
143
|
+
attributes: dict[str, AttributeValue] = {}
|
|
144
|
+
if exception:
|
|
145
|
+
attributes["exception.type"] = exception.type
|
|
146
|
+
attributes["exception.message"] = exception.message
|
|
147
|
+
attributes["exception.stacktrace"] = exception.stacktrace
|
|
148
|
+
if truncated:
|
|
149
|
+
attributes["logify.truncated"] = True
|
|
150
|
+
return ParsedRecord(level, body, attributes)
|