without-logging 0.0.0__tar.gz → 0.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- without_logging-0.0.2/PKG-INFO +84 -0
- without_logging-0.0.2/README.md +63 -0
- without_logging-0.0.2/pyproject.toml +31 -0
- without_logging-0.0.2/pyproject.toml.orig +32 -0
- without_logging-0.0.2/src/without_logging/__init__.py +39 -0
- without_logging-0.0.2/src/without_logging/capture.py +136 -0
- without_logging-0.0.2/src/without_logging/context.py +61 -0
- without_logging-0.0.2/src/without_logging/processors.py +48 -0
- without_logging-0.0.2/src/without_logging/record.py +154 -0
- without_logging-0.0.2/src/without_logging/renderers.py +143 -0
- without_logging-0.0.2/src/without_logging/sinks.py +214 -0
- without_logging-0.0.0/PKG-INFO +0 -5
- without_logging-0.0.0/pyproject.toml +0 -9
- /without_logging-0.0.0/src/without_logging/__init__.py → /without_logging-0.0.2/src/without_logging/py.typed +0 -0
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: without-logging
|
|
3
|
+
Version: 0.0.2
|
|
4
|
+
Summary: A without pipeline for logs: stdlib log records parsed into immutable values, filtered and enriched as processors, drained to a sink.
|
|
5
|
+
Author: Josh Karpel
|
|
6
|
+
Author-email: Josh Karpel <josh.karpel@gmail.com>
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
9
|
+
Classifier: Framework :: AsyncIO
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
16
|
+
Classifier: Topic :: System :: Logging
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
Requires-Dist: without-core==0.0.2
|
|
19
|
+
Requires-Python: >=3.14
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
22
|
+
# without-logging
|
|
23
|
+
|
|
24
|
+
A `without` pipeline for logs. Logger calls (yours and every third-party
|
|
25
|
+
library's) ultimately produce a stream of records; this package parses each one
|
|
26
|
+
into an immutable `Record` value, lets you filter and enrich them as ordinary
|
|
27
|
+
processors, and drains the result into a sink you own.
|
|
28
|
+
|
|
29
|
+
The two stdlib logging pain points it targets: the impenetrable, mutable,
|
|
30
|
+
noun-heavy configuration, and the monolithic handlers that bundle unrelated
|
|
31
|
+
decisions (when to flush, how to rotate, how to format) into one object. Here a
|
|
32
|
+
record is a value, each stage is a `Processor`, and the sink is whatever you
|
|
33
|
+
compose.
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
import logging
|
|
37
|
+
|
|
38
|
+
from without import compose, from_selector, from_sink
|
|
39
|
+
from without_logging import Level, at_least, capture
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
async def write(record):
|
|
43
|
+
print(f"{record.timestamp:%H:%M:%S} {record.level_name} {record.message}")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
pipeline = compose(from_selector(at_least(Level.WARNING)), from_sink(write))
|
|
47
|
+
|
|
48
|
+
async with capture(pipeline): # attaches to the root logger for the block
|
|
49
|
+
logging.getLogger("app").warning("disk almost full", extra={"free_pct": 3})
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
`capture` is the one impure piece: it activates a handler on the root logger,
|
|
53
|
+
turns the pushed records into a `Stream`, and runs your pipeline against them for
|
|
54
|
+
the life of the block. Everything upstream of the sink is pure and testable
|
|
55
|
+
without touching the logging machinery.
|
|
56
|
+
|
|
57
|
+
To write to a file without paying a thread hop per line, `offload` runs a blocking
|
|
58
|
+
writer on a single dedicated thread, fed by a queue that delivers items in bursts
|
|
59
|
+
(so the writer flushes when it catches up, no flush-frequency knob). Writers are
|
|
60
|
+
named by destination: `to_rotating_file` writes to a file (owning the byte count and
|
|
61
|
+
clock, rotating on any combination of size (`max_bytes`), a relative interval
|
|
62
|
+
(`max_age`), and absolute wall-clock boundaries (`schedule=at_times(...)`)), and
|
|
63
|
+
`to_stream` writes to a caller-owned text stream (`sys.stderr`, a socket) without
|
|
64
|
+
closing it. Both take strings (render a `Record` to text with a `from_map` in front):
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
from datetime import timedelta
|
|
68
|
+
|
|
69
|
+
from without import compose, from_map, from_selector
|
|
70
|
+
from without_logging import Level, at_least, offload, to_rotating_file
|
|
71
|
+
|
|
72
|
+
writer = to_rotating_file(lambda i, when: directory / f"app.{i}.log", max_bytes=64 << 20, max_age=timedelta(hours=1))
|
|
73
|
+
async with offload(writer) as sink:
|
|
74
|
+
lines = compose(from_map(render), sink) # Record -> str -> file
|
|
75
|
+
async with capture(compose(from_selector(at_least(Level.WARNING)), lines)):
|
|
76
|
+
... # WARNING+ records written off the event loop, rotated by size and time
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
See the
|
|
80
|
+
[`without-logging` guide](https://without.help/without-logging/)
|
|
81
|
+
(with the [API reference](https://without.help/reference/without_logging/))
|
|
82
|
+
for the design narrative: why stdlib becomes a one-way source, why filtering is
|
|
83
|
+
just the core `from_selector` builder, and where fan-out to multiple sinks slots
|
|
84
|
+
in.
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# without-logging
|
|
2
|
+
|
|
3
|
+
A `without` pipeline for logs. Logger calls (yours and every third-party
|
|
4
|
+
library's) ultimately produce a stream of records; this package parses each one
|
|
5
|
+
into an immutable `Record` value, lets you filter and enrich them as ordinary
|
|
6
|
+
processors, and drains the result into a sink you own.
|
|
7
|
+
|
|
8
|
+
The two stdlib logging pain points it targets: the impenetrable, mutable,
|
|
9
|
+
noun-heavy configuration, and the monolithic handlers that bundle unrelated
|
|
10
|
+
decisions (when to flush, how to rotate, how to format) into one object. Here a
|
|
11
|
+
record is a value, each stage is a `Processor`, and the sink is whatever you
|
|
12
|
+
compose.
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
import logging
|
|
16
|
+
|
|
17
|
+
from without import compose, from_selector, from_sink
|
|
18
|
+
from without_logging import Level, at_least, capture
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
async def write(record):
|
|
22
|
+
print(f"{record.timestamp:%H:%M:%S} {record.level_name} {record.message}")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
pipeline = compose(from_selector(at_least(Level.WARNING)), from_sink(write))
|
|
26
|
+
|
|
27
|
+
async with capture(pipeline): # attaches to the root logger for the block
|
|
28
|
+
logging.getLogger("app").warning("disk almost full", extra={"free_pct": 3})
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
`capture` is the one impure piece: it activates a handler on the root logger,
|
|
32
|
+
turns the pushed records into a `Stream`, and runs your pipeline against them for
|
|
33
|
+
the life of the block. Everything upstream of the sink is pure and testable
|
|
34
|
+
without touching the logging machinery.
|
|
35
|
+
|
|
36
|
+
To write to a file without paying a thread hop per line, `offload` runs a blocking
|
|
37
|
+
writer on a single dedicated thread, fed by a queue that delivers items in bursts
|
|
38
|
+
(so the writer flushes when it catches up, no flush-frequency knob). Writers are
|
|
39
|
+
named by destination: `to_rotating_file` writes to a file (owning the byte count and
|
|
40
|
+
clock, rotating on any combination of size (`max_bytes`), a relative interval
|
|
41
|
+
(`max_age`), and absolute wall-clock boundaries (`schedule=at_times(...)`)), and
|
|
42
|
+
`to_stream` writes to a caller-owned text stream (`sys.stderr`, a socket) without
|
|
43
|
+
closing it. Both take strings (render a `Record` to text with a `from_map` in front):
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from datetime import timedelta
|
|
47
|
+
|
|
48
|
+
from without import compose, from_map, from_selector
|
|
49
|
+
from without_logging import Level, at_least, offload, to_rotating_file
|
|
50
|
+
|
|
51
|
+
writer = to_rotating_file(lambda i, when: directory / f"app.{i}.log", max_bytes=64 << 20, max_age=timedelta(hours=1))
|
|
52
|
+
async with offload(writer) as sink:
|
|
53
|
+
lines = compose(from_map(render), sink) # Record -> str -> file
|
|
54
|
+
async with capture(compose(from_selector(at_least(Level.WARNING)), lines)):
|
|
55
|
+
... # WARNING+ records written off the event loop, rotated by size and time
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
See the
|
|
59
|
+
[`without-logging` guide](https://without.help/without-logging/)
|
|
60
|
+
(with the [API reference](https://without.help/reference/without_logging/))
|
|
61
|
+
for the design narrative: why stdlib becomes a one-way source, why filtering is
|
|
62
|
+
just the core `from_selector` builder, and where fan-out to multiple sinks slots
|
|
63
|
+
in.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.11.32,<0.12"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "without-logging"
|
|
7
|
+
version = "0.0.2"
|
|
8
|
+
description = "A without pipeline for logs: stdlib log records parsed into immutable values, filtered and enriched as processors, drained to a sink."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.14"
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
14
|
+
"Framework :: AsyncIO",
|
|
15
|
+
"Intended Audience :: Developers",
|
|
16
|
+
"Operating System :: OS Independent",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
19
|
+
"Programming Language :: Python :: 3.14",
|
|
20
|
+
"Topic :: Software Development :: Libraries",
|
|
21
|
+
"Topic :: System :: Logging",
|
|
22
|
+
"Typing :: Typed",
|
|
23
|
+
]
|
|
24
|
+
dependencies = ["without-core==0.0.2"]
|
|
25
|
+
|
|
26
|
+
[[project.authors]]
|
|
27
|
+
name = "Josh Karpel"
|
|
28
|
+
email = "josh.karpel@gmail.com"
|
|
29
|
+
|
|
30
|
+
[tool.uv.sources.without-core]
|
|
31
|
+
workspace = true
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.11.32,<0.12"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "without-logging"
|
|
7
|
+
version = "0.0.2"
|
|
8
|
+
description = "A without pipeline for logs: stdlib log records parsed into immutable values, filtered and enriched as processors, drained to a sink."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
authors = [
|
|
12
|
+
{ name = "Josh Karpel", email = "josh.karpel@gmail.com" },
|
|
13
|
+
]
|
|
14
|
+
requires-python = ">=3.14"
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
17
|
+
"Framework :: AsyncIO",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
22
|
+
"Programming Language :: Python :: 3.14",
|
|
23
|
+
"Topic :: Software Development :: Libraries",
|
|
24
|
+
"Topic :: System :: Logging",
|
|
25
|
+
"Typing :: Typed",
|
|
26
|
+
]
|
|
27
|
+
dependencies = [
|
|
28
|
+
"without-core==0.0.2",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[tool.uv.sources]
|
|
32
|
+
without-core = { workspace = true }
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from without_logging.capture import CaptureHandler
|
|
2
|
+
from without_logging.capture import capture
|
|
3
|
+
from without_logging.context import bind
|
|
4
|
+
from without_logging.context import merge_context
|
|
5
|
+
from without_logging.processors import add_fields
|
|
6
|
+
from without_logging.processors import at_least
|
|
7
|
+
from without_logging.record import Level
|
|
8
|
+
from without_logging.record import Record
|
|
9
|
+
from without_logging.record import parse_record
|
|
10
|
+
from without_logging.renderers import exception_to_dict
|
|
11
|
+
from without_logging.renderers import exception_to_text
|
|
12
|
+
from without_logging.renderers import iso_timestamp
|
|
13
|
+
from without_logging.renderers import render_console
|
|
14
|
+
from without_logging.renderers import render_json
|
|
15
|
+
from without_logging.sinks import at_times
|
|
16
|
+
from without_logging.sinks import offload
|
|
17
|
+
from without_logging.sinks import to_rotating_file
|
|
18
|
+
from without_logging.sinks import to_stream
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"CaptureHandler",
|
|
22
|
+
"Level",
|
|
23
|
+
"Record",
|
|
24
|
+
"add_fields",
|
|
25
|
+
"at_least",
|
|
26
|
+
"at_times",
|
|
27
|
+
"bind",
|
|
28
|
+
"capture",
|
|
29
|
+
"exception_to_dict",
|
|
30
|
+
"exception_to_text",
|
|
31
|
+
"iso_timestamp",
|
|
32
|
+
"merge_context",
|
|
33
|
+
"offload",
|
|
34
|
+
"parse_record",
|
|
35
|
+
"render_console",
|
|
36
|
+
"render_json",
|
|
37
|
+
"to_rotating_file",
|
|
38
|
+
"to_stream",
|
|
39
|
+
]
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import logging
|
|
5
|
+
import threading
|
|
6
|
+
from collections.abc import AsyncIterator
|
|
7
|
+
from collections.abc import Callable
|
|
8
|
+
from contextlib import asynccontextmanager
|
|
9
|
+
|
|
10
|
+
from without.interfaces import Sink
|
|
11
|
+
from without.wiring import stream_from_queue
|
|
12
|
+
|
|
13
|
+
from without_logging.context import merge_context
|
|
14
|
+
from without_logging.record import Record
|
|
15
|
+
from without_logging.record import parse_record
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def parse_with_bound_context(log_record: logging.LogRecord) -> Record:
|
|
19
|
+
"""`capture`'s default parser: `parse_record`, then `merge_context` folds in any bound context."""
|
|
20
|
+
return merge_context(parse_record(log_record))
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class CaptureHandler(logging.Handler):
|
|
24
|
+
"""
|
|
25
|
+
A stdlib handler that parses each `LogRecord` and offers it to an asyncio queue.
|
|
26
|
+
|
|
27
|
+
This is the one-way bridge at the ingestion edge. Stdlib `logging` is the
|
|
28
|
+
de-facto narrow waist every third-party library already writes to, so making
|
|
29
|
+
it a *source* (rather than fighting it) is the whole trick: a pushed
|
|
30
|
+
`LogRecord` is parsed to a `Record` here and dropped onto the queue that
|
|
31
|
+
`stream_from_queue` turns into the pull-based stream the pipeline consumes.
|
|
32
|
+
Control only ever flows outward from stdlib into `without`, never back.
|
|
33
|
+
|
|
34
|
+
A log call MAY happen off the event loop thread, so the parsed value is handed
|
|
35
|
+
to the loop with `call_soon_threadsafe`, which writes the loop's self-pipe to
|
|
36
|
+
wake it. A log call *on* the loop thread (an async handler logging) does not
|
|
37
|
+
need that wakeup, so those take the cheaper `call_soon`; the construction thread
|
|
38
|
+
is the loop thread (`capture` builds this right after `get_running_loop`), so its
|
|
39
|
+
id captured here identifies loop-thread emits. The queue is bounded, so a burst that
|
|
40
|
+
outruns the sink is dropped rather than growing memory without limit;
|
|
41
|
+
`dropped` counts how many so an operator can see it. The overflow policy is a
|
|
42
|
+
boundary decision the app owns: raise `capacity`, or pass `capacity=None` for
|
|
43
|
+
an unbounded queue that never drops (at the cost of the bound).
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
def __init__(
|
|
47
|
+
self,
|
|
48
|
+
loop: asyncio.AbstractEventLoop,
|
|
49
|
+
queue: asyncio.Queue[Record],
|
|
50
|
+
parse: Callable[[logging.LogRecord], Record],
|
|
51
|
+
) -> None:
|
|
52
|
+
super().__init__()
|
|
53
|
+
self._loop = loop
|
|
54
|
+
self._loop_thread_id = threading.get_ident()
|
|
55
|
+
self._queue = queue
|
|
56
|
+
self._parse = parse
|
|
57
|
+
self.dropped = 0
|
|
58
|
+
|
|
59
|
+
def emit(self, log_record: logging.LogRecord) -> None:
|
|
60
|
+
record = self._parse(log_record)
|
|
61
|
+
if threading.get_ident() == self._loop_thread_id:
|
|
62
|
+
self._loop.call_soon(self._offer, record)
|
|
63
|
+
else:
|
|
64
|
+
self._loop.call_soon_threadsafe(self._offer, record)
|
|
65
|
+
|
|
66
|
+
def _offer(self, record: Record) -> None:
|
|
67
|
+
try:
|
|
68
|
+
self._queue.put_nowait(record)
|
|
69
|
+
except asyncio.QueueFull, asyncio.QueueShutDown:
|
|
70
|
+
self.dropped += 1
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@asynccontextmanager
|
|
74
|
+
async def capture(
|
|
75
|
+
sink: Sink[Record],
|
|
76
|
+
*,
|
|
77
|
+
logger: logging.Logger | None = None,
|
|
78
|
+
level: int = logging.INFO,
|
|
79
|
+
capacity: int | None = 1024,
|
|
80
|
+
parse: Callable[[logging.LogRecord], Record] = parse_with_bound_context,
|
|
81
|
+
) -> AsyncIterator[CaptureHandler]:
|
|
82
|
+
"""
|
|
83
|
+
Capture stdlib log records into `sink` for the duration of the block.
|
|
84
|
+
|
|
85
|
+
The imperative shell of the pipeline, and the only impure piece: it attaches a
|
|
86
|
+
`CaptureHandler` to `logger` (the root logger by default, so third-party logs
|
|
87
|
+
are captured too), runs `sink(stream_from_queue(queue))` in a background task,
|
|
88
|
+
and on exit detaches, flushes pending records, shuts the queue so the stream
|
|
89
|
+
ends gracefully, and joins the task. `sink` is a fully wired pipeline, e.g.
|
|
90
|
+
`compose(from_selector(at_least(Level.WARNING)), from_sink(write))`;
|
|
91
|
+
everything upstream of the queue in it is pure and testable on its own.
|
|
92
|
+
|
|
93
|
+
`level` sets the capture threshold. stdlib applies *two* gates: the origin
|
|
94
|
+
logger's effective level (checked where a record is created, and defaulting to
|
|
95
|
+
WARNING on the root, which is what usually swallows INFO) and each handler's
|
|
96
|
+
own level. The logger gate is the load-bearing one: a sub-threshold record is
|
|
97
|
+
dropped at creation before any handler runs, so setting the handler level
|
|
98
|
+
alone would capture nothing new. So this temporarily lowers `logger`'s level
|
|
99
|
+
to `level` (restoring it on exit, scoped to the block) *and* sets the handler
|
|
100
|
+
to `level`, the latter keeping the threshold exact for records that propagate
|
|
101
|
+
up from a descendant logger pinned to its own lower level.
|
|
102
|
+
|
|
103
|
+
`parse` is the enrichment of the *emit phase*, the first of the pipeline's two
|
|
104
|
+
phases split by the queue: it turns each `LogRecord` into a `Record`
|
|
105
|
+
*synchronously* in the handler's `emit`, on the logging call's own task
|
|
106
|
+
(stdlib's shape), where the *sink phase* (`sink`, async, draining the queue) has
|
|
107
|
+
not yet begun. It defaults to `parse_with_bound_context` (`parse_record` then
|
|
108
|
+
`merge_context`), so any fields bound with `bind` are stamped on here, where
|
|
109
|
+
they are still live (the sink phase has left the caller's context). A custom
|
|
110
|
+
parser composes `merge_context` the same way, `merge_context(my_parse(log_record))`,
|
|
111
|
+
to keep binds, or omits it to opt out.
|
|
112
|
+
|
|
113
|
+
The handler itself is yielded so a caller can read `handler.dropped` after
|
|
114
|
+
the block to see how many records overflowed the queue.
|
|
115
|
+
"""
|
|
116
|
+
target = logger if logger is not None else logging.getLogger()
|
|
117
|
+
loop = asyncio.get_running_loop()
|
|
118
|
+
queue: asyncio.Queue[Record] = asyncio.Queue(capacity if capacity is not None else 0)
|
|
119
|
+
handler = CaptureHandler(loop, queue, parse)
|
|
120
|
+
handler.setLevel(level)
|
|
121
|
+
|
|
122
|
+
async def run_sink() -> None:
|
|
123
|
+
await sink(stream_from_queue(queue))
|
|
124
|
+
|
|
125
|
+
previous_level = target.level
|
|
126
|
+
target.setLevel(level)
|
|
127
|
+
target.addHandler(handler)
|
|
128
|
+
drain = asyncio.create_task(run_sink())
|
|
129
|
+
try:
|
|
130
|
+
yield handler
|
|
131
|
+
finally:
|
|
132
|
+
target.removeHandler(handler)
|
|
133
|
+
target.setLevel(previous_level)
|
|
134
|
+
await asyncio.sleep(0) # let already-scheduled emits reach the queue before it shuts
|
|
135
|
+
queue.shutdown()
|
|
136
|
+
await drain
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
from contextlib import contextmanager
|
|
6
|
+
from contextvars import ContextVar
|
|
7
|
+
from types import MappingProxyType
|
|
8
|
+
|
|
9
|
+
from without_logging.record import Record
|
|
10
|
+
|
|
11
|
+
_bound: ContextVar[Mapping[str, object]] = ContextVar("without_logging_bound_context", default=MappingProxyType({}))
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@contextmanager
|
|
15
|
+
def bind(**fields: object) -> Iterator[None]:
|
|
16
|
+
"""
|
|
17
|
+
Bind `fields` onto every record logged within this block, on this task.
|
|
18
|
+
|
|
19
|
+
The call-site half of structured context, the equivalent of structlog's
|
|
20
|
+
`bind_contextvars`: `with bind(request_id=...)` stamps those fields onto every
|
|
21
|
+
record logged inside the block. Scoping is explicit and lexical: the fields
|
|
22
|
+
are in effect for the block and gone after it, leaving no shared place
|
|
23
|
+
mutated. It writes a task-local `ContextVar`, so concurrent tasks each see
|
|
24
|
+
only their own binds, and nested binds accumulate (an inner `bind` merges onto
|
|
25
|
+
the outer and is undone on exit; a key set by both takes the inner value
|
|
26
|
+
inside the inner block).
|
|
27
|
+
|
|
28
|
+
Binding alone does nothing until the `capture` side reads it: pair this with
|
|
29
|
+
`merge_context`, which merges the bound fields onto each record at the parse
|
|
30
|
+
edge, where they are still live (the handler's `emit` runs synchronously on
|
|
31
|
+
the logging call's task). A downstream processor cannot read them, having left
|
|
32
|
+
the caller's context for the sink task.
|
|
33
|
+
"""
|
|
34
|
+
merged = MappingProxyType({**_bound.get(), **fields})
|
|
35
|
+
token = _bound.set(merged)
|
|
36
|
+
try:
|
|
37
|
+
yield
|
|
38
|
+
finally:
|
|
39
|
+
_bound.reset(token)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def merge_context(record: Record) -> Record:
|
|
43
|
+
"""
|
|
44
|
+
Merge the context bound by `bind` onto a record: the read half of call-site context.
|
|
45
|
+
|
|
46
|
+
A `Record -> Record` enrichment, *composed on top of* a parser rather than
|
|
47
|
+
wrapping it: `capture`'s default parser applies it after `parse_record`, and a
|
|
48
|
+
custom parser composes it the same way,
|
|
49
|
+
`merge_context(parse_record(log_record))`. A field the log call set explicitly
|
|
50
|
+
via `extra=` wins over a bound field of the same name (the more specific
|
|
51
|
+
value), so `bind` supplies defaults, never overrides a per-call value.
|
|
52
|
+
|
|
53
|
+
It reads the task-local `ContextVar` that `bind` writes, so it MUST run at the
|
|
54
|
+
parse edge, in the handler's `emit` on the caller's task, not as a pipeline
|
|
55
|
+
`Processor`: the pipeline runs in the sink task, having left that context,
|
|
56
|
+
where the bind would read as empty.
|
|
57
|
+
"""
|
|
58
|
+
bound = _bound.get()
|
|
59
|
+
if not bound:
|
|
60
|
+
return record
|
|
61
|
+
return record.with_fields(**{key: value for key, value in bound.items() if key not in record.fields})
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Awaitable
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
|
|
6
|
+
from without.interfaces import Processor
|
|
7
|
+
from without.interfaces import from_map
|
|
8
|
+
|
|
9
|
+
from without_logging.record import Record
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def at_least(level: int) -> Callable[[Record], Awaitable[bool]]:
|
|
13
|
+
"""
|
|
14
|
+
A predicate matching records at `level` or more severe: the level threshold.
|
|
15
|
+
|
|
16
|
+
Pair it with the core `from_selector` to keep only those records (or
|
|
17
|
+
`from_filter` to drop them). It is `async` to match the one color of
|
|
18
|
+
predicate `from_selector`/`from_filter` expect, though the decision itself
|
|
19
|
+
just compares integers. Because a `Record.level` is a plain `int`, a `Level`
|
|
20
|
+
member reads naturally as the argument (`at_least(Level.WARNING)`), but any
|
|
21
|
+
numeric level works.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
async def is_severe_enough(record: Record) -> bool:
|
|
25
|
+
return record.level >= level
|
|
26
|
+
|
|
27
|
+
return is_severe_enough
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def add_fields(**fields: object) -> Processor[Record, Record]:
|
|
31
|
+
"""
|
|
32
|
+
A processor that merges static `fields` onto every record: enrichment.
|
|
33
|
+
|
|
34
|
+
Enrichment *is* expressible as a `from_map` (one record in, one enriched
|
|
35
|
+
record out), so it uses the core builder directly. Enriching from a *shared
|
|
36
|
+
behavior* (a value the whole pipeline sees the latest of: the current config
|
|
37
|
+
revision, a sampling rate) is the same shape, reading it via `current()`
|
|
38
|
+
inside the step. Per-call-site context (a request or trace id, which is
|
|
39
|
+
task-local) is *not* recoverable here: the pipeline runs in the sink task,
|
|
40
|
+
having left the caller's context. Bind it at the edge instead with
|
|
41
|
+
`bind`/`merge_context`, which read it while `emit` is still on the caller's
|
|
42
|
+
task (see the guide).
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
async def enrich(record: Record) -> Record:
|
|
46
|
+
return record.with_fields(**fields)
|
|
47
|
+
|
|
48
|
+
return from_map(enrich)
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from dataclasses import replace
|
|
7
|
+
from datetime import UTC
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
from enum import IntEnum
|
|
10
|
+
from traceback import TracebackException
|
|
11
|
+
from types import MappingProxyType
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Level(IntEnum):
|
|
15
|
+
"""
|
|
16
|
+
The five standard severities as an ordered value, for writing filters.
|
|
17
|
+
|
|
18
|
+
Members are the stdlib numeric levels, so a `Record.level` (a plain `int`)
|
|
19
|
+
can be compared straight against them: `record.level >= Level.WARNING`.
|
|
20
|
+
Keeping `Record.level` an `int` rather than this enum is deliberate: a
|
|
21
|
+
third-party library that logs at a non-standard numeric level is carried
|
|
22
|
+
through as that number, not rejected, so the enum is a vocabulary for
|
|
23
|
+
predicates rather than a closed set the record must belong to.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
DEBUG = logging.DEBUG
|
|
27
|
+
INFO = logging.INFO
|
|
28
|
+
WARNING = logging.WARNING
|
|
29
|
+
ERROR = logging.ERROR
|
|
30
|
+
CRITICAL = logging.CRITICAL
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# The attributes stdlib's LogRecord factory sets. Anything on a record outside
|
|
34
|
+
# this set arrived via `extra=` (or was set by a filter) and is structured data,
|
|
35
|
+
# so it becomes a Record field. The names here split two ways: the caller-supplied
|
|
36
|
+
# content of the log call (`msg`/`args`, `levelno`, `name`, `exc_info`, `created`)
|
|
37
|
+
# is lifted into typed fields, and the rest is factory-synthesized ambient metadata
|
|
38
|
+
# (source location `pathname`/`lineno`/`funcName`; `thread`/`process`/`taskName`)
|
|
39
|
+
# that is deliberately dropped: it is gated by module flags (`logThreads`,
|
|
40
|
+
# `logProcesses`, `logAsyncioTasks`) and a swappable factory, so it may be `None` or
|
|
41
|
+
# absent, and it exists for stdlib's formatter `%(...)s` placeholders rather than as
|
|
42
|
+
# a stable per-event guarantee. A caller who wants some of it lifts it into `fields`
|
|
43
|
+
# with a custom `parse=` on `capture`.
|
|
44
|
+
RESERVED: frozenset[str] = frozenset(
|
|
45
|
+
{
|
|
46
|
+
"name",
|
|
47
|
+
"msg",
|
|
48
|
+
"args",
|
|
49
|
+
"levelname",
|
|
50
|
+
"levelno",
|
|
51
|
+
"pathname",
|
|
52
|
+
"filename",
|
|
53
|
+
"module",
|
|
54
|
+
"exc_info",
|
|
55
|
+
"exc_text",
|
|
56
|
+
"stack_info",
|
|
57
|
+
"lineno",
|
|
58
|
+
"funcName",
|
|
59
|
+
"created",
|
|
60
|
+
"msecs",
|
|
61
|
+
"relativeCreated",
|
|
62
|
+
"thread",
|
|
63
|
+
"threadName",
|
|
64
|
+
"processName",
|
|
65
|
+
"process",
|
|
66
|
+
"taskName",
|
|
67
|
+
"message",
|
|
68
|
+
"asctime",
|
|
69
|
+
}
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass(frozen=True, slots=True)
|
|
74
|
+
class Record:
|
|
75
|
+
"""
|
|
76
|
+
One log event as an immutable value, parsed from a stdlib `LogRecord`.
|
|
77
|
+
|
|
78
|
+
Where a `LogRecord` is a mutable bag that formatters and filters rewrite in
|
|
79
|
+
place, a `Record` is a value: it always means the same thing, so it can be
|
|
80
|
+
filtered, enriched, fanned out, and sunk without any stage disturbing
|
|
81
|
+
another's copy. Enrichment returns a *new* record (`with_fields`) rather than
|
|
82
|
+
mutating this one. `fields` is the record's structured data, exposed read-only:
|
|
83
|
+
*seeded* from the log call's `extra=` at the parse edge, then *grown* by
|
|
84
|
+
enrichment (`add_fields`, a `bind` merged by `merge_context`, any `with_fields`).
|
|
85
|
+
It is named for what it holds, not for `extra=` alone, because those later
|
|
86
|
+
sources write here too. `exception` is a `TracebackException` captured at the
|
|
87
|
+
ingestion edge (`None` when the event carried none): a *structured* value,
|
|
88
|
+
not a rendered string, so how the traceback is formatted stays a downstream
|
|
89
|
+
boundary the app owns (`"".join(exc.format())` for the stdlib text, or walk
|
|
90
|
+
`exc.stack` for JSON frames), the same way rendering the record itself is.
|
|
91
|
+
Extracting it drops all references to the live frames, so the record stays a
|
|
92
|
+
self-contained value with no traceback pinned to it.
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
timestamp: datetime
|
|
96
|
+
level: int
|
|
97
|
+
logger: str
|
|
98
|
+
message: str
|
|
99
|
+
exception: TracebackException | None
|
|
100
|
+
fields: Mapping[str, object]
|
|
101
|
+
|
|
102
|
+
@property
|
|
103
|
+
def level_name(self) -> str:
|
|
104
|
+
return logging.getLevelName(self.level)
|
|
105
|
+
|
|
106
|
+
def with_fields(self, **fields: object) -> Record:
|
|
107
|
+
"""Return a copy of this record with `fields` merged onto its own."""
|
|
108
|
+
return replace(self, fields=MappingProxyType({**self.fields, **fields}))
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def extract_exception(log_record: logging.LogRecord) -> TracebackException | None:
|
|
112
|
+
"""
|
|
113
|
+
Capture a `LogRecord`'s exception as a structured value, or `None` when it carries none.
|
|
114
|
+
|
|
115
|
+
A live traceback is a *place*, not a value: it pins its frames and their
|
|
116
|
+
locals in memory, is mutable, and cannot ride the queue to another thread
|
|
117
|
+
safely (stdlib's own `QueueHandler` flattens it to a string for exactly this
|
|
118
|
+
reason). So the exception is turned into a value at the ingestion edge, while
|
|
119
|
+
it is still live, the same way `getMessage()` resolves the message. Unlike the
|
|
120
|
+
message, which has one sensible rendering, a traceback has many (locals,
|
|
121
|
+
chaining, style), so this keeps the *structure* rather than pre-rendering:
|
|
122
|
+
`TracebackException.from_exception` extracts frame summaries and drops all
|
|
123
|
+
references to the live frames, leaving a value the app can format however it
|
|
124
|
+
likes downstream (`format` for stdlib text, `stack` for JSON frames).
|
|
125
|
+
"""
|
|
126
|
+
if log_record.exc_info:
|
|
127
|
+
_, exc_value, _ = log_record.exc_info
|
|
128
|
+
if exc_value is not None:
|
|
129
|
+
return TracebackException.from_exception(exc_value)
|
|
130
|
+
return None
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def parse_record(log_record: logging.LogRecord) -> Record:
|
|
134
|
+
"""
|
|
135
|
+
Parse a mutable stdlib `LogRecord` into an immutable `Record` value.
|
|
136
|
+
|
|
137
|
+
The ingestion boundary: this is where a pushed `LogRecord`, from the app's
|
|
138
|
+
own calls or any third-party library's, becomes the typed value the rest of
|
|
139
|
+
the pipeline works with. It is a pure function (`LogRecord -> Record`), so the
|
|
140
|
+
whole translation is testable without touching the logging machinery. The
|
|
141
|
+
structured `fields` are every attribute on the record outside the standard
|
|
142
|
+
envelope (`RESERVED`), which is exactly what `logging`'s `extra=` merges in;
|
|
143
|
+
the message is rendered and any exception is captured as a structured value
|
|
144
|
+
here (see `extract_exception`) so nothing live is carried past this point.
|
|
145
|
+
"""
|
|
146
|
+
fields = {key: value for key, value in log_record.__dict__.items() if key not in RESERVED}
|
|
147
|
+
return Record(
|
|
148
|
+
timestamp=datetime.fromtimestamp(log_record.created, tz=UTC),
|
|
149
|
+
level=log_record.levelno,
|
|
150
|
+
logger=log_record.name,
|
|
151
|
+
message=log_record.getMessage(),
|
|
152
|
+
exception=extract_exception(log_record),
|
|
153
|
+
fields=MappingProxyType(fields),
|
|
154
|
+
)
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
from datetime import datetime
|
|
6
|
+
from traceback import TracebackException
|
|
7
|
+
|
|
8
|
+
from without.interfaces import Processor
|
|
9
|
+
from without.interfaces import from_map
|
|
10
|
+
|
|
11
|
+
from without_logging.record import Record
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def exception_to_dict(exception: TracebackException) -> dict[str, object]:
|
|
15
|
+
"""
|
|
16
|
+
Convert a captured `TracebackException` into a JSON-serializable dict: structured traceback.
|
|
17
|
+
|
|
18
|
+
Each frame becomes `{file, line, function, code}` (from the extracted
|
|
19
|
+
`FrameSummary` values, no live frames), and a chained exception is nested under
|
|
20
|
+
`cause`, following Python's own rule: an explicit `raise ... from` (`__cause__`)
|
|
21
|
+
takes precedence, else the implicit `__context__` unless it was suppressed. This
|
|
22
|
+
is the read half `render_json` uses, exposed so a custom JSON renderer can reuse
|
|
23
|
+
it.
|
|
24
|
+
"""
|
|
25
|
+
cause: dict[str, object] | None = None
|
|
26
|
+
if exception.__cause__ is not None:
|
|
27
|
+
cause = exception_to_dict(exception.__cause__)
|
|
28
|
+
elif exception.__context__ is not None and not exception.__suppress_context__:
|
|
29
|
+
cause = exception_to_dict(exception.__context__)
|
|
30
|
+
result: dict[str, object] = {
|
|
31
|
+
"type": exception.exc_type_str,
|
|
32
|
+
"message": "".join(exception.format_exception_only()).strip(),
|
|
33
|
+
"frames": [
|
|
34
|
+
{"file": frame.filename, "line": frame.lineno, "function": frame.name, "code": frame.line}
|
|
35
|
+
for frame in exception.stack
|
|
36
|
+
],
|
|
37
|
+
}
|
|
38
|
+
if cause is not None:
|
|
39
|
+
result["cause"] = cause
|
|
40
|
+
return result
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def exception_to_text(exception: TracebackException) -> str:
|
|
44
|
+
"""
|
|
45
|
+
Render a captured `TracebackException` to its full traceback text.
|
|
46
|
+
|
|
47
|
+
stdlib's own multi-line rendering (`format`), chained cause/context and all,
|
|
48
|
+
as a single string. The flat-string alternative to the structured
|
|
49
|
+
`exception_to_dict` when encoding a record's exception, and what
|
|
50
|
+
`render_console` uses.
|
|
51
|
+
"""
|
|
52
|
+
return "".join(exception.format()).rstrip("\n")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def iso_timestamp(when: datetime) -> str:
|
|
56
|
+
"""The default timestamp format for the renderers: ISO 8601 (`datetime.isoformat`)."""
|
|
57
|
+
return when.isoformat()
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def render_json(
|
|
61
|
+
*,
|
|
62
|
+
timestamp: Callable[[datetime], object] = iso_timestamp,
|
|
63
|
+
exception: Callable[[TracebackException], object] = exception_to_dict,
|
|
64
|
+
) -> Processor[Record, str]:
|
|
65
|
+
"""
|
|
66
|
+
A `Record -> str` renderer emitting one JSON object per record: structured logging.
|
|
67
|
+
|
|
68
|
+
The envelope (`timestamp` ISO-8601, `level`, `logger`, `message`) and every
|
|
69
|
+
field flat at the top level, so a log aggregator can index them directly; the
|
|
70
|
+
envelope wins a name clash, so a field cannot shadow `level` or `message`.
|
|
71
|
+
|
|
72
|
+
`timestamp` chooses how the time is encoded (default `iso_timestamp`; pass, say,
|
|
73
|
+
a `lambda when: when.timestamp()` for an epoch number). `exception` chooses how a
|
|
74
|
+
traceback is encoded: `exception_to_dict` (default, structured frames, stays
|
|
75
|
+
queryable) or `exception_to_text` (the flat traceback string), or any
|
|
76
|
+
`TracebackException -> <json value>` of your own. A
|
|
77
|
+
non-serializable *field* value is coerced with `str` (`default=str`): the log
|
|
78
|
+
JSON is machine-consumed, so a clean indexable value beats a Python `repr`, and
|
|
79
|
+
`str` still falls back to the `repr` form for an object without its own
|
|
80
|
+
`__str__` (`render_console` uses `repr`, matching its human-scanning medium). A
|
|
81
|
+
stray value therefore never tears down the log pipeline.
|
|
82
|
+
|
|
83
|
+
Optional and opt-in: the core ships no mandatory formatter (encoding is the app's
|
|
84
|
+
boundary), so compose this before a string sink yourself, e.g.
|
|
85
|
+
`compose(render_json(), offload(to_stream(sys.stdout)))`.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
async def render(record: Record) -> str:
|
|
89
|
+
payload: dict[str, object] = {
|
|
90
|
+
"timestamp": timestamp(record.timestamp),
|
|
91
|
+
"level": record.level_name,
|
|
92
|
+
"logger": record.logger,
|
|
93
|
+
"message": record.message,
|
|
94
|
+
}
|
|
95
|
+
for key, value in record.fields.items():
|
|
96
|
+
payload.setdefault(key, value)
|
|
97
|
+
if record.exception is not None:
|
|
98
|
+
payload["exception"] = exception(record.exception)
|
|
99
|
+
return json.dumps(payload, default=str)
|
|
100
|
+
|
|
101
|
+
return from_map(render)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def render_console(
|
|
105
|
+
*,
|
|
106
|
+
timestamp: Callable[[datetime], str] = iso_timestamp,
|
|
107
|
+
) -> Processor[Record, str]:
|
|
108
|
+
"""
|
|
109
|
+
A `Record -> str` renderer emitting one human-readable line per record.
|
|
110
|
+
|
|
111
|
+
`TIMESTAMP LEVEL logger "message" {key=value, ...}`: the message is quoted and
|
|
112
|
+
the fields grouped in braces (each value `repr`'d), so the free-text message and
|
|
113
|
+
the structured fields never blur into each other or the leading metadata. The
|
|
114
|
+
message is quoted with `json.dumps` and each field value is `repr`'d, so a control
|
|
115
|
+
character in either (a newline from a decoded request path, say) is escaped rather
|
|
116
|
+
than forging a new log line. The braces are omitted when there are no
|
|
117
|
+
fields. An exception is appended as its full traceback text (`exception_to_text`,
|
|
118
|
+
cause chain and all), indented, on following lines. `timestamp` chooses the time
|
|
119
|
+
format (default `iso_timestamp`). No coloring: wrap the line, or write your own
|
|
120
|
+
`from_map(Record -> str)`, if you want ANSI.
|
|
121
|
+
|
|
122
|
+
Optional and opt-in, like `render_json`: `compose(render_console(),
|
|
123
|
+
offload(to_stream(sys.stderr)))`.
|
|
124
|
+
"""
|
|
125
|
+
|
|
126
|
+
async def render(record: Record) -> str:
|
|
127
|
+
quoted = json.dumps(record.message, ensure_ascii=False) # pragma: no mutate - None acts as False here
|
|
128
|
+
parts = [
|
|
129
|
+
timestamp(record.timestamp),
|
|
130
|
+
record.level_name,
|
|
131
|
+
record.logger,
|
|
132
|
+
quoted,
|
|
133
|
+
]
|
|
134
|
+
if record.fields:
|
|
135
|
+
fields = ", ".join(f"{key}={value!r}" for key, value in record.fields.items())
|
|
136
|
+
parts.append("{" + fields + "}")
|
|
137
|
+
summary = " ".join(parts)
|
|
138
|
+
if record.exception is None:
|
|
139
|
+
return summary
|
|
140
|
+
body = "\n".join(f" {line}" if line else "" for line in exception_to_text(record.exception).splitlines())
|
|
141
|
+
return f"{summary}\n{body}"
|
|
142
|
+
|
|
143
|
+
return from_map(render)
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import os
|
|
5
|
+
import queue
|
|
6
|
+
from collections.abc import AsyncIterator
|
|
7
|
+
from collections.abc import Callable
|
|
8
|
+
from collections.abc import Iterator
|
|
9
|
+
from contextlib import asynccontextmanager
|
|
10
|
+
from datetime import UTC
|
|
11
|
+
from datetime import datetime
|
|
12
|
+
from datetime import time
|
|
13
|
+
from datetime import timedelta
|
|
14
|
+
from datetime import tzinfo
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import TextIO
|
|
17
|
+
|
|
18
|
+
from without.interfaces import Sink
|
|
19
|
+
from without.interfaces import Stream
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@asynccontextmanager
|
|
23
|
+
async def offload[T](work: Callable[[Iterator[list[T]]], None]) -> AsyncIterator[Sink[T]]:
|
|
24
|
+
"""
|
|
25
|
+
Run a blocking `work` on a dedicated thread, fed by the yielded async `Sink`.
|
|
26
|
+
|
|
27
|
+
The point is efficient blocking I/O under heavy logging. An async file library
|
|
28
|
+
(`aiofiles`, `anyio`) hops to a worker thread *per operation*, so a busy log
|
|
29
|
+
stream pays that round-trip on every write. Here a single long-lived thread
|
|
30
|
+
owns the resource and does *all* the I/O: `work` is plain blocking Python that
|
|
31
|
+
consumes the items, and the yielded `Sink` just drops each onto a thread-safe
|
|
32
|
+
queue the worker drains. No per-item thread hop, and the "processor body" stays
|
|
33
|
+
ordinary synchronous code.
|
|
34
|
+
|
|
35
|
+
Items arrive in *bursts*: each element of the iterator is everything available
|
|
36
|
+
on the queue at that instant (at least one item, blocking for the first). A
|
|
37
|
+
burst boundary is therefore exactly the moment the worker has caught up, which
|
|
38
|
+
is where a writer flushes: under load bursts are large and flushes are few, and
|
|
39
|
+
when idle each burst is a single item flushed at once. So durability needs no
|
|
40
|
+
flush-frequency knob; it falls out of the queue's own backlog.
|
|
41
|
+
|
|
42
|
+
Lifecycle is bounded by the `with` block: the thread starts on entry, and on
|
|
43
|
+
exit the queue is shut down so the worker drains what remains and then ends,
|
|
44
|
+
and the thread is joined (so a file is closed before the block returns). If
|
|
45
|
+
`work` raises, that surfaces when the block exits.
|
|
46
|
+
|
|
47
|
+
Nest this *outside* the consumer that drives the sink, so the worker outlives
|
|
48
|
+
the draining. With `capture`, that means `async with offload(...) as writer:`
|
|
49
|
+
around `async with capture(..., writer):`, not the other way round.
|
|
50
|
+
|
|
51
|
+
The queue is unbounded in this first cut: the async side never blocks or drops,
|
|
52
|
+
at the cost of growing memory if the worker cannot keep up with a sustained
|
|
53
|
+
burst (a stalled disk). A bounded, drop-counting variant is a deliberate
|
|
54
|
+
follow-up. The bidirectional case (a full `Processor` bridged onto a thread,
|
|
55
|
+
with an output queue as well) is intentionally out of scope; this covers the
|
|
56
|
+
terminal-sink need (writing) without that extra state.
|
|
57
|
+
"""
|
|
58
|
+
items: queue.Queue[T] = queue.Queue()
|
|
59
|
+
|
|
60
|
+
def drain() -> Iterator[list[T]]:
|
|
61
|
+
while True:
|
|
62
|
+
try:
|
|
63
|
+
batch = [items.get()]
|
|
64
|
+
except queue.ShutDown:
|
|
65
|
+
return
|
|
66
|
+
# Take everything queued at this instant into the burst. No Empty to catch and no
|
|
67
|
+
# get_nowait-in-a-try for control flow: the worker is the queue's *only* consumer, so
|
|
68
|
+
# qsize() is an exact count of items still present to take (a producer may add more,
|
|
69
|
+
# which simply waits for the next burst; nobody else removes), and each get succeeds.
|
|
70
|
+
batch.extend(items.get_nowait() for _ in range(items.qsize()))
|
|
71
|
+
yield batch
|
|
72
|
+
|
|
73
|
+
worker = asyncio.create_task(asyncio.to_thread(work, drain()))
|
|
74
|
+
|
|
75
|
+
async def sink(inputs: Stream[T]) -> None:
|
|
76
|
+
async for item in inputs:
|
|
77
|
+
items.put_nowait(item)
|
|
78
|
+
|
|
79
|
+
try:
|
|
80
|
+
yield sink
|
|
81
|
+
finally:
|
|
82
|
+
items.shutdown()
|
|
83
|
+
await worker
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def now_utc() -> datetime:
|
|
87
|
+
return datetime.now(UTC)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def at_times(*times: time, tz: tzinfo = UTC) -> Callable[[datetime], datetime]:
|
|
91
|
+
"""
|
|
92
|
+
A `schedule` for `to_rotating_file`: the next occurrence of any of these wall-clock times.
|
|
93
|
+
|
|
94
|
+
Given a moment, returns the soonest of `times` strictly after it, interpreted in
|
|
95
|
+
timezone `tz`. So `at_times(time(0, 0))` rotates daily at midnight `tz`, and
|
|
96
|
+
`at_times(time(0, 0), time(12, 0))` twice a day. The list of times is the
|
|
97
|
+
recurring cycle of rotation boundaries, resolved to a concrete datetime against
|
|
98
|
+
the *actual* current time each call, so nothing goes stale and a boundary missed
|
|
99
|
+
while idle collapses to a single rotation. Strictly-after (not at-or-after) is
|
|
100
|
+
what lets the writer advance past a boundary it just hit without looping on it.
|
|
101
|
+
"""
|
|
102
|
+
if not times:
|
|
103
|
+
raise ValueError("at_times requires at least one time-of-day")
|
|
104
|
+
ordered = sorted(times)
|
|
105
|
+
|
|
106
|
+
def schedule(after: datetime) -> datetime:
|
|
107
|
+
local = after.astimezone(tz)
|
|
108
|
+
for moment in ordered:
|
|
109
|
+
candidate = datetime.combine(local.date(), moment, tzinfo=tz)
|
|
110
|
+
if candidate > local:
|
|
111
|
+
return candidate
|
|
112
|
+
return datetime.combine(local.date() + timedelta(days=1), ordered[0], tzinfo=tz)
|
|
113
|
+
|
|
114
|
+
return schedule
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def to_rotating_file(
|
|
118
|
+
name: Callable[[int, datetime], Path],
|
|
119
|
+
*,
|
|
120
|
+
max_bytes: int | None = None,
|
|
121
|
+
max_age: timedelta | None = None,
|
|
122
|
+
schedule: Callable[[datetime], datetime] | None = None,
|
|
123
|
+
now: Callable[[], datetime] = now_utc,
|
|
124
|
+
) -> Callable[[Iterator[list[str]]], None]:
|
|
125
|
+
"""
|
|
126
|
+
A blocking worker (for `offload`) that appends lines to a file, rotating by size and/or time.
|
|
127
|
+
|
|
128
|
+
Rotation lives in the worker deliberately: it is the one place with all the
|
|
129
|
+
information to decide, timed with its own writes. Size is a function of the
|
|
130
|
+
bytes written, so a size limit cannot be an independent, decoupled trigger; the
|
|
131
|
+
worker keeps an exact byte count (writing UTF-8 bytes directly) rather than
|
|
132
|
+
polling a lagging on-disk size, and reads the clock the same way. So `max_bytes`
|
|
133
|
+
(size), `max_age` (elapsed since the file opened, a *relative* interval),
|
|
134
|
+
`schedule` (the next *absolute* wall-clock boundary, e.g. midnight, via a
|
|
135
|
+
next-boundary function such as `at_times`), and any combination (rotate on
|
|
136
|
+
whichever trips first) all work here, which the stdlib's separate size and time
|
|
137
|
+
handlers cannot do at once. At least one policy is required: with no limit the
|
|
138
|
+
file would never rotate (an unbounded file), which is not the affordance the
|
|
139
|
+
name suggests, so that case raises `ValueError` rather than being silently
|
|
140
|
+
allowed.
|
|
141
|
+
|
|
142
|
+
`schedule(t)` returns the next rotation boundary strictly after `t`; the writer
|
|
143
|
+
samples it at each open and rotates once `now()` reaches it (missed boundaries
|
|
144
|
+
collapse to one rotation). It is the pure, thread-synchronous form of "a stream
|
|
145
|
+
of rotation datetimes": the worker already owns the clock, so it generates the
|
|
146
|
+
stream itself rather than sampling an async one.
|
|
147
|
+
|
|
148
|
+
`name` turns a rotation index and the rotation time into a path
|
|
149
|
+
(`lambda i, when: directory / f"app.{i}.log"`); index 0 is the initial file,
|
|
150
|
+
appended to if it already exists, and the time is that file's open time drawn
|
|
151
|
+
from the same injected clock, so timestamped names stay consistent with the
|
|
152
|
+
rotation decision. `now` is the clock, injected so time-based rotation is
|
|
153
|
+
deterministic under test. This writes strings
|
|
154
|
+
(render a `Record` to text in a `from_map` in front), owns the newline framing,
|
|
155
|
+
and flushes at each burst boundary. The rotation decision is made *before* each
|
|
156
|
+
write (so a size limit is not overshot, bar a single line larger than
|
|
157
|
+
`max_bytes`), and it is inlined and short-circuiting: `now()` is read only when a
|
|
158
|
+
size check has not already forced the rotation and an age limit is set.
|
|
159
|
+
"""
|
|
160
|
+
if max_bytes is None and max_age is None and schedule is None:
|
|
161
|
+
raise ValueError("to_rotating_file needs at least one rotation policy: max_bytes, max_age, or schedule")
|
|
162
|
+
|
|
163
|
+
def next_boundary(opened: datetime) -> datetime | None:
|
|
164
|
+
return schedule(opened) if schedule is not None else None
|
|
165
|
+
|
|
166
|
+
def work(batches: Iterator[list[str]]) -> None:
|
|
167
|
+
index = 0
|
|
168
|
+
opened = now()
|
|
169
|
+
boundary = next_boundary(opened)
|
|
170
|
+
file = name(index, opened).open("ab")
|
|
171
|
+
written = file.seek(0, os.SEEK_END)
|
|
172
|
+
try:
|
|
173
|
+
for batch in batches:
|
|
174
|
+
for line in batch:
|
|
175
|
+
data = f"{line}\n".encode()
|
|
176
|
+
if written > 0 and (
|
|
177
|
+
(max_bytes is not None and written + len(data) > max_bytes)
|
|
178
|
+
or (max_age is not None and now() - opened >= max_age)
|
|
179
|
+
or (boundary is not None and now() >= boundary)
|
|
180
|
+
):
|
|
181
|
+
file.close()
|
|
182
|
+
index += 1
|
|
183
|
+
opened = now()
|
|
184
|
+
boundary = next_boundary(opened)
|
|
185
|
+
file = name(index, opened).open("ab")
|
|
186
|
+
written = file.seek(0, os.SEEK_END)
|
|
187
|
+
written += file.write(data)
|
|
188
|
+
file.flush()
|
|
189
|
+
finally:
|
|
190
|
+
file.close()
|
|
191
|
+
|
|
192
|
+
return work
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def to_stream(stream: TextIO) -> Callable[[Iterator[list[str]]], None]:
|
|
196
|
+
"""
|
|
197
|
+
A blocking worker (for `offload`) that appends each string as a line to an open text stream.
|
|
198
|
+
|
|
199
|
+
The destination-shaped sibling of `to_rotating_file`, for a stream the caller
|
|
200
|
+
already owns: `sys.stderr`, a socket's text wrapper, an in-memory buffer. It
|
|
201
|
+
writes strings (render a `Record` to text with a `from_map` in front), owns the
|
|
202
|
+
newline framing, and flushes at each burst boundary. Unlike `to_rotating_file` it
|
|
203
|
+
does *not* open or close the stream: the caller's stream outlives this and may be
|
|
204
|
+
shared, so closing it (`sys.stderr`, say) is not this worker's to do.
|
|
205
|
+
"""
|
|
206
|
+
|
|
207
|
+
def work(batches: Iterator[list[str]]) -> None:
|
|
208
|
+
for batch in batches:
|
|
209
|
+
for line in batch:
|
|
210
|
+
stream.write(line)
|
|
211
|
+
stream.write("\n")
|
|
212
|
+
stream.flush()
|
|
213
|
+
|
|
214
|
+
return work
|
without_logging-0.0.0/PKG-INFO
DELETED
|
@@ -1,9 +0,0 @@
|
|
|
1
|
-
[build-system]
|
|
2
|
-
requires = ["uv_build>=0.11,<0.12"]
|
|
3
|
-
build-backend = "uv_build"
|
|
4
|
-
|
|
5
|
-
[project]
|
|
6
|
-
name = "without-logging"
|
|
7
|
-
version = "0.0.0"
|
|
8
|
-
description = "Placeholder reserving the PyPI project name; the first real release supersedes it."
|
|
9
|
-
requires-python = ">=3.9"
|
|
File without changes
|