without-logging 0.0.0__tar.gz → 0.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,84 @@
1
+ Metadata-Version: 2.4
2
+ Name: without-logging
3
+ Version: 0.0.2
4
+ Summary: A without pipeline for logs: stdlib log records parsed into immutable values, filtered and enriched as processors, drained to a sink.
5
+ Author: Josh Karpel
6
+ Author-email: Josh Karpel <josh.karpel@gmail.com>
7
+ License-Expression: MIT
8
+ Classifier: Development Status :: 2 - Pre-Alpha
9
+ Classifier: Framework :: AsyncIO
10
+ Classifier: Intended Audience :: Developers
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3 :: Only
14
+ Classifier: Programming Language :: Python :: 3.14
15
+ Classifier: Topic :: Software Development :: Libraries
16
+ Classifier: Topic :: System :: Logging
17
+ Classifier: Typing :: Typed
18
+ Requires-Dist: without-core==0.0.2
19
+ Requires-Python: >=3.14
20
+ Description-Content-Type: text/markdown
21
+
22
+ # without-logging
23
+
24
+ A `without` pipeline for logs. Logger calls (yours and every third-party
25
+ library's) ultimately produce a stream of records; this package parses each one
26
+ into an immutable `Record` value, lets you filter and enrich them as ordinary
27
+ processors, and drains the result into a sink you own.
28
+
29
+ The two stdlib logging pain points it targets: the impenetrable, mutable,
30
+ noun-heavy configuration, and the monolithic handlers that bundle unrelated
31
+ decisions (when to flush, how to rotate, how to format) into one object. Here a
32
+ record is a value, each stage is a `Processor`, and the sink is whatever you
33
+ compose.
34
+
35
+ ```python
36
+ import logging
37
+
38
+ from without import compose, from_selector, from_sink
39
+ from without_logging import Level, at_least, capture
40
+
41
+
42
+ async def write(record):
43
+ print(f"{record.timestamp:%H:%M:%S} {record.level_name} {record.message}")
44
+
45
+
46
+ pipeline = compose(from_selector(at_least(Level.WARNING)), from_sink(write))
47
+
48
+ async with capture(pipeline): # attaches to the root logger for the block
49
+ logging.getLogger("app").warning("disk almost full", extra={"free_pct": 3})
50
+ ```
51
+
52
+ `capture` is the one impure piece: it activates a handler on the root logger,
53
+ turns the pushed records into a `Stream`, and runs your pipeline against them for
54
+ the life of the block. Everything upstream of the sink is pure and testable
55
+ without touching the logging machinery.
56
+
57
+ To write to a file without paying a thread hop per line, `offload` runs a blocking
58
+ writer on a single dedicated thread, fed by a queue that delivers items in bursts
59
+ (so the writer flushes when it catches up, no flush-frequency knob). Writers are
60
+ named by destination: `to_rotating_file` writes to a file (owning the byte count and
61
+ clock, rotating on any combination of size (`max_bytes`), a relative interval
62
+ (`max_age`), and absolute wall-clock boundaries (`schedule=at_times(...)`)), and
63
+ `to_stream` writes to a caller-owned text stream (`sys.stderr`, a socket) without
64
+ closing it. Both take strings (render a `Record` to text with a `from_map` in front):
65
+
66
+ ```python
67
+ from datetime import timedelta
68
+
69
+ from without import compose, from_map, from_selector
70
+ from without_logging import Level, at_least, offload, to_rotating_file
71
+
72
+ writer = to_rotating_file(lambda i, when: directory / f"app.{i}.log", max_bytes=64 << 20, max_age=timedelta(hours=1))
73
+ async with offload(writer) as sink:
74
+ lines = compose(from_map(render), sink) # Record -> str -> file
75
+ async with capture(compose(from_selector(at_least(Level.WARNING)), lines)):
76
+ ... # WARNING+ records written off the event loop, rotated by size and time
77
+ ```
78
+
79
+ See the
80
+ [`without-logging` guide](https://without.help/without-logging/)
81
+ (with the [API reference](https://without.help/reference/without_logging/))
82
+ for the design narrative: why stdlib becomes a one-way source, why filtering is
83
+ just the core `from_selector` builder, and where fan-out to multiple sinks slots
84
+ in.
@@ -0,0 +1,63 @@
1
+ # without-logging
2
+
3
+ A `without` pipeline for logs. Logger calls (yours and every third-party
4
+ library's) ultimately produce a stream of records; this package parses each one
5
+ into an immutable `Record` value, lets you filter and enrich them as ordinary
6
+ processors, and drains the result into a sink you own.
7
+
8
+ The two stdlib logging pain points it targets: the impenetrable, mutable,
9
+ noun-heavy configuration, and the monolithic handlers that bundle unrelated
10
+ decisions (when to flush, how to rotate, how to format) into one object. Here a
11
+ record is a value, each stage is a `Processor`, and the sink is whatever you
12
+ compose.
13
+
14
+ ```python
15
+ import logging
16
+
17
+ from without import compose, from_selector, from_sink
18
+ from without_logging import Level, at_least, capture
19
+
20
+
21
+ async def write(record):
22
+ print(f"{record.timestamp:%H:%M:%S} {record.level_name} {record.message}")
23
+
24
+
25
+ pipeline = compose(from_selector(at_least(Level.WARNING)), from_sink(write))
26
+
27
+ async with capture(pipeline): # attaches to the root logger for the block
28
+ logging.getLogger("app").warning("disk almost full", extra={"free_pct": 3})
29
+ ```
30
+
31
+ `capture` is the one impure piece: it activates a handler on the root logger,
32
+ turns the pushed records into a `Stream`, and runs your pipeline against them for
33
+ the life of the block. Everything upstream of the sink is pure and testable
34
+ without touching the logging machinery.
35
+
36
+ To write to a file without paying a thread hop per line, `offload` runs a blocking
37
+ writer on a single dedicated thread, fed by a queue that delivers items in bursts
38
+ (so the writer flushes when it catches up, no flush-frequency knob). Writers are
39
+ named by destination: `to_rotating_file` writes to a file (owning the byte count and
40
+ clock, rotating on any combination of size (`max_bytes`), a relative interval
41
+ (`max_age`), and absolute wall-clock boundaries (`schedule=at_times(...)`)), and
42
+ `to_stream` writes to a caller-owned text stream (`sys.stderr`, a socket) without
43
+ closing it. Both take strings (render a `Record` to text with a `from_map` in front):
44
+
45
+ ```python
46
+ from datetime import timedelta
47
+
48
+ from without import compose, from_map, from_selector
49
+ from without_logging import Level, at_least, offload, to_rotating_file
50
+
51
+ writer = to_rotating_file(lambda i, when: directory / f"app.{i}.log", max_bytes=64 << 20, max_age=timedelta(hours=1))
52
+ async with offload(writer) as sink:
53
+ lines = compose(from_map(render), sink) # Record -> str -> file
54
+ async with capture(compose(from_selector(at_least(Level.WARNING)), lines)):
55
+ ... # WARNING+ records written off the event loop, rotated by size and time
56
+ ```
57
+
58
+ See the
59
+ [`without-logging` guide](https://without.help/without-logging/)
60
+ (with the [API reference](https://without.help/reference/without_logging/))
61
+ for the design narrative: why stdlib becomes a one-way source, why filtering is
62
+ just the core `from_selector` builder, and where fan-out to multiple sinks slots
63
+ in.
@@ -0,0 +1,31 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.11.32,<0.12"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "without-logging"
7
+ version = "0.0.2"
8
+ description = "A without pipeline for logs: stdlib log records parsed into immutable values, filtered and enriched as processors, drained to a sink."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ requires-python = ">=3.14"
12
+ classifiers = [
13
+ "Development Status :: 2 - Pre-Alpha",
14
+ "Framework :: AsyncIO",
15
+ "Intended Audience :: Developers",
16
+ "Operating System :: OS Independent",
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3 :: Only",
19
+ "Programming Language :: Python :: 3.14",
20
+ "Topic :: Software Development :: Libraries",
21
+ "Topic :: System :: Logging",
22
+ "Typing :: Typed",
23
+ ]
24
+ dependencies = ["without-core==0.0.2"]
25
+
26
+ [[project.authors]]
27
+ name = "Josh Karpel"
28
+ email = "josh.karpel@gmail.com"
29
+
30
+ [tool.uv.sources.without-core]
31
+ workspace = true
@@ -0,0 +1,32 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.11.32,<0.12"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "without-logging"
7
+ version = "0.0.2"
8
+ description = "A without pipeline for logs: stdlib log records parsed into immutable values, filtered and enriched as processors, drained to a sink."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ authors = [
12
+ { name = "Josh Karpel", email = "josh.karpel@gmail.com" },
13
+ ]
14
+ requires-python = ">=3.14"
15
+ classifiers = [
16
+ "Development Status :: 2 - Pre-Alpha",
17
+ "Framework :: AsyncIO",
18
+ "Intended Audience :: Developers",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3 :: Only",
22
+ "Programming Language :: Python :: 3.14",
23
+ "Topic :: Software Development :: Libraries",
24
+ "Topic :: System :: Logging",
25
+ "Typing :: Typed",
26
+ ]
27
+ dependencies = [
28
+ "without-core==0.0.2",
29
+ ]
30
+
31
+ [tool.uv.sources]
32
+ without-core = { workspace = true }
@@ -0,0 +1,39 @@
1
+ from without_logging.capture import CaptureHandler
2
+ from without_logging.capture import capture
3
+ from without_logging.context import bind
4
+ from without_logging.context import merge_context
5
+ from without_logging.processors import add_fields
6
+ from without_logging.processors import at_least
7
+ from without_logging.record import Level
8
+ from without_logging.record import Record
9
+ from without_logging.record import parse_record
10
+ from without_logging.renderers import exception_to_dict
11
+ from without_logging.renderers import exception_to_text
12
+ from without_logging.renderers import iso_timestamp
13
+ from without_logging.renderers import render_console
14
+ from without_logging.renderers import render_json
15
+ from without_logging.sinks import at_times
16
+ from without_logging.sinks import offload
17
+ from without_logging.sinks import to_rotating_file
18
+ from without_logging.sinks import to_stream
19
+
20
+ __all__ = [
21
+ "CaptureHandler",
22
+ "Level",
23
+ "Record",
24
+ "add_fields",
25
+ "at_least",
26
+ "at_times",
27
+ "bind",
28
+ "capture",
29
+ "exception_to_dict",
30
+ "exception_to_text",
31
+ "iso_timestamp",
32
+ "merge_context",
33
+ "offload",
34
+ "parse_record",
35
+ "render_console",
36
+ "render_json",
37
+ "to_rotating_file",
38
+ "to_stream",
39
+ ]
@@ -0,0 +1,136 @@
1
+ from __future__ import annotations
2
+
3
+ import asyncio
4
+ import logging
5
+ import threading
6
+ from collections.abc import AsyncIterator
7
+ from collections.abc import Callable
8
+ from contextlib import asynccontextmanager
9
+
10
+ from without.interfaces import Sink
11
+ from without.wiring import stream_from_queue
12
+
13
+ from without_logging.context import merge_context
14
+ from without_logging.record import Record
15
+ from without_logging.record import parse_record
16
+
17
+
18
+ def parse_with_bound_context(log_record: logging.LogRecord) -> Record:
19
+ """`capture`'s default parser: `parse_record`, then `merge_context` folds in any bound context."""
20
+ return merge_context(parse_record(log_record))
21
+
22
+
23
+ class CaptureHandler(logging.Handler):
24
+ """
25
+ A stdlib handler that parses each `LogRecord` and offers it to an asyncio queue.
26
+
27
+ This is the one-way bridge at the ingestion edge. Stdlib `logging` is the
28
+ de-facto narrow waist every third-party library already writes to, so making
29
+ it a *source* (rather than fighting it) is the whole trick: a pushed
30
+ `LogRecord` is parsed to a `Record` here and dropped onto the queue that
31
+ `stream_from_queue` turns into the pull-based stream the pipeline consumes.
32
+ Control only ever flows outward from stdlib into `without`, never back.
33
+
34
+ A log call MAY happen off the event loop thread, so the parsed value is handed
35
+ to the loop with `call_soon_threadsafe`, which writes the loop's self-pipe to
36
+ wake it. A log call *on* the loop thread (an async handler logging) does not
37
+ need that wakeup, so those take the cheaper `call_soon`; the construction thread
38
+ is the loop thread (`capture` builds this right after `get_running_loop`), so its
39
+ id captured here identifies loop-thread emits. The queue is bounded, so a burst that
40
+ outruns the sink is dropped rather than growing memory without limit;
41
+ `dropped` counts how many so an operator can see it. The overflow policy is a
42
+ boundary decision the app owns: raise `capacity`, or pass `capacity=None` for
43
+ an unbounded queue that never drops (at the cost of the bound).
44
+ """
45
+
46
+ def __init__(
47
+ self,
48
+ loop: asyncio.AbstractEventLoop,
49
+ queue: asyncio.Queue[Record],
50
+ parse: Callable[[logging.LogRecord], Record],
51
+ ) -> None:
52
+ super().__init__()
53
+ self._loop = loop
54
+ self._loop_thread_id = threading.get_ident()
55
+ self._queue = queue
56
+ self._parse = parse
57
+ self.dropped = 0
58
+
59
+ def emit(self, log_record: logging.LogRecord) -> None:
60
+ record = self._parse(log_record)
61
+ if threading.get_ident() == self._loop_thread_id:
62
+ self._loop.call_soon(self._offer, record)
63
+ else:
64
+ self._loop.call_soon_threadsafe(self._offer, record)
65
+
66
+ def _offer(self, record: Record) -> None:
67
+ try:
68
+ self._queue.put_nowait(record)
69
+ except asyncio.QueueFull, asyncio.QueueShutDown:
70
+ self.dropped += 1
71
+
72
+
73
+ @asynccontextmanager
74
+ async def capture(
75
+ sink: Sink[Record],
76
+ *,
77
+ logger: logging.Logger | None = None,
78
+ level: int = logging.INFO,
79
+ capacity: int | None = 1024,
80
+ parse: Callable[[logging.LogRecord], Record] = parse_with_bound_context,
81
+ ) -> AsyncIterator[CaptureHandler]:
82
+ """
83
+ Capture stdlib log records into `sink` for the duration of the block.
84
+
85
+ The imperative shell of the pipeline, and the only impure piece: it attaches a
86
+ `CaptureHandler` to `logger` (the root logger by default, so third-party logs
87
+ are captured too), runs `sink(stream_from_queue(queue))` in a background task,
88
+ and on exit detaches, flushes pending records, shuts the queue so the stream
89
+ ends gracefully, and joins the task. `sink` is a fully wired pipeline, e.g.
90
+ `compose(from_selector(at_least(Level.WARNING)), from_sink(write))`;
91
+ everything upstream of the queue in it is pure and testable on its own.
92
+
93
+ `level` sets the capture threshold. stdlib applies *two* gates: the origin
94
+ logger's effective level (checked where a record is created, and defaulting to
95
+ WARNING on the root, which is what usually swallows INFO) and each handler's
96
+ own level. The logger gate is the load-bearing one: a sub-threshold record is
97
+ dropped at creation before any handler runs, so setting the handler level
98
+ alone would capture nothing new. So this temporarily lowers `logger`'s level
99
+ to `level` (restoring it on exit, scoped to the block) *and* sets the handler
100
+ to `level`, the latter keeping the threshold exact for records that propagate
101
+ up from a descendant logger pinned to its own lower level.
102
+
103
+ `parse` is the enrichment of the *emit phase*, the first of the pipeline's two
104
+ phases split by the queue: it turns each `LogRecord` into a `Record`
105
+ *synchronously* in the handler's `emit`, on the logging call's own task
106
+ (stdlib's shape), where the *sink phase* (`sink`, async, draining the queue) has
107
+ not yet begun. It defaults to `parse_with_bound_context` (`parse_record` then
108
+ `merge_context`), so any fields bound with `bind` are stamped on here, where
109
+ they are still live (the sink phase has left the caller's context). A custom
110
+ parser composes `merge_context` the same way, `merge_context(my_parse(log_record))`,
111
+ to keep binds, or omits it to opt out.
112
+
113
+ The handler itself is yielded so a caller can read `handler.dropped` after
114
+ the block to see how many records overflowed the queue.
115
+ """
116
+ target = logger if logger is not None else logging.getLogger()
117
+ loop = asyncio.get_running_loop()
118
+ queue: asyncio.Queue[Record] = asyncio.Queue(capacity if capacity is not None else 0)
119
+ handler = CaptureHandler(loop, queue, parse)
120
+ handler.setLevel(level)
121
+
122
+ async def run_sink() -> None:
123
+ await sink(stream_from_queue(queue))
124
+
125
+ previous_level = target.level
126
+ target.setLevel(level)
127
+ target.addHandler(handler)
128
+ drain = asyncio.create_task(run_sink())
129
+ try:
130
+ yield handler
131
+ finally:
132
+ target.removeHandler(handler)
133
+ target.setLevel(previous_level)
134
+ await asyncio.sleep(0) # let already-scheduled emits reach the queue before it shuts
135
+ queue.shutdown()
136
+ await drain
@@ -0,0 +1,61 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterator
4
+ from collections.abc import Mapping
5
+ from contextlib import contextmanager
6
+ from contextvars import ContextVar
7
+ from types import MappingProxyType
8
+
9
+ from without_logging.record import Record
10
+
11
+ _bound: ContextVar[Mapping[str, object]] = ContextVar("without_logging_bound_context", default=MappingProxyType({}))
12
+
13
+
14
+ @contextmanager
15
+ def bind(**fields: object) -> Iterator[None]:
16
+ """
17
+ Bind `fields` onto every record logged within this block, on this task.
18
+
19
+ The call-site half of structured context, the equivalent of structlog's
20
+ `bind_contextvars`: `with bind(request_id=...)` stamps those fields onto every
21
+ record logged inside the block. Scoping is explicit and lexical: the fields
22
+ are in effect for the block and gone after it, leaving no shared place
23
+ mutated. It writes a task-local `ContextVar`, so concurrent tasks each see
24
+ only their own binds, and nested binds accumulate (an inner `bind` merges onto
25
+ the outer and is undone on exit; a key set by both takes the inner value
26
+ inside the inner block).
27
+
28
+ Binding alone does nothing until the `capture` side reads it: pair this with
29
+ `merge_context`, which merges the bound fields onto each record at the parse
30
+ edge, where they are still live (the handler's `emit` runs synchronously on
31
+ the logging call's task). A downstream processor cannot read them, having left
32
+ the caller's context for the sink task.
33
+ """
34
+ merged = MappingProxyType({**_bound.get(), **fields})
35
+ token = _bound.set(merged)
36
+ try:
37
+ yield
38
+ finally:
39
+ _bound.reset(token)
40
+
41
+
42
+ def merge_context(record: Record) -> Record:
43
+ """
44
+ Merge the context bound by `bind` onto a record: the read half of call-site context.
45
+
46
+ A `Record -> Record` enrichment, *composed on top of* a parser rather than
47
+ wrapping it: `capture`'s default parser applies it after `parse_record`, and a
48
+ custom parser composes it the same way,
49
+ `merge_context(parse_record(log_record))`. A field the log call set explicitly
50
+ via `extra=` wins over a bound field of the same name (the more specific
51
+ value), so `bind` supplies defaults, never overrides a per-call value.
52
+
53
+ It reads the task-local `ContextVar` that `bind` writes, so it MUST run at the
54
+ parse edge, in the handler's `emit` on the caller's task, not as a pipeline
55
+ `Processor`: the pipeline runs in the sink task, having left that context,
56
+ where the bind would read as empty.
57
+ """
58
+ bound = _bound.get()
59
+ if not bound:
60
+ return record
61
+ return record.with_fields(**{key: value for key, value in bound.items() if key not in record.fields})
@@ -0,0 +1,48 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Awaitable
4
+ from collections.abc import Callable
5
+
6
+ from without.interfaces import Processor
7
+ from without.interfaces import from_map
8
+
9
+ from without_logging.record import Record
10
+
11
+
12
+ def at_least(level: int) -> Callable[[Record], Awaitable[bool]]:
13
+ """
14
+ A predicate matching records at `level` or more severe: the level threshold.
15
+
16
+ Pair it with the core `from_selector` to keep only those records (or
17
+ `from_filter` to drop them). It is `async` to match the one color of
18
+ predicate `from_selector`/`from_filter` expect, though the decision itself
19
+ just compares integers. Because a `Record.level` is a plain `int`, a `Level`
20
+ member reads naturally as the argument (`at_least(Level.WARNING)`), but any
21
+ numeric level works.
22
+ """
23
+
24
+ async def is_severe_enough(record: Record) -> bool:
25
+ return record.level >= level
26
+
27
+ return is_severe_enough
28
+
29
+
30
+ def add_fields(**fields: object) -> Processor[Record, Record]:
31
+ """
32
+ A processor that merges static `fields` onto every record: enrichment.
33
+
34
+ Enrichment *is* expressible as a `from_map` (one record in, one enriched
35
+ record out), so it uses the core builder directly. Enriching from a *shared
36
+ behavior* (a value the whole pipeline sees the latest of: the current config
37
+ revision, a sampling rate) is the same shape, reading it via `current()`
38
+ inside the step. Per-call-site context (a request or trace id, which is
39
+ task-local) is *not* recoverable here: the pipeline runs in the sink task,
40
+ having left the caller's context. Bind it at the edge instead with
41
+ `bind`/`merge_context`, which read it while `emit` is still on the caller's
42
+ task (see the guide).
43
+ """
44
+
45
+ async def enrich(record: Record) -> Record:
46
+ return record.with_fields(**fields)
47
+
48
+ return from_map(enrich)
@@ -0,0 +1,154 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ from collections.abc import Mapping
5
+ from dataclasses import dataclass
6
+ from dataclasses import replace
7
+ from datetime import UTC
8
+ from datetime import datetime
9
+ from enum import IntEnum
10
+ from traceback import TracebackException
11
+ from types import MappingProxyType
12
+
13
+
14
+ class Level(IntEnum):
15
+ """
16
+ The five standard severities as an ordered value, for writing filters.
17
+
18
+ Members are the stdlib numeric levels, so a `Record.level` (a plain `int`)
19
+ can be compared straight against them: `record.level >= Level.WARNING`.
20
+ Keeping `Record.level` an `int` rather than this enum is deliberate: a
21
+ third-party library that logs at a non-standard numeric level is carried
22
+ through as that number, not rejected, so the enum is a vocabulary for
23
+ predicates rather than a closed set the record must belong to.
24
+ """
25
+
26
+ DEBUG = logging.DEBUG
27
+ INFO = logging.INFO
28
+ WARNING = logging.WARNING
29
+ ERROR = logging.ERROR
30
+ CRITICAL = logging.CRITICAL
31
+
32
+
33
+ # The attributes stdlib's LogRecord factory sets. Anything on a record outside
34
+ # this set arrived via `extra=` (or was set by a filter) and is structured data,
35
+ # so it becomes a Record field. The names here split two ways: the caller-supplied
36
+ # content of the log call (`msg`/`args`, `levelno`, `name`, `exc_info`, `created`)
37
+ # is lifted into typed fields, and the rest is factory-synthesized ambient metadata
38
+ # (source location `pathname`/`lineno`/`funcName`; `thread`/`process`/`taskName`)
39
+ # that is deliberately dropped: it is gated by module flags (`logThreads`,
40
+ # `logProcesses`, `logAsyncioTasks`) and a swappable factory, so it may be `None` or
41
+ # absent, and it exists for stdlib's formatter `%(...)s` placeholders rather than as
42
+ # a stable per-event guarantee. A caller who wants some of it lifts it into `fields`
43
+ # with a custom `parse=` on `capture`.
44
+ RESERVED: frozenset[str] = frozenset(
45
+ {
46
+ "name",
47
+ "msg",
48
+ "args",
49
+ "levelname",
50
+ "levelno",
51
+ "pathname",
52
+ "filename",
53
+ "module",
54
+ "exc_info",
55
+ "exc_text",
56
+ "stack_info",
57
+ "lineno",
58
+ "funcName",
59
+ "created",
60
+ "msecs",
61
+ "relativeCreated",
62
+ "thread",
63
+ "threadName",
64
+ "processName",
65
+ "process",
66
+ "taskName",
67
+ "message",
68
+ "asctime",
69
+ }
70
+ )
71
+
72
+
73
+ @dataclass(frozen=True, slots=True)
74
+ class Record:
75
+ """
76
+ One log event as an immutable value, parsed from a stdlib `LogRecord`.
77
+
78
+ Where a `LogRecord` is a mutable bag that formatters and filters rewrite in
79
+ place, a `Record` is a value: it always means the same thing, so it can be
80
+ filtered, enriched, fanned out, and sunk without any stage disturbing
81
+ another's copy. Enrichment returns a *new* record (`with_fields`) rather than
82
+ mutating this one. `fields` is the record's structured data, exposed read-only:
83
+ *seeded* from the log call's `extra=` at the parse edge, then *grown* by
84
+ enrichment (`add_fields`, a `bind` merged by `merge_context`, any `with_fields`).
85
+ It is named for what it holds, not for `extra=` alone, because those later
86
+ sources write here too. `exception` is a `TracebackException` captured at the
87
+ ingestion edge (`None` when the event carried none): a *structured* value,
88
+ not a rendered string, so how the traceback is formatted stays a downstream
89
+ boundary the app owns (`"".join(exc.format())` for the stdlib text, or walk
90
+ `exc.stack` for JSON frames), the same way rendering the record itself is.
91
+ Extracting it drops all references to the live frames, so the record stays a
92
+ self-contained value with no traceback pinned to it.
93
+ """
94
+
95
+ timestamp: datetime
96
+ level: int
97
+ logger: str
98
+ message: str
99
+ exception: TracebackException | None
100
+ fields: Mapping[str, object]
101
+
102
+ @property
103
+ def level_name(self) -> str:
104
+ return logging.getLevelName(self.level)
105
+
106
+ def with_fields(self, **fields: object) -> Record:
107
+ """Return a copy of this record with `fields` merged onto its own."""
108
+ return replace(self, fields=MappingProxyType({**self.fields, **fields}))
109
+
110
+
111
+ def extract_exception(log_record: logging.LogRecord) -> TracebackException | None:
112
+ """
113
+ Capture a `LogRecord`'s exception as a structured value, or `None` when it carries none.
114
+
115
+ A live traceback is a *place*, not a value: it pins its frames and their
116
+ locals in memory, is mutable, and cannot ride the queue to another thread
117
+ safely (stdlib's own `QueueHandler` flattens it to a string for exactly this
118
+ reason). So the exception is turned into a value at the ingestion edge, while
119
+ it is still live, the same way `getMessage()` resolves the message. Unlike the
120
+ message, which has one sensible rendering, a traceback has many (locals,
121
+ chaining, style), so this keeps the *structure* rather than pre-rendering:
122
+ `TracebackException.from_exception` extracts frame summaries and drops all
123
+ references to the live frames, leaving a value the app can format however it
124
+ likes downstream (`format` for stdlib text, `stack` for JSON frames).
125
+ """
126
+ if log_record.exc_info:
127
+ _, exc_value, _ = log_record.exc_info
128
+ if exc_value is not None:
129
+ return TracebackException.from_exception(exc_value)
130
+ return None
131
+
132
+
133
+ def parse_record(log_record: logging.LogRecord) -> Record:
134
+ """
135
+ Parse a mutable stdlib `LogRecord` into an immutable `Record` value.
136
+
137
+ The ingestion boundary: this is where a pushed `LogRecord`, from the app's
138
+ own calls or any third-party library's, becomes the typed value the rest of
139
+ the pipeline works with. It is a pure function (`LogRecord -> Record`), so the
140
+ whole translation is testable without touching the logging machinery. The
141
+ structured `fields` are every attribute on the record outside the standard
142
+ envelope (`RESERVED`), which is exactly what `logging`'s `extra=` merges in;
143
+ the message is rendered and any exception is captured as a structured value
144
+ here (see `extract_exception`) so nothing live is carried past this point.
145
+ """
146
+ fields = {key: value for key, value in log_record.__dict__.items() if key not in RESERVED}
147
+ return Record(
148
+ timestamp=datetime.fromtimestamp(log_record.created, tz=UTC),
149
+ level=log_record.levelno,
150
+ logger=log_record.name,
151
+ message=log_record.getMessage(),
152
+ exception=extract_exception(log_record),
153
+ fields=MappingProxyType(fields),
154
+ )
@@ -0,0 +1,143 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections.abc import Callable
5
+ from datetime import datetime
6
+ from traceback import TracebackException
7
+
8
+ from without.interfaces import Processor
9
+ from without.interfaces import from_map
10
+
11
+ from without_logging.record import Record
12
+
13
+
14
+ def exception_to_dict(exception: TracebackException) -> dict[str, object]:
15
+ """
16
+ Convert a captured `TracebackException` into a JSON-serializable dict: structured traceback.
17
+
18
+ Each frame becomes `{file, line, function, code}` (from the extracted
19
+ `FrameSummary` values, no live frames), and a chained exception is nested under
20
+ `cause`, following Python's own rule: an explicit `raise ... from` (`__cause__`)
21
+ takes precedence, else the implicit `__context__` unless it was suppressed. This
22
+ is the read half `render_json` uses, exposed so a custom JSON renderer can reuse
23
+ it.
24
+ """
25
+ cause: dict[str, object] | None = None
26
+ if exception.__cause__ is not None:
27
+ cause = exception_to_dict(exception.__cause__)
28
+ elif exception.__context__ is not None and not exception.__suppress_context__:
29
+ cause = exception_to_dict(exception.__context__)
30
+ result: dict[str, object] = {
31
+ "type": exception.exc_type_str,
32
+ "message": "".join(exception.format_exception_only()).strip(),
33
+ "frames": [
34
+ {"file": frame.filename, "line": frame.lineno, "function": frame.name, "code": frame.line}
35
+ for frame in exception.stack
36
+ ],
37
+ }
38
+ if cause is not None:
39
+ result["cause"] = cause
40
+ return result
41
+
42
+
43
+ def exception_to_text(exception: TracebackException) -> str:
44
+ """
45
+ Render a captured `TracebackException` to its full traceback text.
46
+
47
+ stdlib's own multi-line rendering (`format`), chained cause/context and all,
48
+ as a single string. The flat-string alternative to the structured
49
+ `exception_to_dict` when encoding a record's exception, and what
50
+ `render_console` uses.
51
+ """
52
+ return "".join(exception.format()).rstrip("\n")
53
+
54
+
55
+ def iso_timestamp(when: datetime) -> str:
56
+ """The default timestamp format for the renderers: ISO 8601 (`datetime.isoformat`)."""
57
+ return when.isoformat()
58
+
59
+
60
+ def render_json(
61
+ *,
62
+ timestamp: Callable[[datetime], object] = iso_timestamp,
63
+ exception: Callable[[TracebackException], object] = exception_to_dict,
64
+ ) -> Processor[Record, str]:
65
+ """
66
+ A `Record -> str` renderer emitting one JSON object per record: structured logging.
67
+
68
+ The envelope (`timestamp` ISO-8601, `level`, `logger`, `message`) and every
69
+ field flat at the top level, so a log aggregator can index them directly; the
70
+ envelope wins a name clash, so a field cannot shadow `level` or `message`.
71
+
72
+ `timestamp` chooses how the time is encoded (default `iso_timestamp`; pass, say,
73
+ a `lambda when: when.timestamp()` for an epoch number). `exception` chooses how a
74
+ traceback is encoded: `exception_to_dict` (default, structured frames, stays
75
+ queryable) or `exception_to_text` (the flat traceback string), or any
76
+ `TracebackException -> <json value>` of your own. A
77
+ non-serializable *field* value is coerced with `str` (`default=str`): the log
78
+ JSON is machine-consumed, so a clean indexable value beats a Python `repr`, and
79
+ `str` still falls back to the `repr` form for an object without its own
80
+ `__str__` (`render_console` uses `repr`, matching its human-scanning medium). A
81
+ stray value therefore never tears down the log pipeline.
82
+
83
+ Optional and opt-in: the core ships no mandatory formatter (encoding is the app's
84
+ boundary), so compose this before a string sink yourself, e.g.
85
+ `compose(render_json(), offload(to_stream(sys.stdout)))`.
86
+ """
87
+
88
+ async def render(record: Record) -> str:
89
+ payload: dict[str, object] = {
90
+ "timestamp": timestamp(record.timestamp),
91
+ "level": record.level_name,
92
+ "logger": record.logger,
93
+ "message": record.message,
94
+ }
95
+ for key, value in record.fields.items():
96
+ payload.setdefault(key, value)
97
+ if record.exception is not None:
98
+ payload["exception"] = exception(record.exception)
99
+ return json.dumps(payload, default=str)
100
+
101
+ return from_map(render)
102
+
103
+
104
+ def render_console(
105
+ *,
106
+ timestamp: Callable[[datetime], str] = iso_timestamp,
107
+ ) -> Processor[Record, str]:
108
+ """
109
+ A `Record -> str` renderer emitting one human-readable line per record.
110
+
111
+ `TIMESTAMP LEVEL logger "message" {key=value, ...}`: the message is quoted and
112
+ the fields grouped in braces (each value `repr`'d), so the free-text message and
113
+ the structured fields never blur into each other or the leading metadata. The
114
+ message is quoted with `json.dumps` and each field value is `repr`'d, so a control
115
+ character in either (a newline from a decoded request path, say) is escaped rather
116
+ than forging a new log line. The braces are omitted when there are no
117
+ fields. An exception is appended as its full traceback text (`exception_to_text`,
118
+ cause chain and all), indented, on following lines. `timestamp` chooses the time
119
+ format (default `iso_timestamp`). No coloring: wrap the line, or write your own
120
+ `from_map(Record -> str)`, if you want ANSI.
121
+
122
+ Optional and opt-in, like `render_json`: `compose(render_console(),
123
+ offload(to_stream(sys.stderr)))`.
124
+ """
125
+
126
+ async def render(record: Record) -> str:
127
+ quoted = json.dumps(record.message, ensure_ascii=False) # pragma: no mutate - None acts as False here
128
+ parts = [
129
+ timestamp(record.timestamp),
130
+ record.level_name,
131
+ record.logger,
132
+ quoted,
133
+ ]
134
+ if record.fields:
135
+ fields = ", ".join(f"{key}={value!r}" for key, value in record.fields.items())
136
+ parts.append("{" + fields + "}")
137
+ summary = " ".join(parts)
138
+ if record.exception is None:
139
+ return summary
140
+ body = "\n".join(f" {line}" if line else "" for line in exception_to_text(record.exception).splitlines())
141
+ return f"{summary}\n{body}"
142
+
143
+ return from_map(render)
@@ -0,0 +1,214 @@
1
+ from __future__ import annotations
2
+
3
+ import asyncio
4
+ import os
5
+ import queue
6
+ from collections.abc import AsyncIterator
7
+ from collections.abc import Callable
8
+ from collections.abc import Iterator
9
+ from contextlib import asynccontextmanager
10
+ from datetime import UTC
11
+ from datetime import datetime
12
+ from datetime import time
13
+ from datetime import timedelta
14
+ from datetime import tzinfo
15
+ from pathlib import Path
16
+ from typing import TextIO
17
+
18
+ from without.interfaces import Sink
19
+ from without.interfaces import Stream
20
+
21
+
22
+ @asynccontextmanager
23
+ async def offload[T](work: Callable[[Iterator[list[T]]], None]) -> AsyncIterator[Sink[T]]:
24
+ """
25
+ Run a blocking `work` on a dedicated thread, fed by the yielded async `Sink`.
26
+
27
+ The point is efficient blocking I/O under heavy logging. An async file library
28
+ (`aiofiles`, `anyio`) hops to a worker thread *per operation*, so a busy log
29
+ stream pays that round-trip on every write. Here a single long-lived thread
30
+ owns the resource and does *all* the I/O: `work` is plain blocking Python that
31
+ consumes the items, and the yielded `Sink` just drops each onto a thread-safe
32
+ queue the worker drains. No per-item thread hop, and the "processor body" stays
33
+ ordinary synchronous code.
34
+
35
+ Items arrive in *bursts*: each element of the iterator is everything available
36
+ on the queue at that instant (at least one item, blocking for the first). A
37
+ burst boundary is therefore exactly the moment the worker has caught up, which
38
+ is where a writer flushes: under load bursts are large and flushes are few, and
39
+ when idle each burst is a single item flushed at once. So durability needs no
40
+ flush-frequency knob; it falls out of the queue's own backlog.
41
+
42
+ Lifecycle is bounded by the `with` block: the thread starts on entry, and on
43
+ exit the queue is shut down so the worker drains what remains and then ends,
44
+ and the thread is joined (so a file is closed before the block returns). If
45
+ `work` raises, that surfaces when the block exits.
46
+
47
+ Nest this *outside* the consumer that drives the sink, so the worker outlives
48
+ the draining. With `capture`, that means `async with offload(...) as writer:`
49
+ around `async with capture(..., writer):`, not the other way round.
50
+
51
+ The queue is unbounded in this first cut: the async side never blocks or drops,
52
+ at the cost of growing memory if the worker cannot keep up with a sustained
53
+ burst (a stalled disk). A bounded, drop-counting variant is a deliberate
54
+ follow-up. The bidirectional case (a full `Processor` bridged onto a thread,
55
+ with an output queue as well) is intentionally out of scope; this covers the
56
+ terminal-sink need (writing) without that extra state.
57
+ """
58
+ items: queue.Queue[T] = queue.Queue()
59
+
60
+ def drain() -> Iterator[list[T]]:
61
+ while True:
62
+ try:
63
+ batch = [items.get()]
64
+ except queue.ShutDown:
65
+ return
66
+ # Take everything queued at this instant into the burst. No Empty to catch and no
67
+ # get_nowait-in-a-try for control flow: the worker is the queue's *only* consumer, so
68
+ # qsize() is an exact count of items still present to take (a producer may add more,
69
+ # which simply waits for the next burst; nobody else removes), and each get succeeds.
70
+ batch.extend(items.get_nowait() for _ in range(items.qsize()))
71
+ yield batch
72
+
73
+ worker = asyncio.create_task(asyncio.to_thread(work, drain()))
74
+
75
+ async def sink(inputs: Stream[T]) -> None:
76
+ async for item in inputs:
77
+ items.put_nowait(item)
78
+
79
+ try:
80
+ yield sink
81
+ finally:
82
+ items.shutdown()
83
+ await worker
84
+
85
+
86
+ def now_utc() -> datetime:
87
+ return datetime.now(UTC)
88
+
89
+
90
+ def at_times(*times: time, tz: tzinfo = UTC) -> Callable[[datetime], datetime]:
91
+ """
92
+ A `schedule` for `to_rotating_file`: the next occurrence of any of these wall-clock times.
93
+
94
+ Given a moment, returns the soonest of `times` strictly after it, interpreted in
95
+ timezone `tz`. So `at_times(time(0, 0))` rotates daily at midnight `tz`, and
96
+ `at_times(time(0, 0), time(12, 0))` twice a day. The list of times is the
97
+ recurring cycle of rotation boundaries, resolved to a concrete datetime against
98
+ the *actual* current time each call, so nothing goes stale and a boundary missed
99
+ while idle collapses to a single rotation. Strictly-after (not at-or-after) is
100
+ what lets the writer advance past a boundary it just hit without looping on it.
101
+ """
102
+ if not times:
103
+ raise ValueError("at_times requires at least one time-of-day")
104
+ ordered = sorted(times)
105
+
106
+ def schedule(after: datetime) -> datetime:
107
+ local = after.astimezone(tz)
108
+ for moment in ordered:
109
+ candidate = datetime.combine(local.date(), moment, tzinfo=tz)
110
+ if candidate > local:
111
+ return candidate
112
+ return datetime.combine(local.date() + timedelta(days=1), ordered[0], tzinfo=tz)
113
+
114
+ return schedule
115
+
116
+
117
+ def to_rotating_file(
118
+ name: Callable[[int, datetime], Path],
119
+ *,
120
+ max_bytes: int | None = None,
121
+ max_age: timedelta | None = None,
122
+ schedule: Callable[[datetime], datetime] | None = None,
123
+ now: Callable[[], datetime] = now_utc,
124
+ ) -> Callable[[Iterator[list[str]]], None]:
125
+ """
126
+ A blocking worker (for `offload`) that appends lines to a file, rotating by size and/or time.
127
+
128
+ Rotation lives in the worker deliberately: it is the one place with all the
129
+ information to decide, timed with its own writes. Size is a function of the
130
+ bytes written, so a size limit cannot be an independent, decoupled trigger; the
131
+ worker keeps an exact byte count (writing UTF-8 bytes directly) rather than
132
+ polling a lagging on-disk size, and reads the clock the same way. So `max_bytes`
133
+ (size), `max_age` (elapsed since the file opened, a *relative* interval),
134
+ `schedule` (the next *absolute* wall-clock boundary, e.g. midnight, via a
135
+ next-boundary function such as `at_times`), and any combination (rotate on
136
+ whichever trips first) all work here, which the stdlib's separate size and time
137
+ handlers cannot do at once. At least one policy is required: with no limit the
138
+ file would never rotate (an unbounded file), which is not the affordance the
139
+ name suggests, so that case raises `ValueError` rather than being silently
140
+ allowed.
141
+
142
+ `schedule(t)` returns the next rotation boundary strictly after `t`; the writer
143
+ samples it at each open and rotates once `now()` reaches it (missed boundaries
144
+ collapse to one rotation). It is the pure, thread-synchronous form of "a stream
145
+ of rotation datetimes": the worker already owns the clock, so it generates the
146
+ stream itself rather than sampling an async one.
147
+
148
+ `name` turns a rotation index and the rotation time into a path
149
+ (`lambda i, when: directory / f"app.{i}.log"`); index 0 is the initial file,
150
+ appended to if it already exists, and the time is that file's open time drawn
151
+ from the same injected clock, so timestamped names stay consistent with the
152
+ rotation decision. `now` is the clock, injected so time-based rotation is
153
+ deterministic under test. This writes strings
154
+ (render a `Record` to text in a `from_map` in front), owns the newline framing,
155
+ and flushes at each burst boundary. The rotation decision is made *before* each
156
+ write (so a size limit is not overshot, bar a single line larger than
157
+ `max_bytes`), and it is inlined and short-circuiting: `now()` is read only when a
158
+ size check has not already forced the rotation and an age limit is set.
159
+ """
160
+ if max_bytes is None and max_age is None and schedule is None:
161
+ raise ValueError("to_rotating_file needs at least one rotation policy: max_bytes, max_age, or schedule")
162
+
163
+ def next_boundary(opened: datetime) -> datetime | None:
164
+ return schedule(opened) if schedule is not None else None
165
+
166
+ def work(batches: Iterator[list[str]]) -> None:
167
+ index = 0
168
+ opened = now()
169
+ boundary = next_boundary(opened)
170
+ file = name(index, opened).open("ab")
171
+ written = file.seek(0, os.SEEK_END)
172
+ try:
173
+ for batch in batches:
174
+ for line in batch:
175
+ data = f"{line}\n".encode()
176
+ if written > 0 and (
177
+ (max_bytes is not None and written + len(data) > max_bytes)
178
+ or (max_age is not None and now() - opened >= max_age)
179
+ or (boundary is not None and now() >= boundary)
180
+ ):
181
+ file.close()
182
+ index += 1
183
+ opened = now()
184
+ boundary = next_boundary(opened)
185
+ file = name(index, opened).open("ab")
186
+ written = file.seek(0, os.SEEK_END)
187
+ written += file.write(data)
188
+ file.flush()
189
+ finally:
190
+ file.close()
191
+
192
+ return work
193
+
194
+
195
+ def to_stream(stream: TextIO) -> Callable[[Iterator[list[str]]], None]:
196
+ """
197
+ A blocking worker (for `offload`) that appends each string as a line to an open text stream.
198
+
199
+ The destination-shaped sibling of `to_rotating_file`, for a stream the caller
200
+ already owns: `sys.stderr`, a socket's text wrapper, an in-memory buffer. It
201
+ writes strings (render a `Record` to text with a `from_map` in front), owns the
202
+ newline framing, and flushes at each burst boundary. Unlike `to_rotating_file` it
203
+ does *not* open or close the stream: the caller's stream outlives this and may be
204
+ shared, so closing it (`sys.stderr`, say) is not this worker's to do.
205
+ """
206
+
207
+ def work(batches: Iterator[list[str]]) -> None:
208
+ for batch in batches:
209
+ for line in batch:
210
+ stream.write(line)
211
+ stream.write("\n")
212
+ stream.flush()
213
+
214
+ return work
@@ -1,5 +0,0 @@
1
- Metadata-Version: 2.3
2
- Name: without-logging
3
- Version: 0.0.0
4
- Summary: Placeholder reserving the PyPI project name; the first real release supersedes it.
5
- Requires-Python: >=3.9
@@ -1,9 +0,0 @@
1
- [build-system]
2
- requires = ["uv_build>=0.11,<0.12"]
3
- build-backend = "uv_build"
4
-
5
- [project]
6
- name = "without-logging"
7
- version = "0.0.0"
8
- description = "Placeholder reserving the PyPI project name; the first real release supersedes it."
9
- requires-python = ">=3.9"