ledgence-worker 0.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ledgence/worker/__init__.py +81 -0
- ledgence/worker/_logging.py +132 -0
- ledgence/worker/_observations.py +88 -0
- ledgence/worker/_protocol.py +101 -0
- ledgence/worker/bootstrap.py +489 -0
- ledgence/worker/otel.py +49 -0
- ledgence/worker/py.typed +0 -0
- ledgence/worker/workflow.py +1339 -0
- ledgence_worker-0.4.1.dist-info/METADATA +563 -0
- ledgence_worker-0.4.1.dist-info/RECORD +12 -0
- ledgence_worker-0.4.1.dist-info/WHEEL +4 -0
- ledgence_worker-0.4.1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Small, dependency-free helpers for Ledgence Python programs (MIT)."""
|
|
2
|
+
|
|
3
|
+
from contextvars import ContextVar
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True)
|
|
9
|
+
class TraceContext:
|
|
10
|
+
"""Invocation-local W3C carrier, separate from the immutable event origin."""
|
|
11
|
+
|
|
12
|
+
traceparent: str
|
|
13
|
+
tracestate: Optional[str] = None
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass(frozen=True)
|
|
17
|
+
class InvocationContext:
|
|
18
|
+
"""Envelope identity for the invocation currently handled by this process."""
|
|
19
|
+
|
|
20
|
+
event_id: str
|
|
21
|
+
attempt_id: str
|
|
22
|
+
source: Optional[str] = None
|
|
23
|
+
tenant_id: Optional[str] = None
|
|
24
|
+
namespace: Optional[str] = None
|
|
25
|
+
run_id: Optional[str] = None
|
|
26
|
+
task_id: Optional[str] = None
|
|
27
|
+
attempt_no: Optional[int] = None
|
|
28
|
+
processing_context: Optional[TraceContext] = None
|
|
29
|
+
workflow_id: Optional[str] = None
|
|
30
|
+
activation_id: Optional[str] = None
|
|
31
|
+
parent_workflow_id: Optional[str] = None
|
|
32
|
+
root_workflow_id: Optional[str] = None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
_invocation: ContextVar[Optional[InvocationContext]] = ContextVar(
|
|
36
|
+
"ledgence_invocation", default=None
|
|
37
|
+
)
|
|
38
|
+
_shutdown_callbacks = []
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def current_invocation() -> InvocationContext:
|
|
42
|
+
"""Return this invocation's identifiers; fail when no invocation is active."""
|
|
43
|
+
value = _invocation.get()
|
|
44
|
+
if value is None:
|
|
45
|
+
raise RuntimeError("no Ledgence invocation is active")
|
|
46
|
+
return value
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def register_shutdown(callback):
|
|
50
|
+
"""Call an application-owned provider's shutdown before the closing ACK.
|
|
51
|
+
|
|
52
|
+
Called once on graceful protocol shutdown. The parent process's shutdown
|
|
53
|
+
deadline bounds callbacks; cancellation/crashes do not promise a flush.
|
|
54
|
+
"""
|
|
55
|
+
if not callable(callback):
|
|
56
|
+
raise TypeError("shutdown callback must be callable")
|
|
57
|
+
if callback not in _shutdown_callbacks:
|
|
58
|
+
_shutdown_callbacks.append(callback)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _shutdown():
|
|
62
|
+
import traceback
|
|
63
|
+
|
|
64
|
+
callbacks = _shutdown_callbacks[:]
|
|
65
|
+
_shutdown_callbacks.clear()
|
|
66
|
+
for callback in callbacks:
|
|
67
|
+
try:
|
|
68
|
+
callback()
|
|
69
|
+
except Exception:
|
|
70
|
+
traceback.print_exc()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def get_logger(name):
|
|
74
|
+
"""Return a standard logger with a contextual, best-effort Ledgence handler."""
|
|
75
|
+
from ._logging import get_logger as configured_logger
|
|
76
|
+
|
|
77
|
+
return configured_logger(name)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
__all__ = ["InvocationContext", "TraceContext", "current_invocation", "get_logger",
|
|
81
|
+
"register_shutdown"]
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""Bounded best-effort structured log snapshots (MIT)."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
import threading
|
|
6
|
+
from itertools import islice
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
|
|
9
|
+
from . import _invocation
|
|
10
|
+
|
|
11
|
+
_sink = None
|
|
12
|
+
_lock = threading.Lock()
|
|
13
|
+
_record_lock = threading.Lock()
|
|
14
|
+
_MAX_TEXT = 4096
|
|
15
|
+
_MAX_ATTRIBUTES = 32
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _bounded(value, depth=0, budget=None):
|
|
19
|
+
# Limit the whole attribute tree, not just each branch. A branching object
|
|
20
|
+
# must not multiply per-value allowances into an unbounded encoded record.
|
|
21
|
+
if budget is None:
|
|
22
|
+
budget = [64, _MAX_TEXT]
|
|
23
|
+
budget[0] -= 1
|
|
24
|
+
if budget[0] < 0:
|
|
25
|
+
return "[truncated]"
|
|
26
|
+
if value is None or isinstance(value, (bool, int, float)):
|
|
27
|
+
return value
|
|
28
|
+
if isinstance(value, str):
|
|
29
|
+
text = value[:max(0, budget[1])]
|
|
30
|
+
budget[1] -= len(text)
|
|
31
|
+
return text
|
|
32
|
+
if depth >= 4:
|
|
33
|
+
return "[truncated]"
|
|
34
|
+
if isinstance(value, dict):
|
|
35
|
+
result = {}
|
|
36
|
+
for key, item in islice(value.items(), _MAX_ATTRIBUTES):
|
|
37
|
+
if budget[0] <= 0:
|
|
38
|
+
break
|
|
39
|
+
if isinstance(key, str):
|
|
40
|
+
bounded_key = key[:min(256, max(0, budget[1]))]
|
|
41
|
+
budget[1] -= len(bounded_key)
|
|
42
|
+
result[bounded_key] = _bounded(item, depth + 1, budget)
|
|
43
|
+
return result
|
|
44
|
+
if isinstance(value, (list, tuple)):
|
|
45
|
+
result = []
|
|
46
|
+
for item in value[:_MAX_ATTRIBUTES]:
|
|
47
|
+
if budget[0] <= 0:
|
|
48
|
+
break
|
|
49
|
+
result.append(_bounded(item, depth + 1, budget))
|
|
50
|
+
return result
|
|
51
|
+
return "[unsupported]"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _snapshot(record):
|
|
55
|
+
from .otel import _active_ids, _api
|
|
56
|
+
|
|
57
|
+
frame = {"v": getattr(_sink, "protocol_version", 2), "type": "log",
|
|
58
|
+
"time": datetime.fromtimestamp(record.created, timezone.utc).isoformat(),
|
|
59
|
+
"severity": record.levelname[:32], "logger": record.name[:256],
|
|
60
|
+
"message": record.getMessage()[:_MAX_TEXT],
|
|
61
|
+
"attributes": _bounded(getattr(record, "attributes", {}))}
|
|
62
|
+
invocation = _invocation.get()
|
|
63
|
+
if invocation is not None:
|
|
64
|
+
frame["invocation"] = {key: getattr(invocation, key) for key in (
|
|
65
|
+
"event_id", "attempt_id", "source", "tenant_id", "namespace",
|
|
66
|
+
"run_id", "task_id", "attempt_no") if getattr(invocation, key) is not None}
|
|
67
|
+
if frame["v"] >= 3:
|
|
68
|
+
frame["invocation"].update({key: getattr(invocation, key) for key in
|
|
69
|
+
("workflow_id", "activation_id")
|
|
70
|
+
if getattr(invocation, key) is not None})
|
|
71
|
+
if invocation.parent_workflow_id is not None:
|
|
72
|
+
frame["invocation"].update(parent_workflow_id=invocation.parent_workflow_id,
|
|
73
|
+
root_workflow_id=invocation.root_workflow_id)
|
|
74
|
+
active = _active_ids()
|
|
75
|
+
if active is None and _api is None and invocation is not None:
|
|
76
|
+
carrier = invocation.processing_context
|
|
77
|
+
if carrier is not None:
|
|
78
|
+
parts = carrier.traceparent.split("-")
|
|
79
|
+
active = {"trace_id": parts[1], "span_id": parts[2]}
|
|
80
|
+
if active is not None:
|
|
81
|
+
frame.update(active)
|
|
82
|
+
return frame
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class _Handler(logging.Handler):
|
|
86
|
+
def emit(self, record):
|
|
87
|
+
sink = _sink
|
|
88
|
+
if sink is None:
|
|
89
|
+
return
|
|
90
|
+
try:
|
|
91
|
+
# Propagation can visit several Ledgence handlers for the same
|
|
92
|
+
# LogRecord. Claim that event once, including a best-effort drop,
|
|
93
|
+
# without suppressing application handlers or retaining the sink.
|
|
94
|
+
# Re-dispatching the same record remains the same telemetry event;
|
|
95
|
+
# each ordinary logger call creates its own independent record.
|
|
96
|
+
with _record_lock:
|
|
97
|
+
if record.__dict__.get("_ledgence_log_processed", False):
|
|
98
|
+
return
|
|
99
|
+
record.__dict__["_ledgence_log_processed"] = True
|
|
100
|
+
# Encoding happens synchronously: mutable attributes, context changes,
|
|
101
|
+
# and later delivery cannot relabel a record as another invocation.
|
|
102
|
+
frame = _snapshot(record)
|
|
103
|
+
limit = sink.log_limit
|
|
104
|
+
mandatory = dict(frame, message="", attributes={})
|
|
105
|
+
if len(json.dumps(mandatory, ensure_ascii=False, allow_nan=False,
|
|
106
|
+
separators=(",", ":")).encode("utf-8")) + 1 > limit:
|
|
107
|
+
sink.drop_log()
|
|
108
|
+
return
|
|
109
|
+
# Shrink optional content, never identity. Encoding includes newline.
|
|
110
|
+
for _ in range(15):
|
|
111
|
+
encoded = json.dumps(frame, ensure_ascii=False, allow_nan=False,
|
|
112
|
+
separators=(",", ":")).encode("utf-8") + b"\n"
|
|
113
|
+
if len(encoded) <= limit:
|
|
114
|
+
sink.offer_log(encoded)
|
|
115
|
+
return
|
|
116
|
+
frame["attributes"] = {}
|
|
117
|
+
frame["message"] = frame["message"][:len(frame["message"]) // 2]
|
|
118
|
+
sink.drop_log()
|
|
119
|
+
except Exception:
|
|
120
|
+
# Malformed messages/attributes are telemetry loss, never task failure.
|
|
121
|
+
sink.drop_log()
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def get_logger(name):
|
|
125
|
+
logger = logging.getLogger(name)
|
|
126
|
+
with _lock:
|
|
127
|
+
if not any(isinstance(handler, _Handler) for handler in logger.handlers):
|
|
128
|
+
logger.addHandler(_Handler())
|
|
129
|
+
if logger.level == logging.NOTSET:
|
|
130
|
+
logger.setLevel(logging.INFO)
|
|
131
|
+
# Keep propagation and existing application/root handlers untouched.
|
|
132
|
+
return logger
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Bounded optional invocation measurements, independent of replay (MIT).
|
|
2
|
+
|
|
3
|
+
Measurements describe this Python process, including all its threads, excluding
|
|
4
|
+
child processes. Memory is a lifetime high-water mark, never an invocation peak.
|
|
5
|
+
An abrupt exit may lose the whole buffer. It contains no inputs or outputs.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from contextvars import ContextVar
|
|
9
|
+
import json
|
|
10
|
+
import sys
|
|
11
|
+
import time
|
|
12
|
+
|
|
13
|
+
_current = ContextVar("ledgence_invocation_observations", default=None)
|
|
14
|
+
MAX_LOCAL_STEPS = 128
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _usage():
|
|
18
|
+
try:
|
|
19
|
+
import resource
|
|
20
|
+
value = resource.getrusage(resource.RUSAGE_SELF)
|
|
21
|
+
rss = value.ru_maxrss * (1024 if sys.platform.startswith("linux") else 1)
|
|
22
|
+
if sys.platform != "darwin" and not sys.platform.startswith("linux"):
|
|
23
|
+
rss = None # Units are deliberately not guessed on other platforms.
|
|
24
|
+
return (round(value.ru_utime * 1_000_000), round(value.ru_stime * 1_000_000), rss)
|
|
25
|
+
except (ImportError, OSError, ValueError):
|
|
26
|
+
return (None, None, None)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Observations:
|
|
30
|
+
def __init__(self):
|
|
31
|
+
self.started_at = time.time_ns() // 1_000_000
|
|
32
|
+
self.started = time.monotonic_ns()
|
|
33
|
+
self.usage = _usage()
|
|
34
|
+
self.locals = []
|
|
35
|
+
self.truncated = False
|
|
36
|
+
|
|
37
|
+
def begin(self, binding, replayed=False):
|
|
38
|
+
if len(self.locals) >= MAX_LOCAL_STEPS:
|
|
39
|
+
self.truncated = True
|
|
40
|
+
return None
|
|
41
|
+
record = {"key": binding["key"], "callable": binding["callable"],
|
|
42
|
+
"started_at_ms": time.time_ns() // 1_000_000,
|
|
43
|
+
"elapsed_us": 0, "state": "replayed" if replayed else "cancelled"}
|
|
44
|
+
self.locals.append(record)
|
|
45
|
+
return (record, time.monotonic_ns())
|
|
46
|
+
|
|
47
|
+
@staticmethod
|
|
48
|
+
def finish(token, state):
|
|
49
|
+
if token is not None:
|
|
50
|
+
record, started = token
|
|
51
|
+
record.update(elapsed_us=max(0, (time.monotonic_ns() - started) // 1000), state=state)
|
|
52
|
+
|
|
53
|
+
def snapshot(self):
|
|
54
|
+
usage = _usage()
|
|
55
|
+
def delta(index):
|
|
56
|
+
before, after = self.usage[index], usage[index]
|
|
57
|
+
return None if before is None or after is None or after < before else after - before
|
|
58
|
+
value = {"runtime_started_at_ms": self.started_at,
|
|
59
|
+
"runtime_elapsed_us": max(0, (time.monotonic_ns() - self.started) // 1000),
|
|
60
|
+
"process_cpu_user_us": delta(0), "process_cpu_system_us": delta(1),
|
|
61
|
+
"process_lifetime_peak_rss_bytes": usage[2],
|
|
62
|
+
"local_steps": list(self.locals), "local_steps_truncated": self.truncated}
|
|
63
|
+
# Escaped identifier bytes count too. Always keep resources even when the
|
|
64
|
+
# optional local buffer must be truncated; never alter the replay journal.
|
|
65
|
+
while len(json.dumps(value, ensure_ascii=False, separators=(",", ":")).encode()) > 128 * 1024:
|
|
66
|
+
value["local_steps"].pop()
|
|
67
|
+
value["local_steps_truncated"] = True
|
|
68
|
+
return value
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def begin_local(binding, replayed=False):
|
|
72
|
+
observations = _current.get()
|
|
73
|
+
if observations is None:
|
|
74
|
+
return None
|
|
75
|
+
try:
|
|
76
|
+
return observations.begin(binding, replayed)
|
|
77
|
+
except Exception:
|
|
78
|
+
observations.truncated = True
|
|
79
|
+
return None
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def finish_local(token, state):
|
|
83
|
+
try:
|
|
84
|
+
Observations.finish(token, state)
|
|
85
|
+
except Exception:
|
|
86
|
+
observations = _current.get()
|
|
87
|
+
if observations is not None:
|
|
88
|
+
observations.truncated = True
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""The v2 protocol's single writer and bounded optional log queue (MIT)."""
|
|
2
|
+
|
|
3
|
+
from collections import deque
|
|
4
|
+
import threading
|
|
5
|
+
|
|
6
|
+
MAX_LOG_RECORDS = 64
|
|
7
|
+
MAX_LOG_BYTES = 1024 * 1024
|
|
8
|
+
MAX_LOG_FRAME_BYTES = 16 * 1024
|
|
9
|
+
_MAX_DROPPED = (1 << 64) - 1
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class ProtocolWriter:
|
|
13
|
+
def __init__(self, protocol, frame_limit):
|
|
14
|
+
self.log_limit = min(frame_limit, MAX_LOG_FRAME_BYTES)
|
|
15
|
+
self._protocol = protocol
|
|
16
|
+
self._condition = threading.Condition()
|
|
17
|
+
self._logs = deque()
|
|
18
|
+
self._log_bytes = 0
|
|
19
|
+
self._control = None
|
|
20
|
+
self._error = None
|
|
21
|
+
self._ready = False
|
|
22
|
+
self._stopped = False
|
|
23
|
+
self._dropped = 0
|
|
24
|
+
self._thread = threading.Thread(target=self._run, name="ledgence-protocol", daemon=True)
|
|
25
|
+
self._thread.start()
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def dropped_logs(self):
|
|
29
|
+
with self._condition:
|
|
30
|
+
return self._dropped
|
|
31
|
+
|
|
32
|
+
def drop_log(self):
|
|
33
|
+
with self._condition:
|
|
34
|
+
self._dropped = min(_MAX_DROPPED, self._dropped + 1)
|
|
35
|
+
|
|
36
|
+
def offer_log(self, encoded):
|
|
37
|
+
with self._condition:
|
|
38
|
+
if (self._stopped or self._error is not None
|
|
39
|
+
or len(encoded) > self.log_limit
|
|
40
|
+
or len(self._logs) >= MAX_LOG_RECORDS
|
|
41
|
+
or self._log_bytes + len(encoded) > MAX_LOG_BYTES):
|
|
42
|
+
self._dropped = min(_MAX_DROPPED, self._dropped + 1)
|
|
43
|
+
return False
|
|
44
|
+
self._logs.append(encoded)
|
|
45
|
+
self._log_bytes += len(encoded)
|
|
46
|
+
self._condition.notify()
|
|
47
|
+
return True
|
|
48
|
+
|
|
49
|
+
def control(self, encoded, closing=False):
|
|
50
|
+
done = threading.Event()
|
|
51
|
+
with self._condition:
|
|
52
|
+
if self._error is not None:
|
|
53
|
+
raise self._error
|
|
54
|
+
if self._control is not None:
|
|
55
|
+
raise RuntimeError("concurrent protocol control writes")
|
|
56
|
+
if closing:
|
|
57
|
+
self._stopped = True
|
|
58
|
+
self._dropped = min(_MAX_DROPPED, self._dropped + len(self._logs))
|
|
59
|
+
self._logs.clear()
|
|
60
|
+
self._log_bytes = 0
|
|
61
|
+
self._control = (encoded, done)
|
|
62
|
+
self._condition.notify()
|
|
63
|
+
done.wait() # IPC backpressure only; optional logs never occupy this slot.
|
|
64
|
+
if self._error is not None:
|
|
65
|
+
raise self._error
|
|
66
|
+
|
|
67
|
+
def _run(self):
|
|
68
|
+
while True:
|
|
69
|
+
with self._condition:
|
|
70
|
+
while self._control is None and not (self._ready and self._logs):
|
|
71
|
+
if self._stopped:
|
|
72
|
+
return
|
|
73
|
+
self._condition.wait()
|
|
74
|
+
if self._control is not None:
|
|
75
|
+
encoded, done = self._control
|
|
76
|
+
self._control = None
|
|
77
|
+
else:
|
|
78
|
+
encoded = self._logs.popleft()
|
|
79
|
+
self._log_bytes -= len(encoded)
|
|
80
|
+
done = None
|
|
81
|
+
try:
|
|
82
|
+
pending = memoryview(encoded)
|
|
83
|
+
while pending:
|
|
84
|
+
count = self._protocol.write(pending)
|
|
85
|
+
if count is None or count <= 0:
|
|
86
|
+
raise OSError("protocol output pipe closed")
|
|
87
|
+
pending = pending[count:]
|
|
88
|
+
except Exception as error:
|
|
89
|
+
with self._condition:
|
|
90
|
+
self._error = error
|
|
91
|
+
if self._control is not None:
|
|
92
|
+
self._control[1].set()
|
|
93
|
+
self._logs.clear()
|
|
94
|
+
self._log_bytes = 0
|
|
95
|
+
if done is not None:
|
|
96
|
+
done.set()
|
|
97
|
+
return
|
|
98
|
+
if done is not None:
|
|
99
|
+
with self._condition:
|
|
100
|
+
self._ready = True
|
|
101
|
+
done.set()
|