foundry-testing-actor 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,193 @@
1
+ """Correlation ids — the one place this actor decides what identifies a unit of work.
2
+
3
+ WHY THIS EXISTS. Before it, exactly one call site (the engine's own `_invoke_claude`) passed
4
+ `extra={"task_id": ...}`, so the session transcript was the only thing in Loki that could be
5
+ drilled to a task. Everything else this actor logs — the framework's own `do_POST door=…` access
6
+ line, every git/buildctl step, every failure — arrived with no identity at all, which is exactly
7
+ when you most want one. This module makes carrying the ids the default rather than a thing each
8
+ call site has to remember.
9
+
10
+ TWO IDS, AND WHERE EACH COMES FROM.
11
+
12
+ * `correlation_id` — THE TRACE ID, not a second identifier invented here. One orchestration
13
+ call is one trace: the ORCHESTRATING actor's own `_call_door` injects W3C `traceparent`, and
14
+ `HttpMailbox.do_POST` extracts it into the SERVER span it opens around `Actor.receive()` — so
15
+ every actor in the pipeline, across every retry attempt, is already inside one trace before a
16
+ line of this module runs. Reading `trace.get_current_span()` and formatting its
17
+ trace id is therefore not "generating a correlation id", it is *naming the one that already
18
+ crossed the wire*. It also means a `correlation_id` in Loki pastes straight into Tempo.
19
+ A locally-minted uuid is the fallback for the no-SDK / no-active-span case only.
20
+ * `task_id` — `TASK-NNN`, off the caller's own payload. Spans retries and every peer, where the
21
+ trace id spans one orchestration run; neither subsumes the other, so both are carried.
22
+
23
+ A CALL THAT NEVER REACHES A HANDLER still gets a `correlation_id`. `Actor.receive()` refuses an
24
+ undeclared door or a schema-violating payload before any handler runs, and the binding answers an
25
+ unknown path or an unparseable body without ever calling `receive()` — so nothing of ours binds
26
+ for any of them. `CorrelationFilter` therefore falls back to the active span's trace id, which
27
+ since ADR-PASH-0005 (`papeete-actor-synchronous-messaging-http` 0.4.0) is open across every one of
28
+ those paths. That fallback is the whole reason the binding needed changing: a consumer can stamp
29
+ its own records, never the ones written on its behalf before its code ran.
30
+
31
+ WHY A HANDLER FILTER, NOT `extra=` AT EVERY CALL SITE. A `logging.Filter` on the ROOT LOGGER would
32
+ only see records logged directly to it — a record from any named logger reaches the root's
33
+ HANDLERS without ever being filtered by the root logger itself. So `install()` attaches to the
34
+ handlers, where every record does pass, including ones from code this repo does not own
35
+ (`papeete_actor_synchronous_messaging_http.mailbox`'s access log, most importantly). Nothing has
36
+ to opt in, which is the whole point: correlation ids everywhere means everywhere, not everywhere
37
+ someone remembered.
38
+
39
+ WHY `bind()` NEVER RESETS. `HttpMailbox` serves on a `ThreadingHTTPServer` — one fresh thread per
40
+ request, never pooled — and a `ContextVar` set inside a thread is invisible to every other thread
41
+ and dies with it. So request scope is thread scope here, for free, and NOT resetting is what lets
42
+ the ids reach the lines emitted *after* the handler returns: `do_POST`'s own access log sits in a
43
+ `finally` outside the span, and would otherwise be the one line in the whole request with no
44
+ identity. A second `bind()` in the same thread overwrites, which is what an attempt counter wants.
45
+ """
46
+ from __future__ import annotations
47
+
48
+ import contextvars
49
+ import json
50
+ import logging
51
+ import time
52
+ import uuid
53
+ from contextlib import contextmanager
54
+
55
+ try: # the same guard papeete-actor-synchronous-messaging-
56
+ from opentelemetry import trace # http's own _tracing.py keeps: the API is a no-op
57
+ except ImportError: # without an SDK, and absent entirely in a bare test
58
+ trace = None # environment. Neither is an error here.
59
+
60
+ _FIELDS: contextvars.ContextVar[dict] = contextvars.ContextVar("papeete_correlation", default={})
61
+
62
+ STEP_LOGGER = "pipeline"
63
+
64
+
65
+ def _trace_id() -> str | None:
66
+ """The active span's trace id, or None — no SDK, no span, or an invalid context."""
67
+ if trace is None:
68
+ return None
69
+ context = trace.get_current_span().get_span_context()
70
+ return format(context.trace_id, "032x") if context.is_valid else None
71
+
72
+
73
+ def correlation_id() -> str:
74
+ """The active trace id, or a fresh uuid when there is no SDK/span to read one from."""
75
+ return _trace_id() or uuid.uuid4().hex
76
+
77
+
78
+ def current() -> dict:
79
+ return dict(_FIELDS.get())
80
+
81
+
82
+ def bind(**fields) -> None:
83
+ """Add ids to THIS THREAD's correlation context, for the rest of the thread's life.
84
+
85
+ Deliberately not a context manager — see this module's own docstring for why the binding
86
+ outliving the handler is the point, not an oversight."""
87
+ _FIELDS.set({**_FIELDS.get(), **{k: v for k, v in fields.items() if v is not None}})
88
+
89
+
90
+ class CorrelationFilter(logging.Filter):
91
+ """Stamps every record passing a handler with the current context's ids.
92
+
93
+ Never overwrites: a call site that passed its own `extra={"task_id": ...}` keeps it, so this
94
+ is additive to existing behaviour rather than a replacement for it. The stamped attributes
95
+ are ordinary record attributes, which is precisely what OTel's own `LoggingHandler` turns
96
+ into OTLP log attributes — and Loki, in turn, into structured metadata a `| task_id = "…"`
97
+ matcher filters on with no parser in front of it."""
98
+
99
+ def filter(self, record: logging.LogRecord) -> bool:
100
+ for key, value in _FIELDS.get().items():
101
+ if not hasattr(record, key):
102
+ setattr(record, key, value)
103
+ if not hasattr(record, "correlation_id"):
104
+ # NOTHING BOUND — which is precisely the shape of a call the framework turned away
105
+ # before any handler of ours could run: an undeclared door, a payload the card's own
106
+ # `request_schema` rejects, a body that is not JSON. Since ADR-PASH-0005 the HTTP
107
+ # binding holds its SERVER span open across all of those, parented on the caller's
108
+ # own `traceparent`, so the active trace id here is the very value `bind()` would
109
+ # have used. Falling back to it files those records under the run that caused them
110
+ # instead of under nothing at all. `task_id` genuinely cannot be recovered this way
111
+ # and is left absent: a payload the schema rejected may not carry one.
112
+ fallback = _trace_id()
113
+ if fallback:
114
+ record.correlation_id = fallback
115
+ return True
116
+
117
+
118
+ # The ids Loki gets as structured metadata are invisible in `kubectl logs`, which is still the
119
+ # first place anyone looks when the telemetry backend itself is what's under suspicion.
120
+ CONSOLE_IDS = ("correlation_id", "task_id", "attempt")
121
+
122
+
123
+ class ConsoleFormatter(logging.Formatter):
124
+ """Appends whichever correlation ids a record actually carries.
125
+
126
+ A plain `%(task_id)s` in the format string would instead raise `Formatting field not found`
127
+ for every record emitted outside a request — the startup line, anything on a thread that never
128
+ bound — which is exactly how a logging change takes a process down. Nothing is defaulted onto
129
+ the record either: an id that isn't there stays absent, so the OTLP side never carries a
130
+ placeholder value the dashboard would then have to filter back out."""
131
+
132
+ def format(self, record: logging.LogRecord) -> str:
133
+ line = super().format(record)
134
+ ids = " ".join(
135
+ f"{key}={getattr(record, key)}" for key in CONSOLE_IDS if hasattr(record, key)
136
+ )
137
+ return f"{line} [{ids}]" if ids else line
138
+
139
+
140
+ def install() -> None:
141
+ """Attach the filter to every handler on the root logger.
142
+
143
+ Call AFTER `papeete_observability.configure()` and after any console handler is added — this
144
+ walks the handlers that exist at the moment it runs."""
145
+ correlation_filter = CorrelationFilter()
146
+ for handler in logging.getLogger().handlers:
147
+ handler.addFilter(correlation_filter)
148
+
149
+
150
+ # ── the step vocabulary ───────────────────────────────────────────────────────────────────────
151
+ #
152
+ # One JSON object per step, so the dashboard reads them with the same `| json` it already uses for
153
+ # the session transcript, and tells the two apart by `event`. `phase` is start / ok / failed —
154
+ # a start line with no matching ok is a step that was still running when the pod went away, which
155
+ # a duration-only record could never show.
156
+
157
+ def _emit(level: int, record: dict) -> None:
158
+ # `level` goes INTO the JSON body as well as onto the record. Severity does reach Loki through
159
+ # OTLP, but exactly which field it lands in is the ingester's business, not something a
160
+ # dashboard query should be pinned to; a `level` the emitter wrote itself is one `| json` away
161
+ # in every backend, which is what the product dashboard's own failure panel filters on.
162
+ body = {"level": logging.getLevelName(level).lower(), **record}
163
+ logging.getLogger(STEP_LOGGER).log(level, json.dumps(body, default=str, ensure_ascii=False))
164
+
165
+
166
+ def event(name: str, *, level: int = logging.INFO, **fields) -> None:
167
+ """A point in the pipeline with no duration — a verdict, a published ref, a PR url.
168
+
169
+ `level` is keyword-only — a best-effort step that failed without stopping the run
170
+ (`logging.WARNING`) still reads as one `event` to the dashboard's `| json`, and lands in the
171
+ body as `level` (see `_emit`), so `**fields` must not carry a key of that name."""
172
+ _emit(level, {"event": "event", "step": name, **fields})
173
+
174
+
175
+ @contextmanager
176
+ def stage(name: str, **fields):
177
+ """A step with a beginning and an end, logged as both — and as `failed` with the exception's
178
+ own text if it raises, before the exception continues on its way untouched."""
179
+ started = time.monotonic()
180
+ _emit(logging.INFO, {"event": "step", "step": name, "phase": "start", **fields})
181
+ try:
182
+ yield
183
+ except BaseException as e: # noqa: BLE001 — re-raised below
184
+ _emit(logging.ERROR, {
185
+ "event": "step", "step": name, "phase": "failed",
186
+ "duration_ms": round((time.monotonic() - started) * 1000),
187
+ "error": f"{type(e).__name__}: {e}", **fields,
188
+ })
189
+ raise
190
+ _emit(logging.INFO, {
191
+ "event": "step", "step": name, "phase": "ok",
192
+ "duration_ms": round((time.monotonic() - started) * 1000), **fields,
193
+ })