netbox-opentelemetry-plugin 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- netbox_opentelemetry_plugin/__init__.py +47 -0
- netbox_opentelemetry_plugin/bootstrap.py +613 -0
- netbox_opentelemetry_plugin/conf.py +665 -0
- netbox_opentelemetry_plugin/middleware.py +58 -0
- netbox_opentelemetry_plugin/modules/__init__.py +0 -0
- netbox_opentelemetry_plugin/modules/audit.py +306 -0
- netbox_opentelemetry_plugin/modules/base.py +36 -0
- netbox_opentelemetry_plugin/modules/logs.py +62 -0
- netbox_opentelemetry_plugin/modules/rq.py +427 -0
- netbox_opentelemetry_plugin/modules/runtime.py +69 -0
- netbox_opentelemetry_plugin/modules/traces.py +93 -0
- netbox_opentelemetry_plugin/otel.py +983 -0
- netbox_opentelemetry_plugin/version.py +1 -0
- netbox_opentelemetry_plugin-0.1.0.dist-info/METADATA +125 -0
- netbox_opentelemetry_plugin-0.1.0.dist-info/RECORD +17 -0
- netbox_opentelemetry_plugin-0.1.0.dist-info/WHEEL +4 -0
- netbox_opentelemetry_plugin-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
|
|
3
|
+
from netbox.plugins import PluginConfig
|
|
4
|
+
|
|
5
|
+
from .version import __version__
|
|
6
|
+
|
|
7
|
+
logger = logging.getLogger("netbox_opentelemetry_plugin")
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _django_settings():
|
|
11
|
+
from django.conf import settings
|
|
12
|
+
|
|
13
|
+
return settings
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class NetBoxOpenTelemetryConfig(PluginConfig):
|
|
17
|
+
name = "netbox_opentelemetry_plugin"
|
|
18
|
+
verbose_name = "NetBox OpenTelemetry"
|
|
19
|
+
description = "Export NetBox logs, audit records, traces and metrics over OTLP"
|
|
20
|
+
version = __version__
|
|
21
|
+
base_url = "opentelemetry"
|
|
22
|
+
min_version = "4.7.0"
|
|
23
|
+
max_version = "4.7.99"
|
|
24
|
+
# Always registered: NetBox reads it while loading settings, before ready(). Without a
|
|
25
|
+
# recording span (traces off) it returns immediately.
|
|
26
|
+
middleware = ["netbox_opentelemetry_plugin.middleware.RequestSpanMiddleware"]
|
|
27
|
+
# Defaults live in conf.DEFAULTS. NetBox merges default_settings one level deep only, and
|
|
28
|
+
# pre-filled defaults would hide which values were set explicitly (needed for OTEL_* fallback).
|
|
29
|
+
default_settings = {}
|
|
30
|
+
|
|
31
|
+
def ready(self):
|
|
32
|
+
super().ready()
|
|
33
|
+
try:
|
|
34
|
+
from . import bootstrap
|
|
35
|
+
|
|
36
|
+
settings = _django_settings()
|
|
37
|
+
release = getattr(settings, "RELEASE", None)
|
|
38
|
+
bootstrap.install(
|
|
39
|
+
settings.PLUGINS_CONFIG.get(self.name, {}),
|
|
40
|
+
netbox_version=getattr(release, "version", "unknown"),
|
|
41
|
+
)
|
|
42
|
+
except Exception:
|
|
43
|
+
# The plugin must never prevent NetBox from starting.
|
|
44
|
+
logger.warning("OpenTelemetry setup failed; export disabled", exc_info=True)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
config = NetBoxOpenTelemetryConfig
|
|
@@ -0,0 +1,613 @@
|
|
|
1
|
+
"""Per-process installation of the plugin's OTel pipeline.
|
|
2
|
+
|
|
3
|
+
install() is safe to call more than once: the second call in the same process returns the
|
|
4
|
+
existing context. Nothing in here raises for configuration or exporter problems; those are
|
|
5
|
+
logged as warnings on the plugin logger and NetBox keeps running.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import atexit
|
|
11
|
+
import contextlib
|
|
12
|
+
import logging
|
|
13
|
+
import os
|
|
14
|
+
import re
|
|
15
|
+
import sys
|
|
16
|
+
import threading
|
|
17
|
+
import time
|
|
18
|
+
from collections.abc import Mapping, Sequence
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
|
|
21
|
+
from . import conf, otel
|
|
22
|
+
from .modules.audit import AuditModule
|
|
23
|
+
from .modules.base import Context, Module
|
|
24
|
+
from .modules.logs import LogsModule
|
|
25
|
+
from .modules.rq import RqModule
|
|
26
|
+
from .modules.runtime import RuntimeModule
|
|
27
|
+
from .modules.traces import TracesModule
|
|
28
|
+
from .version import __version__
|
|
29
|
+
|
|
30
|
+
logger = logging.getLogger(otel.PLUGIN_LOGGER)
|
|
31
|
+
|
|
32
|
+
ROLE_WEB = "web"
|
|
33
|
+
ROLE_RQWORKER = "rqworker"
|
|
34
|
+
ROLE_RQ_HORSE = "rq_horse"
|
|
35
|
+
ROLE_MANAGEMENT = "management"
|
|
36
|
+
ROLE_RUNSERVER_PARENT = "runserver_parent"
|
|
37
|
+
|
|
38
|
+
# Short management commands never start trace exporter threads (SPEC 4.1). The RQ work-horse
|
|
39
|
+
# inherits the rqworker's provider and rebuilds it after fork.
|
|
40
|
+
TRACE_ROLES = frozenset({ROLE_WEB, ROLE_RQWORKER})
|
|
41
|
+
|
|
42
|
+
# Metrics run only in the long-lived web and rqworker processes (SPEC 4.1, 6.4). A forked RQ
|
|
43
|
+
# work-horse records into a no-op provider and never exports; job metrics come from its parent.
|
|
44
|
+
METRIC_ROLES = frozenset({ROLE_WEB, ROLE_RQWORKER})
|
|
45
|
+
|
|
46
|
+
# A bulk edit or bulk import can write thousands of ObjectChange rows inside a single commit; the
|
|
47
|
+
# default BatchLogRecordProcessor queue (2048) is sized for scattered log lines, not that burst.
|
|
48
|
+
# Sized generously so a single bulk operation cannot overrun it and silently drop audit records.
|
|
49
|
+
AUDIT_QUEUE_SIZE = 20_000
|
|
50
|
+
|
|
51
|
+
# user:password@ in URLs or host strings, including percent-encoded credentials. Deliberately
|
|
52
|
+
# favours false positives: any "word:word@" shape is masked, not just valid userinfo. Only ever
|
|
53
|
+
# applied to a message already bounded by _MESSAGE_LIMIT (see _describe) because this pattern can
|
|
54
|
+
# backtrack catastrophically on long inputs with a ":" but no "@" (for example "a" * n + ":" + "b" * n).
|
|
55
|
+
_USERINFO = re.compile(r"[A-Za-z0-9._~%!$&'()*+,;=-]+:[^\s/@'\"]+@")
|
|
56
|
+
|
|
57
|
+
# scheme://TOKEN@host userinfo with no colon (a bare token, not a user:password pair). Also only
|
|
58
|
+
# ever applied to a message already bounded by _MESSAGE_LIMIT, for the same reason as _USERINFO.
|
|
59
|
+
_TOKEN_USERINFO = re.compile(r"(//)[^\s/@'\"]+@")
|
|
60
|
+
|
|
61
|
+
_MESSAGE_LIMIT = 2000
|
|
62
|
+
|
|
63
|
+
UWSGI_THREADS_WARNING = (
|
|
64
|
+
"OpenTelemetry: uWSGI is running without thread support, so the exporter's background thread "
|
|
65
|
+
"cannot run and nothing will be exported. Set `enable-threads = true` in the uWSGI configuration."
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class _State:
|
|
71
|
+
pid: int
|
|
72
|
+
context: Context | None
|
|
73
|
+
modules: list[Module] = field(default_factory=list)
|
|
74
|
+
owns_logger_provider: bool = False
|
|
75
|
+
owns_tracer_provider: bool = False
|
|
76
|
+
metrics_pipeline: otel.MetricsPipeline | None = None
|
|
77
|
+
netbox_version: str = "unknown"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
_state: _State | None = None
|
|
81
|
+
_lock = threading.RLock()
|
|
82
|
+
_atexit_registered = False
|
|
83
|
+
_next_fork_role: str | None = None
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def detect_role(argv: Sequence[str], env: Mapping[str, str]) -> str:
|
|
87
|
+
if len(argv) >= 2 and os.path.basename(argv[0]) == "manage.py":
|
|
88
|
+
command = argv[1]
|
|
89
|
+
if command == "rqworker":
|
|
90
|
+
return ROLE_RQWORKER
|
|
91
|
+
if command == "runserver":
|
|
92
|
+
# The autoreloader parent only watches files; the child it spawns has RUN_MAIN=true.
|
|
93
|
+
if "--noreload" in argv or env.get("RUN_MAIN") == "true":
|
|
94
|
+
return ROLE_WEB
|
|
95
|
+
return ROLE_RUNSERVER_PARENT
|
|
96
|
+
return ROLE_MANAGEMENT
|
|
97
|
+
return ROLE_WEB
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def install(
|
|
101
|
+
user_config: Mapping | None,
|
|
102
|
+
*,
|
|
103
|
+
env: Mapping[str, str] | None = None,
|
|
104
|
+
argv: Sequence[str] | None = None,
|
|
105
|
+
netbox_version: str = "unknown",
|
|
106
|
+
) -> Context | None:
|
|
107
|
+
global _state, _atexit_registered
|
|
108
|
+
env = os.environ if env is None else env
|
|
109
|
+
argv = sys.argv if argv is None else argv
|
|
110
|
+
|
|
111
|
+
with _lock:
|
|
112
|
+
if _state is not None and _state.pid == os.getpid():
|
|
113
|
+
return _state.context
|
|
114
|
+
|
|
115
|
+
settings = conf.resolve(user_config, env)
|
|
116
|
+
for message in settings.warnings:
|
|
117
|
+
logger.warning("OpenTelemetry: %s", message)
|
|
118
|
+
|
|
119
|
+
role = detect_role(argv, env)
|
|
120
|
+
if not settings.enabled or role == ROLE_RUNSERVER_PARENT:
|
|
121
|
+
_state = _State(pid=os.getpid(), context=None)
|
|
122
|
+
return None
|
|
123
|
+
|
|
124
|
+
if logger.isEnabledFor(logging.DEBUG):
|
|
125
|
+
logger.debug("OpenTelemetry resolved config: %s", settings.redacted())
|
|
126
|
+
|
|
127
|
+
resource = _build_resource(settings, role, netbox_version)
|
|
128
|
+
ctx = Context(settings=settings, role=role, resource=resource)
|
|
129
|
+
state = _State(pid=os.getpid(), context=ctx, netbox_version=netbox_version)
|
|
130
|
+
|
|
131
|
+
if settings.log_exporter is not None:
|
|
132
|
+
_setup_logger_provider(ctx, state)
|
|
133
|
+
|
|
134
|
+
if settings.traces.enabled and role in TRACE_ROLES:
|
|
135
|
+
_setup_tracer_provider(ctx, state)
|
|
136
|
+
|
|
137
|
+
if settings.metrics.enabled and role in METRIC_ROLES:
|
|
138
|
+
_setup_meter_provider(ctx, state)
|
|
139
|
+
|
|
140
|
+
for module in _candidate_modules(ctx):
|
|
141
|
+
if not module.enabled(settings):
|
|
142
|
+
continue
|
|
143
|
+
try:
|
|
144
|
+
module.install(ctx)
|
|
145
|
+
except Exception as exc:
|
|
146
|
+
logger.warning("OpenTelemetry: %s module disabled: %s", module.name, _describe(exc, settings))
|
|
147
|
+
continue
|
|
148
|
+
state.modules.append(module)
|
|
149
|
+
|
|
150
|
+
uwsgi_module = _uwsgi_module()
|
|
151
|
+
if uwsgi_module is not None:
|
|
152
|
+
try:
|
|
153
|
+
_integrate_uwsgi(uwsgi_module)
|
|
154
|
+
except Exception as exc:
|
|
155
|
+
logger.warning("OpenTelemetry: uWSGI integration failed: %s", _describe(exc, settings))
|
|
156
|
+
|
|
157
|
+
_state = state
|
|
158
|
+
if not _atexit_registered:
|
|
159
|
+
atexit.register(shutdown)
|
|
160
|
+
_atexit_registered = True
|
|
161
|
+
return ctx
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _build_resource(settings: conf.Settings, role: str, netbox_version: str):
|
|
165
|
+
return otel.build_resource(
|
|
166
|
+
settings.service_name,
|
|
167
|
+
settings.resource_attributes,
|
|
168
|
+
service_version=netbox_version,
|
|
169
|
+
plugin_version=__version__,
|
|
170
|
+
role=role,
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def shutdown() -> None:
|
|
175
|
+
global _state
|
|
176
|
+
with _lock:
|
|
177
|
+
state = _state
|
|
178
|
+
if state is None or state.pid != os.getpid():
|
|
179
|
+
return
|
|
180
|
+
_state = None
|
|
181
|
+
ctx = state.context
|
|
182
|
+
# Remove handlers first so records emitted during shutdown do not hit a closed provider.
|
|
183
|
+
for module in reversed(state.modules):
|
|
184
|
+
try:
|
|
185
|
+
module.shutdown()
|
|
186
|
+
except Exception as exc:
|
|
187
|
+
logger.warning("OpenTelemetry: %s module shutdown failed: %s", module.name, type(exc).__name__)
|
|
188
|
+
if state.metrics_pipeline is not None:
|
|
189
|
+
pipeline, state.metrics_pipeline = state.metrics_pipeline, None
|
|
190
|
+
try:
|
|
191
|
+
pipeline.shutdown(ctx.settings.metrics.exporter.timeout)
|
|
192
|
+
except Exception as exc:
|
|
193
|
+
logger.warning("OpenTelemetry: meter provider shutdown failed: %s", type(exc).__name__)
|
|
194
|
+
if ctx is not None and ctx.meter_provider is not None:
|
|
195
|
+
# Later measurements (for example from a request racing the shutdown) go nowhere.
|
|
196
|
+
with contextlib.suppress(Exception):
|
|
197
|
+
ctx.meter_provider.set_delegate(otel.noop_meter_provider())
|
|
198
|
+
if state.owns_tracer_provider and ctx is not None and ctx.tracer_provider is not None:
|
|
199
|
+
try:
|
|
200
|
+
ctx.tracer_provider.shutdown()
|
|
201
|
+
except Exception as exc:
|
|
202
|
+
logger.warning("OpenTelemetry: tracer provider shutdown failed: %s", type(exc).__name__)
|
|
203
|
+
if state.owns_logger_provider and ctx is not None and ctx.logger_provider is not None:
|
|
204
|
+
try:
|
|
205
|
+
ctx.logger_provider.shutdown()
|
|
206
|
+
except Exception as exc:
|
|
207
|
+
logger.warning("OpenTelemetry: logger provider shutdown failed: %s", type(exc).__name__)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def reinit_after_fork() -> None:
|
|
211
|
+
"""Rebuild per-process OTel state in a forked child.
|
|
212
|
+
|
|
213
|
+
The SDK restarts its batch threads after fork but keeps the parent's service.instance.id and
|
|
214
|
+
shares the parent's exporter connection. This gives the child its own Resource, exporter and
|
|
215
|
+
LoggerProvider. Idempotent: a second call in the same process does nothing. If this process
|
|
216
|
+
owned its LoggerProvider and the rebuild fails, it detaches the logging handler and points the
|
|
217
|
+
tracer provider at a detached provider rather than keep using the inherited, now-orphaned
|
|
218
|
+
providers.
|
|
219
|
+
|
|
220
|
+
This never calls shutdown() (or anything else) on an object inherited from the parent: any
|
|
221
|
+
lock inside such an object (a threading.Lock, a Condition, an SSL/urllib3 connection pool
|
|
222
|
+
mutex, a gRPC channel's internal state) was copied by fork() in whatever state it happened to
|
|
223
|
+
be in at that instant. If some other parent thread held that lock at fork time, the copy in
|
|
224
|
+
the child is born locked with no owner able to release it, and touching it deadlocks the
|
|
225
|
+
child forever. We simply stop referencing the inherited provider and let its worker thread and
|
|
226
|
+
exporter connection sit idle and unused; that idle thread/connection is the accepted cost.
|
|
227
|
+
"""
|
|
228
|
+
global _next_fork_role
|
|
229
|
+
with _lock:
|
|
230
|
+
role_hint, _next_fork_role = _next_fork_role, None
|
|
231
|
+
state = _state
|
|
232
|
+
if state is None or state.pid == os.getpid():
|
|
233
|
+
return
|
|
234
|
+
state.pid = os.getpid()
|
|
235
|
+
ctx = state.context
|
|
236
|
+
if ctx is None:
|
|
237
|
+
return
|
|
238
|
+
try:
|
|
239
|
+
_rebuild_for_child(ctx, state, role_hint)
|
|
240
|
+
except Exception as exc:
|
|
241
|
+
logger.warning("OpenTelemetry: re-initialisation after fork failed: %s", _describe(exc, ctx.settings))
|
|
242
|
+
if state.owns_tracer_provider and ctx.tracer_provider is not None:
|
|
243
|
+
# Same reason as below: never export through the inherited exporter connection.
|
|
244
|
+
state.owns_tracer_provider = False
|
|
245
|
+
with contextlib.suppress(Exception):
|
|
246
|
+
ctx.tracer_provider.set_delegate(otel.detached_tracer_provider())
|
|
247
|
+
role = role_hint or ctx.role
|
|
248
|
+
if ctx.meter_provider is not None and (state.metrics_pipeline is not None or role not in METRIC_ROLES):
|
|
249
|
+
# No exporter of our own in this process: record nowhere rather than into inherited state.
|
|
250
|
+
state.metrics_pipeline = None
|
|
251
|
+
with contextlib.suppress(Exception):
|
|
252
|
+
ctx.meter_provider.set_delegate(otel.noop_meter_provider())
|
|
253
|
+
if state.owns_logger_provider:
|
|
254
|
+
# This process could not build its own provider. The inherited one shares the parent's
|
|
255
|
+
# exporter connection, so stop exporting from this process instead of using it.
|
|
256
|
+
state.owns_logger_provider = False
|
|
257
|
+
ctx.logger_provider = None
|
|
258
|
+
for module in state.modules:
|
|
259
|
+
with contextlib.suppress(Exception):
|
|
260
|
+
module.after_fork(ctx)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def set_next_fork_role(role: str | None) -> None:
|
|
264
|
+
"""Label the child of the next fork (the RQ integration sets this around fork_work_horse).
|
|
265
|
+
|
|
266
|
+
One-shot: the hint is consumed by the next fork and then cleared, in both the parent process
|
|
267
|
+
(see _after_fork_in_parent) and the child (see reinit_after_fork), so it applies to exactly
|
|
268
|
+
one fork.
|
|
269
|
+
"""
|
|
270
|
+
global _next_fork_role
|
|
271
|
+
_next_fork_role = role
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def force_flush(timeout: float) -> bool:
|
|
275
|
+
"""Flush every provider of this process (logs, traces and, where this process owns one, the
|
|
276
|
+
metrics pipeline) in parallel, waiting at most `timeout` seconds in total. Never raises.
|
|
277
|
+
|
|
278
|
+
True means every flush call returned within the deadline, or there was nothing to flush: no
|
|
279
|
+
installed state, a process whose PID does not match the installed state (treated as nothing
|
|
280
|
+
to flush here), or no providers. It does not mean the records were exported: a failure inside
|
|
281
|
+
a provider's force_flush is swallowed and still counts as that flush having returned, since
|
|
282
|
+
the flush call itself did not hang past the deadline. False is returned when the deadline
|
|
283
|
+
passed before every flush finished, or when a helper thread itself could not be started.
|
|
284
|
+
|
|
285
|
+
Each provider's flush runs on its own helper thread, in parallel, so a stuck export cannot
|
|
286
|
+
hold the caller beyond the deadline.
|
|
287
|
+
"""
|
|
288
|
+
state = _state
|
|
289
|
+
if state is None or state.pid != os.getpid() or state.context is None:
|
|
290
|
+
return True
|
|
291
|
+
ctx = state.context
|
|
292
|
+
flushables = [p for p in (ctx.logger_provider, ctx.tracer_provider) if p is not None]
|
|
293
|
+
if state.metrics_pipeline is not None:
|
|
294
|
+
# Never in a work-horse: it has no pipeline (SPEC 6.4).
|
|
295
|
+
flushables.append(state.metrics_pipeline)
|
|
296
|
+
if not flushables:
|
|
297
|
+
return True
|
|
298
|
+
deadline = time.monotonic() + timeout
|
|
299
|
+
done_events = []
|
|
300
|
+
for flushable in flushables:
|
|
301
|
+
done = threading.Event()
|
|
302
|
+
|
|
303
|
+
def run(flushable=flushable, done=done) -> None:
|
|
304
|
+
try:
|
|
305
|
+
flushable.force_flush(timeout_millis=int(timeout * 1000))
|
|
306
|
+
except Exception:
|
|
307
|
+
pass
|
|
308
|
+
finally:
|
|
309
|
+
done.set()
|
|
310
|
+
|
|
311
|
+
try:
|
|
312
|
+
threading.Thread(target=run, name="otel-flush", daemon=True).start()
|
|
313
|
+
except Exception:
|
|
314
|
+
# Thread creation can fail (resource limits, interpreter finalization). Nothing was
|
|
315
|
+
# started, so there is nothing to wait on: report the flush as not completed.
|
|
316
|
+
return False
|
|
317
|
+
done_events.append(done)
|
|
318
|
+
return all(done.wait(max(0.0, deadline - time.monotonic())) for done in done_events)
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _rebuild_for_child(ctx: Context, state: _State, role_hint: str | None) -> None:
|
|
322
|
+
"""Build a new role, Resource and (if we own them) LoggerProvider, tracer SDK provider and metrics
|
|
323
|
+
pipeline for this process. The metrics pipeline is rebuilt only in the web and rqworker roles; a
|
|
324
|
+
work-horse gets a no-op meter provider instead.
|
|
325
|
+
|
|
326
|
+
Nothing is assigned onto ctx until every new provider has been built successfully, so a
|
|
327
|
+
failure here (for example the exporter config referencing a now-unreadable certificate file)
|
|
328
|
+
leaves ctx.role, ctx.resource, ctx.logger_provider and ctx.tracer_provider exactly as they
|
|
329
|
+
were: still consistent with each other, still the values inherited from the parent at fork
|
|
330
|
+
time.
|
|
331
|
+
|
|
332
|
+
The role changes only when the parent announced the fork (see `set_next_fork_role`); an
|
|
333
|
+
unannounced fork of an rqworker process (for example the RQ scheduler's own child) keeps the
|
|
334
|
+
rqworker role.
|
|
335
|
+
|
|
336
|
+
The old logger provider (if we owned one) is never shut down here: see reinit_after_fork for
|
|
337
|
+
why. The tracer provider is different: we own the SwitchableTracerProvider handed to
|
|
338
|
+
instrumentors for the process lifetime, and only replace the SDK provider behind it, so there
|
|
339
|
+
is nothing of ours to shut down; the old SDK provider is simply dropped, for the same
|
|
340
|
+
never-touch-inherited-state reason. The metrics pipeline is handled the same way: the inherited
|
|
341
|
+
one (export thread gone, locks copied) is dropped, never flushed or shut down, and only the
|
|
342
|
+
delegate of the SwitchableMeterProvider changes.
|
|
343
|
+
"""
|
|
344
|
+
role = role_hint or ctx.role
|
|
345
|
+
resource = _build_resource(ctx.settings, role, state.netbox_version)
|
|
346
|
+
# Build everything first, assign afterwards: a failure leaves ctx exactly as inherited.
|
|
347
|
+
new_logger_provider = None
|
|
348
|
+
if state.owns_logger_provider and ctx.logger_provider is not None:
|
|
349
|
+
# A fresh exporter gives the child its own HTTP session or gRPC channel instead of
|
|
350
|
+
# sharing the parent's keep-alive connections.
|
|
351
|
+
exporter = otel.build_log_exporter(ctx.settings.log_exporter)
|
|
352
|
+
max_queue_size = AUDIT_QUEUE_SIZE if ctx.settings.audit.enabled else None
|
|
353
|
+
new_logger_provider = otel.build_logger_provider(resource, exporter, max_queue_size=max_queue_size)
|
|
354
|
+
# Child-owned objects not yet referenced anywhere else: shut them down if a later build fails or
|
|
355
|
+
# they leak silently, unlike the inherited providers still on ctx, which this function never touches.
|
|
356
|
+
built = []
|
|
357
|
+
if new_logger_provider is not None:
|
|
358
|
+
built.append(new_logger_provider)
|
|
359
|
+
new_tracer_provider = None
|
|
360
|
+
new_pipeline = None
|
|
361
|
+
try:
|
|
362
|
+
if state.owns_tracer_provider and ctx.tracer_provider is not None:
|
|
363
|
+
new_tracer_provider = _build_tracer_provider(ctx.settings, resource)
|
|
364
|
+
built.append(new_tracer_provider)
|
|
365
|
+
if state.metrics_pipeline is not None and role in METRIC_ROLES:
|
|
366
|
+
new_pipeline = _build_metrics_pipeline(ctx.settings, resource)
|
|
367
|
+
except Exception:
|
|
368
|
+
for obj in built:
|
|
369
|
+
with contextlib.suppress(Exception):
|
|
370
|
+
obj.shutdown()
|
|
371
|
+
raise
|
|
372
|
+
try:
|
|
373
|
+
if new_logger_provider is not None:
|
|
374
|
+
ctx.logger_provider = new_logger_provider
|
|
375
|
+
if new_tracer_provider is not None:
|
|
376
|
+
# Instrumentors hold the switchable provider; only the SDK provider behind it changes.
|
|
377
|
+
ctx.tracer_provider.set_delegate(new_tracer_provider)
|
|
378
|
+
if ctx.meter_provider is not None:
|
|
379
|
+
if role not in METRIC_ROLES:
|
|
380
|
+
# An RQ work-horse exports no metrics (SPEC 6.4). It must not record into the inherited
|
|
381
|
+
# SDK provider either: the parent's export thread may have held its locks at fork time.
|
|
382
|
+
state.metrics_pipeline = None
|
|
383
|
+
with contextlib.suppress(Exception):
|
|
384
|
+
ctx.meter_provider.set_delegate(otel.noop_meter_provider())
|
|
385
|
+
elif new_pipeline is not None:
|
|
386
|
+
# The inherited pipeline (thread gone, locks copied) is dropped, never shut down.
|
|
387
|
+
ctx.meter_provider.set_delegate(new_pipeline.provider)
|
|
388
|
+
state.metrics_pipeline = new_pipeline
|
|
389
|
+
except Exception:
|
|
390
|
+
# A swap failed. The caller's failure path switches this process to no-op providers and
|
|
391
|
+
# drops the new ones, so stop their threads here (all child-owned, never inherited).
|
|
392
|
+
for obj in built:
|
|
393
|
+
with contextlib.suppress(Exception):
|
|
394
|
+
obj.shutdown()
|
|
395
|
+
if new_pipeline is not None:
|
|
396
|
+
with contextlib.suppress(Exception):
|
|
397
|
+
new_pipeline.shutdown(0)
|
|
398
|
+
raise
|
|
399
|
+
ctx.role = role
|
|
400
|
+
ctx.resource = resource
|
|
401
|
+
for module in state.modules:
|
|
402
|
+
try:
|
|
403
|
+
module.after_fork(ctx)
|
|
404
|
+
except Exception as exc:
|
|
405
|
+
logger.warning("OpenTelemetry: %s module after-fork failed: %s", module.name, _describe(exc, ctx.settings))
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def _before_fork() -> None:
|
|
409
|
+
# Holding the lock across fork() means no other thread can be half way through install()
|
|
410
|
+
# or shutdown() when the child's copy of the state is taken.
|
|
411
|
+
_lock.acquire()
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _after_fork_in_parent() -> None:
|
|
415
|
+
global _next_fork_role
|
|
416
|
+
# Some embedders run the parent hook without the before hook. Fork hooks must never raise,
|
|
417
|
+
# so releasing a lock we may not hold is tolerated rather than propagated.
|
|
418
|
+
with contextlib.suppress(RuntimeError):
|
|
419
|
+
_lock.release()
|
|
420
|
+
# The hint is a copy-on-write page shared with the child at fork time; clearing it here only
|
|
421
|
+
# affects this (parent) process's own memory. It makes the hint apply to exactly one fork in
|
|
422
|
+
# the parent too, matching the child-side clear in reinit_after_fork.
|
|
423
|
+
_next_fork_role = None
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _after_fork_in_child() -> None:
|
|
427
|
+
global _lock
|
|
428
|
+
# The child must not reuse a lock whose state was copied from the parent.
|
|
429
|
+
_lock = threading.RLock()
|
|
430
|
+
reinit_after_fork()
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
if hasattr(os, "register_at_fork"):
|
|
434
|
+
os.register_at_fork(
|
|
435
|
+
before=_before_fork,
|
|
436
|
+
after_in_parent=_after_fork_in_parent,
|
|
437
|
+
after_in_child=_after_fork_in_child,
|
|
438
|
+
)
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def _setup_logger_provider(ctx: Context, state: _State) -> None:
|
|
442
|
+
existing = otel.existing_logger_provider()
|
|
443
|
+
if existing is not None:
|
|
444
|
+
logger.info("OpenTelemetry: reusing the LoggerProvider configured outside the plugin")
|
|
445
|
+
ctx.logger_provider = existing
|
|
446
|
+
return
|
|
447
|
+
try:
|
|
448
|
+
exporter = otel.build_log_exporter(ctx.settings.log_exporter)
|
|
449
|
+
max_queue_size = AUDIT_QUEUE_SIZE if ctx.settings.audit.enabled else None
|
|
450
|
+
ctx.logger_provider = otel.build_logger_provider(ctx.resource, exporter, max_queue_size=max_queue_size)
|
|
451
|
+
state.owns_logger_provider = True
|
|
452
|
+
except Exception as exc:
|
|
453
|
+
logger.warning("OpenTelemetry: log export disabled: could not build exporter: %s", _describe(exc, ctx.settings))
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def _build_tracer_provider(settings: conf.Settings, resource):
|
|
457
|
+
cfg = settings.traces
|
|
458
|
+
return otel.build_tracer_provider(
|
|
459
|
+
resource, otel.build_span_exporter(cfg.exporter), otel.build_sampler(cfg.sampler, cfg.sampler_arg)
|
|
460
|
+
)
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def _setup_tracer_provider(ctx: Context, state: _State) -> None:
|
|
464
|
+
existing = otel.existing_tracer_provider()
|
|
465
|
+
if existing is not None:
|
|
466
|
+
logger.info("OpenTelemetry: reusing the TracerProvider configured outside the plugin")
|
|
467
|
+
ctx.tracer_provider = existing
|
|
468
|
+
return
|
|
469
|
+
try:
|
|
470
|
+
provider = _build_tracer_provider(ctx.settings, ctx.resource)
|
|
471
|
+
except Exception as exc:
|
|
472
|
+
logger.warning(
|
|
473
|
+
"OpenTelemetry: trace export disabled: could not build exporter: %s", _describe(exc, ctx.settings)
|
|
474
|
+
)
|
|
475
|
+
return
|
|
476
|
+
ctx.tracer_provider = otel.SwitchableTracerProvider(provider)
|
|
477
|
+
state.owns_tracer_provider = True
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def _build_metrics_pipeline(settings: conf.Settings, resource) -> otel.MetricsPipeline:
|
|
481
|
+
cfg = settings.metrics
|
|
482
|
+
return otel.MetricsPipeline(
|
|
483
|
+
resource, otel.build_metric_exporter(cfg.exporter), interval=cfg.export_interval, timeout=cfg.exporter.timeout
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def _setup_meter_provider(ctx: Context, state: _State) -> None:
|
|
488
|
+
existing = otel.existing_meter_provider()
|
|
489
|
+
if existing is not None:
|
|
490
|
+
logger.info("OpenTelemetry: reusing the MeterProvider configured outside the plugin")
|
|
491
|
+
# Wrapped all the same, so the plugin's instruments can be switched off in a forked work-horse.
|
|
492
|
+
ctx.meter_provider = otel.SwitchableMeterProvider(existing)
|
|
493
|
+
return
|
|
494
|
+
try:
|
|
495
|
+
pipeline = _build_metrics_pipeline(ctx.settings, ctx.resource)
|
|
496
|
+
except Exception as exc:
|
|
497
|
+
logger.warning(
|
|
498
|
+
"OpenTelemetry: metric export disabled: could not build exporter: %s", _describe(exc, ctx.settings)
|
|
499
|
+
)
|
|
500
|
+
return
|
|
501
|
+
ctx.meter_provider = otel.SwitchableMeterProvider(pipeline.provider)
|
|
502
|
+
state.metrics_pipeline = pipeline
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def _candidate_modules(ctx: Context) -> list[Module]:
|
|
506
|
+
modules: list[Module] = []
|
|
507
|
+
if ctx.logger_provider is not None:
|
|
508
|
+
modules.append(LogsModule())
|
|
509
|
+
# The audit receiver also counts changes, which works without a log pipeline.
|
|
510
|
+
if ctx.logger_provider is not None or ctx.meter_provider is not None:
|
|
511
|
+
modules.append(AuditModule())
|
|
512
|
+
# One set of instrumentors serves spans and HTTP metrics.
|
|
513
|
+
if ctx.tracer_provider is not None or ctx.meter_provider is not None:
|
|
514
|
+
modules.append(TracesModule())
|
|
515
|
+
# RQ: worker wraps in the rqworker process; the enqueue wrap wherever spans are recorded.
|
|
516
|
+
if ctx.role == ROLE_RQWORKER or ctx.tracer_provider is not None:
|
|
517
|
+
modules.append(RqModule())
|
|
518
|
+
if ctx.meter_provider is not None:
|
|
519
|
+
modules.append(RuntimeModule())
|
|
520
|
+
return modules
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _describe(exc: BaseException, settings: conf.Settings) -> str:
|
|
524
|
+
"""Format exc for a log message, redacting exporter header values. Never raises.
|
|
525
|
+
|
|
526
|
+
Called from inside except blocks, including in the at-fork hook, so a badly behaved exception
|
|
527
|
+
(for example one whose __str__ itself raises) must never turn into an exception escaping the
|
|
528
|
+
hook that is handling it.
|
|
529
|
+
"""
|
|
530
|
+
try:
|
|
531
|
+
message = str(exc)
|
|
532
|
+
# Run the known-literal replacements on the FULL message first: str.replace is linear, so
|
|
533
|
+
# this is safe on arbitrarily long input, and it means a secret that would straddle the
|
|
534
|
+
# truncation cut below is still matched and redacted in full.
|
|
535
|
+
for exporter in settings.exporters():
|
|
536
|
+
message = message.replace(exporter.endpoint, conf._redact_userinfo(exporter.endpoint))
|
|
537
|
+
for value in exporter.headers.values():
|
|
538
|
+
if value:
|
|
539
|
+
message = message.replace(value, conf.REDACTED)
|
|
540
|
+
if len(message) > _MESSAGE_LIMIT:
|
|
541
|
+
message = message[:_MESSAGE_LIMIT]
|
|
542
|
+
# Drop a partial token left dangling at the cut (for example the prefix of a secret
|
|
543
|
+
# that was not caught above, such as URL userinfo) by cutting back to the last
|
|
544
|
+
# whitespace in the kept text. If the kept text has no whitespace at all, we cannot
|
|
545
|
+
# tell whether it ends mid-token, so drop it entirely rather than risk leaking a
|
|
546
|
+
# prefix of a secret.
|
|
547
|
+
cut = None
|
|
548
|
+
for i in range(len(message) - 1, -1, -1):
|
|
549
|
+
if message[i].isspace():
|
|
550
|
+
cut = i
|
|
551
|
+
break
|
|
552
|
+
message = message[:cut] if cut is not None else ""
|
|
553
|
+
message += " [truncated]"
|
|
554
|
+
# Only applied to the now-bounded text: these patterns can backtrack catastrophically on
|
|
555
|
+
# long input containing ":" or "//" but no "@".
|
|
556
|
+
if "@" in message:
|
|
557
|
+
message = _USERINFO.sub(f"{conf.REDACTED}@", message)
|
|
558
|
+
message = _TOKEN_USERINFO.sub(rf"\1{conf.REDACTED}@", message)
|
|
559
|
+
return f"{type(exc).__name__}: {message}"
|
|
560
|
+
except Exception:
|
|
561
|
+
return type(exc).__name__
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
def _uwsgi_module():
|
|
565
|
+
"""Return the uwsgi module when running inside uWSGI, otherwise None.
|
|
566
|
+
|
|
567
|
+
The module is provided by the uWSGI runtime itself and never exists as an installed package.
|
|
568
|
+
"""
|
|
569
|
+
try:
|
|
570
|
+
import uwsgi
|
|
571
|
+
except Exception:
|
|
572
|
+
return None
|
|
573
|
+
return uwsgi
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
def _integrate_uwsgi(uwsgi_module) -> None:
|
|
577
|
+
# uWSGI forks workers in C and only runs Python's at-fork hooks with py-call-osafterfork.
|
|
578
|
+
# post_fork_hook is called in every worker after fork. If both fire, the second
|
|
579
|
+
# re-initialisation is a no-op because it is keyed on the PID.
|
|
580
|
+
previous = getattr(uwsgi_module, "post_fork_hook", None)
|
|
581
|
+
if not getattr(previous, "_netbox_otel", False):
|
|
582
|
+
|
|
583
|
+
def post_fork_hook():
|
|
584
|
+
try:
|
|
585
|
+
if previous is not None:
|
|
586
|
+
previous()
|
|
587
|
+
finally:
|
|
588
|
+
_after_fork_in_child()
|
|
589
|
+
|
|
590
|
+
post_fork_hook._netbox_otel = True
|
|
591
|
+
uwsgi_module.post_fork_hook = post_fork_hook
|
|
592
|
+
opt = getattr(uwsgi_module, "opt", None) or {}
|
|
593
|
+
if uwsgi_threads_disabled(opt, "pyuwsgi" in sys.modules):
|
|
594
|
+
logger.warning(UWSGI_THREADS_WARNING)
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
def uwsgi_threads_disabled(opt: Mapping, embedded_in_python: bool) -> bool:
|
|
598
|
+
"""True when uWSGI will not run threads started by the application.
|
|
599
|
+
|
|
600
|
+
pyuwsgi (the PyPI package: uWSGI embedded in an already running interpreter) always has
|
|
601
|
+
thread support. The classic uwsgi binary needs enable-threads, which --threads implies.
|
|
602
|
+
"""
|
|
603
|
+
if embedded_in_python:
|
|
604
|
+
return False
|
|
605
|
+
return not (_truthy(opt.get("enable-threads")) or _truthy(opt.get("threads")))
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def _truthy(value) -> bool:
|
|
609
|
+
if isinstance(value, bytes):
|
|
610
|
+
value = value.decode(errors="ignore")
|
|
611
|
+
if isinstance(value, str):
|
|
612
|
+
return value.strip().lower() not in ("", "0", "false", "no", "off")
|
|
613
|
+
return bool(value)
|