hyperprobe-agent 1.12.19__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,730 @@
1
+ import sys
2
+ import os
3
+ import time
4
+ import re
5
+ import json
6
+ import math
7
+ import weakref
8
+ import threading
9
+ import json
10
+ from hyperprobe.core.serializer import serialize
11
+ from hyperprobe.core.evaluator import ProbeEvaluator
12
+ from hyperprobe.core.logger import get_logger
13
+
14
+ logger = get_logger("hyperprobe:monitor")
15
+ from hyperprobe.core.trace_extractor import extract_trace_context
16
+ from hyperprobe.protos import agent_pb2
17
+
18
+ class MonitoringEngine:
19
+ TOOL_ID_CANDIDATES = ( 3, 4)
20
+ PENDING_POLL_INTERVAL_SEC = 0.5
21
+ DURATION_TTL_SECONDS = 60.0
22
+ _JS_FLOAT_PREFIX_RE = re.compile(
23
+ r"^\s*([+-]?(?:(?:[0-9]+\.?[0-9]*)|(?:\.[0-9]+))(?:[eE][+-]?[0-9]+)?)"
24
+ )
25
+
26
+ def __init__(
27
+ self,
28
+ quota_manager,
29
+ safety_monitor,
30
+ on_capture,
31
+ custom_set_trace_id=None
32
+ ):
33
+ self.quota_manager = quota_manager
34
+ self.safety_monitor = safety_monitor
35
+ self.on_capture = on_capture
36
+ self.custom_set_trace_id = custom_set_trace_id
37
+ self.tool_id = None
38
+ self._closed = False
39
+
40
+ self.lock = threading.RLock() # Reentrant Lock for safety during callbacks
41
+
42
+ # Memory-safe weak references to pivoted code objects
43
+ self.pivoted_codes = weakref.WeakSet()
44
+ self.pivoted_locations = set() # (filepath, line_number)
45
+ self.pending_pivots = set() # (filepath, line_number)
46
+
47
+ self.active_probes = {}
48
+ self.secondary_duration_probes = {}
49
+ self.instrumented_files = set()
50
+ self.is_active = False
51
+ self.is_suspended = False
52
+ self.global_config = {}
53
+ self.total_hits = 0
54
+ self.total_skips = 0
55
+
56
+ self.duration_starts = {}
57
+ self._next_duration_cleanup = time.monotonic() + self.DURATION_TTL_SECONDS
58
+
59
+ # Cached resolved file paths
60
+ self._file_cache = {}
61
+
62
+ self._claim_tool()
63
+
64
+ def _claim_tool(self):
65
+ for tool_id in self.TOOL_ID_CANDIDATES:
66
+ claimed = False
67
+ try:
68
+ if sys.monitoring.get_tool(tool_id) is not None:
69
+ continue
70
+ sys.monitoring.use_tool_id(tool_id, "hyperprobe")
71
+ claimed = True
72
+ sys.monitoring.register_callback(
73
+ tool_id,
74
+ sys.monitoring.events.LINE,
75
+ self._line_callback,
76
+ )
77
+ self.tool_id = tool_id
78
+ return
79
+ except ValueError:
80
+ continue
81
+ except Exception as e:
82
+ logger.forceError(f"\033[1m\033[33m⚠️ [HyperProbe] CRITICAL FAIL ON TOOL ID {tool_id}: {type(e).__name__} - {str(e)}")
83
+ if claimed:
84
+ try:
85
+ sys.monitoring.free_tool_id(tool_id)
86
+ except Exception:
87
+ pass
88
+ self.tool_id = None
89
+ continue
90
+
91
+ logger.forceError("\033[1m\033[33m⚠️ [HyperProbe] No available sys.monitoring tool ID for agent. Running application uninstrumented.")
92
+
93
+ def set_global_config(self, config):
94
+ with self.lock:
95
+ self.global_config = dict(config or {})
96
+
97
+ def suspend(self):
98
+ """Disable instrumentation without discarding the configured probes."""
99
+ with self.lock:
100
+ if self._closed or self.is_suspended:
101
+ return
102
+ self.is_suspended = True
103
+ if self.tool_id is not None:
104
+ for code in list(self.pivoted_codes):
105
+ try:
106
+ sys.monitoring.set_local_events(self.tool_id, code, 0)
107
+ except (RuntimeError, ValueError):
108
+ pass
109
+ self._stop_monitoring()
110
+
111
+ def resume(self):
112
+ """Re-enable instrumentation after safety recovery."""
113
+ with self.lock:
114
+ if self._closed:
115
+ return
116
+ self.pending_pivots.update(self.pivoted_locations)
117
+ self.is_suspended = False
118
+ self._update_monitoring_state()
119
+
120
+ def get_stats(self):
121
+ with self.lock:
122
+ self._cleanup_durations_locked(time.monotonic())
123
+ stats = {"hits": self.total_hits, "skips": self.total_skips}
124
+ self.total_hits = 0
125
+ self.total_skips = 0
126
+ return stats
127
+
128
+ def _resolve_filename_cached(self, filename):
129
+ resolved = self._file_cache.get(filename)
130
+ if resolved is None:
131
+ resolved = os.path.abspath(os.path.realpath(filename))
132
+ self._file_cache[filename] = resolved
133
+ return resolved
134
+
135
+ def set_probes(self, probes):
136
+ with self.lock:
137
+ if self._closed:
138
+ return
139
+
140
+ new_active_probes = {}
141
+ new_secondary_duration_probes = {}
142
+ new_instrumented_files = set()
143
+ new_pending_pivots = set()
144
+ now = time.monotonic()
145
+
146
+ for probe in probes:
147
+ filepath = self._resolve_filepath(probe.runtime_location)
148
+ loc_key = (filepath, probe.runtime_line)
149
+ if loc_key not in new_active_probes:
150
+ new_active_probes[loc_key] = []
151
+ new_active_probes[loc_key].append(probe)
152
+ new_instrumented_files.add(filepath)
153
+
154
+ # Only search for files we haven't pivoted yet
155
+ if loc_key not in self.pivoted_locations:
156
+ new_pending_pivots.add(loc_key)
157
+
158
+ if (
159
+ getattr(probe, "type", agent_pb2.PROBE_TYPE_UNSPECIFIED)
160
+ == agent_pb2.PROBE_TYPE_DURATION
161
+ ):
162
+ secondary_line = getattr(probe, "secondary_runtime_line", 0)
163
+ if secondary_line:
164
+ secondary_location = (
165
+ getattr(probe, "secondary_runtime_location", "")
166
+ or probe.runtime_location
167
+ )
168
+ secondary_filepath = self._resolve_filepath(secondary_location)
169
+ secondary_key = (secondary_filepath, secondary_line)
170
+ if secondary_key not in new_secondary_duration_probes:
171
+ new_secondary_duration_probes[secondary_key] = []
172
+ new_secondary_duration_probes[secondary_key].append(probe)
173
+ new_instrumented_files.add(secondary_filepath)
174
+ if secondary_key not in self.pivoted_locations:
175
+ new_pending_pivots.add(secondary_key)
176
+
177
+ target_locations = set(new_active_probes) | set(new_secondary_duration_probes)
178
+
179
+ # Clean stale pivoted locations
180
+ self.pivoted_locations = {
181
+ loc for loc in self.pivoted_locations if loc in target_locations
182
+ }
183
+
184
+ # Atomic swap to guarantee thread safety
185
+ self.active_probes = new_active_probes
186
+ self.secondary_duration_probes = new_secondary_duration_probes
187
+ self.instrumented_files = new_instrumented_files
188
+ self.pending_pivots = new_pending_pivots
189
+ logger.info(
190
+ f"[HyperProbe] set_probes count={len(probes)} "
191
+ f"pending={len(self.pending_pivots)} files={len(self.instrumented_files)}"
192
+ )
193
+
194
+ self._update_monitoring_state()
195
+
196
+ def _update_monitoring_state(self):
197
+ # Must be called under a lock
198
+ if self._closed or self.tool_id is None:
199
+ return
200
+
201
+ # Clean local events for code objects from untargeted files
202
+ for code in list(self.pivoted_codes):
203
+ resolved_filename = self._resolve_filename_cached(code.co_filename)
204
+ if resolved_filename not in self.instrumented_files:
205
+ try:
206
+ sys.monitoring.set_local_events(self.tool_id, code, 0)
207
+ except ValueError:
208
+ pass
209
+ self.pivoted_codes.discard(code)
210
+
211
+ if self.is_suspended:
212
+ return
213
+
214
+ # Update global searchlight line callbacks (Keep the callback registered)
215
+ if self.pending_pivots:
216
+ logger.info(f"[HyperProbe] pending pivots remaining : {len(self.pending_pivots)}. starting global search.")
217
+ sys.monitoring.set_events(self.tool_id, sys.monitoring.events.LINE)
218
+ self.is_active = True
219
+ else:
220
+ logger.info(f"[HyperProbe] pending pivots remaining : {len(self.pending_pivots)}. stopped global search.")
221
+ sys.monitoring.set_events(self.tool_id, 0)
222
+ self.is_active = False
223
+
224
+ def _stop_monitoring(self):
225
+ # Disable global search events.
226
+ if self.tool_id is None:
227
+ return
228
+ try:
229
+ sys.monitoring.set_events(self.tool_id, 0)
230
+ except (RuntimeError, ValueError) as e:
231
+ logger.error(
232
+ f"[HyperProbe] Failed to disable monitoring: {type(e).__name__}: {e}"
233
+ )
234
+ else:
235
+ logger.info("[HyperProbe] Monitoring disabled.")
236
+ self.is_active = False
237
+
238
+ def close(self):
239
+ with self.lock:
240
+ if self._closed:
241
+ return
242
+ self._closed = True
243
+ self._stop_monitoring()
244
+
245
+ for code in list(self.pivoted_codes):
246
+ try:
247
+ sys.monitoring.set_local_events(self.tool_id, code, 0)
248
+ except (RuntimeError, ValueError):
249
+ pass
250
+
251
+ self.pivoted_codes.clear()
252
+ self.pivoted_locations.clear()
253
+ self.pending_pivots.clear()
254
+
255
+ if self.tool_id is not None:
256
+ try:
257
+ sys.monitoring.register_callback(
258
+ self.tool_id,
259
+ sys.monitoring.events.LINE,
260
+ None,
261
+ )
262
+ except Exception:
263
+ pass
264
+ try:
265
+ sys.monitoring.free_tool_id(self.tool_id)
266
+ except Exception:
267
+ pass
268
+ self.tool_id = None
269
+
270
+ def _line_callback(self, code, line_number):
271
+ with self.lock:
272
+ if self._closed or self.is_suspended:
273
+ return None
274
+
275
+ filepath = self._resolve_filename_cached(code.co_filename)
276
+ if filepath not in self.instrumented_files:
277
+ return None
278
+
279
+ loc_key = (filepath, line_number)
280
+
281
+ # Catch and Pivot Searchlight logic
282
+ if loc_key in self.pending_pivots:
283
+ try:
284
+ sys.monitoring.set_local_events(self.tool_id, code, sys.monitoring.events.LINE)
285
+ self.pivoted_codes.add(code)
286
+ self.pivoted_locations.add(loc_key)
287
+ except ValueError:
288
+ pass
289
+ self.pending_pivots.discard(loc_key)
290
+ self._update_monitoring_state()
291
+
292
+ probes_list = self.active_probes.get(loc_key)
293
+ secondary_probes_list = self.secondary_duration_probes.get(loc_key)
294
+ if not probes_list and not secondary_probes_list:
295
+ return None
296
+
297
+ # Execute hit capture outside the self.lock to avoid serialization contention
298
+ try:
299
+ # Under native PEP 669, sys._getframe(1) directly gets the monitored frame
300
+ frame = sys._getframe(1)
301
+ if frame:
302
+ for probe in probes_list or ():
303
+ self._handle_probe_hit(probe, frame, is_secondary=False)
304
+ for probe in secondary_probes_list or ():
305
+ self._handle_probe_hit(probe, frame, is_secondary=True)
306
+ except Exception:
307
+ pass
308
+
309
+ return None
310
+
311
+ def _handle_probe_hit(self, probe, frame, is_secondary=False):
312
+ probe_type = getattr(probe, "type", agent_pb2.PROBE_TYPE_UNSPECIFIED)
313
+ if probe_type in (
314
+ agent_pb2.PROBE_TYPE_COUNTER,
315
+ agent_pb2.PROBE_TYPE_METRIC,
316
+ agent_pb2.PROBE_TYPE_DURATION,
317
+ ):
318
+ self._handle_metric_probe_hit(probe, frame, is_secondary)
319
+ return
320
+
321
+ start_time = time.perf_counter()
322
+ # Check admission before touching the frame or evaluating user code.
323
+ if not self.quota_manager.can_evaluate():
324
+ with self.lock:
325
+ self.total_skips += 1
326
+ logger.error(f"[HyperProbe] Probe ID={probe.id} hit rate-limited by QuotaManager.")
327
+ return
328
+
329
+ globals_dict = frame.f_globals
330
+ locals_dict = frame.f_locals
331
+
332
+ # Get config snapshot under lock
333
+ with self.lock:
334
+ config_snapshot = self.global_config.copy() if self.global_config else {}
335
+
336
+ if probe.condition:
337
+ try:
338
+ cond_val = ProbeEvaluator.safe_eval(probe.condition, globals_dict, locals_dict)
339
+ if not cond_val:
340
+ return
341
+ except Exception as e:
342
+ self._report_error(probe.id, f"Condition evaluation failed: {type(e).__name__}: {str(e)}")
343
+ return
344
+
345
+ logger.info(f"[HyperProbe] Hit probe ID={probe.id}! Capturing telemetry...")
346
+
347
+ # Prepare parameters from snapshot
348
+ max_depth = getattr(probe, 'max_object_depth', None) or config_snapshot.get('max_object_depth', 3)
349
+ max_array_length = getattr(probe, 'max_array_length', None) or config_snapshot.get('max_array_length', 3)
350
+ max_object_properties = getattr(probe, 'max_object_properties', None) or config_snapshot.get('max_object_properties', 50)
351
+ max_string_length = getattr(probe, 'max_string_length', None) or config_snapshot.get('max_string_length', 1024)
352
+ stack_frame_depth = getattr(probe, 'stack_frame_depth', None) or config_snapshot.get('stack_frame_depth', 3)
353
+
354
+ redact_keys = config_snapshot.get('redact_keys', [])
355
+ redact_values = config_snapshot.get('redact_values', [])
356
+
357
+ redact_keys_re = re.compile("|".join(f"(?:{p})" for p in redact_keys), re.IGNORECASE) if redact_keys else None
358
+ redact_values_re = re.compile("|".join(f"(?:{p})" for p in redact_values), re.IGNORECASE) if redact_values else None
359
+
360
+ # Build telemetry structure
361
+ event = {
362
+ "probe_id": probe.id,
363
+ "timestamp_ms": int(time.time() * 1000),
364
+ "stack_frames": [],
365
+ "captured_vars_json": "",
366
+ "watch_results_json": "",
367
+ "evaluated_log": "",
368
+ "metric_value": 0.0,
369
+ "capture_error": "",
370
+ "trace_id": None
371
+ }
372
+
373
+ if getattr(probe, 'should_capture_trace_id', False):
374
+ trace_id = extract_trace_context(self.custom_set_trace_id)
375
+ if trace_id:
376
+ event["trace_id"] = trace_id
377
+ logger.info(f"[HyperProbe] trace id found : {trace_id}")
378
+
379
+ try:
380
+ # SNAPSHOT CAPTURE
381
+ if probe.type == 1:
382
+ serialization_context = {}
383
+
384
+ if getattr(probe, 'watch_expressions', None):
385
+ watches_result = ProbeEvaluator.evaluate_watches(probe.watch_expressions, globals_dict, locals_dict)
386
+ serialized_watches = {}
387
+ for watch_name, watch_value in watches_result.items():
388
+ watch_path = f"watch[{json.dumps(str(watch_name))}]"
389
+ serialized_watches[watch_name] = serialize(
390
+ watch_value,
391
+ max_depth=max_depth,
392
+ max_array_length=max_array_length,
393
+ max_object_properties=max_object_properties,
394
+ max_string_length=max_string_length,
395
+ redact_keys_re=redact_keys_re,
396
+ redact_values_re=redact_values_re,
397
+ visited=serialization_context,
398
+ path=watch_path,
399
+ )
400
+ event["watch_results_json"] = json.dumps(serialized_watches)
401
+
402
+ curr_frame = frame
403
+ depth = 0
404
+ wrapped_vars = []
405
+
406
+ while curr_frame and depth < stack_frame_depth:
407
+ event["stack_frames"].append({
408
+ "function_name": curr_frame.f_code.co_name,
409
+ "file_name": curr_frame.f_code.co_filename,
410
+ "line_number": curr_frame.f_lineno,
411
+ "column_number": 0
412
+ })
413
+
414
+ clean_locals = {k: v for k, v in curr_frame.f_locals.items() if not k.startswith('_') and k != 'hyperprobe'}
415
+ locals_path = f"frame[{depth}].scopes[0]"
416
+ serialized_locals = serialize(
417
+ clean_locals,
418
+ max_depth=max_depth,
419
+ max_array_length=max_array_length,
420
+ max_object_properties=max_object_properties,
421
+ max_string_length=max_string_length,
422
+ redact_keys_re=redact_keys_re,
423
+ redact_values_re=redact_values_re,
424
+ visited=serialization_context,
425
+ path=locals_path,
426
+ )
427
+
428
+ wrapped_vars.append([
429
+ {
430
+ "type": "local",
431
+ "name": "Local",
432
+ "vars": serialized_locals
433
+ }
434
+ ])
435
+
436
+ curr_frame = curr_frame.f_back
437
+ depth += 1
438
+
439
+ event["captured_vars_json"] = json.dumps(wrapped_vars)
440
+
441
+ # LOG TEMPLATE CAPTURE
442
+ elif probe.type == 2:
443
+ evaluated = ProbeEvaluator.evaluate_log_template(probe.template, globals_dict, locals_dict)
444
+ if len(evaluated) > max_string_length:
445
+ evaluated = evaluated[:max_string_length] + f"... [Truncated: +{len(evaluated) - max_string_length} more chars]"
446
+ if redact_values_re and redact_values_re.search(evaluated):
447
+ evaluated = redact_values_re.sub("[REDACTED Value]", evaluated)
448
+ event["evaluated_log"] = evaluated
449
+
450
+ # Safety budget calculation
451
+ # Fire callback
452
+ with self.lock:
453
+ self.total_hits += 1
454
+ self.on_capture(event)
455
+ logger.info(f"[HyperProbe] Successfully captured and queued telemetry for probe ID={probe.id}.")
456
+
457
+ except Exception as err:
458
+ err_msg = f"Capture failed: {type(err).__name__}: {str(err)}"
459
+ logger.error(f"[HyperProbe] Error during capture: {err_msg}")
460
+ event["capture_error"] = err_msg
461
+ with self.lock:
462
+ self.total_hits += 1
463
+ self.on_capture(event)
464
+ finally:
465
+ duration_ms = (time.perf_counter() - start_time) * 1000.0
466
+ self.safety_monitor.report_pause_duration(duration_ms)
467
+
468
+ def _handle_metric_probe_hit(self, probe, frame, is_secondary):
469
+ """Handle counter, metric, and duration probes without snapshot capture."""
470
+ handler_started_at = time.perf_counter()
471
+ globals_dict = frame.f_globals
472
+ locals_dict = frame.f_locals
473
+
474
+ try:
475
+ if probe.condition:
476
+ try:
477
+ if not ProbeEvaluator.safe_eval(
478
+ probe.condition, globals_dict, locals_dict
479
+ ):
480
+ return
481
+ except Exception as error:
482
+ self._report_metric_error(probe, str(error))
483
+ return
484
+
485
+ if probe.type == agent_pb2.PROBE_TYPE_COUNTER:
486
+ if not self._consume_evaluation_quota(probe.id):
487
+ return
488
+ self._emit_metric_event(probe, metric_value=1.0)
489
+ return
490
+
491
+ if probe.type == agent_pb2.PROBE_TYPE_METRIC:
492
+ metric_expression = getattr(probe, "metric_expression", "")
493
+ if not metric_expression:
494
+ # The backend validates this field. Node also emits nothing when
495
+ # an empty expression reaches the SDK.
496
+ return
497
+ try:
498
+ raw_value = ProbeEvaluator.safe_eval(
499
+ metric_expression, globals_dict, locals_dict
500
+ )
501
+ except Exception as error:
502
+ self._report_metric_error(probe, str(error))
503
+ return
504
+
505
+ if not self._consume_evaluation_quota(probe.id):
506
+ return
507
+
508
+ metric_value = self._coerce_metric_value(raw_value)
509
+ if metric_value is None:
510
+ self._emit_metric_event(
511
+ probe,
512
+ capture_error=f"Metric evaluation failed: {raw_value}",
513
+ )
514
+ else:
515
+ self._emit_metric_event(probe, metric_value=metric_value)
516
+ return
517
+
518
+ correlation_expression = getattr(
519
+ probe, "correlation_expression", ""
520
+ )
521
+ if correlation_expression:
522
+ try:
523
+ correlation_value = ProbeEvaluator.safe_eval(
524
+ correlation_expression, globals_dict, locals_dict
525
+ )
526
+ except Exception as error:
527
+ self._report_metric_error(probe, f"Error: {error}")
528
+ return
529
+ else:
530
+ correlation_value = None
531
+
532
+ try:
533
+ correlation_key = self._normalize_correlation_key(correlation_value)
534
+ except (TypeError, ValueError) as error:
535
+ self._report_metric_error(probe, str(error))
536
+ return
537
+
538
+ if not is_secondary:
539
+ self._start_duration(probe.id, correlation_key)
540
+ return
541
+
542
+ duration_ms = self._finish_duration(probe.id, correlation_key)
543
+ if duration_ms is None:
544
+ return
545
+ if not self._consume_evaluation_quota(probe.id):
546
+ return
547
+ self._emit_metric_event(probe, metric_value=duration_ms)
548
+ finally:
549
+ try:
550
+ self.safety_monitor.report_pause_duration(
551
+ (time.perf_counter() - handler_started_at) * 1000.0
552
+ )
553
+ except Exception:
554
+ # Safety accounting must never affect application execution.
555
+ pass
556
+
557
+ def _consume_evaluation_quota(self, probe_id):
558
+ if self.quota_manager.can_evaluate():
559
+ return True
560
+ with self.lock:
561
+ self.total_skips += 1
562
+ logger.error(
563
+ f"[HyperProbe] Probe ID={probe_id} hit rate-limited by QuotaManager."
564
+ )
565
+ return False
566
+
567
+ def _build_metric_event(self, probe, metric_value=0.0, capture_error=""):
568
+ event = {
569
+ "probe_id": probe.id,
570
+ "timestamp_ms": int(time.time() * 1000),
571
+ "stack_frames": [],
572
+ "captured_vars_json": "",
573
+ "watch_results_json": "",
574
+ "evaluated_log": "",
575
+ "metric_value": float(metric_value),
576
+ "capture_error": capture_error,
577
+ "trace_id": None,
578
+ }
579
+ if getattr(probe, "should_capture_trace_id", False):
580
+ event["trace_id"] = extract_trace_context(self.custom_set_trace_id)
581
+ return event
582
+
583
+ def _emit_metric_event(self, probe, metric_value=0.0, capture_error=""):
584
+ event = self._build_metric_event(probe, metric_value, capture_error)
585
+
586
+ # no need to check for bandwidth here.
587
+ # bandwidth check is owned by on_capture()
588
+
589
+ # payload_size = len(
590
+ # json.dumps(event, separators=(",", ":"), ensure_ascii=False).encode(
591
+ # "utf-8"
592
+ # )
593
+ # )
594
+
595
+ # if not self.quota_manager.can_send(payload_size):
596
+ # with self.lock:
597
+ # self.total_skips += 1
598
+ # return False
599
+
600
+ with self.lock:
601
+ self.total_hits += 1
602
+ self.on_capture(event)
603
+
604
+ def _report_metric_error(self, probe, error_message):
605
+ self._emit_metric_event(probe, capture_error=error_message)
606
+
607
+ @classmethod
608
+ def _coerce_metric_value(cls, value):
609
+ """Mirror Node's finite-number/parseFloat behavior for common values."""
610
+ if isinstance(value, bool):
611
+ return None
612
+
613
+ if isinstance(value, (int, float)):
614
+ try:
615
+ parsed = float(value)
616
+ except (OverflowError, TypeError, ValueError):
617
+ return None
618
+ return parsed if math.isfinite(parsed) else None
619
+
620
+ if isinstance(value, str):
621
+ match = cls._JS_FLOAT_PREFIX_RE.match(value)
622
+ if not match:
623
+ return None
624
+ try:
625
+ parsed = float(match.group(1))
626
+ except (OverflowError, ValueError):
627
+ return None
628
+ return parsed if math.isfinite(parsed) else None
629
+
630
+ return None
631
+
632
+ @staticmethod
633
+ def _node_type_name(value):
634
+ if isinstance(value, bool):
635
+ return "boolean"
636
+ if isinstance(value, str):
637
+ return "string"
638
+ if isinstance(value, (int, float)):
639
+ return "number"
640
+ if value is None:
641
+ return "object"
642
+ return "object"
643
+
644
+ @classmethod
645
+ def _normalize_correlation_key(cls, value):
646
+ if value is None:
647
+ normalized = "static-singleton"
648
+ elif isinstance(value, bool):
649
+ raise TypeError(
650
+ "Correlation expression must evaluate to a string or number, "
651
+ f"got {cls._node_type_name(value)}"
652
+ )
653
+ elif isinstance(value, str):
654
+ if value.startswith("Error: "):
655
+ raise ValueError(value)
656
+ normalized = value
657
+ elif isinstance(value, int):
658
+ normalized = str(value)
659
+ elif isinstance(value, float):
660
+ if math.isnan(value):
661
+ normalized = "NaN"
662
+ elif math.isinf(value):
663
+ normalized = "Infinity" if value > 0 else "-Infinity"
664
+ elif value.is_integer():
665
+ normalized = str(int(value))
666
+ else:
667
+ normalized = str(value)
668
+ else:
669
+ raise TypeError(
670
+ "Correlation expression must evaluate to a string or number, "
671
+ f"got {cls._node_type_name(value)}"
672
+ )
673
+
674
+ return normalized
675
+
676
+ def _start_duration(self, probe_id, correlation_key):
677
+ start_time = time.perf_counter()
678
+ created_at = time.monotonic()
679
+ duration_key = (probe_id, correlation_key)
680
+ with self.lock:
681
+ self._cleanup_durations_locked(created_at)
682
+ self.duration_starts[duration_key] = (start_time, created_at)
683
+
684
+ def _finish_duration(self, probe_id, correlation_key):
685
+ end_time = time.perf_counter()
686
+ duration_key = (probe_id, correlation_key)
687
+ with self.lock:
688
+ self._cleanup_durations_locked(time.monotonic())
689
+ entry = self.duration_starts.pop(duration_key, None)
690
+ if entry is None:
691
+ return None
692
+ return (end_time - entry[0]) * 1000.0
693
+
694
+ def _cleanup_durations_locked(self, now):
695
+ if now < self._next_duration_cleanup:
696
+ return
697
+
698
+ for duration_key, (_, created_at) in list(self.duration_starts.items()):
699
+ if now - created_at > self.DURATION_TTL_SECONDS:
700
+ del self.duration_starts[duration_key]
701
+
702
+ self._next_duration_cleanup = now + self.DURATION_TTL_SECONDS
703
+
704
+ def _report_error(self, probe_id, err_msg):
705
+ event = {
706
+ "probe_id": probe_id,
707
+ "timestamp_ms": int(time.time() * 1000),
708
+ "stack_frames": [],
709
+ "captured_vars_json": "",
710
+ "watch_results_json": "",
711
+ "evaluated_log": "",
712
+ "metric_value": 0.0,
713
+ "capture_error": err_msg,
714
+ "trace_id": None
715
+ }
716
+ with self.lock:
717
+ self.total_hits += 1
718
+ self.on_capture(event)
719
+
720
+ def _resolve_filepath(self, runtime_location):
721
+ if os.path.isabs(runtime_location):
722
+ return os.path.abspath(os.path.realpath(runtime_location))
723
+ cwd = os.getcwd()
724
+ parts = runtime_location.split('/')
725
+ for i in range(len(parts)):
726
+ suffix = os.path.join(*parts[i:])
727
+ attempt = os.path.abspath(os.path.realpath(os.path.join(cwd, suffix)))
728
+ if os.path.exists(attempt):
729
+ return attempt
730
+ return os.path.abspath(os.path.realpath(os.path.join(cwd, runtime_location)))