scope-analytics 0.1.4__tar.gz → 0.1.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scope_analytics-0.1.4/scope_analytics.egg-info → scope_analytics-0.1.5}/PKG-INFO +16 -5
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/README.md +13 -4
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/__init__.py +101 -3
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/cli.py +59 -2
- scope_analytics-0.1.5/scope_analytics/client.py +241 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/config.py +5 -1
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/context.py +6 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/events.py +65 -2
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/middleware.py +63 -6
- scope_analytics-0.1.5/scope_analytics/patches/_raw_response.py +137 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/patches/anthropic_patch.py +18 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/patches/openai_patch.py +18 -24
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/queue.py +89 -14
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/supported_versions.py +38 -11
- {scope_analytics-0.1.4 → scope_analytics-0.1.5/scope_analytics.egg-info}/PKG-INFO +16 -5
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics.egg-info/SOURCES.txt +7 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics.egg-info/requires.txt +2 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/setup.py +5 -1
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_capture_contract.py +376 -0
- scope_analytics-0.1.5/tests/test_cli_exec.py +250 -0
- scope_analytics-0.1.5/tests/test_client_ip.py +96 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_coverage_honesty.py +6 -0
- scope_analytics-0.1.5/tests/test_discard_reply.py +139 -0
- scope_analytics-0.1.5/tests/test_fork_safety.py +582 -0
- scope_analytics-0.1.5/tests/test_frameworks_langchain.py +215 -0
- scope_analytics-0.1.5/tests/test_manual_event_and_debug_output.py +43 -0
- scope_analytics-0.1.5/tests/test_query_scrub.py +122 -0
- scope_analytics-0.1.4/scope_analytics/client.py +0 -155
- scope_analytics-0.1.4/tests/test_cli_exec.py +0 -130
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/LICENSE +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/MANIFEST.in +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/auto.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/deployment.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/patches/__init__.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/patches/_capture.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/patches/_streaming.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/patches/gemini_patch.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/patches/google_genai_patch.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics/patches/http_patch.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics.egg-info/dependency_links.txt +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics.egg-info/entry_points.txt +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/scope_analytics.egg-info/top_level.txt +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/setup.cfg +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_capture_contract_google.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_deployment.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_http_patch.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_identify.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_identity_bridge.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_patch_versions.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_redaction.py +0 -0
- {scope_analytics-0.1.4 → scope_analytics-0.1.5}/tests/test_user_facing.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scope-analytics
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.5
|
|
4
4
|
Summary: AI-powered analytics SDK for backend applications with automatic LLM tracking
|
|
5
5
|
Home-page: https://scopeai.dev
|
|
6
6
|
Author: Scope AI
|
|
@@ -27,6 +27,8 @@ Requires-Dist: pytest-mock>=3.10.0; extra == "dev"
|
|
|
27
27
|
Requires-Dist: requests>=2.25.0; extra == "dev"
|
|
28
28
|
Requires-Dist: aiohttp>=3.8.0; extra == "dev"
|
|
29
29
|
Requires-Dist: httpx2>=2.0.0; extra == "dev"
|
|
30
|
+
Requires-Dist: gunicorn>=20.0; extra == "dev"
|
|
31
|
+
Requires-Dist: flask>=2.0; extra == "dev"
|
|
30
32
|
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
31
33
|
Requires-Dist: flake8>=6.0.0; extra == "dev"
|
|
32
34
|
Provides-Extra: openai
|
|
@@ -91,15 +93,24 @@ export SCOPE_API_KEY="sk_live_..."
|
|
|
91
93
|
# Instead of: uvicorn main:app --reload
|
|
92
94
|
# Run: scope-run uvicorn main:app --reload
|
|
93
95
|
|
|
94
|
-
# Instead of: gunicorn app:app -w 4
|
|
96
|
+
# Instead of: gunicorn app:app -w 4 (Flask, Django)
|
|
95
97
|
# Run: scope-run gunicorn app:app -w 4
|
|
96
98
|
|
|
99
|
+
# Instead of: gunicorn main:app -w 4 -k uvicorn_worker.UvicornWorker (FastAPI; pip install uvicorn-worker)
|
|
100
|
+
# Run: scope-run gunicorn main:app -w 4 -k uvicorn_worker.UvicornWorker
|
|
101
|
+
|
|
97
102
|
# Instead of: flask run --port 5000
|
|
98
103
|
# Run: scope-run flask run --port 5000
|
|
99
104
|
```
|
|
100
105
|
|
|
101
106
|
That's it! Your LLM calls are now automatically tracked AND correlated
|
|
102
|
-
with frontend sessions (for FastAPI, Flask, Django).
|
|
107
|
+
with frontend sessions (for FastAPI, Flask, Django). Workers forked by gunicorn or
|
|
108
|
+
Celery ship on their own timer (0.1.5). LangChain / LangGraph, LlamaIndex, PydanticAI,
|
|
109
|
+
Instructor and the OpenAI Agents SDK are captured through the SDK they call
|
|
110
|
+
(LangChain needs 0.1.5).
|
|
111
|
+
|
|
112
|
+
**Serverless / notebooks** — no start command to prefix: create `scope = ScopeAnalytics()`
|
|
113
|
+
at module import and call `scope.flush(timeout=2.0)` before the handler returns (0.1.5).
|
|
103
114
|
|
|
104
115
|
### Option B: Code-Based Installation
|
|
105
116
|
|
|
@@ -176,6 +187,7 @@ scope-run --dry-run uvicorn main:app
|
|
|
176
187
|
| `SCOPE_DEBUG` | No | Set to 'true' for debug logging |
|
|
177
188
|
| `SCOPE_ENVIRONMENT` | No | Environment name (default: production) |
|
|
178
189
|
| `SCOPE_CAPTURE_HTTP` | No | Capture outbound calls to other services (default: on; `false` turns it off) |
|
|
190
|
+
| `HTTPS_PROXY` / `NO_PROXY` / `SSL_CERT_FILE` | No | Honoured for the SDK's own connection to Scope (read from the environment only) |
|
|
179
191
|
|
|
180
192
|
## Configuration (Code-Based)
|
|
181
193
|
|
|
@@ -230,8 +242,7 @@ def my_request_handler(request):
|
|
|
230
242
|
Add the frontend SDK to link user interactions with backend LLM calls:
|
|
231
243
|
|
|
232
244
|
```html
|
|
233
|
-
<script src="https://cdn.scopeai.dev/v1/sdk.js"
|
|
234
|
-
data-api-key="pk_live_your_public_key"></script>
|
|
245
|
+
<script src="https://cdn.scopeai.dev/v1/sdk.js?token=pk_live_your_public_key" async></script>
|
|
235
246
|
```
|
|
236
247
|
|
|
237
248
|
The frontend SDK automatically:
|
|
@@ -38,15 +38,24 @@ export SCOPE_API_KEY="sk_live_..."
|
|
|
38
38
|
# Instead of: uvicorn main:app --reload
|
|
39
39
|
# Run: scope-run uvicorn main:app --reload
|
|
40
40
|
|
|
41
|
-
# Instead of: gunicorn app:app -w 4
|
|
41
|
+
# Instead of: gunicorn app:app -w 4 (Flask, Django)
|
|
42
42
|
# Run: scope-run gunicorn app:app -w 4
|
|
43
43
|
|
|
44
|
+
# Instead of: gunicorn main:app -w 4 -k uvicorn_worker.UvicornWorker (FastAPI; pip install uvicorn-worker)
|
|
45
|
+
# Run: scope-run gunicorn main:app -w 4 -k uvicorn_worker.UvicornWorker
|
|
46
|
+
|
|
44
47
|
# Instead of: flask run --port 5000
|
|
45
48
|
# Run: scope-run flask run --port 5000
|
|
46
49
|
```
|
|
47
50
|
|
|
48
51
|
That's it! Your LLM calls are now automatically tracked AND correlated
|
|
49
|
-
with frontend sessions (for FastAPI, Flask, Django).
|
|
52
|
+
with frontend sessions (for FastAPI, Flask, Django). Workers forked by gunicorn or
|
|
53
|
+
Celery ship on their own timer (0.1.5). LangChain / LangGraph, LlamaIndex, PydanticAI,
|
|
54
|
+
Instructor and the OpenAI Agents SDK are captured through the SDK they call
|
|
55
|
+
(LangChain needs 0.1.5).
|
|
56
|
+
|
|
57
|
+
**Serverless / notebooks** — no start command to prefix: create `scope = ScopeAnalytics()`
|
|
58
|
+
at module import and call `scope.flush(timeout=2.0)` before the handler returns (0.1.5).
|
|
50
59
|
|
|
51
60
|
### Option B: Code-Based Installation
|
|
52
61
|
|
|
@@ -123,6 +132,7 @@ scope-run --dry-run uvicorn main:app
|
|
|
123
132
|
| `SCOPE_DEBUG` | No | Set to 'true' for debug logging |
|
|
124
133
|
| `SCOPE_ENVIRONMENT` | No | Environment name (default: production) |
|
|
125
134
|
| `SCOPE_CAPTURE_HTTP` | No | Capture outbound calls to other services (default: on; `false` turns it off) |
|
|
135
|
+
| `HTTPS_PROXY` / `NO_PROXY` / `SSL_CERT_FILE` | No | Honoured for the SDK's own connection to Scope (read from the environment only) |
|
|
126
136
|
|
|
127
137
|
## Configuration (Code-Based)
|
|
128
138
|
|
|
@@ -177,8 +187,7 @@ def my_request_handler(request):
|
|
|
177
187
|
Add the frontend SDK to link user interactions with backend LLM calls:
|
|
178
188
|
|
|
179
189
|
```html
|
|
180
|
-
<script src="https://cdn.scopeai.dev/v1/sdk.js"
|
|
181
|
-
data-api-key="pk_live_your_public_key"></script>
|
|
190
|
+
<script src="https://cdn.scopeai.dev/v1/sdk.js?token=pk_live_your_public_key" async></script>
|
|
182
191
|
```
|
|
183
192
|
|
|
184
193
|
The frontend SDK automatically:
|
|
@@ -5,6 +5,7 @@ AI-powered analytics with automatic LLM conversation tracking
|
|
|
5
5
|
|
|
6
6
|
import atexit
|
|
7
7
|
import logging
|
|
8
|
+
import sys
|
|
8
9
|
from typing import Optional
|
|
9
10
|
|
|
10
11
|
from .config import ScopeConfig, DEFAULT_REDACT_PATTERNS
|
|
@@ -45,7 +46,7 @@ def _resolve_version() -> str:
|
|
|
45
46
|
|
|
46
47
|
__version__ = _resolve_version()
|
|
47
48
|
__all__ = [
|
|
48
|
-
"ScopeAnalytics",
|
|
49
|
+
"ScopeAnalytics", # .flush(timeout=) ships queued events now (serverless, notebooks)
|
|
49
50
|
"ScopeContext",
|
|
50
51
|
"ScopeConfig",
|
|
51
52
|
"DEFAULT_REDACT_PATTERNS", # default credential-scrub patterns (on by default; [] to opt out)
|
|
@@ -184,6 +185,7 @@ class ScopeAnalytics:
|
|
|
184
185
|
flush_callback=self._flush_events,
|
|
185
186
|
config=self.config,
|
|
186
187
|
event_enricher=self.deployment.stamp,
|
|
188
|
+
on_fork=self._after_fork,
|
|
187
189
|
)
|
|
188
190
|
|
|
189
191
|
# Initialize patchers
|
|
@@ -200,10 +202,16 @@ class ScopeAnalytics:
|
|
|
200
202
|
# in it would claim LLM coverage this SDK does not have.
|
|
201
203
|
self.outbound_http_libraries = []
|
|
202
204
|
self._llm_sdks_detected = [] # Recognized SDKs whose import succeeded (≠ patched)
|
|
205
|
+
self._celery_hooked = False
|
|
203
206
|
|
|
204
207
|
# Start queue background thread
|
|
205
208
|
self.queue.start()
|
|
206
209
|
|
|
210
|
+
# A Celery worker already loaded? Then the prefork children's shutdown signal
|
|
211
|
+
# flushes their last batch (see _connect_celery_shutdown). Workers started through
|
|
212
|
+
# scope-run load Celery AFTER this point; the fork hook connects it in each child.
|
|
213
|
+
self._connect_celery_shutdown()
|
|
214
|
+
|
|
207
215
|
# Register shutdown hook
|
|
208
216
|
atexit.register(self.shutdown)
|
|
209
217
|
|
|
@@ -380,6 +388,87 @@ class ScopeAnalytics:
|
|
|
380
388
|
# Coverage reporting must never break startup.
|
|
381
389
|
self.config.log(f"Coverage-gap check failed: {e}")
|
|
382
390
|
|
|
391
|
+
def _after_fork(self):
|
|
392
|
+
"""Runs in a forked CHILD once the queue has restarted its flush thread there.
|
|
393
|
+
The inherited ship client shares its pooled sockets with the parent — two
|
|
394
|
+
processes writing one TCP connection interleave bytes — so the child gets its
|
|
395
|
+
own. The background session id is per process, so the child mints its own. And
|
|
396
|
+
a Celery child flushes on its own shutdown signal (see _connect_celery_shutdown)."""
|
|
397
|
+
try:
|
|
398
|
+
# The inherited client is dropped, never closed: httpx's pool close() takes
|
|
399
|
+
# the pool's thread lock, and if the parent's flush thread held it at the
|
|
400
|
+
# instant of the fork the child would block forever inside this hook (the
|
|
401
|
+
# thread that holds it does not exist here). Unreferenced, the sockets close
|
|
402
|
+
# by garbage collection, which takes no pool lock.
|
|
403
|
+
self.client = ScopeAPIClient(self.config)
|
|
404
|
+
except Exception as e: # noqa: BLE001 - never let a fork hook break the child
|
|
405
|
+
self.config.log(f"Could not rebuild the ship client after fork: {e}")
|
|
406
|
+
ScopeContext.reset_background_session()
|
|
407
|
+
# A receiver connected in the parent is inherited and works here (it closes over
|
|
408
|
+
# this same instance), so connect only if nothing was connected before the fork
|
|
409
|
+
# — e.g. the scope-run path, where Celery was not yet imported at init.
|
|
410
|
+
self._connect_celery_shutdown()
|
|
411
|
+
|
|
412
|
+
def _connect_celery_shutdown(self):
|
|
413
|
+
"""Celery's prefork children exit through os._exit(), which skips atexit — so
|
|
414
|
+
their last partial batch (under batch_size, under batch_timeout) would be lost.
|
|
415
|
+
When Celery is loaded (checked in sys.modules; never imported here to find out),
|
|
416
|
+
the child's ``worker_process_shutdown`` signal flushes with a hard two-second
|
|
417
|
+
limit, so a broken network can never hold a worker's shutdown."""
|
|
418
|
+
if self._celery_hooked or "celery" not in sys.modules:
|
|
419
|
+
return
|
|
420
|
+
try:
|
|
421
|
+
from celery import signals
|
|
422
|
+
except Exception: # noqa: BLE001 - a half-imported celery is not ours to break
|
|
423
|
+
return
|
|
424
|
+
|
|
425
|
+
def _on_worker_process_shutdown(**_kwargs):
|
|
426
|
+
self.flush(timeout=2.0)
|
|
427
|
+
|
|
428
|
+
# Celery holds signal receivers weakly by default; keep a strong reference.
|
|
429
|
+
self._celery_shutdown_handler = _on_worker_process_shutdown
|
|
430
|
+
try:
|
|
431
|
+
signals.worker_process_shutdown.connect(_on_worker_process_shutdown, weak=False)
|
|
432
|
+
self._celery_hooked = True
|
|
433
|
+
self.config.log("Celery worker_process_shutdown hook connected")
|
|
434
|
+
except Exception as e: # noqa: BLE001
|
|
435
|
+
self.config.log(f"Could not connect the Celery shutdown hook: {e}")
|
|
436
|
+
|
|
437
|
+
def flush(self, timeout: Optional[float] = None):
|
|
438
|
+
"""
|
|
439
|
+
Ship every queued event now.
|
|
440
|
+
|
|
441
|
+
For the places the background timer can't be trusted to fire: a serverless
|
|
442
|
+
handler about to be frozen, a notebook cell, a worker process about to exit.
|
|
443
|
+
Without ``timeout`` the events are shipped on the calling thread and the call
|
|
444
|
+
returns when they are delivered (the client's normal request timeout applies).
|
|
445
|
+
With ``timeout`` the call returns within that many seconds whatever the network
|
|
446
|
+
does: the delivery runs on a helper thread with that limit per request and the
|
|
447
|
+
caller waits for it only until the deadline, so a slow batch still lands if the
|
|
448
|
+
process lives on, and a stuck one never holds the caller past the budget.
|
|
449
|
+
Batches already in flight on the batch-size path are waited for within the same
|
|
450
|
+
budget. Capture stays live afterwards — unlike shutdown(), which also unpatches.
|
|
451
|
+
|
|
452
|
+
Args:
|
|
453
|
+
timeout: Optional hard limit, in seconds, on how long this call may take.
|
|
454
|
+
"""
|
|
455
|
+
if not self.enabled:
|
|
456
|
+
return
|
|
457
|
+
if timeout is None:
|
|
458
|
+
self.queue.flush(synchronous=True)
|
|
459
|
+
self.queue.wait_inflight(self.client.ship_timeout or 30.0)
|
|
460
|
+
return
|
|
461
|
+
import threading
|
|
462
|
+
import time
|
|
463
|
+
deadline = time.monotonic() + timeout
|
|
464
|
+
worker = threading.Thread(
|
|
465
|
+
target=self.queue.flush, kwargs={"synchronous": True, "timeout": timeout},
|
|
466
|
+
daemon=True, name="ScopeAnalytics-Flush",
|
|
467
|
+
)
|
|
468
|
+
worker.start()
|
|
469
|
+
worker.join(max(0.0, deadline - time.monotonic()))
|
|
470
|
+
self.queue.wait_inflight(max(0.0, deadline - time.monotonic()))
|
|
471
|
+
|
|
383
472
|
def track_event(self, event_type: str, properties: dict):
|
|
384
473
|
"""
|
|
385
474
|
Manually track an event
|
|
@@ -409,6 +498,11 @@ class ScopeAnalytics:
|
|
|
409
498
|
# libraries — was silently dropped at validation.
|
|
410
499
|
from datetime import datetime, timezone
|
|
411
500
|
event.setdefault("timestamp", datetime.now(timezone.utc).isoformat())
|
|
501
|
+
# The two stamps every auto-captured event carries, so a manual event is the same
|
|
502
|
+
# to the SDK-status and environment views (a manual-only install used to show no
|
|
503
|
+
# version at all).
|
|
504
|
+
event.setdefault("sdk_version", self.config.sdk_version)
|
|
505
|
+
event.setdefault("environment", self.config.environment)
|
|
412
506
|
if "session_id" not in event:
|
|
413
507
|
# generate_temp (NOT ensure_session_id): ensure_ would PIN the temp id
|
|
414
508
|
# into the contextvar as a side effect — in a long-lived worker with no
|
|
@@ -478,15 +572,19 @@ class ScopeAnalytics:
|
|
|
478
572
|
if traits:
|
|
479
573
|
self.config.log(f"User traits: {traits}")
|
|
480
574
|
|
|
481
|
-
def _flush_events(self, events: list):
|
|
575
|
+
def _flush_events(self, events: list, timeout: Optional[float] = None):
|
|
482
576
|
"""
|
|
483
577
|
Callback for flushing events to API
|
|
484
578
|
Called by event queue when batch is ready
|
|
485
579
|
|
|
486
580
|
Args:
|
|
487
581
|
events: List of events to flush
|
|
582
|
+
timeout: Per-request limit for this batch only (set by ``flush(timeout=)``)
|
|
488
583
|
"""
|
|
489
|
-
|
|
584
|
+
# The limit is passed on only when set, so a one-argument stand-in for ship_events
|
|
585
|
+
# (tests, custom sinks) keeps working.
|
|
586
|
+
success = (self.client.ship_events(events) if timeout is None
|
|
587
|
+
else self.client.ship_events(events, timeout=timeout))
|
|
490
588
|
|
|
491
589
|
if not success:
|
|
492
590
|
self.config.log(f"⚠️ Failed to ship {len(events)} events")
|
|
@@ -21,6 +21,7 @@ Environment Variables:
|
|
|
21
21
|
|
|
22
22
|
import os
|
|
23
23
|
import sys
|
|
24
|
+
import shutil
|
|
24
25
|
import subprocess
|
|
25
26
|
import argparse
|
|
26
27
|
import tempfile
|
|
@@ -88,8 +89,31 @@ import os
|
|
|
88
89
|
if os.environ.get('SCOPE_AUTO_INSTRUMENT') == 'true':
|
|
89
90
|
try:
|
|
90
91
|
import scope_analytics.auto
|
|
91
|
-
|
|
92
|
-
|
|
92
|
+
_launcher = os.environ.get('SCOPE_RUN_VERSION')
|
|
93
|
+
_running = getattr(scope_analytics, '__version__', None)
|
|
94
|
+
if _launcher and _running and _launcher != _running:
|
|
95
|
+
# The app's interpreter found a DIFFERENT copy of scope-analytics than the one
|
|
96
|
+
# scope-run was launched from (a global uvicorn next to an un-activated venv is
|
|
97
|
+
# the usual shape). That copy is what instruments the app; say so.
|
|
98
|
+
import sys
|
|
99
|
+
print(
|
|
100
|
+
"[Scope SDK] scope-run is scope-analytics " + _launcher + " but the interpreter "
|
|
101
|
+
"running your app (" + sys.executable + ") imported scope-analytics " + _running
|
|
102
|
+
+ " — that copy is what instruments the app. To run " + _launcher + " here: "
|
|
103
|
+
+ sys.executable + " -m pip install scope-analytics==" + _launcher,
|
|
104
|
+
file=sys.stderr,
|
|
105
|
+
)
|
|
106
|
+
except ImportError as e:
|
|
107
|
+
# The interpreter running the app is not the one scope-analytics was installed into
|
|
108
|
+
# (a pyenv shim, a system python, a multi-stage image). The user ran scope-run to be
|
|
109
|
+
# instrumented, so silence here would be the worst outcome: say it, once, loudly.
|
|
110
|
+
import sys
|
|
111
|
+
print(
|
|
112
|
+
"[Scope SDK] scope-analytics could not be imported by this interpreter ("
|
|
113
|
+
+ sys.executable + "): " + str(e) + " — NOTHING is instrumented. Install it for "
|
|
114
|
+
"this interpreter: " + sys.executable + " -m pip install scope-analytics",
|
|
115
|
+
file=sys.stderr,
|
|
116
|
+
)
|
|
93
117
|
except Exception as e:
|
|
94
118
|
import sys
|
|
95
119
|
print(f"[Scope SDK] Warning: Auto-instrumentation failed: {e}", file=sys.stderr)
|
|
@@ -187,6 +211,38 @@ Environment Variables:
|
|
|
187
211
|
env = os.environ.copy()
|
|
188
212
|
env['SCOPE_AUTO_INSTRUMENT'] = 'true'
|
|
189
213
|
|
|
214
|
+
# The command is resolved on PATH, so scope-run invoked from a virtualenv that is not
|
|
215
|
+
# activated (its absolute path in a Procfile, a Dockerfile, a systemd unit) used to
|
|
216
|
+
# launch whatever `uvicorn` / `gunicorn` came first on PATH — a different interpreter
|
|
217
|
+
# with a different scope-analytics, or none — and nothing said so. The launcher's own
|
|
218
|
+
# bin directory goes first, which is exactly what activating the virtualenv does; and
|
|
219
|
+
# the sitecustomize compares this version with the one the app's interpreter imports,
|
|
220
|
+
# saying so on stderr if they differ.
|
|
221
|
+
launcher_bin = os.path.dirname(os.path.abspath(sys.executable))
|
|
222
|
+
env['PATH'] = launcher_bin + os.pathsep + env.get('PATH', '')
|
|
223
|
+
version = _get_version()
|
|
224
|
+
if version != 'unknown' and not version.startswith('0.0.0'):
|
|
225
|
+
env['SCOPE_RUN_VERSION'] = version
|
|
226
|
+
|
|
227
|
+
# macOS: a Python process that runs a thread (the SDK's flush thread) and then forks
|
|
228
|
+
# workers — gunicorn, Celery's prefork pool, multiprocessing — trips the Objective-C
|
|
229
|
+
# runtime's fork-safety check in every child the first time it touches a system
|
|
230
|
+
# framework (the app's own HTTP client doing its proxy lookup is enough), and the child
|
|
231
|
+
# aborts: "+[NSCharacterSet initialize] may have been in progress in another thread when
|
|
232
|
+
# fork() was called". Measured with `scope-run gunicorn -w 2`: 0 of 10 runs healthy
|
|
233
|
+
# without this, 8 of 8 with it. This variable is the documented remedy for Python-and-fork
|
|
234
|
+
# on macOS; it means nothing anywhere else, so a Linux deployment is untouched. A value
|
|
235
|
+
# the user already set wins — and when scope-run sets it, it says so: nothing silent.
|
|
236
|
+
if sys.platform == 'darwin' and 'OBJC_DISABLE_INITIALIZE_FORK_SAFETY' not in env:
|
|
237
|
+
env['OBJC_DISABLE_INITIALIZE_FORK_SAFETY'] = 'YES'
|
|
238
|
+
print(
|
|
239
|
+
"[Scope SDK] macOS: set OBJC_DISABLE_INITIALIZE_FORK_SAFETY=YES for this command — "
|
|
240
|
+
"without it a server that forks workers (gunicorn, Celery) aborts in every child on "
|
|
241
|
+
"macOS's Objective-C fork-safety check. Set the variable yourself, to any value, to "
|
|
242
|
+
"choose otherwise.",
|
|
243
|
+
file=sys.stderr,
|
|
244
|
+
)
|
|
245
|
+
|
|
190
246
|
if args.debug:
|
|
191
247
|
env['SCOPE_DEBUG'] = 'true'
|
|
192
248
|
|
|
@@ -211,6 +267,7 @@ Environment Variables:
|
|
|
211
267
|
print("[Scope SDK] Dry run - would execute:")
|
|
212
268
|
print(f" SCOPE_AUTO_INSTRUMENT=true")
|
|
213
269
|
print(f" PYTHONPATH={env['PYTHONPATH']}")
|
|
270
|
+
print(f" {executable} resolves to: {shutil.which(executable, path=env['PATH']) or '(not found on PATH)'}")
|
|
214
271
|
if args.debug:
|
|
215
272
|
print(f" SCOPE_DEBUG=true")
|
|
216
273
|
print(f" {' '.join(final_command)}")
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
"""
|
|
2
|
+
HTTP client for shipping events to Scope Analytics API
|
|
3
|
+
Handles async communication with the backend API
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import ssl
|
|
9
|
+
import urllib.request
|
|
10
|
+
from urllib.parse import urlsplit
|
|
11
|
+
|
|
12
|
+
import httpx
|
|
13
|
+
from typing import List, Dict, Any, Optional
|
|
14
|
+
|
|
15
|
+
from .context import ScopeContext
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class ScopeAPIClient:
|
|
19
|
+
"""
|
|
20
|
+
Async HTTP client for Scope Analytics API
|
|
21
|
+
|
|
22
|
+
Handles:
|
|
23
|
+
- Event shipping to /api/events endpoint
|
|
24
|
+
- Authentication with secret API key
|
|
25
|
+
- Retry logic with exponential backoff
|
|
26
|
+
- Error handling
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
def __init__(self, config):
|
|
30
|
+
"""
|
|
31
|
+
Initialize API client
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
config: SDK configuration
|
|
35
|
+
"""
|
|
36
|
+
self.config = config
|
|
37
|
+
self.endpoint = f"{config.endpoint}/api/events"
|
|
38
|
+
|
|
39
|
+
# Per-request timeout override; shutdown sets this to a small value so the
|
|
40
|
+
# final synchronous flush can't hold process exit for the full 30s.
|
|
41
|
+
self.ship_timeout = None
|
|
42
|
+
|
|
43
|
+
# Discard reasons already said (see _note_discard) — once each, not per batch.
|
|
44
|
+
self._warned_discard_reasons = set()
|
|
45
|
+
|
|
46
|
+
# Create HTTP client
|
|
47
|
+
self.client = self._build_http_client(config)
|
|
48
|
+
|
|
49
|
+
self.config.log(f"API client initialized: {self.endpoint}")
|
|
50
|
+
|
|
51
|
+
@staticmethod
|
|
52
|
+
def ship_settings(endpoint: str) -> Dict[str, Any]:
|
|
53
|
+
"""What the ship client takes from the environment: the proxy for the endpoint's
|
|
54
|
+
scheme (``HTTPS_PROXY`` / ``HTTP_PROXY`` / ``ALL_PROXY``, with ``NO_PROXY`` honoured)
|
|
55
|
+
and a custom CA bundle (``SSL_CERT_FILE`` / ``SSL_CERT_DIR``). Read from the
|
|
56
|
+
environment ONLY — never from the operating system's proxy service — see
|
|
57
|
+
_build_http_client for why. Pure, so it can be tested without a client."""
|
|
58
|
+
parts = urlsplit(endpoint)
|
|
59
|
+
proxies = urllib.request.getproxies_environment()
|
|
60
|
+
proxy = proxies.get(parts.scheme) or proxies.get("all")
|
|
61
|
+
if proxy and urllib.request.proxy_bypass_environment(parts.hostname or "", proxies):
|
|
62
|
+
proxy = None
|
|
63
|
+
return {
|
|
64
|
+
"proxy": proxy or None,
|
|
65
|
+
"cafile": os.environ.get("SSL_CERT_FILE") or None,
|
|
66
|
+
"capath": os.environ.get("SSL_CERT_DIR") or None,
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
@classmethod
|
|
70
|
+
def _build_http_client(cls, config) -> httpx.Client:
|
|
71
|
+
"""Scope's own ship client, configured from the environment WITHOUT asking the
|
|
72
|
+
operating system for its proxy settings.
|
|
73
|
+
|
|
74
|
+
httpx's default (``trust_env=True``) resolves proxies through urllib's
|
|
75
|
+
``getproxies()``, which on macOS falls through to the system proxy service when no
|
|
76
|
+
``*_proxy`` variable is set. That call leaves CoreFoundation state behind that is
|
|
77
|
+
not fork-safe: a server that forks its workers after this SDK initialised in the
|
|
78
|
+
master (gunicorn without --preload) then had every worker crash inside the SAME
|
|
79
|
+
lookup — made by the app's own OpenAI/Anthropic client — intermittently. So the
|
|
80
|
+
three things ``trust_env`` would read are read here directly (proxies, NO_PROXY, a
|
|
81
|
+
custom CA bundle) and the OS proxy service is never consulted. The fourth,
|
|
82
|
+
``.netrc``, does not apply: every request carries the Bearer header."""
|
|
83
|
+
settings = cls.ship_settings(config.endpoint)
|
|
84
|
+
verify: Any = True
|
|
85
|
+
if settings["cafile"] or settings["capath"]:
|
|
86
|
+
verify = ssl.create_default_context(cafile=settings["cafile"], capath=settings["capath"])
|
|
87
|
+
transport_kwargs: Dict[str, Any] = {"verify": verify}
|
|
88
|
+
if settings["proxy"]:
|
|
89
|
+
transport_kwargs["proxy"] = httpx.Proxy(settings["proxy"])
|
|
90
|
+
return httpx.Client(
|
|
91
|
+
timeout=30.0,
|
|
92
|
+
headers={
|
|
93
|
+
"Authorization": f"Bearer {config.api_key}",
|
|
94
|
+
"Content-Type": "application/json",
|
|
95
|
+
"User-Agent": f"scope-analytics-python/{config.sdk_version}",
|
|
96
|
+
},
|
|
97
|
+
trust_env=False,
|
|
98
|
+
transport=httpx.HTTPTransport(**transport_kwargs),
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
def ship_events(self, events: List[Dict[str, Any]], timeout: Optional[float] = None) -> bool:
|
|
102
|
+
"""
|
|
103
|
+
Ship batch of events to API
|
|
104
|
+
|
|
105
|
+
Args:
|
|
106
|
+
events: List of event dictionaries
|
|
107
|
+
timeout: Per-request limit for THIS call only (``flush(timeout=)``); None means
|
|
108
|
+
the client's own ``ship_timeout`` (None there = httpx's default).
|
|
109
|
+
|
|
110
|
+
Returns:
|
|
111
|
+
True if successful, False otherwise
|
|
112
|
+
"""
|
|
113
|
+
if not events:
|
|
114
|
+
return True
|
|
115
|
+
|
|
116
|
+
# Shipping is Scope's OWN outbound HTTP. The marker below is what stops the
|
|
117
|
+
# outbound-HTTP patcher from capturing this POST — which would enqueue an event,
|
|
118
|
+
# which would trigger another POST, forever. It is set HERE rather than at the
|
|
119
|
+
# queue because this method is also called directly by the final shutdown flush,
|
|
120
|
+
# and because a contextvar set on the caller's thread would not be visible on the
|
|
121
|
+
# queue's background thread (contexts do not cross threads); setting it inside the
|
|
122
|
+
# call means it is always set on whichever thread is actually shipping.
|
|
123
|
+
with ScopeContext.scope_internal():
|
|
124
|
+
return self._ship(events, timeout)
|
|
125
|
+
|
|
126
|
+
def _ship(self, events: List[Dict[str, Any]], timeout: Optional[float] = None) -> bool:
|
|
127
|
+
try:
|
|
128
|
+
self.config.log(f"Shipping {len(events)} events to {self.endpoint}")
|
|
129
|
+
|
|
130
|
+
# Prepare payload
|
|
131
|
+
payload = {
|
|
132
|
+
"events": events,
|
|
133
|
+
"source": self.config.sdk_source,
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
# Strict-first serialization with a LOUD lossy fallback: events are
|
|
137
|
+
# sanitized at build time, so the strict path should always win — but a
|
|
138
|
+
# single stray non-JSON value must degrade to its string form (with a
|
|
139
|
+
# warning), never lose the whole batch.
|
|
140
|
+
try:
|
|
141
|
+
body = json.dumps(payload)
|
|
142
|
+
except (TypeError, ValueError):
|
|
143
|
+
body = json.dumps(payload, default=str)
|
|
144
|
+
self.config.warn(
|
|
145
|
+
"Non-JSON value in event batch — coerced via str(). "
|
|
146
|
+
"This indicates an event-sanitization gap; please report it."
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
# Send POST request
|
|
150
|
+
kwargs = {"content": body}
|
|
151
|
+
limit = timeout if timeout is not None else self.ship_timeout
|
|
152
|
+
if limit is not None:
|
|
153
|
+
kwargs["timeout"] = limit
|
|
154
|
+
response = self.client.post(self.endpoint, **kwargs)
|
|
155
|
+
|
|
156
|
+
# Check response
|
|
157
|
+
if response.status_code == 200:
|
|
158
|
+
self._note_discard(response, len(events))
|
|
159
|
+
self.config.log(f"✅ Successfully shipped {len(events)} events")
|
|
160
|
+
return True
|
|
161
|
+
else:
|
|
162
|
+
self.config.log(
|
|
163
|
+
f"❌ Failed to ship events: HTTP {response.status_code} - {response.text}"
|
|
164
|
+
)
|
|
165
|
+
return False
|
|
166
|
+
|
|
167
|
+
except httpx.TimeoutException:
|
|
168
|
+
self.config.log("❌ Request timeout while shipping events")
|
|
169
|
+
return False
|
|
170
|
+
|
|
171
|
+
except httpx.HTTPError as e:
|
|
172
|
+
self.config.log(f"❌ HTTP error while shipping events: {e}")
|
|
173
|
+
return False
|
|
174
|
+
|
|
175
|
+
except Exception as e:
|
|
176
|
+
self.config.log(f"❌ Unexpected error while shipping events: {e}")
|
|
177
|
+
return False
|
|
178
|
+
|
|
179
|
+
def _note_discard(self, response, count: int) -> None:
|
|
180
|
+
"""A 200 can still mean "received and thrown away": the ingest service answers
|
|
181
|
+
``{"status": "discarded", "reason": ...}`` for a batch it will not keep — an IP the
|
|
182
|
+
project excludes, for instance. Reading only the status code made such batches
|
|
183
|
+
vanish without a word (the SDK logged "shipped"; the dashboard showed nothing).
|
|
184
|
+
Said once per reason, with where to look."""
|
|
185
|
+
try:
|
|
186
|
+
body = response.json()
|
|
187
|
+
except Exception: # noqa: BLE001 - a non-JSON 200 is a stored batch as far as we know
|
|
188
|
+
return
|
|
189
|
+
if not isinstance(body, dict) or body.get("status") != "discarded":
|
|
190
|
+
return
|
|
191
|
+
reason = str(body.get("reason") or "unspecified")
|
|
192
|
+
if reason in self._warned_discard_reasons:
|
|
193
|
+
return
|
|
194
|
+
self._warned_discard_reasons.add(reason)
|
|
195
|
+
# The ingest's IP-exclusion reasons are user_excluded_ip and auto_excluded_login_ip
|
|
196
|
+
# (datacenter_ip exists but never fires for backend events); anything else gets the
|
|
197
|
+
# generic pointer rather than a guess.
|
|
198
|
+
if reason.endswith("_ip"):
|
|
199
|
+
where = (
|
|
200
|
+
"Events from this machine's IP are excluded for this project — see Setup → "
|
|
201
|
+
"Security for this project in the Scope dashboard ('Your own traffic' and "
|
|
202
|
+
"'Exclusions') and remove the exclusion to see them."
|
|
203
|
+
)
|
|
204
|
+
else:
|
|
205
|
+
where = "Check the project's settings in the Scope dashboard."
|
|
206
|
+
self.config.warn(
|
|
207
|
+
f"Scope received {count} event(s) but DISCARDED them (reason: {reason}). {where} "
|
|
208
|
+
f"This is said once per reason."
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
def test_connection(self) -> bool:
|
|
212
|
+
"""
|
|
213
|
+
Test connection to API
|
|
214
|
+
|
|
215
|
+
Returns:
|
|
216
|
+
True if connection successful, False otherwise
|
|
217
|
+
"""
|
|
218
|
+
try:
|
|
219
|
+
self.config.log("Testing API connection...")
|
|
220
|
+
|
|
221
|
+
# Same reason as ship_events: Scope's own traffic is never captured.
|
|
222
|
+
with ScopeContext.scope_internal():
|
|
223
|
+
# Simple health check (could be a ping endpoint)
|
|
224
|
+
# For now, just verify we can reach the endpoint
|
|
225
|
+
response = self.client.get(f"{self.config.endpoint}/health")
|
|
226
|
+
|
|
227
|
+
if response.status_code == 200:
|
|
228
|
+
self.config.log("✅ API connection successful")
|
|
229
|
+
return True
|
|
230
|
+
else:
|
|
231
|
+
self.config.log(f"⚠️ API returned status {response.status_code}")
|
|
232
|
+
return False
|
|
233
|
+
|
|
234
|
+
except Exception as e:
|
|
235
|
+
self.config.log(f"⚠️ Could not connect to API: {e}")
|
|
236
|
+
return False
|
|
237
|
+
|
|
238
|
+
def close(self):
|
|
239
|
+
"""Close HTTP client"""
|
|
240
|
+
self.client.close()
|
|
241
|
+
self.config.log("API client closed")
|
|
@@ -149,7 +149,11 @@ class ScopeConfig:
|
|
|
149
149
|
def log(self, message: str):
|
|
150
150
|
"""Log debug message if debug mode enabled"""
|
|
151
151
|
if self.debug:
|
|
152
|
-
|
|
152
|
+
# flush: under gunicorn, in a container, or anywhere stdout is a pipe or a
|
|
153
|
+
# file, Python buffers prints — and a worker that exits through os._exit()
|
|
154
|
+
# never writes the buffer out, so SCOPE_DEBUG=true showed nothing from any
|
|
155
|
+
# worker (the master's lines appeared; the "Captured …" lines never did).
|
|
156
|
+
print(f"[Scope SDK] {message}", flush=True)
|
|
153
157
|
|
|
154
158
|
def warn(self, message: str):
|
|
155
159
|
"""Always-visible warning. Deliberately NOT gated on ``debug``.
|
|
@@ -202,6 +202,12 @@ class ScopeContext:
|
|
|
202
202
|
ScopeContext._background_session_id = f"temp_bg_{uuid.uuid4().hex[:12]}"
|
|
203
203
|
return ScopeContext._background_session_id
|
|
204
204
|
|
|
205
|
+
@staticmethod
|
|
206
|
+
def reset_background_session() -> None:
|
|
207
|
+
"""Forget this process's background session id. Called in a forked child, which
|
|
208
|
+
is a new process and must not share its parent's id."""
|
|
209
|
+
ScopeContext._background_session_id = None
|
|
210
|
+
|
|
205
211
|
@staticmethod
|
|
206
212
|
def generate_temp_session_id() -> str:
|
|
207
213
|
"""
|