scope-analytics 0.1.2__tar.gz → 0.1.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. {scope_analytics-0.1.2/scope_analytics.egg-info → scope_analytics-0.1.4}/PKG-INFO +8 -1
  2. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/README.md +4 -0
  3. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/__init__.py +67 -1
  4. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/cli.py +18 -1
  5. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/client.py +18 -3
  6. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/config.py +29 -0
  7. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/context.py +107 -0
  8. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/events.py +154 -16
  9. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/patches/_streaming.py +18 -4
  10. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/patches/anthropic_patch.py +21 -6
  11. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/patches/gemini_patch.py +10 -2
  12. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/patches/google_genai_patch.py +10 -2
  13. scope_analytics-0.1.4/scope_analytics/patches/http_patch.py +612 -0
  14. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/patches/openai_patch.py +10 -2
  15. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/queue.py +30 -1
  16. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/supported_versions.py +23 -13
  17. {scope_analytics-0.1.2 → scope_analytics-0.1.4/scope_analytics.egg-info}/PKG-INFO +8 -1
  18. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics.egg-info/SOURCES.txt +5 -1
  19. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics.egg-info/requires.txt +3 -0
  20. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/setup.py +9 -1
  21. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/tests/test_capture_contract.py +30 -0
  22. scope_analytics-0.1.4/tests/test_cli_exec.py +130 -0
  23. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/tests/test_coverage_honesty.py +12 -9
  24. scope_analytics-0.1.4/tests/test_http_patch.py +822 -0
  25. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/tests/test_patch_versions.py +31 -2
  26. scope_analytics-0.1.4/tests/test_user_facing.py +63 -0
  27. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/LICENSE +0 -0
  28. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/MANIFEST.in +0 -0
  29. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/auto.py +0 -0
  30. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/deployment.py +0 -0
  31. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/middleware.py +0 -0
  32. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/patches/__init__.py +0 -0
  33. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics/patches/_capture.py +0 -0
  34. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics.egg-info/dependency_links.txt +0 -0
  35. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics.egg-info/entry_points.txt +0 -0
  36. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/scope_analytics.egg-info/top_level.txt +0 -0
  37. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/setup.cfg +0 -0
  38. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/tests/test_capture_contract_google.py +0 -0
  39. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/tests/test_deployment.py +0 -0
  40. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/tests/test_identify.py +0 -0
  41. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/tests/test_identity_bridge.py +0 -0
  42. {scope_analytics-0.1.2 → scope_analytics-0.1.4}/tests/test_redaction.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scope-analytics
3
- Version: 0.1.2
3
+ Version: 0.1.4
4
4
  Summary: AI-powered analytics SDK for backend applications with automatic LLM tracking
5
5
  Home-page: https://scopeai.dev
6
6
  Author: Scope AI
@@ -24,6 +24,9 @@ Provides-Extra: dev
24
24
  Requires-Dist: pytest>=7.0.0; extra == "dev"
25
25
  Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
26
26
  Requires-Dist: pytest-mock>=3.10.0; extra == "dev"
27
+ Requires-Dist: requests>=2.25.0; extra == "dev"
28
+ Requires-Dist: aiohttp>=3.8.0; extra == "dev"
29
+ Requires-Dist: httpx2>=2.0.0; extra == "dev"
27
30
  Requires-Dist: black>=23.0.0; extra == "dev"
28
31
  Requires-Dist: flake8>=6.0.0; extra == "dev"
29
32
  Provides-Extra: openai
@@ -55,6 +58,7 @@ AI-powered analytics for backend applications with **zero-code LLM conversation
55
58
  ## Features
56
59
 
57
60
  - **Automatic LLM Tracking**: Captures OpenAI, Anthropic, and Gemini calls automatically
61
+ - **Outbound Call Tracking**: Records the calls your app makes to other services (httpx, httpx2, requests, aiohttp, urllib) — destination, status, latency and failures. Never the request or response bodies.
58
62
  - **Session Correlation**: Links backend events to frontend user sessions via `X-Scope-Session-ID` header
59
63
  - **Conversation Intelligence**: Classifies LLM calls as user-facing vs background jobs
60
64
  - **Zero Code Changes**: Drop-in integration with automatic monkey-patching
@@ -171,6 +175,7 @@ scope-run --dry-run uvicorn main:app
171
175
  | `SCOPE_ENDPOINT` | No | Custom API endpoint |
172
176
  | `SCOPE_DEBUG` | No | Set to 'true' for debug logging |
173
177
  | `SCOPE_ENVIRONMENT` | No | Environment name (default: production) |
178
+ | `SCOPE_CAPTURE_HTTP` | No | Capture outbound calls to other services (default: on; `false` turns it off) |
174
179
 
175
180
  ## Configuration (Code-Based)
176
181
 
@@ -181,6 +186,8 @@ scope = ScopeAnalytics(
181
186
  auto_patch=True, # Optional: Auto-patch LLM libraries (default: True)
182
187
  batch_size=10, # Optional: Events per batch (default: 10)
183
188
  batch_timeout_seconds=5, # Optional: Max wait time (default: 5)
189
+ capture_http=True, # Optional: Capture outbound calls (default: True,
190
+ # independent of auto_patch)
184
191
  debug=False, # Optional: Enable debug logging
185
192
  environment="production", # Optional: Environment name
186
193
  )
@@ -5,6 +5,7 @@ AI-powered analytics for backend applications with **zero-code LLM conversation
5
5
  ## Features
6
6
 
7
7
  - **Automatic LLM Tracking**: Captures OpenAI, Anthropic, and Gemini calls automatically
8
+ - **Outbound Call Tracking**: Records the calls your app makes to other services (httpx, httpx2, requests, aiohttp, urllib) — destination, status, latency and failures. Never the request or response bodies.
8
9
  - **Session Correlation**: Links backend events to frontend user sessions via `X-Scope-Session-ID` header
9
10
  - **Conversation Intelligence**: Classifies LLM calls as user-facing vs background jobs
10
11
  - **Zero Code Changes**: Drop-in integration with automatic monkey-patching
@@ -121,6 +122,7 @@ scope-run --dry-run uvicorn main:app
121
122
  | `SCOPE_ENDPOINT` | No | Custom API endpoint |
122
123
  | `SCOPE_DEBUG` | No | Set to 'true' for debug logging |
123
124
  | `SCOPE_ENVIRONMENT` | No | Environment name (default: production) |
125
+ | `SCOPE_CAPTURE_HTTP` | No | Capture outbound calls to other services (default: on; `false` turns it off) |
124
126
 
125
127
  ## Configuration (Code-Based)
126
128
 
@@ -131,6 +133,8 @@ scope = ScopeAnalytics(
131
133
  auto_patch=True, # Optional: Auto-patch LLM libraries (default: True)
132
134
  batch_size=10, # Optional: Events per batch (default: 10)
133
135
  batch_timeout_seconds=5, # Optional: Max wait time (default: 5)
136
+ capture_http=True, # Optional: Capture outbound calls (default: True,
137
+ # independent of auto_patch)
134
138
  debug=False, # Optional: Enable debug logging
135
139
  environment="production", # Optional: Environment name
136
140
  )
@@ -17,6 +17,7 @@ from .patches.openai_patch import OpenAIPatcher
17
17
  from .patches.anthropic_patch import AnthropicPatcher
18
18
  from .patches.gemini_patch import GeminiPatcher
19
19
  from .patches.google_genai_patch import GoogleGenaiPatcher
20
+ from .patches.http_patch import HTTPPatcher
20
21
  from .middleware import (
21
22
  ScopeSessionMiddleware,
22
23
  FlaskScopeMiddleware,
@@ -102,6 +103,7 @@ class ScopeAnalytics:
102
103
  max_queue_size: int = 1000,
103
104
  debug: bool = False,
104
105
  environment: Optional[str] = None,
106
+ capture_http: Optional[bool] = None,
105
107
  ):
106
108
  """
107
109
  Initialize Scope Analytics SDK
@@ -121,6 +123,9 @@ class ScopeAnalytics:
121
123
  max_queue_size: Max events to queue (default: 1000)
122
124
  debug: Enable debug logging (default: False)
123
125
  environment: Environment name (default: production)
126
+ capture_http: Capture the outbound calls this app makes to other services
127
+ (default: on; SCOPE_CAPTURE_HTTP=false turns it off). Destination, status,
128
+ latency and failures only — never request or response bodies.
124
129
  """
125
130
  # Defined before anything can fail so every instance — including a
126
131
  # disabled shell — carries it, and the no-op guards below never AttributeError.
@@ -141,6 +146,7 @@ class ScopeAnalytics:
141
146
  max_queue_size=max_queue_size,
142
147
  debug=debug,
143
148
  environment=environment,
149
+ capture_http=capture_http,
144
150
  )
145
151
  except ValueError as e:
146
152
  # self.config doesn't exist yet, so go straight to the same
@@ -157,6 +163,8 @@ class ScopeAnalytics:
157
163
  # middleware's SDK.
158
164
  self.patches = []
159
165
  self.uncovered_llm_libraries = []
166
+ self.outbound_http_libraries = []
167
+ self._warned_ship_failure = False
160
168
  self._llm_sdks_detected = []
161
169
  return
162
170
 
@@ -183,8 +191,14 @@ class ScopeAnalytics:
183
191
  self.anthropic_patcher = AnthropicPatcher(self)
184
192
  self.gemini_patcher = GeminiPatcher(self) # legacy google-generativeai (EOL)
185
193
  self.google_genai_patcher = GoogleGenaiPatcher(self) # unified google-genai (modern)
186
- self.patches = [] # List of successfully applied patches
194
+ self.http_patcher = HTTPPatcher(self) # outbound calls to other services
195
+ self._warned_ship_failure = False # the delivery-failure warning fires once
196
+ self.patches = [] # List of successfully applied LLM patches (LLM ONLY — see below)
187
197
  self.uncovered_llm_libraries = [] # Known-uncovered LLM libs detected at startup
198
+ # Outbound HTTP coverage is tracked separately and deliberately NOT in self.patches:
199
+ # that list is reported to the backend as `patched_llm_libraries`, and an HTTP client
200
+ # in it would claim LLM coverage this SDK does not have.
201
+ self.outbound_http_libraries = []
188
202
  self._llm_sdks_detected = [] # Recognized SDKs whose import succeeded (≠ patched)
189
203
 
190
204
  # Start queue background thread
@@ -203,6 +217,9 @@ class ScopeAnalytics:
203
217
  if self.config.auto_patch:
204
218
  self._apply_patches()
205
219
 
220
+ # Outbound capture runs on its own switch (see _apply_http_capture).
221
+ self._apply_http_capture()
222
+
206
223
  # Emit a single deployment_detected event when this boot's commit differs
207
224
  # from the previous boot's (PRODUCT.md §4.2). No-op when no SHA is known.
208
225
  self._emit_deployment_detected()
@@ -220,6 +237,12 @@ class ScopeAnalytics:
220
237
  coverage = {
221
238
  "patched_llm_libraries": [name_map.get(p, p) for p in self.patches],
222
239
  "uncovered_llm_libraries": list(self.uncovered_llm_libraries),
240
+ # Which HTTP clients this deploy captures outbound calls through.
241
+ # Empty when capture is switched off — the record that distinguishes
242
+ # "this app makes no outbound calls" from "we were not looking".
243
+ # Stored on the deploy event; the coverage report reads live
244
+ # presence today and could read this receipt as well.
245
+ "outbound_http_libraries": list(self.outbound_http_libraries),
223
246
  }
224
247
  event = self.deployment.build_deployment_event(coverage=coverage)
225
248
  if self.event_formatter.validate_event(event):
@@ -301,6 +324,32 @@ class ScopeAnalytics:
301
324
 
302
325
  self._report_coverage_gaps()
303
326
 
327
+ def _apply_http_capture(self):
328
+ """Capture the calls this app makes TO other services.
329
+
330
+ Its own switch, deliberately not `auto_patch`: that one is about instrumenting LLM
331
+ libraries, and an app that turns it off (custom provider handling, an unsupported
332
+ SDK) still wants to know which services it calls and how they are behaving.
333
+ SCOPE_CAPTURE_HTTP=false is the knob for this one.
334
+
335
+ Coverage is tracked apart from self.patches, which is reported to the backend as
336
+ `patched_llm_libraries`: a patched HTTP client says nothing about whether this
337
+ app's LLM calls are covered, and letting it silence the "no LLM SDK detected"
338
+ warning would hide a real gap.
339
+ """
340
+ if not self.config.capture_http:
341
+ self.config.log("Outbound HTTP capture disabled (SCOPE_CAPTURE_HTTP=false)")
342
+ return
343
+ if self.http_patcher.patch():
344
+ self.outbound_http_libraries = list(self.http_patcher.patched_libraries)
345
+ else:
346
+ # httpx ships with this SDK, so reaching here means the shape we instrument
347
+ # has moved — outbound calls silently stop being captured. Say so.
348
+ self.config.warn(
349
+ "Scope could not instrument any HTTP client — the outbound calls this app "
350
+ "makes to other services will NOT be captured. Set SCOPE_DEBUG=true for details."
351
+ )
352
+
304
353
  def _report_coverage_gaps(self):
305
354
  """Coverage honesty at startup (rule b): a known gap must never look like
306
355
  working coverage.
@@ -441,6 +490,18 @@ class ScopeAnalytics:
441
490
 
442
491
  if not success:
443
492
  self.config.log(f"⚠️ Failed to ship {len(events)} events")
493
+ if not self._warned_ship_failure:
494
+ # Event LOSS: a batch the backend refused (rate limit, auth, outage) is
495
+ # dropped, and until now that was visible only in debug mode. It matters
496
+ # more since outbound capture: more events means the per-IP ingest limit
497
+ # is reachable, and a dropped batch takes llm_call events with it.
498
+ self._warned_ship_failure = True
499
+ self.config.warn(
500
+ f"Could not deliver {len(events)} events to Scope — they were dropped. "
501
+ f"Check the API key and the endpoint; if this is volume, "
502
+ f"SCOPE_CAPTURE_HTTP=false turns off outbound HTTP capture. "
503
+ f"This warning is shown once."
504
+ )
444
505
 
445
506
  def shutdown(self):
446
507
  """
@@ -460,6 +521,11 @@ class ScopeAnalytics:
460
521
  if self.client:
461
522
  self.client.ship_timeout = 5.0
462
523
 
524
+ # Remove the outbound-HTTP wrappers FIRST: anything captured after the queue
525
+ # stops is appended to a queue that will never ship it, and the customer's own
526
+ # atexit handlers can still be making calls at this point.
527
+ self.http_patcher.unpatch()
528
+
463
529
  # Stop queue and flush remaining events
464
530
  if self.queue:
465
531
  self.queue.stop()
@@ -221,8 +221,25 @@ Environment Variables:
221
221
  print(f"[Scope SDK] Starting with auto-instrumentation...")
222
222
  print(f"[Scope SDK] Command: {' '.join(final_command)}")
223
223
 
224
- # Execute the wrapped command
224
+ # Execute the wrapped command.
225
+ #
226
+ # POSIX: replace this process (exec) instead of spawning a child. scope-run's whole job is
227
+ # done once the environment is prepared, and staying resident as a wrapper breaks graceful
228
+ # shutdown in containers: as PID 1 it neither installs handlers nor forwards signals, so
229
+ # `docker stop`/Cloud Run/K8s SIGTERM is swallowed, the runtime escalates to SIGKILL, and
230
+ # the SDK's synchronous shutdown flush never runs — silently losing the final event batch
231
+ # on every instance stop. After exec the app itself is PID 1 and owns its signals.
232
+ #
233
+ # The sitecustomize temp dir outlives this process by design — reload/fork respawns
234
+ # (e.g. `uvicorn --reload`) re-import it from PYTHONPATH — so no cleanup is attempted
235
+ # (atexit couldn't run in a replaced process anyway; the OS tmp reaper collects it).
236
+ #
237
+ # Windows: exec* detaches from the waiting console/parent, so keep the subprocess wrapper.
225
238
  try:
239
+ if os.name == 'posix':
240
+ sys.stdout.flush()
241
+ sys.stderr.flush()
242
+ os.execvpe(final_command[0], final_command, env)
226
243
  result = subprocess.run(final_command, env=env)
227
244
  return result.returncode
228
245
  except FileNotFoundError:
@@ -8,6 +8,8 @@ import json
8
8
  import httpx
9
9
  from typing import List, Dict, Any
10
10
 
11
+ from .context import ScopeContext
12
+
11
13
 
12
14
  class ScopeAPIClient:
13
15
  """
@@ -59,6 +61,17 @@ class ScopeAPIClient:
59
61
  if not events:
60
62
  return True
61
63
 
64
+ # Shipping is Scope's OWN outbound HTTP. The marker below is what stops the
65
+ # outbound-HTTP patcher from capturing this POST — which would enqueue an event,
66
+ # which would trigger another POST, forever. It is set HERE rather than at the
67
+ # queue because this method is also called directly by the final shutdown flush,
68
+ # and because a contextvar set on the caller's thread would not be visible on the
69
+ # queue's background thread (contexts do not cross threads); setting it inside the
70
+ # call means it is always set on whichever thread is actually shipping.
71
+ with ScopeContext.scope_internal():
72
+ return self._ship(events)
73
+
74
+ def _ship(self, events: List[Dict[str, Any]]) -> bool:
62
75
  try:
63
76
  self.config.log(f"Shipping {len(events)} events to {self.endpoint}")
64
77
 
@@ -119,9 +132,11 @@ class ScopeAPIClient:
119
132
  try:
120
133
  self.config.log("Testing API connection...")
121
134
 
122
- # Simple health check (could be a ping endpoint)
123
- # For now, just verify we can reach the endpoint
124
- response = self.client.get(f"{self.config.endpoint}/health")
135
+ # Same reason as ship_events: Scope's own traffic is never captured.
136
+ with ScopeContext.scope_internal():
137
+ # Simple health check (could be a ping endpoint)
138
+ # For now, just verify we can reach the endpoint
139
+ response = self.client.get(f"{self.config.endpoint}/health")
125
140
 
126
141
  if response.status_code == 200:
127
142
  self.config.log("✅ API connection successful")
@@ -27,6 +27,20 @@ DEFAULT_REDACT_PATTERNS = [
27
27
  ]
28
28
 
29
29
 
30
+ def _env_flag(name: str, default: bool) -> bool:
31
+ """Read a boolean SCOPE_* env var. Unset (or unrecognized) keeps the default, so a
32
+ typo'd value can never silently turn a default-on feature off."""
33
+ raw = os.getenv(name)
34
+ if raw is None:
35
+ return default
36
+ value = raw.strip().lower()
37
+ if value in ("true", "1", "yes", "on"):
38
+ return True
39
+ if value in ("false", "0", "no", "off"):
40
+ return False
41
+ return default
42
+
43
+
30
44
  class ScopeConfig:
31
45
  """Configuration for Scope Analytics SDK"""
32
46
 
@@ -41,6 +55,7 @@ class ScopeConfig:
41
55
  debug: bool = False,
42
56
  environment: Optional[str] = None,
43
57
  redact_patterns: Optional[list] = None,
58
+ capture_http: Optional[bool] = None,
44
59
  ):
45
60
  """
46
61
  Initialize Scope SDK configuration
@@ -54,6 +69,9 @@ class ScopeConfig:
54
69
  max_queue_size: Maximum events to queue (oldest dropped if exceeded)
55
70
  debug: Enable debug logging
56
71
  environment: Environment name (production, staging, development)
72
+ capture_http: Capture the outbound HTTP calls the app makes to other services.
73
+ Default None = on (SCOPE_CAPTURE_HTTP=false turns it off). Method, URL, status,
74
+ latency and errors only — never request or response bodies.
57
75
  redact_patterns: Regex patterns to scrub from prompt/response/bodies. Default None =
58
76
  credential scrub ON (DEFAULT_REDACT_PATTERNS — password/api_key/token/secret in
59
77
  key=value form). Pass [] to opt out and store raw at full fidelity, or your own
@@ -81,6 +99,16 @@ class ScopeConfig:
81
99
  # Patching
82
100
  self.auto_patch = auto_patch
83
101
 
102
+ # Outbound HTTP capture — ON by default, because an install that silently skips
103
+ # the app's calls to other services would leave a hole the user never hears about
104
+ # (an upstream API's failures would simply not exist). Off is one env var away for
105
+ # an app whose outbound volume it doesn't want: SCOPE_CAPTURE_HTTP=false.
106
+ self.capture_http = (
107
+ _env_flag("SCOPE_CAPTURE_HTTP", default=True)
108
+ if capture_http is None
109
+ else bool(capture_http)
110
+ )
111
+
84
112
  # Batching
85
113
  self.batch_size = batch_size
86
114
  self.batch_timeout_seconds = batch_timeout_seconds
@@ -109,6 +137,7 @@ class ScopeConfig:
109
137
  return {
110
138
  "endpoint": self.endpoint,
111
139
  "auto_patch": self.auto_patch,
140
+ "capture_http": self.capture_http,
112
141
  "batch_size": self.batch_size,
113
142
  "batch_timeout_seconds": self.batch_timeout_seconds,
114
143
  "max_queue_size": self.max_queue_size,
@@ -4,6 +4,8 @@ Handles session ID propagation across async boundaries
4
4
  """
5
5
 
6
6
  import contextvars
7
+ import functools
8
+ import inspect
7
9
  from typing import Optional
8
10
  from contextlib import contextmanager
9
11
  import uuid
@@ -48,6 +50,25 @@ _in_scope_context: contextvars.ContextVar[bool] = contextvars.ContextVar(
48
50
  'scope_in_scope_context', default=False
49
51
  )
50
52
 
53
+ # "A provider call this SDK is already capturing is running right now."
54
+ #
55
+ # The LLM patchers set this for the duration of the provider method they wrap (and
56
+ # for the deferred stream helpers, while the stream is being opened/consumed inside
57
+ # our own proxies). The outbound-HTTP patcher reads it and stays silent: the HTTP
58
+ # request underneath a captured provider call is already recorded as the llm_call
59
+ # itself, and recording it a second time as an external_api_call would double every
60
+ # AI app's dominant traffic.
61
+ #
62
+ # WHY A FLAG AND NOT A HOST LIST: a list of provider hostnames is wrong in both
63
+ # directions. It hides real outbound calls when an app points its OpenAI client at
64
+ # Azure, OpenRouter, a proxy or a local model (base_url overrides are common), and it
65
+ # silences calls to a provider we did NOT manage to instrument — exactly the gap the
66
+ # user needs to see. The flag excludes precisely what we actually captured, and it
67
+ # stays correct when a new provider is added.
68
+ _provider_call_in_flight: contextvars.ContextVar[bool] = contextvars.ContextVar(
69
+ 'scope_provider_call_in_flight', default=False
70
+ )
71
+
51
72
 
52
73
  class ScopeContext:
53
74
  """
@@ -165,6 +186,22 @@ class ScopeContext:
165
186
  thread_id_context.set(None)
166
187
  user_agent_context.set(None)
167
188
 
189
+ # One id for everything this process does with no visitor attached. Machine-driven
190
+ # events (a cron job's outbound calls, a worker's retries) have no session, but the
191
+ # event schema needs one — and minting a fresh temp id per call would file 10,000
192
+ # calls as 10,000 one-event "sessions", which is what every session count in the
193
+ # product would then report. One per process is the truthful shape: this was one
194
+ # background process, not ten thousand visits. Still `temp_`, so it reads as "no
195
+ # visitor" everywhere a temp session already does.
196
+ _background_session_id = None
197
+
198
+ @staticmethod
199
+ def background_session_id() -> str:
200
+ """The stable temp session id for this process's context-less events."""
201
+ if ScopeContext._background_session_id is None:
202
+ ScopeContext._background_session_id = f"temp_bg_{uuid.uuid4().hex[:12]}"
203
+ return ScopeContext._background_session_id
204
+
168
205
  @staticmethod
169
206
  def generate_temp_session_id() -> str:
170
207
  """
@@ -241,6 +278,33 @@ class ScopeContext:
241
278
  """
242
279
  _in_scope_context.set(False)
243
280
 
281
+ # =========================================================================
282
+ # Provider-call marker - lets outbound-HTTP capture skip what is already an llm_call
283
+ # =========================================================================
284
+
285
+ @staticmethod
286
+ def is_provider_call_in_flight() -> bool:
287
+ """True while a provider call this SDK captures as an ``llm_call`` is executing.
288
+
289
+ Read by the outbound-HTTP patcher so one LLM request is one event, not two.
290
+ """
291
+ return _provider_call_in_flight.get()
292
+
293
+ @staticmethod
294
+ @contextmanager
295
+ def provider_call():
296
+ """Mark the enclosed block as a provider call Scope already captures.
297
+
298
+ Re-entrant (nested provider calls restore the previous value), context-local
299
+ (a concurrent request or task is unaffected), and total — the flag is always
300
+ restored, including when the provider raises.
301
+ """
302
+ token = _provider_call_in_flight.set(True)
303
+ try:
304
+ yield
305
+ finally:
306
+ _provider_call_in_flight.reset(token)
307
+
244
308
  @staticmethod
245
309
  @contextmanager
246
310
  def scope_internal():
@@ -260,3 +324,46 @@ class ScopeContext:
260
324
  yield
261
325
  finally:
262
326
  _in_scope_context.set(previous)
327
+
328
+
329
+ def mark_provider_call(func):
330
+ """Wrap a provider method so everything it does — including the HTTP request it
331
+ issues underneath — runs marked as "already captured as an llm_call".
332
+
333
+ Applied by every LLM patcher at the moment it wraps a provider method, so the
334
+ outbound-HTTP patcher records one event per LLM call instead of two. Transparent:
335
+ the provider's own return value, exceptions and signature pass through untouched.
336
+
337
+ Two shapes are handled because provider SDKs use both. A true ``async def`` is
338
+ wrapped as a coroutine function. A *sync* function that RETURNS an awaitable —
339
+ openai's ``@required_args`` decorator produces exactly this, so
340
+ ``iscoroutinefunction`` reads False on its async resources — has the mark
341
+ re-entered around the await, because the flag must be set while the request is
342
+ actually in flight, not merely while the coroutine is being constructed.
343
+
344
+ A sync function that returns a lazily-opened STREAM is deliberately not covered
345
+ here: nothing is requested yet at return time. Those are marked where the stream
346
+ is opened and consumed instead (patches/_streaming.py, and the anthropic stream
347
+ helper), which is the moment the connection is really made.
348
+ """
349
+ if inspect.iscoroutinefunction(func):
350
+
351
+ @functools.wraps(func)
352
+ async def async_marked(*args, **kwargs):
353
+ with ScopeContext.provider_call():
354
+ return await func(*args, **kwargs)
355
+
356
+ return async_marked
357
+
358
+ @functools.wraps(func)
359
+ def marked(*args, **kwargs):
360
+ with ScopeContext.provider_call():
361
+ result = func(*args, **kwargs)
362
+ if inspect.isawaitable(result):
363
+ async def awaited():
364
+ with ScopeContext.provider_call():
365
+ return await result
366
+ return awaited()
367
+ return result
368
+
369
+ return marked