stacktrace-cli 0.3.1__py3-none-any.whl → 0.4.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. stacktrace_cli/__init__.py +1 -1
  2. stacktrace_cli/analysis.py +242 -29
  3. stacktrace_cli/build.py +130 -0
  4. stacktrace_cli/cli.py +176 -67
  5. stacktrace_cli/correlate/acquire.py +23 -2
  6. stacktrace_cli/correlate/orchestrate.py +73 -9
  7. stacktrace_cli/correlate/render.py +7 -5
  8. stacktrace_cli/daemon/__init__.py +1 -0
  9. stacktrace_cli/daemon/cli.py +106 -0
  10. stacktrace_cli/daemon/client.py +134 -0
  11. stacktrace_cli/daemon/composition.py +133 -0
  12. stacktrace_cli/daemon/observer.py +173 -0
  13. stacktrace_cli/daemon/paths.py +35 -0
  14. stacktrace_cli/daemon/protocol.py +29 -0
  15. stacktrace_cli/daemon/runtime.py +222 -0
  16. stacktrace_cli/daemon/server.py +164 -0
  17. stacktrace_cli/daemon/service.py +154 -0
  18. stacktrace_cli/daemon/store.py +368 -0
  19. stacktrace_cli/detector/blocked.py +450 -0
  20. stacktrace_cli/detector/cache.py +245 -9
  21. stacktrace_cli/detector/deterministic.py +363 -511
  22. stacktrace_cli/detector/finding.py +178 -15
  23. stacktrace_cli/detector/history.py +210 -0
  24. stacktrace_cli/detector/jev.py +213 -0
  25. stacktrace_cli/detector/priors.py +53 -285
  26. stacktrace_cli/detector/prompts/__init__.py +20 -0
  27. stacktrace_cli/detector/prompts/jev-v1/stacktrace-deceptive-completion.json +16 -0
  28. stacktrace_cli/detector/prompts/jev-v2/stacktrace-deceptive-completion.json +16 -0
  29. stacktrace_cli/detector/reasoning.py +861 -308
  30. stacktrace_cli/detector/render.py +92 -7
  31. stacktrace_cli/detector/rules.py +69 -56
  32. stacktrace_cli/detector/run.py +432 -78
  33. stacktrace_cli/detector/secrets.py +8 -6
  34. stacktrace_cli/detector/windows.py +293 -0
  35. stacktrace_cli/monitor/reasoning.py +153 -21
  36. stacktrace_cli/monitor/render.py +439 -10
  37. stacktrace_cli/monitor/server.py +242 -21
  38. stacktrace_cli/monitor/site/app.js +1952 -197
  39. stacktrace_cli/monitor/site/index.html +43 -2
  40. stacktrace_cli/monitor/site/styles.css +162 -1
  41. stacktrace_cli/monitor/state.py +72 -1
  42. stacktrace_cli/monitor/watch.py +269 -57
  43. stacktrace_cli/options.py +107 -0
  44. stacktrace_cli/private_state.py +38 -0
  45. stacktrace_cli/remote/cli.py +20 -10
  46. stacktrace_cli/remote/detect_payload.py +37 -5
  47. stacktrace_cli/remote/redact.py +3 -3
  48. stacktrace_cli/remote/spool.py +3 -2
  49. stacktrace_cli/remote/sync_detect.py +15 -1
  50. stacktrace_cli/remote/upload_contract.py +7 -0
  51. stacktrace_cli/sessions/access.py +24 -0
  52. stacktrace_cli/sessions/index.py +30 -0
  53. stacktrace_cli/sessions/protocols.py +21 -0
  54. stacktrace_cli/sessions/render.py +15 -9
  55. stacktrace_cli/telemetry/__init__.py +11 -0
  56. stacktrace_cli/telemetry/cli.py +67 -0
  57. stacktrace_cli/telemetry/events.py +266 -0
  58. stacktrace_cli/telemetry/posthog.py +126 -0
  59. stacktrace_cli/telemetry/state.py +190 -0
  60. {stacktrace_cli-0.3.1.dist-info → stacktrace_cli-0.4.1.dist-info}/METADATA +56 -24
  61. stacktrace_cli-0.4.1.dist-info/RECORD +90 -0
  62. {stacktrace_cli-0.3.1.dist-info → stacktrace_cli-0.4.1.dist-info}/WHEEL +1 -1
  63. stacktrace_cli/detector/markers.py +0 -158
  64. stacktrace_cli/detector/prompts/v1/stacktrace-injected-instruction-followed.md +0 -14
  65. stacktrace_cli/detector/prompts/v1/stacktrace-intent-drift.md +0 -11
  66. stacktrace_cli-0.3.1.dist-info/RECORD +0 -67
  67. {stacktrace_cli-0.3.1.dist-info → stacktrace_cli-0.4.1.dist-info}/entry_points.txt +0 -0
@@ -1,3 +1,3 @@
1
1
  """The `stacktrace` command-line interface — detection and response for AI agents."""
2
2
 
3
- __version__ = "0.3.1"
3
+ __version__ = "0.4.1"
@@ -19,19 +19,22 @@ path's tests already pin.
19
19
 
20
20
  from __future__ import annotations
21
21
 
22
+ import re
22
23
  from collections.abc import Iterator
23
24
  from dataclasses import dataclass, replace
24
25
  from datetime import UTC, datetime, timedelta
25
26
  from pathlib import Path
26
- from typing import Any
27
+ from typing import Any, Final
27
28
 
28
29
  from stacktrace_cli.correlate.orchestrate import Acquired, acquire_correlated_view
29
30
  from stacktrace_cli.correlate.project_map import parse_mapping
30
31
  from stacktrace_cli.correlate.record import CorrelatedSession, CorrelatedView, Coverage
31
32
  from stacktrace_cli.detector.cache import VerdictCache
33
+ from stacktrace_cli.detector.history import DriftHistory
32
34
  from stacktrace_cli.detector.run import (
33
35
  DEFAULT_BUDGET,
34
36
  DEFAULT_SAMPLE_BUDGET,
37
+ NO_ANALYZER,
35
38
  DetectorRun,
36
39
  run_detector,
37
40
  )
@@ -61,41 +64,91 @@ class Analysis:
61
64
  placed: bool = True
62
65
 
63
66
 
67
+ #: A count and an optional unit, anchored. Anchored rather than searched
68
+ #: because the tolerance this parser owes the upload path is about the *unit*
69
+ #: being omissible, not about the value being loosely shaped: `int` alone would
70
+ #: read `'5 m'` as five (it strips whitespace before parsing) and a signed
71
+ #: `'-1h'` as a negative window, so the shape is pinned here and the count is
72
+ #: parsed from digits already known to be digits.
73
+ _SINCE_SPELLING: Final = re.compile(r"^(\d+)([mhd]?)$")
74
+
75
+ #: The unit suffixes, and the `timedelta` keyword each one names. The empty
76
+ #: key is the bare count -- `'7'`, which the upload path's config value has
77
+ #: always spelled without a unit and which has always meant days.
78
+ _SINCE_UNITS: Final = {"m": "minutes", "h": "hours", "d": "days", "": "days"}
79
+
80
+ #: One message for every way the spelling can be wrong, because they are one
81
+ #: user mistake: a window that does not name a positive amount of time. Shared
82
+ #: with `cli.py`'s stricter parser so a reader meets the same vocabulary
83
+ #: whichever of the two refused them.
84
+ SINCE_HINT: Final = "like '5m', '2h' or '7d'"
85
+
86
+
64
87
  def parse_since(value: str) -> datetime:
65
- """`7d` or `7` -> an aware UTC cut-off.
88
+ """`5m`, `2h`, `7d` or `7` -> an aware UTC cut-off.
89
+
90
+ Sub-day units are here rather than only at the command line because the
91
+ window reaches this module as a *string* from every caller that has one --
92
+ `monitor` hands it over untouched and parses it on the watcher thread, so a
93
+ unit the command line accepted and this parser did not would raise where no
94
+ request is left to answer.
66
95
 
67
96
  Raises `ValueError`, never a Click or transport exception: this module is
68
97
  below both. The message is the upload path's, which its tests pin.
69
98
  """
70
- text = value.strip().lower()
99
+ match = _SINCE_SPELLING.match(value.strip().lower())
100
+ if not match:
101
+ raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}")
71
102
  try:
72
- days = int(text[:-1]) if text.endswith("d") else int(text)
103
+ count = int(match.group(1))
73
104
  except ValueError:
74
- raise ValueError(f"--since must be a positive day count, not {value!r}") from None
75
- if days <= 0:
76
- raise ValueError(f"--since must be a positive day count, not {value!r}")
105
+ # `int` refuses a string past its default digit-conversion limit. The
106
+ # same user mistake as a non-numeric value, so the same diagnostic.
107
+ raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}") from None
108
+ if count <= 0:
109
+ raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}")
77
110
  try:
78
- return datetime.now(UTC) - timedelta(days=days)
111
+ return datetime.now(UTC) - timedelta(**{_SINCE_UNITS[match.group(2)]: count})
79
112
  except OverflowError:
80
113
  # `timedelta` takes any `int` but overflows past `timedelta.max.days`.
81
114
  # The same user mistake as a non-numeric value, so the same diagnostic.
82
- raise ValueError(f"--since must be a positive day count, not {value!r}") from None
115
+ raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}") from None
83
116
 
84
117
 
85
- def _only(view: SessionView, session_ids: tuple[str, ...]) -> SessionView:
86
- """The window narrowed to named sessions, or the window itself.
118
+ def _only(
119
+ view: SessionView,
120
+ session_ids: tuple[str, ...],
121
+ agent_kinds: tuple[str, ...],
122
+ ) -> SessionView:
123
+ """The window narrowed to named session identities, or the window itself.
87
124
 
88
125
  Narrowed *before* correlation rather than after judgement, because the point
89
126
  is that nothing else can be dispatched: a caller that pays for one named
90
127
  session must not be able to spend that payment on another one the detector
91
- happened to rank higher. `unavailable` is carried through unchanged — a
92
- reader that failed is still a reader that failed, whichever session was
93
- asked for.
128
+ happened to rank higher. The kind and id filters are both applied when the
129
+ caller supplies both halves of the identity. `unavailable` is carried
130
+ through unchanged — a reader that failed is still a reader that failed,
131
+ whichever session was asked for.
132
+
133
+ A kind-anonymous session (`agent_kind is None`) is never dropped by the
134
+ kind filter: `sessions/access.py`'s coverage contract is that a session
135
+ OpenAIDR could not map to a kind is a gap to surface, not a kind the user
136
+ filtered out, and this seam must not reintroduce that drop after
137
+ collection already kept it.
94
138
  """
95
- if not session_ids:
139
+ if not session_ids and not agent_kinds:
96
140
  return view
97
141
  wanted = set(session_ids)
98
- return replace(view, sessions=tuple(s for s in view.sessions if s.session_id in wanted))
142
+ kinds = set(agent_kinds)
143
+ return replace(
144
+ view,
145
+ sessions=tuple(
146
+ session
147
+ for session in view.sessions
148
+ if (not wanted or session.session_id in wanted)
149
+ and (not kinds or session.agent_kind is None or session.agent_kind in kinds)
150
+ ),
151
+ )
99
152
 
100
153
 
101
154
  def analyse(
@@ -111,6 +164,10 @@ def analyse(
111
164
  sample_budget: int = DEFAULT_SAMPLE_BUDGET,
112
165
  cache: VerdictCache | None = None,
113
166
  attach_advisories: Any = None,
167
+ build_all: Any = None,
168
+ analyzer: str = NO_ANALYZER,
169
+ jev_key: str | None = None,
170
+ history: DriftHistory | None = None,
114
171
  ) -> Analysis:
115
172
  """Read this machine's sessions, correlate them, and judge them.
116
173
 
@@ -122,23 +179,111 @@ def analyse(
122
179
  cache through it so osv.dev is asked once per component rather than once per
123
180
  pass; a one-shot command passes nothing and gets the default.
124
181
 
182
+ `build_all` is the same kind of seam for composition builds, which shell out
183
+ and cost seconds each. A long-lived process passes a cache through it so an
184
+ unchanged machine or project is not rebuilt on every pass; a one-shot
185
+ command passes nothing and gets the default.
186
+
125
187
  `session_ids` narrows the window to named sessions. Empty means the whole
126
188
  window, which is what every command passes; monitor's reasoning button is
127
189
  what needs the narrowing, and needs it to be structural.
128
190
 
129
191
  `since` is the window's spelling — `parse_since` reads it — or the cutoff
130
192
  itself, for a caller whose own `--since` spelling is not this one's.
193
+
194
+ `reasoning` switches the third stage on and is off by default; `jev_key`
195
+ hands the analyzer a key a caller already holds, in place of reading
196
+ `TYPESAFE_API_KEY`. Both travel to `run_detector` unchanged, so a
197
+ programmatic caller controls exactly what `detect` controls.
131
198
  """
132
199
  window_end = datetime.now(UTC)
133
200
  window_start = since if isinstance(since, datetime) else parse_since(since)
134
- acquired = _correlated(
201
+ sessions = _collected(
135
202
  agent_kinds=agent_kinds,
136
203
  window_start=window_start,
137
- bom_paths=bom_paths,
138
- project_map=project_map,
139
204
  root=root,
140
205
  session_ids=session_ids,
206
+ )
207
+ return _analyse_sessions(
208
+ sessions,
209
+ window_start=window_start,
210
+ window_end=window_end,
211
+ bom_paths=bom_paths,
212
+ project_map=project_map,
213
+ reasoning=reasoning,
214
+ budget=budget,
215
+ sample_budget=sample_budget,
216
+ cache=cache,
141
217
  attach_advisories=attach_advisories,
218
+ build_all=build_all,
219
+ analyzer=analyzer,
220
+ jev_key=jev_key,
221
+ history=history,
222
+ )
223
+
224
+
225
+ def analyse_sessions(
226
+ sessions: SessionView,
227
+ *,
228
+ agent_kinds: tuple[str, ...] = (),
229
+ since: str | datetime = "7d",
230
+ bom_paths: tuple[Path, ...] = (),
231
+ project_map: tuple[str, ...] = (),
232
+ session_ids: tuple[str, ...] = (),
233
+ reasoning: bool = False,
234
+ budget: int = DEFAULT_BUDGET,
235
+ sample_budget: int = DEFAULT_SAMPLE_BUDGET,
236
+ cache: VerdictCache | None = None,
237
+ attach_advisories: Any = None,
238
+ build_all: Any = None,
239
+ analyzer: str = NO_ANALYZER,
240
+ jev_key: str | None = None,
241
+ history: DriftHistory | None = None,
242
+ ) -> Analysis:
243
+ """Correlate and judge a session view already collected by a long-lived reader."""
244
+ window_end = datetime.now(UTC)
245
+ window_start = since if isinstance(since, datetime) else parse_since(since)
246
+ return _analyse_sessions(
247
+ _only(sessions, session_ids, agent_kinds),
248
+ window_start=window_start,
249
+ window_end=window_end,
250
+ bom_paths=bom_paths,
251
+ project_map=project_map,
252
+ reasoning=reasoning,
253
+ budget=budget,
254
+ sample_budget=sample_budget,
255
+ cache=cache,
256
+ attach_advisories=attach_advisories,
257
+ build_all=build_all,
258
+ analyzer=analyzer,
259
+ jev_key=jev_key,
260
+ history=history,
261
+ )
262
+
263
+
264
+ def _analyse_sessions(
265
+ sessions: SessionView,
266
+ *,
267
+ window_start: datetime,
268
+ window_end: datetime,
269
+ bom_paths: tuple[Path, ...],
270
+ project_map: tuple[str, ...],
271
+ reasoning: bool,
272
+ budget: int,
273
+ sample_budget: int,
274
+ cache: VerdictCache | None,
275
+ attach_advisories: Any,
276
+ build_all: Any,
277
+ analyzer: str,
278
+ jev_key: str | None,
279
+ history: DriftHistory | None,
280
+ ) -> Analysis:
281
+ acquired = _acquire(
282
+ sessions,
283
+ bom_paths=bom_paths,
284
+ project_map=project_map,
285
+ attach_advisories=attach_advisories,
286
+ build_all=build_all,
142
287
  )
143
288
  run = run_detector(
144
289
  acquired.view,
@@ -146,6 +291,9 @@ def analyse(
146
291
  budget=budget,
147
292
  sample_budget=sample_budget,
148
293
  cache=cache,
294
+ analyzer=analyzer,
295
+ jev_key=jev_key,
296
+ history=history,
149
297
  )
150
298
  return Analysis(
151
299
  run=run,
@@ -170,7 +318,11 @@ def _collected(
170
318
  session_ids: tuple[str, ...],
171
319
  ) -> SessionView:
172
320
  """Read the window. Its own function so a caller can read twice."""
173
- return _only(collect_sessions(list(agent_kinds) or None, window_start, root=root), session_ids)
321
+ return _only(
322
+ collect_sessions(list(agent_kinds) or None, window_start, root=root),
323
+ session_ids,
324
+ agent_kinds,
325
+ )
174
326
 
175
327
 
176
328
  def _acquire(
@@ -179,9 +331,14 @@ def _acquire(
179
331
  bom_paths: tuple[Path, ...],
180
332
  project_map: tuple[str, ...],
181
333
  attach_advisories: Any,
334
+ build_all: Any = None,
182
335
  ) -> Acquired:
183
336
  """Correlate what was read."""
184
- hook = {} if attach_advisories is None else {"attach_advisories": attach_advisories}
337
+ hook: dict[str, Any] = {}
338
+ if attach_advisories is not None:
339
+ hook["attach_advisories"] = attach_advisories
340
+ if build_all is not None:
341
+ hook["build_all"] = build_all
185
342
  return acquire_correlated_view(
186
343
  sessions,
187
344
  bom_paths=bom_paths,
@@ -199,6 +356,7 @@ def _correlated(
199
356
  root: Path | None,
200
357
  session_ids: tuple[str, ...],
201
358
  attach_advisories: Any,
359
+ build_all: Any = None,
202
360
  ) -> Acquired:
203
361
  """Everything up to the judging, shared by both entry points."""
204
362
  return _acquire(
@@ -211,6 +369,7 @@ def _correlated(
211
369
  bom_paths=bom_paths,
212
370
  project_map=project_map,
213
371
  attach_advisories=attach_advisories,
372
+ build_all=build_all,
214
373
  )
215
374
 
216
375
 
@@ -258,9 +417,18 @@ def analyse_progressively(
258
417
  preview: timedelta | None = PREVIEW_WINDOW,
259
418
  cache: VerdictCache | None = None,
260
419
  attach_advisories: Any = None,
420
+ analyzer: str = NO_ANALYZER,
421
+ jev_key: str | None = None,
422
+ history: DriftHistory | None = None,
423
+ reasoning: bool = False,
424
+ budget: int = DEFAULT_BUDGET,
261
425
  ) -> Iterator[Analysis]:
262
426
  """The same pipeline, delivered in instalments.
263
427
 
428
+ `history` is accepted and unused: this path never requests reasoning, so
429
+ it never adds a point. It is here so the watcher can forward one options
430
+ dict to both entry points (the test in `tests/monitor/test_watch.py`).
431
+
264
432
  `analyse` answers once, which is right for a command that prints and exits
265
433
  and wrong for a page meant to look alive. Measured over 339 real sessions
266
434
  the work splits about 6.6s to collect and correlate, then about 12ms to
@@ -289,11 +457,13 @@ def analyse_progressively(
289
457
  `cache` is read and never written. A session graded by an earlier
290
458
  `detect --reasoning` shows that grade here; this path never commissions one.
291
459
 
292
- **No `reasoning` parameter, deliberately.** Stage 3's budget is per *run*, so
293
- judging in batches would give each batch its own budget and spend a multiple
294
- of what was authorised. A streaming reasoning run needs a budget shared across
295
- batches; until it has one, this path does not offer the option rather than
296
- offering it wrongly.
460
+ **Reasoning runs once, after the batches, never inside them.** Stage 3's
461
+ budget is per *run*, so asking each batch would give each its own budget and
462
+ spend a multiple of what was authorised — which is why this path refused the
463
+ option entirely until now. Running it once over the whole view keeps the
464
+ budget meaning what the flag said, and keeps the ordering that matters: the
465
+ cheap stages paint first and stage 3 arrives as one more instalment, rather
466
+ than holding the first paint behind several hundred sequential requests.
297
467
  """
298
468
  window_end = datetime.now(UTC)
299
469
  window_start = since if isinstance(since, datetime) else parse_since(since)
@@ -348,20 +518,43 @@ def analyse_progressively(
348
518
  if preview is None:
349
519
  yield _stage(DetectorRun(collection_failures=failures), ())
350
520
 
521
+ # Newest *activity* first, not newest start. This decides which sessions are
522
+ # judged first and therefore what a progressive reader sees first, and
523
+ # `started_at` answers a different question: a session running for a day
524
+ # began long ago and is the most recent thing on the machine. Ordering by
525
+ # its start put the session doing the work behind sessions idle for hours --
526
+ # the same defect the monitor's own session list had, one layer up, so the
527
+ # batch order and the page's order now read the same clock.
351
528
  ordered = sorted(
352
529
  view.sessions,
353
- key=lambda correlated: correlated.session.started_at or datetime.min.replace(tzinfo=UTC),
530
+ key=lambda correlated: (
531
+ correlated.session.last_activity_at
532
+ or correlated.session.started_at
533
+ or datetime.min.replace(tzinfo=UTC)
534
+ ),
354
535
  reverse=True,
355
536
  )
356
537
  step = max(batch, 1)
357
- accumulated = DetectorRun(collection_failures=failures)
538
+ accumulated = DetectorRun(collection_failures=failures, analyzer=analyzer)
358
539
  for index in range(0, len(ordered), step):
359
540
  judged = run_detector(
360
541
  CorrelatedView(sessions=tuple(ordered[index : index + step])),
361
542
  # Read, never written, and never requesting: a session an earlier
362
543
  # `detect --reasoning` graded renders with that grade here, and this
363
- # path commissions nothing (ADR-0026 clause 2).
544
+ # path commissions nothing (ADR-0026 clause 2). The analyzer name
545
+ # is what lets a stored Jev map be served: the cache key carries
546
+ # the analyzer's identity, so a page started with --analyzer jev
547
+ # reads the entries that flag paid for and no others.
548
+ #
549
+ # `jev_key` rides along although this path asks nothing. Today the
550
+ # cache key reads `identity()`, which does not depend on the key,
551
+ # so omitting it would work — but it would leave one call building
552
+ # two differently-configured analyzers under one name, and the next
553
+ # thing to depend on the key would break here and not in the
554
+ # instalment below.
364
555
  cache=cache,
556
+ analyzer=analyzer,
557
+ jev_key=jev_key,
365
558
  )
366
559
  accumulated = DetectorRun(
367
560
  detections=accumulated.detections + judged.detections,
@@ -370,6 +563,26 @@ def analyse_progressively(
370
563
  requested=accumulated.requested + judged.requested,
371
564
  analysed=accumulated.analysed + judged.analysed,
372
565
  cache_hits=accumulated.cache_hits + judged.cache_hits,
566
+ maps=accumulated.maps + judged.maps,
373
567
  collection_failures=failures,
568
+ analyzer=analyzer,
374
569
  )
375
570
  yield _stage(accumulated, tuple(ordered))
571
+
572
+ if not reasoning or not ordered:
573
+ return
574
+
575
+ # One more instalment, one run, one budget. Everything above has already
576
+ # been published, so the page is complete and readable while this is in
577
+ # flight — on a large window it is several hundred sequential requests and
578
+ # holding the first paint behind it is what made the page look hung.
579
+ judged = run_detector(
580
+ view,
581
+ reasoning=True,
582
+ budget=budget,
583
+ cache=cache,
584
+ analyzer=analyzer,
585
+ jev_key=jev_key,
586
+ history=history,
587
+ )
588
+ yield _stage(judged, tuple(ordered))
@@ -0,0 +1,130 @@
1
+ """Which commit this install was built from, when it was not a PyPI release.
2
+
3
+ Read from the install's own PEP 610 record (`direct_url.json` in the
4
+ dist-info), which pip and uv write for every install that did not come from an
5
+ index:
6
+
7
+ - ``pip install git+https://…@branch`` / ``uv tool install git+…`` record the
8
+ resolved commit in ``vcs_info.commit_id``.
9
+ - ``pip install -e .`` / ``uv sync`` record an editable ``file://`` URL; the
10
+ code runs from that checkout, so its current ``HEAD`` is the answer, plus
11
+ ``.dirty`` when the worktree has uncommitted changes.
12
+
13
+ A wheel from PyPI has no record, and a non-editable install from a local
14
+ directory records only the path, whose ``HEAD`` may have moved since the build
15
+ — both answer `None` rather than a commit that may not be the one running.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import functools
21
+ import json
22
+ import os
23
+ import re
24
+ import subprocess
25
+ import threading
26
+ from importlib.metadata import distribution
27
+ from pathlib import Path
28
+ from urllib.parse import unquote, urlparse
29
+
30
+ DISTRIBUTION = "stacktrace-cli"
31
+
32
+ _SHA = re.compile(r"^[0-9a-f]{7,64}$")
33
+ _SHORT = 7
34
+
35
+ #: Serializes `_compute_version_label`'s cache misses. `functools.cache` alone
36
+ #: lets concurrent misses run the wrapped function more than once (verified:
37
+ #: two threads racing an uncached call both entered `build_commit`), so the
38
+ #: daemon's startup warm-up (`daemon/runtime.py`) and a request thread could
39
+ #: each launch their own `git` probe. Held only around the call: the thread
40
+ #: that loses the race blocks briefly here rather than repeating the winner's
41
+ #: work, and finds the answer already cached once it gets in.
42
+ _version_label_lock = threading.Lock()
43
+
44
+
45
+ def version_label() -> str:
46
+ """The version as `--version` prints it: `0.4.0`, or `0.4.0+86eef86` for a
47
+ build from a branch or a pull request.
48
+
49
+ One expression for both places that report it, the CLI and telemetry, so a
50
+ PostHog event and `--version` cannot disagree about which build ran.
51
+ """
52
+ with _version_label_lock:
53
+ return _compute_version_label()
54
+
55
+
56
+ @functools.cache
57
+ def _compute_version_label() -> str:
58
+ from . import __version__
59
+
60
+ commit = build_commit()
61
+ return f"{__version__}+{commit}" if commit else __version__
62
+
63
+
64
+ def build_commit() -> str | None:
65
+ """The short commit this install runs, or `None` for a release install."""
66
+ try:
67
+ return _read_build_commit()
68
+ except Exception: # noqa: BLE001 - direct_url.json is external, untrusted input
69
+ # Every step below reads or parses `direct_url.json`, an external
70
+ # document written by pip or uv. Three narrower guards in a row each
71
+ # missed a new failure mode in turn (`UnicodeDecodeError` from
72
+ # `read_text`'s UTF-8 decode, `urlparse`'s `ValueError` on a malformed
73
+ # URL, `json.loads`'s `RecursionError` on deep nesting) -- the fix is
74
+ # the boundary, not the exception list. Anything this parse can raise
75
+ # reads the same as no record.
76
+ return None
77
+
78
+
79
+ def _read_build_commit() -> str | None:
80
+ raw = distribution(DISTRIBUTION).read_text("direct_url.json")
81
+ if not raw:
82
+ return None
83
+ record = json.loads(raw)
84
+ if not isinstance(record, dict):
85
+ return None
86
+
87
+ vcs = record.get("vcs_info")
88
+ if isinstance(vcs, dict):
89
+ commit = vcs.get("commit_id")
90
+ if vcs.get("vcs") == "git" and isinstance(commit, str) and _SHA.match(commit):
91
+ return commit[:_SHORT]
92
+ return None
93
+
94
+ directory = record.get("dir_info")
95
+ url = record.get("url")
96
+ if isinstance(directory, dict) and directory.get("editable") is True and isinstance(url, str):
97
+ parsed = urlparse(url)
98
+ if parsed.scheme == "file":
99
+ return _checkout_commit(Path(unquote(parsed.path)))
100
+ return None
101
+
102
+
103
+ def _checkout_commit(root: Path) -> str | None:
104
+ # GIT_DIR/GIT_WORK_TREE in the caller's environment override -C, so a
105
+ # caller running us from inside another repo's hook can make git report
106
+ # that repo's HEAD instead of `root`'s.
107
+ env = {k: v for k, v in os.environ.items() if k not in ("GIT_DIR", "GIT_WORK_TREE")}
108
+
109
+ def git(*args: str) -> str | None:
110
+ try:
111
+ done = subprocess.run(
112
+ ["git", "-C", str(root), *args],
113
+ capture_output=True,
114
+ text=True,
115
+ timeout=5,
116
+ check=False,
117
+ env=env,
118
+ )
119
+ except (OSError, subprocess.SubprocessError):
120
+ return None
121
+ return done.stdout if done.returncode == 0 else None
122
+
123
+ head = git("rev-parse", "HEAD")
124
+ if head is None or not _SHA.match(head.strip()):
125
+ return None
126
+ commit = head.strip()[:_SHORT]
127
+ status = git("status", "--porcelain")
128
+ if status is None:
129
+ return None
130
+ return f"{commit}.dirty" if status else commit