stacktrace-cli 0.3.1__py3-none-any.whl → 0.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/analysis.py +242 -29
- stacktrace_cli/build.py +130 -0
- stacktrace_cli/cli.py +176 -67
- stacktrace_cli/correlate/acquire.py +23 -2
- stacktrace_cli/correlate/orchestrate.py +73 -9
- stacktrace_cli/correlate/render.py +7 -5
- stacktrace_cli/daemon/__init__.py +1 -0
- stacktrace_cli/daemon/cli.py +106 -0
- stacktrace_cli/daemon/client.py +134 -0
- stacktrace_cli/daemon/composition.py +133 -0
- stacktrace_cli/daemon/observer.py +173 -0
- stacktrace_cli/daemon/paths.py +35 -0
- stacktrace_cli/daemon/protocol.py +29 -0
- stacktrace_cli/daemon/runtime.py +222 -0
- stacktrace_cli/daemon/server.py +164 -0
- stacktrace_cli/daemon/service.py +154 -0
- stacktrace_cli/daemon/store.py +368 -0
- stacktrace_cli/detector/blocked.py +450 -0
- stacktrace_cli/detector/cache.py +245 -9
- stacktrace_cli/detector/deterministic.py +363 -511
- stacktrace_cli/detector/finding.py +178 -15
- stacktrace_cli/detector/history.py +210 -0
- stacktrace_cli/detector/jev.py +213 -0
- stacktrace_cli/detector/priors.py +53 -285
- stacktrace_cli/detector/prompts/__init__.py +20 -0
- stacktrace_cli/detector/prompts/jev-v1/stacktrace-deceptive-completion.json +16 -0
- stacktrace_cli/detector/prompts/jev-v2/stacktrace-deceptive-completion.json +16 -0
- stacktrace_cli/detector/reasoning.py +861 -308
- stacktrace_cli/detector/render.py +92 -7
- stacktrace_cli/detector/rules.py +69 -56
- stacktrace_cli/detector/run.py +432 -78
- stacktrace_cli/detector/secrets.py +8 -6
- stacktrace_cli/detector/windows.py +293 -0
- stacktrace_cli/monitor/reasoning.py +153 -21
- stacktrace_cli/monitor/render.py +439 -10
- stacktrace_cli/monitor/server.py +242 -21
- stacktrace_cli/monitor/site/app.js +1952 -197
- stacktrace_cli/monitor/site/index.html +43 -2
- stacktrace_cli/monitor/site/styles.css +162 -1
- stacktrace_cli/monitor/state.py +72 -1
- stacktrace_cli/monitor/watch.py +269 -57
- stacktrace_cli/options.py +107 -0
- stacktrace_cli/private_state.py +38 -0
- stacktrace_cli/remote/cli.py +20 -10
- stacktrace_cli/remote/detect_payload.py +37 -5
- stacktrace_cli/remote/redact.py +3 -3
- stacktrace_cli/remote/spool.py +3 -2
- stacktrace_cli/remote/sync_detect.py +15 -1
- stacktrace_cli/remote/upload_contract.py +7 -0
- stacktrace_cli/sessions/access.py +24 -0
- stacktrace_cli/sessions/index.py +30 -0
- stacktrace_cli/sessions/protocols.py +21 -0
- stacktrace_cli/sessions/render.py +15 -9
- stacktrace_cli/telemetry/__init__.py +11 -0
- stacktrace_cli/telemetry/cli.py +67 -0
- stacktrace_cli/telemetry/events.py +266 -0
- stacktrace_cli/telemetry/posthog.py +126 -0
- stacktrace_cli/telemetry/state.py +190 -0
- {stacktrace_cli-0.3.1.dist-info → stacktrace_cli-0.4.1.dist-info}/METADATA +56 -24
- stacktrace_cli-0.4.1.dist-info/RECORD +90 -0
- {stacktrace_cli-0.3.1.dist-info → stacktrace_cli-0.4.1.dist-info}/WHEEL +1 -1
- stacktrace_cli/detector/markers.py +0 -158
- stacktrace_cli/detector/prompts/v1/stacktrace-injected-instruction-followed.md +0 -14
- stacktrace_cli/detector/prompts/v1/stacktrace-intent-drift.md +0 -11
- stacktrace_cli-0.3.1.dist-info/RECORD +0 -67
- {stacktrace_cli-0.3.1.dist-info → stacktrace_cli-0.4.1.dist-info}/entry_points.txt +0 -0
stacktrace_cli/__init__.py
CHANGED
stacktrace_cli/analysis.py
CHANGED
|
@@ -19,19 +19,22 @@ path's tests already pin.
|
|
|
19
19
|
|
|
20
20
|
from __future__ import annotations
|
|
21
21
|
|
|
22
|
+
import re
|
|
22
23
|
from collections.abc import Iterator
|
|
23
24
|
from dataclasses import dataclass, replace
|
|
24
25
|
from datetime import UTC, datetime, timedelta
|
|
25
26
|
from pathlib import Path
|
|
26
|
-
from typing import Any
|
|
27
|
+
from typing import Any, Final
|
|
27
28
|
|
|
28
29
|
from stacktrace_cli.correlate.orchestrate import Acquired, acquire_correlated_view
|
|
29
30
|
from stacktrace_cli.correlate.project_map import parse_mapping
|
|
30
31
|
from stacktrace_cli.correlate.record import CorrelatedSession, CorrelatedView, Coverage
|
|
31
32
|
from stacktrace_cli.detector.cache import VerdictCache
|
|
33
|
+
from stacktrace_cli.detector.history import DriftHistory
|
|
32
34
|
from stacktrace_cli.detector.run import (
|
|
33
35
|
DEFAULT_BUDGET,
|
|
34
36
|
DEFAULT_SAMPLE_BUDGET,
|
|
37
|
+
NO_ANALYZER,
|
|
35
38
|
DetectorRun,
|
|
36
39
|
run_detector,
|
|
37
40
|
)
|
|
@@ -61,41 +64,91 @@ class Analysis:
|
|
|
61
64
|
placed: bool = True
|
|
62
65
|
|
|
63
66
|
|
|
67
|
+
#: A count and an optional unit, anchored. Anchored rather than searched
|
|
68
|
+
#: because the tolerance this parser owes the upload path is about the *unit*
|
|
69
|
+
#: being omissible, not about the value being loosely shaped: `int` alone would
|
|
70
|
+
#: read `'5 m'` as five (it strips whitespace before parsing) and a signed
|
|
71
|
+
#: `'-1h'` as a negative window, so the shape is pinned here and the count is
|
|
72
|
+
#: parsed from digits already known to be digits.
|
|
73
|
+
_SINCE_SPELLING: Final = re.compile(r"^(\d+)([mhd]?)$")
|
|
74
|
+
|
|
75
|
+
#: The unit suffixes, and the `timedelta` keyword each one names. The empty
|
|
76
|
+
#: key is the bare count -- `'7'`, which the upload path's config value has
|
|
77
|
+
#: always spelled without a unit and which has always meant days.
|
|
78
|
+
_SINCE_UNITS: Final = {"m": "minutes", "h": "hours", "d": "days", "": "days"}
|
|
79
|
+
|
|
80
|
+
#: One message for every way the spelling can be wrong, because they are one
|
|
81
|
+
#: user mistake: a window that does not name a positive amount of time. Shared
|
|
82
|
+
#: with `cli.py`'s stricter parser so a reader meets the same vocabulary
|
|
83
|
+
#: whichever of the two refused them.
|
|
84
|
+
SINCE_HINT: Final = "like '5m', '2h' or '7d'"
|
|
85
|
+
|
|
86
|
+
|
|
64
87
|
def parse_since(value: str) -> datetime:
|
|
65
|
-
"""`7d` or `7` -> an aware UTC cut-off.
|
|
88
|
+
"""`5m`, `2h`, `7d` or `7` -> an aware UTC cut-off.
|
|
89
|
+
|
|
90
|
+
Sub-day units are here rather than only at the command line because the
|
|
91
|
+
window reaches this module as a *string* from every caller that has one --
|
|
92
|
+
`monitor` hands it over untouched and parses it on the watcher thread, so a
|
|
93
|
+
unit the command line accepted and this parser did not would raise where no
|
|
94
|
+
request is left to answer.
|
|
66
95
|
|
|
67
96
|
Raises `ValueError`, never a Click or transport exception: this module is
|
|
68
97
|
below both. The message is the upload path's, which its tests pin.
|
|
69
98
|
"""
|
|
70
|
-
|
|
99
|
+
match = _SINCE_SPELLING.match(value.strip().lower())
|
|
100
|
+
if not match:
|
|
101
|
+
raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}")
|
|
71
102
|
try:
|
|
72
|
-
|
|
103
|
+
count = int(match.group(1))
|
|
73
104
|
except ValueError:
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
raise ValueError(f"--since must be a positive
|
|
105
|
+
# `int` refuses a string past its default digit-conversion limit. The
|
|
106
|
+
# same user mistake as a non-numeric value, so the same diagnostic.
|
|
107
|
+
raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}") from None
|
|
108
|
+
if count <= 0:
|
|
109
|
+
raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}")
|
|
77
110
|
try:
|
|
78
|
-
return datetime.now(UTC) - timedelta(
|
|
111
|
+
return datetime.now(UTC) - timedelta(**{_SINCE_UNITS[match.group(2)]: count})
|
|
79
112
|
except OverflowError:
|
|
80
113
|
# `timedelta` takes any `int` but overflows past `timedelta.max.days`.
|
|
81
114
|
# The same user mistake as a non-numeric value, so the same diagnostic.
|
|
82
|
-
raise ValueError(f"--since must be a positive
|
|
115
|
+
raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}") from None
|
|
83
116
|
|
|
84
117
|
|
|
85
|
-
def _only(
|
|
86
|
-
|
|
118
|
+
def _only(
|
|
119
|
+
view: SessionView,
|
|
120
|
+
session_ids: tuple[str, ...],
|
|
121
|
+
agent_kinds: tuple[str, ...],
|
|
122
|
+
) -> SessionView:
|
|
123
|
+
"""The window narrowed to named session identities, or the window itself.
|
|
87
124
|
|
|
88
125
|
Narrowed *before* correlation rather than after judgement, because the point
|
|
89
126
|
is that nothing else can be dispatched: a caller that pays for one named
|
|
90
127
|
session must not be able to spend that payment on another one the detector
|
|
91
|
-
happened to rank higher.
|
|
92
|
-
|
|
93
|
-
|
|
128
|
+
happened to rank higher. The kind and id filters are both applied when the
|
|
129
|
+
caller supplies both halves of the identity. `unavailable` is carried
|
|
130
|
+
through unchanged — a reader that failed is still a reader that failed,
|
|
131
|
+
whichever session was asked for.
|
|
132
|
+
|
|
133
|
+
A kind-anonymous session (`agent_kind is None`) is never dropped by the
|
|
134
|
+
kind filter: `sessions/access.py`'s coverage contract is that a session
|
|
135
|
+
OpenAIDR could not map to a kind is a gap to surface, not a kind the user
|
|
136
|
+
filtered out, and this seam must not reintroduce that drop after
|
|
137
|
+
collection already kept it.
|
|
94
138
|
"""
|
|
95
|
-
if not session_ids:
|
|
139
|
+
if not session_ids and not agent_kinds:
|
|
96
140
|
return view
|
|
97
141
|
wanted = set(session_ids)
|
|
98
|
-
|
|
142
|
+
kinds = set(agent_kinds)
|
|
143
|
+
return replace(
|
|
144
|
+
view,
|
|
145
|
+
sessions=tuple(
|
|
146
|
+
session
|
|
147
|
+
for session in view.sessions
|
|
148
|
+
if (not wanted or session.session_id in wanted)
|
|
149
|
+
and (not kinds or session.agent_kind is None or session.agent_kind in kinds)
|
|
150
|
+
),
|
|
151
|
+
)
|
|
99
152
|
|
|
100
153
|
|
|
101
154
|
def analyse(
|
|
@@ -111,6 +164,10 @@ def analyse(
|
|
|
111
164
|
sample_budget: int = DEFAULT_SAMPLE_BUDGET,
|
|
112
165
|
cache: VerdictCache | None = None,
|
|
113
166
|
attach_advisories: Any = None,
|
|
167
|
+
build_all: Any = None,
|
|
168
|
+
analyzer: str = NO_ANALYZER,
|
|
169
|
+
jev_key: str | None = None,
|
|
170
|
+
history: DriftHistory | None = None,
|
|
114
171
|
) -> Analysis:
|
|
115
172
|
"""Read this machine's sessions, correlate them, and judge them.
|
|
116
173
|
|
|
@@ -122,23 +179,111 @@ def analyse(
|
|
|
122
179
|
cache through it so osv.dev is asked once per component rather than once per
|
|
123
180
|
pass; a one-shot command passes nothing and gets the default.
|
|
124
181
|
|
|
182
|
+
`build_all` is the same kind of seam for composition builds, which shell out
|
|
183
|
+
and cost seconds each. A long-lived process passes a cache through it so an
|
|
184
|
+
unchanged machine or project is not rebuilt on every pass; a one-shot
|
|
185
|
+
command passes nothing and gets the default.
|
|
186
|
+
|
|
125
187
|
`session_ids` narrows the window to named sessions. Empty means the whole
|
|
126
188
|
window, which is what every command passes; monitor's reasoning button is
|
|
127
189
|
what needs the narrowing, and needs it to be structural.
|
|
128
190
|
|
|
129
191
|
`since` is the window's spelling — `parse_since` reads it — or the cutoff
|
|
130
192
|
itself, for a caller whose own `--since` spelling is not this one's.
|
|
193
|
+
|
|
194
|
+
`reasoning` switches the third stage on and is off by default; `jev_key`
|
|
195
|
+
hands the analyzer a key a caller already holds, in place of reading
|
|
196
|
+
`TYPESAFE_API_KEY`. Both travel to `run_detector` unchanged, so a
|
|
197
|
+
programmatic caller controls exactly what `detect` controls.
|
|
131
198
|
"""
|
|
132
199
|
window_end = datetime.now(UTC)
|
|
133
200
|
window_start = since if isinstance(since, datetime) else parse_since(since)
|
|
134
|
-
|
|
201
|
+
sessions = _collected(
|
|
135
202
|
agent_kinds=agent_kinds,
|
|
136
203
|
window_start=window_start,
|
|
137
|
-
bom_paths=bom_paths,
|
|
138
|
-
project_map=project_map,
|
|
139
204
|
root=root,
|
|
140
205
|
session_ids=session_ids,
|
|
206
|
+
)
|
|
207
|
+
return _analyse_sessions(
|
|
208
|
+
sessions,
|
|
209
|
+
window_start=window_start,
|
|
210
|
+
window_end=window_end,
|
|
211
|
+
bom_paths=bom_paths,
|
|
212
|
+
project_map=project_map,
|
|
213
|
+
reasoning=reasoning,
|
|
214
|
+
budget=budget,
|
|
215
|
+
sample_budget=sample_budget,
|
|
216
|
+
cache=cache,
|
|
141
217
|
attach_advisories=attach_advisories,
|
|
218
|
+
build_all=build_all,
|
|
219
|
+
analyzer=analyzer,
|
|
220
|
+
jev_key=jev_key,
|
|
221
|
+
history=history,
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def analyse_sessions(
|
|
226
|
+
sessions: SessionView,
|
|
227
|
+
*,
|
|
228
|
+
agent_kinds: tuple[str, ...] = (),
|
|
229
|
+
since: str | datetime = "7d",
|
|
230
|
+
bom_paths: tuple[Path, ...] = (),
|
|
231
|
+
project_map: tuple[str, ...] = (),
|
|
232
|
+
session_ids: tuple[str, ...] = (),
|
|
233
|
+
reasoning: bool = False,
|
|
234
|
+
budget: int = DEFAULT_BUDGET,
|
|
235
|
+
sample_budget: int = DEFAULT_SAMPLE_BUDGET,
|
|
236
|
+
cache: VerdictCache | None = None,
|
|
237
|
+
attach_advisories: Any = None,
|
|
238
|
+
build_all: Any = None,
|
|
239
|
+
analyzer: str = NO_ANALYZER,
|
|
240
|
+
jev_key: str | None = None,
|
|
241
|
+
history: DriftHistory | None = None,
|
|
242
|
+
) -> Analysis:
|
|
243
|
+
"""Correlate and judge a session view already collected by a long-lived reader."""
|
|
244
|
+
window_end = datetime.now(UTC)
|
|
245
|
+
window_start = since if isinstance(since, datetime) else parse_since(since)
|
|
246
|
+
return _analyse_sessions(
|
|
247
|
+
_only(sessions, session_ids, agent_kinds),
|
|
248
|
+
window_start=window_start,
|
|
249
|
+
window_end=window_end,
|
|
250
|
+
bom_paths=bom_paths,
|
|
251
|
+
project_map=project_map,
|
|
252
|
+
reasoning=reasoning,
|
|
253
|
+
budget=budget,
|
|
254
|
+
sample_budget=sample_budget,
|
|
255
|
+
cache=cache,
|
|
256
|
+
attach_advisories=attach_advisories,
|
|
257
|
+
build_all=build_all,
|
|
258
|
+
analyzer=analyzer,
|
|
259
|
+
jev_key=jev_key,
|
|
260
|
+
history=history,
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _analyse_sessions(
|
|
265
|
+
sessions: SessionView,
|
|
266
|
+
*,
|
|
267
|
+
window_start: datetime,
|
|
268
|
+
window_end: datetime,
|
|
269
|
+
bom_paths: tuple[Path, ...],
|
|
270
|
+
project_map: tuple[str, ...],
|
|
271
|
+
reasoning: bool,
|
|
272
|
+
budget: int,
|
|
273
|
+
sample_budget: int,
|
|
274
|
+
cache: VerdictCache | None,
|
|
275
|
+
attach_advisories: Any,
|
|
276
|
+
build_all: Any,
|
|
277
|
+
analyzer: str,
|
|
278
|
+
jev_key: str | None,
|
|
279
|
+
history: DriftHistory | None,
|
|
280
|
+
) -> Analysis:
|
|
281
|
+
acquired = _acquire(
|
|
282
|
+
sessions,
|
|
283
|
+
bom_paths=bom_paths,
|
|
284
|
+
project_map=project_map,
|
|
285
|
+
attach_advisories=attach_advisories,
|
|
286
|
+
build_all=build_all,
|
|
142
287
|
)
|
|
143
288
|
run = run_detector(
|
|
144
289
|
acquired.view,
|
|
@@ -146,6 +291,9 @@ def analyse(
|
|
|
146
291
|
budget=budget,
|
|
147
292
|
sample_budget=sample_budget,
|
|
148
293
|
cache=cache,
|
|
294
|
+
analyzer=analyzer,
|
|
295
|
+
jev_key=jev_key,
|
|
296
|
+
history=history,
|
|
149
297
|
)
|
|
150
298
|
return Analysis(
|
|
151
299
|
run=run,
|
|
@@ -170,7 +318,11 @@ def _collected(
|
|
|
170
318
|
session_ids: tuple[str, ...],
|
|
171
319
|
) -> SessionView:
|
|
172
320
|
"""Read the window. Its own function so a caller can read twice."""
|
|
173
|
-
return _only(
|
|
321
|
+
return _only(
|
|
322
|
+
collect_sessions(list(agent_kinds) or None, window_start, root=root),
|
|
323
|
+
session_ids,
|
|
324
|
+
agent_kinds,
|
|
325
|
+
)
|
|
174
326
|
|
|
175
327
|
|
|
176
328
|
def _acquire(
|
|
@@ -179,9 +331,14 @@ def _acquire(
|
|
|
179
331
|
bom_paths: tuple[Path, ...],
|
|
180
332
|
project_map: tuple[str, ...],
|
|
181
333
|
attach_advisories: Any,
|
|
334
|
+
build_all: Any = None,
|
|
182
335
|
) -> Acquired:
|
|
183
336
|
"""Correlate what was read."""
|
|
184
|
-
hook
|
|
337
|
+
hook: dict[str, Any] = {}
|
|
338
|
+
if attach_advisories is not None:
|
|
339
|
+
hook["attach_advisories"] = attach_advisories
|
|
340
|
+
if build_all is not None:
|
|
341
|
+
hook["build_all"] = build_all
|
|
185
342
|
return acquire_correlated_view(
|
|
186
343
|
sessions,
|
|
187
344
|
bom_paths=bom_paths,
|
|
@@ -199,6 +356,7 @@ def _correlated(
|
|
|
199
356
|
root: Path | None,
|
|
200
357
|
session_ids: tuple[str, ...],
|
|
201
358
|
attach_advisories: Any,
|
|
359
|
+
build_all: Any = None,
|
|
202
360
|
) -> Acquired:
|
|
203
361
|
"""Everything up to the judging, shared by both entry points."""
|
|
204
362
|
return _acquire(
|
|
@@ -211,6 +369,7 @@ def _correlated(
|
|
|
211
369
|
bom_paths=bom_paths,
|
|
212
370
|
project_map=project_map,
|
|
213
371
|
attach_advisories=attach_advisories,
|
|
372
|
+
build_all=build_all,
|
|
214
373
|
)
|
|
215
374
|
|
|
216
375
|
|
|
@@ -258,9 +417,18 @@ def analyse_progressively(
|
|
|
258
417
|
preview: timedelta | None = PREVIEW_WINDOW,
|
|
259
418
|
cache: VerdictCache | None = None,
|
|
260
419
|
attach_advisories: Any = None,
|
|
420
|
+
analyzer: str = NO_ANALYZER,
|
|
421
|
+
jev_key: str | None = None,
|
|
422
|
+
history: DriftHistory | None = None,
|
|
423
|
+
reasoning: bool = False,
|
|
424
|
+
budget: int = DEFAULT_BUDGET,
|
|
261
425
|
) -> Iterator[Analysis]:
|
|
262
426
|
"""The same pipeline, delivered in instalments.
|
|
263
427
|
|
|
428
|
+
`history` is accepted and unused: this path never requests reasoning, so
|
|
429
|
+
it never adds a point. It is here so the watcher can forward one options
|
|
430
|
+
dict to both entry points (the test in `tests/monitor/test_watch.py`).
|
|
431
|
+
|
|
264
432
|
`analyse` answers once, which is right for a command that prints and exits
|
|
265
433
|
and wrong for a page meant to look alive. Measured over 339 real sessions
|
|
266
434
|
the work splits about 6.6s to collect and correlate, then about 12ms to
|
|
@@ -289,11 +457,13 @@ def analyse_progressively(
|
|
|
289
457
|
`cache` is read and never written. A session graded by an earlier
|
|
290
458
|
`detect --reasoning` shows that grade here; this path never commissions one.
|
|
291
459
|
|
|
292
|
-
**
|
|
293
|
-
|
|
294
|
-
of what was authorised
|
|
295
|
-
|
|
296
|
-
|
|
460
|
+
**Reasoning runs once, after the batches, never inside them.** Stage 3's
|
|
461
|
+
budget is per *run*, so asking each batch would give each its own budget and
|
|
462
|
+
spend a multiple of what was authorised — which is why this path refused the
|
|
463
|
+
option entirely until now. Running it once over the whole view keeps the
|
|
464
|
+
budget meaning what the flag said, and keeps the ordering that matters: the
|
|
465
|
+
cheap stages paint first and stage 3 arrives as one more instalment, rather
|
|
466
|
+
than holding the first paint behind several hundred sequential requests.
|
|
297
467
|
"""
|
|
298
468
|
window_end = datetime.now(UTC)
|
|
299
469
|
window_start = since if isinstance(since, datetime) else parse_since(since)
|
|
@@ -348,20 +518,43 @@ def analyse_progressively(
|
|
|
348
518
|
if preview is None:
|
|
349
519
|
yield _stage(DetectorRun(collection_failures=failures), ())
|
|
350
520
|
|
|
521
|
+
# Newest *activity* first, not newest start. This decides which sessions are
|
|
522
|
+
# judged first and therefore what a progressive reader sees first, and
|
|
523
|
+
# `started_at` answers a different question: a session running for a day
|
|
524
|
+
# began long ago and is the most recent thing on the machine. Ordering by
|
|
525
|
+
# its start put the session doing the work behind sessions idle for hours --
|
|
526
|
+
# the same defect the monitor's own session list had, one layer up, so the
|
|
527
|
+
# batch order and the page's order now read the same clock.
|
|
351
528
|
ordered = sorted(
|
|
352
529
|
view.sessions,
|
|
353
|
-
key=lambda correlated:
|
|
530
|
+
key=lambda correlated: (
|
|
531
|
+
correlated.session.last_activity_at
|
|
532
|
+
or correlated.session.started_at
|
|
533
|
+
or datetime.min.replace(tzinfo=UTC)
|
|
534
|
+
),
|
|
354
535
|
reverse=True,
|
|
355
536
|
)
|
|
356
537
|
step = max(batch, 1)
|
|
357
|
-
accumulated = DetectorRun(collection_failures=failures)
|
|
538
|
+
accumulated = DetectorRun(collection_failures=failures, analyzer=analyzer)
|
|
358
539
|
for index in range(0, len(ordered), step):
|
|
359
540
|
judged = run_detector(
|
|
360
541
|
CorrelatedView(sessions=tuple(ordered[index : index + step])),
|
|
361
542
|
# Read, never written, and never requesting: a session an earlier
|
|
362
543
|
# `detect --reasoning` graded renders with that grade here, and this
|
|
363
|
-
# path commissions nothing (ADR-0026 clause 2).
|
|
544
|
+
# path commissions nothing (ADR-0026 clause 2). The analyzer name
|
|
545
|
+
# is what lets a stored Jev map be served: the cache key carries
|
|
546
|
+
# the analyzer's identity, so a page started with --analyzer jev
|
|
547
|
+
# reads the entries that flag paid for and no others.
|
|
548
|
+
#
|
|
549
|
+
# `jev_key` rides along although this path asks nothing. Today the
|
|
550
|
+
# cache key reads `identity()`, which does not depend on the key,
|
|
551
|
+
# so omitting it would work — but it would leave one call building
|
|
552
|
+
# two differently-configured analyzers under one name, and the next
|
|
553
|
+
# thing to depend on the key would break here and not in the
|
|
554
|
+
# instalment below.
|
|
364
555
|
cache=cache,
|
|
556
|
+
analyzer=analyzer,
|
|
557
|
+
jev_key=jev_key,
|
|
365
558
|
)
|
|
366
559
|
accumulated = DetectorRun(
|
|
367
560
|
detections=accumulated.detections + judged.detections,
|
|
@@ -370,6 +563,26 @@ def analyse_progressively(
|
|
|
370
563
|
requested=accumulated.requested + judged.requested,
|
|
371
564
|
analysed=accumulated.analysed + judged.analysed,
|
|
372
565
|
cache_hits=accumulated.cache_hits + judged.cache_hits,
|
|
566
|
+
maps=accumulated.maps + judged.maps,
|
|
373
567
|
collection_failures=failures,
|
|
568
|
+
analyzer=analyzer,
|
|
374
569
|
)
|
|
375
570
|
yield _stage(accumulated, tuple(ordered))
|
|
571
|
+
|
|
572
|
+
if not reasoning or not ordered:
|
|
573
|
+
return
|
|
574
|
+
|
|
575
|
+
# One more instalment, one run, one budget. Everything above has already
|
|
576
|
+
# been published, so the page is complete and readable while this is in
|
|
577
|
+
# flight — on a large window it is several hundred sequential requests and
|
|
578
|
+
# holding the first paint behind it is what made the page look hung.
|
|
579
|
+
judged = run_detector(
|
|
580
|
+
view,
|
|
581
|
+
reasoning=True,
|
|
582
|
+
budget=budget,
|
|
583
|
+
cache=cache,
|
|
584
|
+
analyzer=analyzer,
|
|
585
|
+
jev_key=jev_key,
|
|
586
|
+
history=history,
|
|
587
|
+
)
|
|
588
|
+
yield _stage(judged, tuple(ordered))
|
stacktrace_cli/build.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Which commit this install was built from, when it was not a PyPI release.
|
|
2
|
+
|
|
3
|
+
Read from the install's own PEP 610 record (`direct_url.json` in the
|
|
4
|
+
dist-info), which pip and uv write for every install that did not come from an
|
|
5
|
+
index:
|
|
6
|
+
|
|
7
|
+
- ``pip install git+https://…@branch`` / ``uv tool install git+…`` record the
|
|
8
|
+
resolved commit in ``vcs_info.commit_id``.
|
|
9
|
+
- ``pip install -e .`` / ``uv sync`` record an editable ``file://`` URL; the
|
|
10
|
+
code runs from that checkout, so its current ``HEAD`` is the answer, plus
|
|
11
|
+
``.dirty`` when the worktree has uncommitted changes.
|
|
12
|
+
|
|
13
|
+
A wheel from PyPI has no record, and a non-editable install from a local
|
|
14
|
+
directory records only the path, whose ``HEAD`` may have moved since the build
|
|
15
|
+
— both answer `None` rather than a commit that may not be the one running.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import functools
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import re
|
|
24
|
+
import subprocess
|
|
25
|
+
import threading
|
|
26
|
+
from importlib.metadata import distribution
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from urllib.parse import unquote, urlparse
|
|
29
|
+
|
|
30
|
+
DISTRIBUTION = "stacktrace-cli"
|
|
31
|
+
|
|
32
|
+
_SHA = re.compile(r"^[0-9a-f]{7,64}$")
|
|
33
|
+
_SHORT = 7
|
|
34
|
+
|
|
35
|
+
#: Serializes `_compute_version_label`'s cache misses. `functools.cache` alone
|
|
36
|
+
#: lets concurrent misses run the wrapped function more than once (verified:
|
|
37
|
+
#: two threads racing an uncached call both entered `build_commit`), so the
|
|
38
|
+
#: daemon's startup warm-up (`daemon/runtime.py`) and a request thread could
|
|
39
|
+
#: each launch their own `git` probe. Held only around the call: the thread
|
|
40
|
+
#: that loses the race blocks briefly here rather than repeating the winner's
|
|
41
|
+
#: work, and finds the answer already cached once it gets in.
|
|
42
|
+
_version_label_lock = threading.Lock()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def version_label() -> str:
|
|
46
|
+
"""The version as `--version` prints it: `0.4.0`, or `0.4.0+86eef86` for a
|
|
47
|
+
build from a branch or a pull request.
|
|
48
|
+
|
|
49
|
+
One expression for both places that report it, the CLI and telemetry, so a
|
|
50
|
+
PostHog event and `--version` cannot disagree about which build ran.
|
|
51
|
+
"""
|
|
52
|
+
with _version_label_lock:
|
|
53
|
+
return _compute_version_label()
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@functools.cache
|
|
57
|
+
def _compute_version_label() -> str:
|
|
58
|
+
from . import __version__
|
|
59
|
+
|
|
60
|
+
commit = build_commit()
|
|
61
|
+
return f"{__version__}+{commit}" if commit else __version__
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def build_commit() -> str | None:
|
|
65
|
+
"""The short commit this install runs, or `None` for a release install."""
|
|
66
|
+
try:
|
|
67
|
+
return _read_build_commit()
|
|
68
|
+
except Exception: # noqa: BLE001 - direct_url.json is external, untrusted input
|
|
69
|
+
# Every step below reads or parses `direct_url.json`, an external
|
|
70
|
+
# document written by pip or uv. Three narrower guards in a row each
|
|
71
|
+
# missed a new failure mode in turn (`UnicodeDecodeError` from
|
|
72
|
+
# `read_text`'s UTF-8 decode, `urlparse`'s `ValueError` on a malformed
|
|
73
|
+
# URL, `json.loads`'s `RecursionError` on deep nesting) -- the fix is
|
|
74
|
+
# the boundary, not the exception list. Anything this parse can raise
|
|
75
|
+
# reads the same as no record.
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _read_build_commit() -> str | None:
|
|
80
|
+
raw = distribution(DISTRIBUTION).read_text("direct_url.json")
|
|
81
|
+
if not raw:
|
|
82
|
+
return None
|
|
83
|
+
record = json.loads(raw)
|
|
84
|
+
if not isinstance(record, dict):
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
vcs = record.get("vcs_info")
|
|
88
|
+
if isinstance(vcs, dict):
|
|
89
|
+
commit = vcs.get("commit_id")
|
|
90
|
+
if vcs.get("vcs") == "git" and isinstance(commit, str) and _SHA.match(commit):
|
|
91
|
+
return commit[:_SHORT]
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
directory = record.get("dir_info")
|
|
95
|
+
url = record.get("url")
|
|
96
|
+
if isinstance(directory, dict) and directory.get("editable") is True and isinstance(url, str):
|
|
97
|
+
parsed = urlparse(url)
|
|
98
|
+
if parsed.scheme == "file":
|
|
99
|
+
return _checkout_commit(Path(unquote(parsed.path)))
|
|
100
|
+
return None
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _checkout_commit(root: Path) -> str | None:
|
|
104
|
+
# GIT_DIR/GIT_WORK_TREE in the caller's environment override -C, so a
|
|
105
|
+
# caller running us from inside another repo's hook can make git report
|
|
106
|
+
# that repo's HEAD instead of `root`'s.
|
|
107
|
+
env = {k: v for k, v in os.environ.items() if k not in ("GIT_DIR", "GIT_WORK_TREE")}
|
|
108
|
+
|
|
109
|
+
def git(*args: str) -> str | None:
|
|
110
|
+
try:
|
|
111
|
+
done = subprocess.run(
|
|
112
|
+
["git", "-C", str(root), *args],
|
|
113
|
+
capture_output=True,
|
|
114
|
+
text=True,
|
|
115
|
+
timeout=5,
|
|
116
|
+
check=False,
|
|
117
|
+
env=env,
|
|
118
|
+
)
|
|
119
|
+
except (OSError, subprocess.SubprocessError):
|
|
120
|
+
return None
|
|
121
|
+
return done.stdout if done.returncode == 0 else None
|
|
122
|
+
|
|
123
|
+
head = git("rev-parse", "HEAD")
|
|
124
|
+
if head is None or not _SHA.match(head.strip()):
|
|
125
|
+
return None
|
|
126
|
+
commit = head.strip()[:_SHORT]
|
|
127
|
+
status = git("status", "--porcelain")
|
|
128
|
+
if status is None:
|
|
129
|
+
return None
|
|
130
|
+
return f"{commit}.dirty" if status else commit
|