stacktrace-cli 0.0.1__py3-none-any.whl → 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. stacktrace_cli/__init__.py +1 -1
  2. stacktrace_cli/__main__.py +10 -18
  3. stacktrace_cli/analysis.py +375 -0
  4. stacktrace_cli/cli.py +470 -0
  5. stacktrace_cli/correlate/__init__.py +1 -0
  6. stacktrace_cli/correlate/acquire.py +369 -0
  7. stacktrace_cli/correlate/composition.py +434 -0
  8. stacktrace_cli/correlate/join.py +426 -0
  9. stacktrace_cli/correlate/observed.py +1836 -0
  10. stacktrace_cli/correlate/orchestrate.py +451 -0
  11. stacktrace_cli/correlate/project_map.py +81 -0
  12. stacktrace_cli/correlate/record.py +287 -0
  13. stacktrace_cli/correlate/render.py +766 -0
  14. stacktrace_cli/detector/__init__.py +1 -0
  15. stacktrace_cli/detector/analyzer.py +207 -0
  16. stacktrace_cli/detector/cache.py +483 -0
  17. stacktrace_cli/detector/deterministic.py +654 -0
  18. stacktrace_cli/detector/finding.py +356 -0
  19. stacktrace_cli/detector/markers.py +158 -0
  20. stacktrace_cli/detector/priors.py +502 -0
  21. stacktrace_cli/detector/prompts/__init__.py +36 -0
  22. stacktrace_cli/detector/prompts/v1/exclusions.md +36 -0
  23. stacktrace_cli/detector/prompts/v1/framing.md +23 -0
  24. stacktrace_cli/detector/prompts/v1/stacktrace-deceptive-completion.md +12 -0
  25. stacktrace_cli/detector/prompts/v1/stacktrace-injected-instruction-followed.md +14 -0
  26. stacktrace_cli/detector/prompts/v1/stacktrace-intent-drift.md +11 -0
  27. stacktrace_cli/detector/reasoning.py +1180 -0
  28. stacktrace_cli/detector/render.py +370 -0
  29. stacktrace_cli/detector/rules.py +136 -0
  30. stacktrace_cli/detector/run.py +821 -0
  31. stacktrace_cli/detector/secrets.py +203 -0
  32. stacktrace_cli/detector/verdict.py +266 -0
  33. stacktrace_cli/monitor/__init__.py +12 -0
  34. stacktrace_cli/monitor/escalate.py +172 -0
  35. stacktrace_cli/monitor/render.py +252 -0
  36. stacktrace_cli/monitor/server.py +477 -0
  37. stacktrace_cli/monitor/site/app.js +1045 -0
  38. stacktrace_cli/monitor/site/fonts/OFL.txt +210 -0
  39. stacktrace_cli/monitor/site/fonts/dm-mono-400-latin.woff2 +0 -0
  40. stacktrace_cli/monitor/site/fonts/dm-mono-500-latin.woff2 +0 -0
  41. stacktrace_cli/monitor/site/fonts/dm-sans-latin.woff2 +0 -0
  42. stacktrace_cli/monitor/site/index.html +104 -0
  43. stacktrace_cli/monitor/site/styles.css +628 -0
  44. stacktrace_cli/monitor/state.py +102 -0
  45. stacktrace_cli/monitor/verdicts.py +52 -0
  46. stacktrace_cli/monitor/watch.py +349 -0
  47. stacktrace_cli/remote/__init__.py +1 -0
  48. stacktrace_cli/remote/cli.py +475 -0
  49. stacktrace_cli/remote/client.py +393 -0
  50. stacktrace_cli/remote/config.py +106 -0
  51. stacktrace_cli/remote/detect_payload.py +334 -0
  52. stacktrace_cli/remote/payload.py +435 -0
  53. stacktrace_cli/remote/policy.py +88 -0
  54. stacktrace_cli/remote/redact.py +660 -0
  55. stacktrace_cli/remote/spool.py +446 -0
  56. stacktrace_cli/remote/sync.py +558 -0
  57. stacktrace_cli/remote/sync_detect.py +446 -0
  58. stacktrace_cli/remote/upload_contract.py +994 -0
  59. stacktrace_cli/sessions/__init__.py +1 -0
  60. stacktrace_cli/sessions/access.py +79 -0
  61. stacktrace_cli/sessions/outcome.py +65 -0
  62. stacktrace_cli/sessions/protocols.py +204 -0
  63. stacktrace_cli/sessions/render.py +295 -0
  64. stacktrace_cli-0.2.0.dist-info/METADATA +228 -0
  65. stacktrace_cli-0.2.0.dist-info/RECORD +67 -0
  66. stacktrace_cli-0.0.1.dist-info/METADATA +0 -39
  67. stacktrace_cli-0.0.1.dist-info/RECORD +0 -6
  68. {stacktrace_cli-0.0.1.dist-info → stacktrace_cli-0.2.0.dist-info}/WHEEL +0 -0
  69. {stacktrace_cli-0.0.1.dist-info → stacktrace_cli-0.2.0.dist-info}/entry_points.txt +0 -0
@@ -1,3 +1,3 @@
1
1
  """Placeholder CLI package for Stacktrace.ai."""
2
2
 
3
- __version__ = "0.0.1"
3
+ __version__ = "0.2.0"
@@ -1,24 +1,16 @@
1
- """Entry point for the ``stacktrace`` command."""
1
+ """Entry point for the ``stacktrace`` command.
2
2
 
3
- from __future__ import annotations
4
-
5
- import argparse
6
-
7
- from . import __version__
8
-
9
- HOMEPAGE = "https://stacktrace.ai"
3
+ `[project.scripts]` still names `stacktrace_cli.__main__:main`, which now
4
+ resolves to the Click group in `stacktrace_cli.cli`, so both the console script
5
+ and `python -m stacktrace_cli` reach the same object and the published
6
+ entry-point string does not change between releases.
7
+ """
10
8
 
9
+ from __future__ import annotations
11
10
 
12
- def main(argv: list[str] | None = None) -> int:
13
- parser = argparse.ArgumentParser(
14
- prog="stacktrace",
15
- description=f"Stacktrace.ai CLI (pre-alpha). See {HOMEPAGE}",
16
- )
17
- parser.add_argument("--version", action="version", version=f"stacktrace {__version__}")
18
- parser.parse_args(argv)
19
- print(f"stacktrace {__version__} — pre-alpha placeholder. See {HOMEPAGE}")
20
- return 0
11
+ from .cli import main
21
12
 
13
+ __all__ = ["main"]
22
14
 
23
15
  if __name__ == "__main__":
24
- raise SystemExit(main())
16
+ main()
@@ -0,0 +1,375 @@
1
+ """The read pipeline, in the one place that knows its order.
2
+
3
+ `run_detector` takes a `CorrelatedView`, not a `SessionView`, so
4
+ sessions -> correlate -> detect is stitched by the caller. It was stitched
5
+ twice — once in `cli.py`'s `detect`, once in `sync_detect.py` — and `monitor`
6
+ would have been the third copy of an order that must not differ between them.
7
+ A page showing different findings from the command, for no reason but a
8
+ divergent call site, is the failure this module exists to prevent.
9
+
10
+ Both halves of the result travel together because neither is recoverable from
11
+ the other: `run` carries the findings, and `view` carries the component a call
12
+ resolved to and the coverage a reader weighs them against.
13
+
14
+ This module knows nothing about Click. Every caller renders an error its own
15
+ way — a `ClickException` at the command line, a `SyncError` on the upload path —
16
+ so the failures here are `ValueError`, and the message is the one the upload
17
+ path's tests already pin.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from collections.abc import Iterator
23
+ from dataclasses import dataclass, replace
24
+ from datetime import UTC, datetime, timedelta
25
+ from pathlib import Path
26
+ from typing import Any
27
+
28
+ from stacktrace_cli.correlate.orchestrate import Acquired, acquire_correlated_view
29
+ from stacktrace_cli.correlate.project_map import parse_mapping
30
+ from stacktrace_cli.correlate.record import CorrelatedSession, CorrelatedView, Coverage
31
+ from stacktrace_cli.detector.cache import VerdictCache
32
+ from stacktrace_cli.detector.run import (
33
+ DEFAULT_BUDGET,
34
+ DEFAULT_SAMPLE_BUDGET,
35
+ DetectorRun,
36
+ run_detector,
37
+ )
38
+ from stacktrace_cli.sessions.access import SessionView, collect_sessions
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class Analysis:
43
+ """One run of the read pipeline, and the window it read."""
44
+
45
+ run: DetectorRun
46
+ #: The correlated record the detector judged. Kept because a consumer needs
47
+ #: what a call resolved to — `monitor` renders it, `sync detect` counts it —
48
+ #: and a `Detection` carries neither the resolution nor the coverage.
49
+ view: CorrelatedView
50
+ #: `parse_since`'s cutoff, retained. Nothing downstream of session
51
+ #: selection needed it before, so nothing kept it.
52
+ window_start: datetime
53
+ #: `now(UTC)` taken *before* collection, so the window never claims to
54
+ #: cover a session that arrived while the pipeline was running.
55
+ window_end: datetime
56
+ #: Whether correlation has run. `False` only on `analyse_progressively`'s
57
+ #: preview stages, which carry the sessions and their built-in activity
58
+ #: before a composition has been built — so a consumer knows that coverage
59
+ #: is not yet countable and that only calls needing no composition are
60
+ #: listed. `analyse` is always placed.
61
+ placed: bool = True
62
+
63
+
64
+ def parse_since(value: str) -> datetime:
65
+ """`7d` or `7` -> an aware UTC cut-off.
66
+
67
+ Raises `ValueError`, never a Click or transport exception: this module is
68
+ below both. The message is the upload path's, which its tests pin.
69
+ """
70
+ text = value.strip().lower()
71
+ try:
72
+ days = int(text[:-1]) if text.endswith("d") else int(text)
73
+ except ValueError:
74
+ raise ValueError(f"--since must be a positive day count, not {value!r}") from None
75
+ if days <= 0:
76
+ raise ValueError(f"--since must be a positive day count, not {value!r}")
77
+ try:
78
+ return datetime.now(UTC) - timedelta(days=days)
79
+ except OverflowError:
80
+ # `timedelta` takes any `int` but overflows past `timedelta.max.days`.
81
+ # The same user mistake as a non-numeric value, so the same diagnostic.
82
+ raise ValueError(f"--since must be a positive day count, not {value!r}") from None
83
+
84
+
85
+ def _only(view: SessionView, session_ids: tuple[str, ...]) -> SessionView:
86
+ """The window narrowed to named sessions, or the window itself.
87
+
88
+ Narrowed *before* correlation rather than after judgement, because the point
89
+ is that nothing else can be dispatched: a caller that pays for one named
90
+ session must not be able to spend that payment on another one the detector
91
+ happened to rank higher. `unavailable` is carried through unchanged — a
92
+ reader that failed is still a reader that failed, whichever session was
93
+ asked for.
94
+ """
95
+ if not session_ids:
96
+ return view
97
+ wanted = set(session_ids)
98
+ return replace(view, sessions=tuple(s for s in view.sessions if s.session_id in wanted))
99
+
100
+
101
+ def analyse(
102
+ *,
103
+ agent_kinds: tuple[str, ...] = (),
104
+ since: str | datetime = "7d",
105
+ bom_paths: tuple[Path, ...] = (),
106
+ project_map: tuple[str, ...] = (),
107
+ root: Path | None = None,
108
+ session_ids: tuple[str, ...] = (),
109
+ escalate: bool = False,
110
+ budget: int = DEFAULT_BUDGET,
111
+ sample_budget: int = DEFAULT_SAMPLE_BUDGET,
112
+ cache: VerdictCache | None = None,
113
+ attach_advisories: Any = None,
114
+ ) -> Analysis:
115
+ """Read this machine's sessions, correlate them, and judge them.
116
+
117
+ `window_end` is taken before collection rather than after: a window that
118
+ ends when the pipeline finished would claim to cover a session that started
119
+ while it ran.
120
+
121
+ `attach_advisories` is correlation's own seam. A long-lived process passes a
122
+ cache through it so osv.dev is asked once per component rather than once per
123
+ pass; a one-shot command passes nothing and gets the default.
124
+
125
+ `session_ids` narrows the window to named sessions. Empty means the whole
126
+ window, which is what every command passes; monitor's escalate button is
127
+ what needs the narrowing, and needs it to be structural.
128
+
129
+ `since` is the window's spelling — `parse_since` reads it — or the cutoff
130
+ itself, for a caller whose own `--since` spelling is not this one's.
131
+ """
132
+ window_end = datetime.now(UTC)
133
+ window_start = since if isinstance(since, datetime) else parse_since(since)
134
+ acquired = _correlated(
135
+ agent_kinds=agent_kinds,
136
+ window_start=window_start,
137
+ bom_paths=bom_paths,
138
+ project_map=project_map,
139
+ root=root,
140
+ session_ids=session_ids,
141
+ attach_advisories=attach_advisories,
142
+ )
143
+ run = run_detector(
144
+ acquired.view,
145
+ escalate=escalate,
146
+ budget=budget,
147
+ sample_budget=sample_budget,
148
+ cache=cache,
149
+ )
150
+ return Analysis(
151
+ run=run,
152
+ view=acquired.view,
153
+ window_start=window_start,
154
+ window_end=window_end,
155
+ )
156
+
157
+
158
+ #: How far back the first, cheap read goes. Collection scales with the window
159
+ #: on a fixed floor — measured at 0.24s for a minute, 0.33s for an hour and
160
+ #: 1.20s for three days — so reading the newest hour first is what puts rows on
161
+ #: screen in a third of a second instead of one and a half.
162
+ PREVIEW_WINDOW = timedelta(hours=1)
163
+
164
+
165
+ def _collected(
166
+ *,
167
+ agent_kinds: tuple[str, ...],
168
+ window_start: datetime,
169
+ root: Path | None,
170
+ session_ids: tuple[str, ...],
171
+ ) -> SessionView:
172
+ """Read the window. Its own function so a caller can read twice."""
173
+ return _only(collect_sessions(list(agent_kinds) or None, window_start, root=root), session_ids)
174
+
175
+
176
+ def _acquire(
177
+ sessions: SessionView,
178
+ *,
179
+ bom_paths: tuple[Path, ...],
180
+ project_map: tuple[str, ...],
181
+ attach_advisories: Any,
182
+ ) -> Acquired:
183
+ """Correlate what was read."""
184
+ hook = {} if attach_advisories is None else {"attach_advisories": attach_advisories}
185
+ return acquire_correlated_view(
186
+ sessions,
187
+ bom_paths=bom_paths,
188
+ project_map=tuple(parse_mapping(mapping) for mapping in project_map),
189
+ **hook,
190
+ )
191
+
192
+
193
+ def _correlated(
194
+ *,
195
+ agent_kinds: tuple[str, ...],
196
+ window_start: datetime,
197
+ bom_paths: tuple[Path, ...],
198
+ project_map: tuple[str, ...],
199
+ root: Path | None,
200
+ session_ids: tuple[str, ...],
201
+ attach_advisories: Any,
202
+ ) -> Acquired:
203
+ """Everything up to the judging, shared by both entry points."""
204
+ return _acquire(
205
+ _collected(
206
+ agent_kinds=agent_kinds,
207
+ window_start=window_start,
208
+ root=root,
209
+ session_ids=session_ids,
210
+ ),
211
+ bom_paths=bom_paths,
212
+ project_map=project_map,
213
+ attach_advisories=attach_advisories,
214
+ )
215
+
216
+
217
+ def _unplaced(sessions: SessionView, *, window_start: datetime, window_end: datetime) -> Analysis:
218
+ """The sessions as read, before anything has been correlated or judged.
219
+
220
+ Correlation is 2.7s of a 4.4s first paint and it places 0.9% of calls; the
221
+ other 99% are built-in tools, which need no composition at all. So this
222
+ carries every session read and lets the render list the calls a
223
+ composition is not needed for — coverage stays uncounted, and no finding is
224
+ claimed, because neither is known yet.
225
+ """
226
+ return Analysis(
227
+ run=DetectorRun(
228
+ collection_failures=run_detector(
229
+ CorrelatedView(unavailable=sessions.unavailable)
230
+ ).collection_failures
231
+ ),
232
+ view=CorrelatedView(
233
+ unavailable=sessions.unavailable,
234
+ sessions=tuple(
235
+ CorrelatedSession(
236
+ session=session,
237
+ composition=None,
238
+ resolutions={},
239
+ coverage=Coverage(total=0, resolved=0, by_outcome={}, by_component_type={}),
240
+ )
241
+ for session in sessions.sessions
242
+ ),
243
+ ),
244
+ window_start=window_start,
245
+ window_end=window_end,
246
+ placed=False,
247
+ )
248
+
249
+
250
+ def analyse_progressively(
251
+ *,
252
+ agent_kinds: tuple[str, ...] = (),
253
+ since: str | datetime = "7d",
254
+ bom_paths: tuple[Path, ...] = (),
255
+ project_map: tuple[str, ...] = (),
256
+ root: Path | None = None,
257
+ batch: int = 8,
258
+ preview: timedelta | None = PREVIEW_WINDOW,
259
+ cache: VerdictCache | None = None,
260
+ attach_advisories: Any = None,
261
+ ) -> Iterator[Analysis]:
262
+ """The same pipeline, delivered in instalments.
263
+
264
+ `analyse` answers once, which is right for a command that prints and exits
265
+ and wrong for a page meant to look alive. Measured over 339 real sessions
266
+ the work splits about 6.6s to collect and correlate, then about 12ms to
267
+ judge each session — so a caller that waits for the whole run shows nothing
268
+ for six seconds while already holding the sessions and their activity.
269
+
270
+ Three kinds of yield, cheapest first.
271
+
272
+ **Preview stages, before correlation** (`placed=False`). Correlation is
273
+ 2.7s of a 4.4s first paint and it places 0.9% of a real window's calls; the
274
+ other 99% are built-in tools, which `join.py`'s predicates answer for in a
275
+ millisecond. So the sessions and those calls go out first — the newest
276
+ `preview` of them, then the whole window — and the page fills in ~0.35s
277
+ rather than ~4.4s. Pass `preview=None` for attributed data only.
278
+
279
+ **Then batches of judged sessions**, newest first, because someone watching
280
+ a machine cares about what just happened and should not have to scroll for
281
+ it. Every one of these carries the *whole* session list: the preview
282
+ already published it, and a later stage carrying fewer would make the count
283
+ on the page go backwards.
284
+
285
+ The last yield equals what `analyse` would have returned. If it did not,
286
+ the page and the command would disagree, which is the thing one pipeline
287
+ exists to prevent.
288
+
289
+ `cache` is read and never written. A session graded by an earlier
290
+ `detect --escalate` shows that grade here; this path never commissions one.
291
+
292
+ **No `escalate` parameter, deliberately.** Stage 3's budget is per *run*, so
293
+ judging in batches would give each batch its own budget and spend a multiple
294
+ of what was authorised. A streaming escalation needs a budget shared across
295
+ batches; until it has one, this path does not offer the option rather than
296
+ offering it wrongly.
297
+ """
298
+ window_end = datetime.now(UTC)
299
+ window_start = since if isinstance(since, datetime) else parse_since(since)
300
+
301
+ def read(cutoff: datetime) -> SessionView:
302
+ return _collected(agent_kinds=agent_kinds, window_start=cutoff, root=root, session_ids=())
303
+
304
+ if preview is not None:
305
+ cutoff = window_end - preview
306
+ # A window already narrower than the preview is read once, not twice.
307
+ if cutoff > window_start:
308
+ yield _unplaced(read(cutoff), window_start=window_start, window_end=window_end)
309
+
310
+ sessions = read(window_start)
311
+ if preview is not None:
312
+ yield _unplaced(sessions, window_start=window_start, window_end=window_end)
313
+
314
+ view = _acquire(
315
+ sessions,
316
+ bom_paths=bom_paths,
317
+ project_map=project_map,
318
+ attach_advisories=attach_advisories,
319
+ ).view
320
+
321
+ def _stage(run: DetectorRun, sessions: tuple[CorrelatedSession, ...]) -> Analysis:
322
+ """One instalment: every session, and the findings judged so far.
323
+
324
+ The whole session list, every time. It grew with the findings until the
325
+ preview stages existed — the point then was that the feed should not
326
+ appear all at once — but the preview publishes every session within
327
+ ~1.6s and a client that paces its own reveal streams them in anyway. A
328
+ stage carrying fewer sessions than the preview did would make the count
329
+ on the page go backwards.
330
+ """
331
+ return Analysis(
332
+ run=run,
333
+ view=replace(view, sessions=sessions),
334
+ window_start=window_start,
335
+ window_end=window_end,
336
+ )
337
+
338
+ # The detector derives these from the view, and a stage that has not judged
339
+ # anything yet still has to carry them: a reader that failed is the one
340
+ # thing a page must never render as silence.
341
+ failures = run_detector(CorrelatedView(unavailable=view.unavailable)).collection_failures
342
+
343
+ # Stage zero: what could not be read, before a session is judged. A reader
344
+ # that failed is the one thing a page must never render as silence, and it
345
+ # is known before the detector has done anything at all. Skipped when a
346
+ # preview already carried it, since that stage said the same and carried
347
+ # the sessions too.
348
+ if preview is None:
349
+ yield _stage(DetectorRun(collection_failures=failures), ())
350
+
351
+ ordered = sorted(
352
+ view.sessions,
353
+ key=lambda correlated: correlated.session.started_at or datetime.min.replace(tzinfo=UTC),
354
+ reverse=True,
355
+ )
356
+ step = max(batch, 1)
357
+ accumulated = DetectorRun(collection_failures=failures)
358
+ for index in range(0, len(ordered), step):
359
+ judged = run_detector(
360
+ CorrelatedView(sessions=tuple(ordered[index : index + step])),
361
+ # Read, never written, and never escalating: a session an earlier
362
+ # `detect --escalate` graded renders with that grade here, and this
363
+ # path commissions nothing (ADR-0026 clause 2).
364
+ cache=cache,
365
+ )
366
+ accumulated = DetectorRun(
367
+ detections=accumulated.detections + judged.detections,
368
+ unknowns=accumulated.unknowns + judged.unknowns,
369
+ sessions=accumulated.sessions + judged.sessions,
370
+ escalated=accumulated.escalated + judged.escalated,
371
+ analysed=accumulated.analysed + judged.analysed,
372
+ cache_hits=accumulated.cache_hits + judged.cache_hits,
373
+ collection_failures=failures,
374
+ )
375
+ yield _stage(accumulated, tuple(ordered))