stacktrace-cli 0.2.2__py3-none-any.whl → 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/analysis.py +9 -9
- stacktrace_cli/cli.py +37 -28
- stacktrace_cli/correlate/acquire.py +154 -2
- stacktrace_cli/detector/analyzer.py +1 -1
- stacktrace_cli/detector/cache.py +15 -15
- stacktrace_cli/detector/deterministic.py +1 -1
- stacktrace_cli/detector/finding.py +3 -3
- stacktrace_cli/detector/markers.py +1 -1
- stacktrace_cli/detector/priors.py +21 -21
- stacktrace_cli/detector/reasoning.py +74 -63
- stacktrace_cli/detector/render.py +22 -25
- stacktrace_cli/detector/rules.py +3 -3
- stacktrace_cli/detector/run.py +65 -65
- stacktrace_cli/monitor/{escalate.py → reasoning.py} +13 -13
- stacktrace_cli/monitor/render.py +2 -2
- stacktrace_cli/monitor/server.py +37 -35
- stacktrace_cli/monitor/site/app.js +22 -22
- stacktrace_cli/monitor/site/index.html +1 -5
- stacktrace_cli/monitor/site/styles.css +4 -9
- stacktrace_cli/monitor/state.py +3 -3
- stacktrace_cli/monitor/verdicts.py +3 -3
- stacktrace_cli/monitor/watch.py +7 -7
- stacktrace_cli/remote/cli.py +8 -9
- stacktrace_cli/remote/sync_detect.py +13 -7
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/METADATA +7 -6
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/RECORD +29 -29
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/WHEEL +0 -0
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/entry_points.txt +0 -0
stacktrace_cli/__init__.py
CHANGED
stacktrace_cli/analysis.py
CHANGED
|
@@ -106,7 +106,7 @@ def analyse(
|
|
|
106
106
|
project_map: tuple[str, ...] = (),
|
|
107
107
|
root: Path | None = None,
|
|
108
108
|
session_ids: tuple[str, ...] = (),
|
|
109
|
-
|
|
109
|
+
reasoning: bool = False,
|
|
110
110
|
budget: int = DEFAULT_BUDGET,
|
|
111
111
|
sample_budget: int = DEFAULT_SAMPLE_BUDGET,
|
|
112
112
|
cache: VerdictCache | None = None,
|
|
@@ -123,7 +123,7 @@ def analyse(
|
|
|
123
123
|
pass; a one-shot command passes nothing and gets the default.
|
|
124
124
|
|
|
125
125
|
`session_ids` narrows the window to named sessions. Empty means the whole
|
|
126
|
-
window, which is what every command passes; monitor's
|
|
126
|
+
window, which is what every command passes; monitor's reasoning button is
|
|
127
127
|
what needs the narrowing, and needs it to be structural.
|
|
128
128
|
|
|
129
129
|
`since` is the window's spelling — `parse_since` reads it — or the cutoff
|
|
@@ -142,7 +142,7 @@ def analyse(
|
|
|
142
142
|
)
|
|
143
143
|
run = run_detector(
|
|
144
144
|
acquired.view,
|
|
145
|
-
|
|
145
|
+
reasoning=reasoning,
|
|
146
146
|
budget=budget,
|
|
147
147
|
sample_budget=sample_budget,
|
|
148
148
|
cache=cache,
|
|
@@ -287,11 +287,11 @@ def analyse_progressively(
|
|
|
287
287
|
exists to prevent.
|
|
288
288
|
|
|
289
289
|
`cache` is read and never written. A session graded by an earlier
|
|
290
|
-
`detect --
|
|
290
|
+
`detect --reasoning` shows that grade here; this path never commissions one.
|
|
291
291
|
|
|
292
|
-
**No `
|
|
292
|
+
**No `reasoning` parameter, deliberately.** Stage 3's budget is per *run*, so
|
|
293
293
|
judging in batches would give each batch its own budget and spend a multiple
|
|
294
|
-
of what was authorised. A streaming
|
|
294
|
+
of what was authorised. A streaming reasoning run needs a budget shared across
|
|
295
295
|
batches; until it has one, this path does not offer the option rather than
|
|
296
296
|
offering it wrongly.
|
|
297
297
|
"""
|
|
@@ -358,8 +358,8 @@ def analyse_progressively(
|
|
|
358
358
|
for index in range(0, len(ordered), step):
|
|
359
359
|
judged = run_detector(
|
|
360
360
|
CorrelatedView(sessions=tuple(ordered[index : index + step])),
|
|
361
|
-
# Read, never written, and never
|
|
362
|
-
# `detect --
|
|
361
|
+
# Read, never written, and never requesting: a session an earlier
|
|
362
|
+
# `detect --reasoning` graded renders with that grade here, and this
|
|
363
363
|
# path commissions nothing (ADR-0026 clause 2).
|
|
364
364
|
cache=cache,
|
|
365
365
|
)
|
|
@@ -367,7 +367,7 @@ def analyse_progressively(
|
|
|
367
367
|
detections=accumulated.detections + judged.detections,
|
|
368
368
|
unknowns=accumulated.unknowns + judged.unknowns,
|
|
369
369
|
sessions=accumulated.sessions + judged.sessions,
|
|
370
|
-
|
|
370
|
+
requested=accumulated.requested + judged.requested,
|
|
371
371
|
analysed=accumulated.analysed + judged.analysed,
|
|
372
372
|
cache_hits=accumulated.cache_hits + judged.cache_hits,
|
|
373
373
|
collection_failures=failures,
|
stacktrace_cli/cli.py
CHANGED
|
@@ -223,16 +223,19 @@ def sessions(
|
|
|
223
223
|
help="Show every detection and all of its evidence. Default: the summary alone.",
|
|
224
224
|
)
|
|
225
225
|
@click.option(
|
|
226
|
-
"--
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
help="
|
|
230
|
-
"
|
|
226
|
+
"--reasoning",
|
|
227
|
+
is_flag=True,
|
|
228
|
+
default=False,
|
|
229
|
+
help="Analyse flagged sessions with a reasoning model, run through the "
|
|
230
|
+
"agent's own CLI. Off by default: it is the only stage that sends anything "
|
|
231
231
|
"session-derived off this machine, and it spends the developer's own "
|
|
232
|
-
"provider quota, so
|
|
233
|
-
"
|
|
234
|
-
"
|
|
235
|
-
"
|
|
232
|
+
"provider quota, so a run that does neither is the one to reach for first. "
|
|
233
|
+
"--reasoning turns it on "
|
|
234
|
+
"and --budget caps a noisy day. Leaving it off does not make the run "
|
|
235
|
+
"offline: correlation runs first either way and matches advisories by "
|
|
236
|
+
"sending the package coordinates of the components a session invoked to "
|
|
237
|
+
"osv.dev. Sessions that qualified are still counted, and named in "
|
|
238
|
+
"--format json, so the run says what it did not do.",
|
|
236
239
|
)
|
|
237
240
|
@click.option(
|
|
238
241
|
"--budget",
|
|
@@ -279,7 +282,7 @@ def detect(
|
|
|
279
282
|
bom_paths: tuple[Path, ...],
|
|
280
283
|
output_format: str,
|
|
281
284
|
detail: bool,
|
|
282
|
-
|
|
285
|
+
reasoning: bool,
|
|
283
286
|
budget: int,
|
|
284
287
|
sample_budget: int,
|
|
285
288
|
cache: bool,
|
|
@@ -295,18 +298,20 @@ def detect(
|
|
|
295
298
|
families carry their own severity ladders, because a stalled loop and a
|
|
296
299
|
leaked credential cannot share one (ADR-0010).
|
|
297
300
|
|
|
298
|
-
Two stages need no model or credential
|
|
299
|
-
the agent's own CLI:
|
|
300
|
-
--
|
|
301
|
+
Two stages need no model or credential, and they are the two that run by
|
|
302
|
+
default. The third sends flagged sessions to the agent's own CLI:
|
|
303
|
+
--reasoning turns it on, capped by --budget. Sessions that qualified for it
|
|
304
|
+
are counted either way, and named in --format json, so a run without it
|
|
305
|
+
says so rather than reading as a clean one.
|
|
301
306
|
|
|
302
307
|
That third stage is the one that sends session content off this machine:
|
|
303
308
|
the agent's own CLI hands the prompts, arguments and results a rule needs
|
|
304
309
|
to the provider it is already authenticated against (ADR-0004 -- provider
|
|
305
310
|
affinity, so a transcript goes back to the vendor that produced it, and
|
|
306
|
-
there is no fallback to any other). --
|
|
307
|
-
that send nothing.
|
|
311
|
+
there is no fallback to any other). Without --reasoning the run is the two
|
|
312
|
+
stages that send nothing.
|
|
308
313
|
|
|
309
|
-
One network call happens before any of them
|
|
314
|
+
One network call happens before any of them, with or without --reasoning:
|
|
310
315
|
correlation matches advisories by sending the package coordinates of the
|
|
311
316
|
components a session invoked to osv.dev. Coordinates only -- never a
|
|
312
317
|
prompt, an argument or a result.
|
|
@@ -321,12 +326,16 @@ def detect(
|
|
|
321
326
|
bom_paths=bom_paths,
|
|
322
327
|
project_map=project_map,
|
|
323
328
|
root=root,
|
|
324
|
-
|
|
329
|
+
reasoning=reasoning,
|
|
325
330
|
budget=budget,
|
|
326
331
|
sample_budget=sample_budget,
|
|
327
|
-
#
|
|
328
|
-
#
|
|
329
|
-
|
|
332
|
+
# Not gated on `reasoning`. A verdict already paid for is an
|
|
333
|
+
# answer this run has, and `reasoning` governs whether new ones may
|
|
334
|
+
# be commissioned, not whether old ones may be read -- the rule
|
|
335
|
+
# `run_detector` states at its cache pass. Gating it here made a
|
|
336
|
+
# session read as graded or ungraded depending on a flag that
|
|
337
|
+
# spent nothing either way.
|
|
338
|
+
cache=VerdictCache(default_directory()) if cache else None,
|
|
330
339
|
)
|
|
331
340
|
except ValueError as error:
|
|
332
341
|
raise click.ClickException(str(error)) from error
|
|
@@ -360,11 +369,11 @@ def detect(
|
|
|
360
369
|
"entirely when nothing on disk has changed.",
|
|
361
370
|
)
|
|
362
371
|
@click.option(
|
|
363
|
-
"--
|
|
372
|
+
"--reasoning",
|
|
373
|
+
is_flag=True,
|
|
364
374
|
default=False,
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
"analysis. Off by default: stage 3 spends your provider quota, and a page "
|
|
375
|
+
help="Let the page analyse one session with a reasoning model, run through "
|
|
376
|
+
"the agent's own CLI. Off by default: stage 3 spends your provider quota, and a page "
|
|
368
377
|
"left open is the wrong place for that to be implicit. Capped by --budget.",
|
|
369
378
|
)
|
|
370
379
|
@click.option(
|
|
@@ -372,7 +381,7 @@ def detect(
|
|
|
372
381
|
type=click.IntRange(min=1),
|
|
373
382
|
default=DEFAULT_BUDGET,
|
|
374
383
|
show_default=True,
|
|
375
|
-
help="
|
|
384
|
+
help="Reasoning runs this monitor may spend in total, if --reasoning is on.",
|
|
376
385
|
)
|
|
377
386
|
@click.option(
|
|
378
387
|
"--no-open", is_flag=True, default=False, help="Print the URL, do not open a browser."
|
|
@@ -413,7 +422,7 @@ def monitor(
|
|
|
413
422
|
port: int,
|
|
414
423
|
host: str,
|
|
415
424
|
interval: float,
|
|
416
|
-
|
|
425
|
+
reasoning: bool,
|
|
417
426
|
budget: int,
|
|
418
427
|
no_open: bool,
|
|
419
428
|
agent_kinds: tuple[str, ...],
|
|
@@ -431,7 +440,7 @@ def monitor(
|
|
|
431
440
|
|
|
432
441
|
Loopback only, and free to leave open: a pass is skipped when nothing has
|
|
433
442
|
changed, advisory lookups are asked once per component, and the two stages
|
|
434
|
-
that need no model are the only ones that run. --
|
|
443
|
+
that need no model are the only ones that run. --reasoning adds a button
|
|
435
444
|
that spends your provider quota, off unless asked for.
|
|
436
445
|
"""
|
|
437
446
|
try:
|
|
@@ -439,7 +448,7 @@ def monitor(
|
|
|
439
448
|
host=host,
|
|
440
449
|
port=port,
|
|
441
450
|
interval=interval,
|
|
442
|
-
|
|
451
|
+
reasoning=reasoning,
|
|
443
452
|
budget=budget,
|
|
444
453
|
open_browser=not no_open,
|
|
445
454
|
echo=click.echo,
|
|
@@ -16,13 +16,23 @@ answer naming a component that was never involved.
|
|
|
16
16
|
|
|
17
17
|
from __future__ import annotations
|
|
18
18
|
|
|
19
|
+
import base64
|
|
20
|
+
import hashlib
|
|
19
21
|
import json
|
|
22
|
+
import os
|
|
20
23
|
import subprocess
|
|
24
|
+
import sys
|
|
21
25
|
import tempfile
|
|
22
26
|
from collections.abc import Callable, Sequence
|
|
23
27
|
from collections.abc import Set as AbstractSet
|
|
24
28
|
from dataclasses import dataclass
|
|
25
29
|
from datetime import UTC, datetime
|
|
30
|
+
from importlib.metadata import (
|
|
31
|
+
Distribution,
|
|
32
|
+
PackageNotFoundError,
|
|
33
|
+
PackagePath,
|
|
34
|
+
distribution,
|
|
35
|
+
)
|
|
26
36
|
from pathlib import Path
|
|
27
37
|
from typing import Any
|
|
28
38
|
|
|
@@ -36,6 +46,140 @@ from stacktrace_cli.correlate.composition import (
|
|
|
36
46
|
|
|
37
47
|
_AGENT_KIND_PROPERTY = "openaca:agent_kind"
|
|
38
48
|
|
|
49
|
+
#: The name OpenACA publishes its console script under. `.exe` on Windows
|
|
50
|
+
#: because that is the launcher installers write there, and this package claims
|
|
51
|
+
#: `Operating System :: OS Independent`.
|
|
52
|
+
_SCRIPT = "openaca.exe" if os.name == "nt" else "openaca"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _openaca() -> str:
|
|
56
|
+
"""The OpenACA console script this package pins, addressed by path.
|
|
57
|
+
|
|
58
|
+
The bare string `"openaca"` is a `PATH` lookup, and `PATH` is not where the
|
|
59
|
+
pinned copy lives: `uv tool install stacktrace-cli` links only the
|
|
60
|
+
`stacktrace` entry point out of the tool environment, so the OpenACA that
|
|
61
|
+
`[project.dependencies]` resolved sits in `bin/` reachable but unnamed. A
|
|
62
|
+
bare name therefore selects whichever global, `pipx` or `--user` copy the
|
|
63
|
+
machine carries — a stale one silently does the work and its version need
|
|
64
|
+
not be the one this package depends on — or, on a machine with none,
|
|
65
|
+
nothing at all. This is the rule `docs/specs/cli-composition.md` states for
|
|
66
|
+
the mounted commands, applied to the one path that reaches OpenACA as a
|
|
67
|
+
subprocess.
|
|
68
|
+
|
|
69
|
+
**The path is the one the pinned distribution's own install recorded, not a
|
|
70
|
+
guess from this interpreter's location.** Python defines separate prefix,
|
|
71
|
+
virtual-environment, user and home installation schemes, and only the first
|
|
72
|
+
two put console scripts beside the interpreter: `pip install --user
|
|
73
|
+
stacktrace-cli` puts them under `site.USER_BASE` while `sys.executable`
|
|
74
|
+
stays the base interpreter. Deriving the path from `sys.executable` alone
|
|
75
|
+
therefore names a file that does not exist under a supported install, and
|
|
76
|
+
`detect` dies at process launch before it can build anything.
|
|
77
|
+
`importlib.metadata` answers instead: it resolves the OpenACA on *this*
|
|
78
|
+
interpreter's `sys.path` — the pinned one, by construction — and that
|
|
79
|
+
install's `RECORD` names the script file it wrote, wherever its scheme put
|
|
80
|
+
it.
|
|
81
|
+
|
|
82
|
+
**The only thing run is a file that install recorded writing, still holding
|
|
83
|
+
the contents it recorded.** Nothing is run for being named `openaca` in a
|
|
84
|
+
directory the distribution merely shares: an installation scheme pairs a
|
|
85
|
+
library directory with a scripts directory, but that scripts directory holds
|
|
86
|
+
the launchers of *every* distribution installed into the scheme and records
|
|
87
|
+
nothing about which one wrote any of them, so co-location establishes where
|
|
88
|
+
installers put scripts in general and never that this distribution put this
|
|
89
|
+
one there. Nor does a recorded path settle it on its own, because a record
|
|
90
|
+
describes what an install wrote rather than reserving where it wrote it: the
|
|
91
|
+
scripts directory stays shared afterwards, so a later install of another
|
|
92
|
+
distribution can overwrite that very file while the record still names it.
|
|
93
|
+
A stale, global, independently managed or overwritten launcher of that name
|
|
94
|
+
reintroduces exactly the version skew this addressing exists to remove, and
|
|
95
|
+
it does so invisibly — a wrong-version OpenACA can return a structurally
|
|
96
|
+
valid BOM. Where ownership cannot be established the answer is a
|
|
97
|
+
`RuntimeError` naming why, which `_build` renders as a legible CLI error.
|
|
98
|
+
|
|
99
|
+
Still the published console script, so the ADR-0007 seam is unchanged: this
|
|
100
|
+
module imports no OpenACA internals and `tests/test_seam_boundary.py` holds
|
|
101
|
+
that line.
|
|
102
|
+
"""
|
|
103
|
+
try:
|
|
104
|
+
installed = distribution("openaca")
|
|
105
|
+
except PackageNotFoundError as error:
|
|
106
|
+
raise RuntimeError(
|
|
107
|
+
"the OpenACA this package pins is not installed on "
|
|
108
|
+
f"{sys.executable}'s import path, so no {_SCRIPT!r} on this machine "
|
|
109
|
+
"can be known to be it; reinstalling stacktrace-cli restores it"
|
|
110
|
+
) from error
|
|
111
|
+
recorded = _recorded_script(installed)
|
|
112
|
+
if recorded is not None:
|
|
113
|
+
return str(recorded)
|
|
114
|
+
raise RuntimeError(
|
|
115
|
+
f"the pinned OpenACA installed at {installed.locate_file('')} records no "
|
|
116
|
+
f"{_SCRIPT!r} console script of its own that still holds the contents it "
|
|
117
|
+
"recorded; a copy found by name alone — by sharing a directory with that "
|
|
118
|
+
"install, or by having replaced a file it wrote — need not be the version "
|
|
119
|
+
"this package depends on, so none is run; reinstalling stacktrace-cli "
|
|
120
|
+
"restores it"
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _recorded_script(installed: Distribution) -> Path | None:
|
|
125
|
+
"""The console script the pinned OpenACA's install recorded, if it did.
|
|
126
|
+
|
|
127
|
+
A `RECORD` entry says the distribution `[project.dependencies]` resolved
|
|
128
|
+
wrote a file of these contents at this path — which is more than a directory
|
|
129
|
+
a scheme suggested can say, and less than the path alone would suggest,
|
|
130
|
+
because the two halves age differently: the path stays shared, the contents
|
|
131
|
+
do not. Scripts live outside `site-packages` and are recorded relative to
|
|
132
|
+
it, so the entry resolves correctly under every scheme.
|
|
133
|
+
|
|
134
|
+
Absent when the install enumerated no files: `importlib.metadata` answers
|
|
135
|
+
`None` when the metadata that lists them is missing — `RECORD` for a
|
|
136
|
+
`dist-info` install — and an `egg-info` install lists the sources it was
|
|
137
|
+
built from rather than the scripts it wrote. That is a distribution that
|
|
138
|
+
cannot name its own launcher, not licence to run someone else's: the only
|
|
139
|
+
other thing available about such an install is where it sits, and a
|
|
140
|
+
directory is shared by everything installed alongside it.
|
|
141
|
+
|
|
142
|
+
Absent, too, when the file at the recorded path is not the file that was
|
|
143
|
+
recorded there. Every candidate is authenticated against its own recorded
|
|
144
|
+
digest, because the two halves of an entry answer different questions: the
|
|
145
|
+
path says where the install wrote a launcher, and the digest says what it
|
|
146
|
+
wrote. Only the second survives another install overwriting the shared
|
|
147
|
+
scripts directory afterwards.
|
|
148
|
+
"""
|
|
149
|
+
for recorded in installed.files or ():
|
|
150
|
+
if recorded.name == _SCRIPT:
|
|
151
|
+
candidate = Path(str(installed.locate_file(recorded))).resolve()
|
|
152
|
+
if _holds_recorded_contents(recorded, candidate):
|
|
153
|
+
return candidate
|
|
154
|
+
return None
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _holds_recorded_contents(recorded: PackagePath, candidate: Path) -> bool:
|
|
158
|
+
"""Whether the file now at a recorded path hashes to what was recorded for it.
|
|
159
|
+
|
|
160
|
+
Fails closed on every reason the question cannot be answered — no digest,
|
|
161
|
+
a digest naming an algorithm this interpreter does not implement, an
|
|
162
|
+
unreadable file — because "we could not check" and "it matches" differ by
|
|
163
|
+
exactly the guarantee this locator exists to give, and the unchecked case
|
|
164
|
+
fails invisibly: a wrong-version OpenACA returns a structurally valid BOM.
|
|
165
|
+
The digest field is optional in a `RECORD` row, so an install can decline
|
|
166
|
+
to say what it wrote; that install cannot then be shown to own the file
|
|
167
|
+
sitting at the path it names, and a reinstall is what restores the
|
|
168
|
+
evidence.
|
|
169
|
+
|
|
170
|
+
Digests are base64url without padding, so the padding is stripped from
|
|
171
|
+
both sides rather than assumed absent from either.
|
|
172
|
+
"""
|
|
173
|
+
digest = recorded.hash
|
|
174
|
+
if digest is None:
|
|
175
|
+
return False
|
|
176
|
+
try:
|
|
177
|
+
computed = hashlib.new(digest.mode, candidate.read_bytes()).digest()
|
|
178
|
+
except (ValueError, TypeError, OSError):
|
|
179
|
+
return False
|
|
180
|
+
encoded = base64.urlsafe_b64encode(computed).decode("ascii")
|
|
181
|
+
return encoded.rstrip("=") == digest.value.rstrip("=")
|
|
182
|
+
|
|
39
183
|
|
|
40
184
|
def load_bom(path: Path) -> tuple[str, Built]:
|
|
41
185
|
"""Read a supplied Agent BOM, or say why it cannot be used.
|
|
@@ -95,7 +239,7 @@ def build_bom(
|
|
|
95
239
|
with tempfile.TemporaryDirectory() as scratch:
|
|
96
240
|
directory = Path(scratch)
|
|
97
241
|
argv = [
|
|
98
|
-
|
|
242
|
+
_openaca(),
|
|
99
243
|
"bom",
|
|
100
244
|
"endpoint",
|
|
101
245
|
"--kind",
|
|
@@ -291,8 +435,16 @@ def _scan(
|
|
|
291
435
|
array, and a family we do not read is a fact about the scan rather than a
|
|
292
436
|
fault in it.
|
|
293
437
|
"""
|
|
438
|
+
try:
|
|
439
|
+
openaca = _openaca()
|
|
440
|
+
except RuntimeError:
|
|
441
|
+
# The same answer as an unreadable response, for the same reason: with no
|
|
442
|
+
# OpenACA that can be known to be the pinned one, this scan did not run.
|
|
443
|
+
# A `--bom` run reaches here without having built anything, so this is
|
|
444
|
+
# the first place that can be discovered — and `{}` would say we looked.
|
|
445
|
+
return None
|
|
294
446
|
result = subprocess.run(
|
|
295
|
-
[
|
|
447
|
+
[openaca, "scan", "bom", "--input", str(bom_path), "--format", "json"],
|
|
296
448
|
capture_output=True,
|
|
297
449
|
text=True,
|
|
298
450
|
check=False,
|
|
@@ -64,7 +64,7 @@ _TIMEOUT = 180
|
|
|
64
64
|
#: run /login"* and exits 1, on a machine whose CLI is authenticated and working.
|
|
65
65
|
#: That is the same premise `--bare` is avoided for (ADR-0004: the developer's
|
|
66
66
|
#: own CLI, already authenticated), broken through the environment instead of
|
|
67
|
-
#: through a flag. The failure mode is the dangerous kind: every
|
|
67
|
+
#: through a flag. The failure mode is the dangerous kind: every reasoning request
|
|
68
68
|
#: returns `analyzer_error`, the pipeline records `Unknown`, and stage 3 looks
|
|
69
69
|
#: like it ran. `LOGNAME` accompanies it as the same identity on systems that
|
|
70
70
|
#: set that instead.
|
stacktrace_cli/detector/cache.py
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Authorized by ADR-0011, which amends ADR-0006's persistence clause. The argument
|
|
4
4
|
is narrow: a reasoning verdict is an **expensive pure function of an immutable,
|
|
5
5
|
ended session**. `stacktrace detect` is a command a person runs repeatedly, each
|
|
6
|
-
|
|
6
|
+
reasoning request costs roughly $0.09 and 9 seconds, and the answer cannot change once
|
|
7
7
|
the session has stopped growing.
|
|
8
8
|
|
|
9
9
|
Three properties carry the whole design, and each exists because its absence was
|
|
@@ -116,7 +116,7 @@ class CachedVerdict:
|
|
|
116
116
|
|
|
117
117
|
@dataclass(frozen=True)
|
|
118
118
|
class CachedOutcome:
|
|
119
|
-
"""One
|
|
119
|
+
"""One reasoning request's stored answer: provenance, plus whichever rules fired.
|
|
120
120
|
|
|
121
121
|
Provenance sits here rather than on each verdict because ADR-0011
|
|
122
122
|
constraint 1 lists it as part of what an entry *carries*, not only what it is
|
|
@@ -132,7 +132,7 @@ class CachedOutcome:
|
|
|
132
132
|
|
|
133
133
|
def cache_key(
|
|
134
134
|
session: SessionLike,
|
|
135
|
-
|
|
135
|
+
request: object,
|
|
136
136
|
rule_ids: Sequence[str],
|
|
137
137
|
prompt_version: str,
|
|
138
138
|
identity: tuple[str, str],
|
|
@@ -143,17 +143,17 @@ def cache_key(
|
|
|
143
143
|
ADR-0011 constraint 3 states them apart, and neither is a field `serialise()`
|
|
144
144
|
shows the analyzer — so a fingerprint built only from analyzer-visible
|
|
145
145
|
content would let two distinct sessions with byte-identical rendered turns
|
|
146
|
-
and
|
|
146
|
+
and reasoning-request context share one verdict.
|
|
147
147
|
|
|
148
148
|
**Reasons and spans are hashed in their given order, never sorted.**
|
|
149
149
|
`_why()` joins each collection in order and puts the result straight into
|
|
150
150
|
the prompt, so the same members in a different order are two different
|
|
151
151
|
prompts. Sorting first would merge them.
|
|
152
152
|
|
|
153
|
-
**Which rule cited which span, not only the merged set.** `
|
|
154
|
-
is a deduplicated union over every
|
|
153
|
+
**Which rule cited which span, not only the merged set.** `request.spans`
|
|
154
|
+
is a deduplicated union over every requesting reason, so it cannot say
|
|
155
155
|
whether a span was the injection marker's evidence or the credential rule's
|
|
156
|
-
— and `serialise_injection()` reads exactly that, from `
|
|
156
|
+
— and `serialise_injection()` reads exactly that, from `request.stage_one`,
|
|
157
157
|
to decide which results and arguments the transcript carries. Two builds
|
|
158
158
|
that attribute overlapping spans differently produce the same union and
|
|
159
159
|
wholly different transcripts, so without this a verdict computed from one
|
|
@@ -181,12 +181,12 @@ def cache_key(
|
|
|
181
181
|
{
|
|
182
182
|
"session_id": session.session_id,
|
|
183
183
|
"turn_count": session.turn_count,
|
|
184
|
-
"reasons": list(getattr(
|
|
185
|
-
"spans": list(getattr(
|
|
186
|
-
"stage_one": _cited_per_rule(
|
|
184
|
+
"reasons": list(getattr(request, "reasons", ()) or ()),
|
|
185
|
+
"spans": list(getattr(request, "spans", ()) or ()),
|
|
186
|
+
"stage_one": _cited_per_rule(request),
|
|
187
187
|
"coverage": [
|
|
188
|
-
getattr(getattr(
|
|
189
|
-
getattr(getattr(
|
|
188
|
+
getattr(getattr(request, "coverage", None), "total", None),
|
|
189
|
+
getattr(getattr(request, "coverage", None), "resolved", None),
|
|
190
190
|
],
|
|
191
191
|
"prompt_version": prompt_version,
|
|
192
192
|
"verdict_logic_version": _VERDICT_LOGIC_VERSION,
|
|
@@ -201,7 +201,7 @@ def cache_key(
|
|
|
201
201
|
return digest.hexdigest()
|
|
202
202
|
|
|
203
203
|
|
|
204
|
-
def _cited_per_rule(
|
|
204
|
+
def _cited_per_rule(request: object) -> list[list[object]]:
|
|
205
205
|
"""Which stage-one rule cited which spans, in the order they were found.
|
|
206
206
|
|
|
207
207
|
Descriptors only, and only the two a prompt reads: `serialise_injection()`
|
|
@@ -212,13 +212,13 @@ def _cited_per_rule(escalation: object) -> list[list[object]]:
|
|
|
212
212
|
|
|
213
213
|
Order is preserved rather than sorted. Nothing downstream reads it — the
|
|
214
214
|
spans become a set — so the worst a preserved order can do is separate two
|
|
215
|
-
|
|
215
|
+
requests that would have produced one prompt, which costs a miss. The
|
|
216
216
|
opposite mistake merges two prompts under one key, and that is the one
|
|
217
217
|
ADR-0011 constraint 5 forbids.
|
|
218
218
|
"""
|
|
219
219
|
return [
|
|
220
220
|
[finding.rule_id, [evidence.span for evidence in finding.evidence]]
|
|
221
|
-
for finding in getattr(
|
|
221
|
+
for finding in getattr(request, "stage_one", ()) or ()
|
|
222
222
|
]
|
|
223
223
|
|
|
224
224
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Every rule here is one a responder can verify by looking at the session. That is
|
|
4
4
|
the bar for this stage, and it is why the stage emits directly rather than
|
|
5
|
-
|
|
5
|
+
requesting reasoning: nothing a slower stage adds changes *this string looks like a
|
|
6
6
|
credential and it went to an outbound call*.
|
|
7
7
|
|
|
8
8
|
Two asymmetries carry most of the precision:
|
|
@@ -150,7 +150,7 @@ class Evidence:
|
|
|
150
150
|
class Verdict:
|
|
151
151
|
"""What produced a finding, so two machines' answers are comparable.
|
|
152
152
|
|
|
153
|
-
Recorded for every stage, not only
|
|
153
|
+
Recorded for every stage, not only requested ones: knowing a finding came
|
|
154
154
|
from `priors` rather than `reasoning` is what tells a reader whether a model
|
|
155
155
|
was involved at all.
|
|
156
156
|
"""
|
|
@@ -340,9 +340,9 @@ class Unknown:
|
|
|
340
340
|
reason: UnknownReason
|
|
341
341
|
spans: tuple[str, ...] = ()
|
|
342
342
|
detail: str = ""
|
|
343
|
-
#: For
|
|
343
|
+
#: For a reasoning request that qualified for stage 3 but never reached it, the
|
|
344
344
|
#: priors that would have been analysed. Kept so a budget-limited or
|
|
345
|
-
#: `--
|
|
345
|
+
#: run without `--reasoning` still shows *why* the session was of interest.
|
|
346
346
|
reasons: tuple[str, ...] = field(default_factory=tuple)
|
|
347
347
|
#: Reasoning-stage only: whether this unknown was raised **after** the
|
|
348
348
|
#: session was actually sent to the analyzer. `run_detector` counts a
|
|
@@ -27,7 +27,7 @@ from data to instruction.
|
|
|
27
27
|
|
|
28
28
|
Finding the characters is not finding an attack, and the difference is
|
|
29
29
|
measurable. On 621 real sessions this rule fired three times — the only
|
|
30
|
-
evidence-backed
|
|
30
|
+
evidence-backed reasoning requests in the whole corpus — and all three were one file:
|
|
31
31
|
another agent-security tool's own detector, whose source reads
|
|
32
32
|
`_BIDI_OVERRIDE_CHARS = frozenset('\u202d\u202e')`. Security work names the
|
|
33
33
|
alphabet of the attacks it looks for, and so do specifications, test fixtures
|