stacktrace-cli 0.2.3__py3-none-any.whl → 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/analysis.py +9 -9
- stacktrace_cli/cli.py +22 -21
- stacktrace_cli/detector/analyzer.py +1 -1
- stacktrace_cli/detector/cache.py +15 -15
- stacktrace_cli/detector/deterministic.py +1 -1
- stacktrace_cli/detector/finding.py +3 -3
- stacktrace_cli/detector/markers.py +1 -1
- stacktrace_cli/detector/priors.py +21 -21
- stacktrace_cli/detector/reasoning.py +74 -63
- stacktrace_cli/detector/render.py +14 -14
- stacktrace_cli/detector/rules.py +3 -3
- stacktrace_cli/detector/run.py +65 -65
- stacktrace_cli/monitor/{escalate.py → reasoning.py} +13 -13
- stacktrace_cli/monitor/render.py +2 -2
- stacktrace_cli/monitor/server.py +37 -35
- stacktrace_cli/monitor/site/app.js +22 -22
- stacktrace_cli/monitor/site/styles.css +4 -4
- stacktrace_cli/monitor/state.py +3 -3
- stacktrace_cli/monitor/verdicts.py +3 -3
- stacktrace_cli/monitor/watch.py +7 -7
- stacktrace_cli/remote/cli.py +8 -8
- stacktrace_cli/remote/sync_detect.py +8 -8
- {stacktrace_cli-0.2.3.dist-info → stacktrace_cli-0.3.1.dist-info}/METADATA +5 -5
- {stacktrace_cli-0.2.3.dist-info → stacktrace_cli-0.3.1.dist-info}/RECORD +27 -27
- {stacktrace_cli-0.2.3.dist-info → stacktrace_cli-0.3.1.dist-info}/WHEEL +0 -0
- {stacktrace_cli-0.2.3.dist-info → stacktrace_cli-0.3.1.dist-info}/entry_points.txt +0 -0
stacktrace_cli/__init__.py
CHANGED
stacktrace_cli/analysis.py
CHANGED
|
@@ -106,7 +106,7 @@ def analyse(
|
|
|
106
106
|
project_map: tuple[str, ...] = (),
|
|
107
107
|
root: Path | None = None,
|
|
108
108
|
session_ids: tuple[str, ...] = (),
|
|
109
|
-
|
|
109
|
+
reasoning: bool = False,
|
|
110
110
|
budget: int = DEFAULT_BUDGET,
|
|
111
111
|
sample_budget: int = DEFAULT_SAMPLE_BUDGET,
|
|
112
112
|
cache: VerdictCache | None = None,
|
|
@@ -123,7 +123,7 @@ def analyse(
|
|
|
123
123
|
pass; a one-shot command passes nothing and gets the default.
|
|
124
124
|
|
|
125
125
|
`session_ids` narrows the window to named sessions. Empty means the whole
|
|
126
|
-
window, which is what every command passes; monitor's
|
|
126
|
+
window, which is what every command passes; monitor's reasoning button is
|
|
127
127
|
what needs the narrowing, and needs it to be structural.
|
|
128
128
|
|
|
129
129
|
`since` is the window's spelling — `parse_since` reads it — or the cutoff
|
|
@@ -142,7 +142,7 @@ def analyse(
|
|
|
142
142
|
)
|
|
143
143
|
run = run_detector(
|
|
144
144
|
acquired.view,
|
|
145
|
-
|
|
145
|
+
reasoning=reasoning,
|
|
146
146
|
budget=budget,
|
|
147
147
|
sample_budget=sample_budget,
|
|
148
148
|
cache=cache,
|
|
@@ -287,11 +287,11 @@ def analyse_progressively(
|
|
|
287
287
|
exists to prevent.
|
|
288
288
|
|
|
289
289
|
`cache` is read and never written. A session graded by an earlier
|
|
290
|
-
`detect --
|
|
290
|
+
`detect --reasoning` shows that grade here; this path never commissions one.
|
|
291
291
|
|
|
292
|
-
**No `
|
|
292
|
+
**No `reasoning` parameter, deliberately.** Stage 3's budget is per *run*, so
|
|
293
293
|
judging in batches would give each batch its own budget and spend a multiple
|
|
294
|
-
of what was authorised. A streaming
|
|
294
|
+
of what was authorised. A streaming reasoning run needs a budget shared across
|
|
295
295
|
batches; until it has one, this path does not offer the option rather than
|
|
296
296
|
offering it wrongly.
|
|
297
297
|
"""
|
|
@@ -358,8 +358,8 @@ def analyse_progressively(
|
|
|
358
358
|
for index in range(0, len(ordered), step):
|
|
359
359
|
judged = run_detector(
|
|
360
360
|
CorrelatedView(sessions=tuple(ordered[index : index + step])),
|
|
361
|
-
# Read, never written, and never
|
|
362
|
-
# `detect --
|
|
361
|
+
# Read, never written, and never requesting: a session an earlier
|
|
362
|
+
# `detect --reasoning` graded renders with that grade here, and this
|
|
363
363
|
# path commissions nothing (ADR-0026 clause 2).
|
|
364
364
|
cache=cache,
|
|
365
365
|
)
|
|
@@ -367,7 +367,7 @@ def analyse_progressively(
|
|
|
367
367
|
detections=accumulated.detections + judged.detections,
|
|
368
368
|
unknowns=accumulated.unknowns + judged.unknowns,
|
|
369
369
|
sessions=accumulated.sessions + judged.sessions,
|
|
370
|
-
|
|
370
|
+
requested=accumulated.requested + judged.requested,
|
|
371
371
|
analysed=accumulated.analysed + judged.analysed,
|
|
372
372
|
cache_hits=accumulated.cache_hits + judged.cache_hits,
|
|
373
373
|
collection_failures=failures,
|
stacktrace_cli/cli.py
CHANGED
|
@@ -223,13 +223,14 @@ def sessions(
|
|
|
223
223
|
help="Show every detection and all of its evidence. Default: the summary alone.",
|
|
224
224
|
)
|
|
225
225
|
@click.option(
|
|
226
|
-
"--
|
|
226
|
+
"--reasoning",
|
|
227
|
+
is_flag=True,
|
|
227
228
|
default=False,
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
"
|
|
231
|
-
"
|
|
232
|
-
"
|
|
229
|
+
help="Analyse flagged sessions with a reasoning model, run through the "
|
|
230
|
+
"agent's own CLI. Off by default: it is the only stage that sends anything "
|
|
231
|
+
"session-derived off this machine, and it spends the developer's own "
|
|
232
|
+
"provider quota, so a run that does neither is the one to reach for first. "
|
|
233
|
+
"--reasoning turns it on "
|
|
233
234
|
"and --budget caps a noisy day. Leaving it off does not make the run "
|
|
234
235
|
"offline: correlation runs first either way and matches advisories by "
|
|
235
236
|
"sending the package coordinates of the components a session invoked to "
|
|
@@ -281,7 +282,7 @@ def detect(
|
|
|
281
282
|
bom_paths: tuple[Path, ...],
|
|
282
283
|
output_format: str,
|
|
283
284
|
detail: bool,
|
|
284
|
-
|
|
285
|
+
reasoning: bool,
|
|
285
286
|
budget: int,
|
|
286
287
|
sample_budget: int,
|
|
287
288
|
cache: bool,
|
|
@@ -299,7 +300,7 @@ def detect(
|
|
|
299
300
|
|
|
300
301
|
Two stages need no model or credential, and they are the two that run by
|
|
301
302
|
default. The third sends flagged sessions to the agent's own CLI:
|
|
302
|
-
--
|
|
303
|
+
--reasoning turns it on, capped by --budget. Sessions that qualified for it
|
|
303
304
|
are counted either way, and named in --format json, so a run without it
|
|
304
305
|
says so rather than reading as a clean one.
|
|
305
306
|
|
|
@@ -307,10 +308,10 @@ def detect(
|
|
|
307
308
|
the agent's own CLI hands the prompts, arguments and results a rule needs
|
|
308
309
|
to the provider it is already authenticated against (ADR-0004 -- provider
|
|
309
310
|
affinity, so a transcript goes back to the vendor that produced it, and
|
|
310
|
-
there is no fallback to any other). Without --
|
|
311
|
+
there is no fallback to any other). Without --reasoning the run is the two
|
|
311
312
|
stages that send nothing.
|
|
312
313
|
|
|
313
|
-
One network call happens before any of them, with or without --
|
|
314
|
+
One network call happens before any of them, with or without --reasoning:
|
|
314
315
|
correlation matches advisories by sending the package coordinates of the
|
|
315
316
|
components a session invoked to osv.dev. Coordinates only -- never a
|
|
316
317
|
prompt, an argument or a result.
|
|
@@ -325,11 +326,11 @@ def detect(
|
|
|
325
326
|
bom_paths=bom_paths,
|
|
326
327
|
project_map=project_map,
|
|
327
328
|
root=root,
|
|
328
|
-
|
|
329
|
+
reasoning=reasoning,
|
|
329
330
|
budget=budget,
|
|
330
331
|
sample_budget=sample_budget,
|
|
331
|
-
# Not gated on `
|
|
332
|
-
# answer this run has, and `
|
|
332
|
+
# Not gated on `reasoning`. A verdict already paid for is an
|
|
333
|
+
# answer this run has, and `reasoning` governs whether new ones may
|
|
333
334
|
# be commissioned, not whether old ones may be read -- the rule
|
|
334
335
|
# `run_detector` states at its cache pass. Gating it here made a
|
|
335
336
|
# session read as graded or ungraded depending on a flag that
|
|
@@ -368,11 +369,11 @@ def detect(
|
|
|
368
369
|
"entirely when nothing on disk has changed.",
|
|
369
370
|
)
|
|
370
371
|
@click.option(
|
|
371
|
-
"--
|
|
372
|
+
"--reasoning",
|
|
373
|
+
is_flag=True,
|
|
372
374
|
default=False,
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
"analysis. Off by default: stage 3 spends your provider quota, and a page "
|
|
375
|
+
help="Let the page analyse one session with a reasoning model, run through "
|
|
376
|
+
"the agent's own CLI. Off by default: stage 3 spends your provider quota, and a page "
|
|
376
377
|
"left open is the wrong place for that to be implicit. Capped by --budget.",
|
|
377
378
|
)
|
|
378
379
|
@click.option(
|
|
@@ -380,7 +381,7 @@ def detect(
|
|
|
380
381
|
type=click.IntRange(min=1),
|
|
381
382
|
default=DEFAULT_BUDGET,
|
|
382
383
|
show_default=True,
|
|
383
|
-
help="
|
|
384
|
+
help="Reasoning runs this monitor may spend in total, if --reasoning is on.",
|
|
384
385
|
)
|
|
385
386
|
@click.option(
|
|
386
387
|
"--no-open", is_flag=True, default=False, help="Print the URL, do not open a browser."
|
|
@@ -421,7 +422,7 @@ def monitor(
|
|
|
421
422
|
port: int,
|
|
422
423
|
host: str,
|
|
423
424
|
interval: float,
|
|
424
|
-
|
|
425
|
+
reasoning: bool,
|
|
425
426
|
budget: int,
|
|
426
427
|
no_open: bool,
|
|
427
428
|
agent_kinds: tuple[str, ...],
|
|
@@ -439,7 +440,7 @@ def monitor(
|
|
|
439
440
|
|
|
440
441
|
Loopback only, and free to leave open: a pass is skipped when nothing has
|
|
441
442
|
changed, advisory lookups are asked once per component, and the two stages
|
|
442
|
-
that need no model are the only ones that run. --
|
|
443
|
+
that need no model are the only ones that run. --reasoning adds a button
|
|
443
444
|
that spends your provider quota, off unless asked for.
|
|
444
445
|
"""
|
|
445
446
|
try:
|
|
@@ -447,7 +448,7 @@ def monitor(
|
|
|
447
448
|
host=host,
|
|
448
449
|
port=port,
|
|
449
450
|
interval=interval,
|
|
450
|
-
|
|
451
|
+
reasoning=reasoning,
|
|
451
452
|
budget=budget,
|
|
452
453
|
open_browser=not no_open,
|
|
453
454
|
echo=click.echo,
|
|
@@ -64,7 +64,7 @@ _TIMEOUT = 180
|
|
|
64
64
|
#: run /login"* and exits 1, on a machine whose CLI is authenticated and working.
|
|
65
65
|
#: That is the same premise `--bare` is avoided for (ADR-0004: the developer's
|
|
66
66
|
#: own CLI, already authenticated), broken through the environment instead of
|
|
67
|
-
#: through a flag. The failure mode is the dangerous kind: every
|
|
67
|
+
#: through a flag. The failure mode is the dangerous kind: every reasoning request
|
|
68
68
|
#: returns `analyzer_error`, the pipeline records `Unknown`, and stage 3 looks
|
|
69
69
|
#: like it ran. `LOGNAME` accompanies it as the same identity on systems that
|
|
70
70
|
#: set that instead.
|
stacktrace_cli/detector/cache.py
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Authorized by ADR-0011, which amends ADR-0006's persistence clause. The argument
|
|
4
4
|
is narrow: a reasoning verdict is an **expensive pure function of an immutable,
|
|
5
5
|
ended session**. `stacktrace detect` is a command a person runs repeatedly, each
|
|
6
|
-
|
|
6
|
+
reasoning request costs roughly $0.09 and 9 seconds, and the answer cannot change once
|
|
7
7
|
the session has stopped growing.
|
|
8
8
|
|
|
9
9
|
Three properties carry the whole design, and each exists because its absence was
|
|
@@ -116,7 +116,7 @@ class CachedVerdict:
|
|
|
116
116
|
|
|
117
117
|
@dataclass(frozen=True)
|
|
118
118
|
class CachedOutcome:
|
|
119
|
-
"""One
|
|
119
|
+
"""One reasoning request's stored answer: provenance, plus whichever rules fired.
|
|
120
120
|
|
|
121
121
|
Provenance sits here rather than on each verdict because ADR-0011
|
|
122
122
|
constraint 1 lists it as part of what an entry *carries*, not only what it is
|
|
@@ -132,7 +132,7 @@ class CachedOutcome:
|
|
|
132
132
|
|
|
133
133
|
def cache_key(
|
|
134
134
|
session: SessionLike,
|
|
135
|
-
|
|
135
|
+
request: object,
|
|
136
136
|
rule_ids: Sequence[str],
|
|
137
137
|
prompt_version: str,
|
|
138
138
|
identity: tuple[str, str],
|
|
@@ -143,17 +143,17 @@ def cache_key(
|
|
|
143
143
|
ADR-0011 constraint 3 states them apart, and neither is a field `serialise()`
|
|
144
144
|
shows the analyzer — so a fingerprint built only from analyzer-visible
|
|
145
145
|
content would let two distinct sessions with byte-identical rendered turns
|
|
146
|
-
and
|
|
146
|
+
and reasoning-request context share one verdict.
|
|
147
147
|
|
|
148
148
|
**Reasons and spans are hashed in their given order, never sorted.**
|
|
149
149
|
`_why()` joins each collection in order and puts the result straight into
|
|
150
150
|
the prompt, so the same members in a different order are two different
|
|
151
151
|
prompts. Sorting first would merge them.
|
|
152
152
|
|
|
153
|
-
**Which rule cited which span, not only the merged set.** `
|
|
154
|
-
is a deduplicated union over every
|
|
153
|
+
**Which rule cited which span, not only the merged set.** `request.spans`
|
|
154
|
+
is a deduplicated union over every requesting reason, so it cannot say
|
|
155
155
|
whether a span was the injection marker's evidence or the credential rule's
|
|
156
|
-
— and `serialise_injection()` reads exactly that, from `
|
|
156
|
+
— and `serialise_injection()` reads exactly that, from `request.stage_one`,
|
|
157
157
|
to decide which results and arguments the transcript carries. Two builds
|
|
158
158
|
that attribute overlapping spans differently produce the same union and
|
|
159
159
|
wholly different transcripts, so without this a verdict computed from one
|
|
@@ -181,12 +181,12 @@ def cache_key(
|
|
|
181
181
|
{
|
|
182
182
|
"session_id": session.session_id,
|
|
183
183
|
"turn_count": session.turn_count,
|
|
184
|
-
"reasons": list(getattr(
|
|
185
|
-
"spans": list(getattr(
|
|
186
|
-
"stage_one": _cited_per_rule(
|
|
184
|
+
"reasons": list(getattr(request, "reasons", ()) or ()),
|
|
185
|
+
"spans": list(getattr(request, "spans", ()) or ()),
|
|
186
|
+
"stage_one": _cited_per_rule(request),
|
|
187
187
|
"coverage": [
|
|
188
|
-
getattr(getattr(
|
|
189
|
-
getattr(getattr(
|
|
188
|
+
getattr(getattr(request, "coverage", None), "total", None),
|
|
189
|
+
getattr(getattr(request, "coverage", None), "resolved", None),
|
|
190
190
|
],
|
|
191
191
|
"prompt_version": prompt_version,
|
|
192
192
|
"verdict_logic_version": _VERDICT_LOGIC_VERSION,
|
|
@@ -201,7 +201,7 @@ def cache_key(
|
|
|
201
201
|
return digest.hexdigest()
|
|
202
202
|
|
|
203
203
|
|
|
204
|
-
def _cited_per_rule(
|
|
204
|
+
def _cited_per_rule(request: object) -> list[list[object]]:
|
|
205
205
|
"""Which stage-one rule cited which spans, in the order they were found.
|
|
206
206
|
|
|
207
207
|
Descriptors only, and only the two a prompt reads: `serialise_injection()`
|
|
@@ -212,13 +212,13 @@ def _cited_per_rule(escalation: object) -> list[list[object]]:
|
|
|
212
212
|
|
|
213
213
|
Order is preserved rather than sorted. Nothing downstream reads it — the
|
|
214
214
|
spans become a set — so the worst a preserved order can do is separate two
|
|
215
|
-
|
|
215
|
+
requests that would have produced one prompt, which costs a miss. The
|
|
216
216
|
opposite mistake merges two prompts under one key, and that is the one
|
|
217
217
|
ADR-0011 constraint 5 forbids.
|
|
218
218
|
"""
|
|
219
219
|
return [
|
|
220
220
|
[finding.rule_id, [evidence.span for evidence in finding.evidence]]
|
|
221
|
-
for finding in getattr(
|
|
221
|
+
for finding in getattr(request, "stage_one", ()) or ()
|
|
222
222
|
]
|
|
223
223
|
|
|
224
224
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Every rule here is one a responder can verify by looking at the session. That is
|
|
4
4
|
the bar for this stage, and it is why the stage emits directly rather than
|
|
5
|
-
|
|
5
|
+
requesting reasoning: nothing a slower stage adds changes *this string looks like a
|
|
6
6
|
credential and it went to an outbound call*.
|
|
7
7
|
|
|
8
8
|
Two asymmetries carry most of the precision:
|
|
@@ -150,7 +150,7 @@ class Evidence:
|
|
|
150
150
|
class Verdict:
|
|
151
151
|
"""What produced a finding, so two machines' answers are comparable.
|
|
152
152
|
|
|
153
|
-
Recorded for every stage, not only
|
|
153
|
+
Recorded for every stage, not only requested ones: knowing a finding came
|
|
154
154
|
from `priors` rather than `reasoning` is what tells a reader whether a model
|
|
155
155
|
was involved at all.
|
|
156
156
|
"""
|
|
@@ -340,9 +340,9 @@ class Unknown:
|
|
|
340
340
|
reason: UnknownReason
|
|
341
341
|
spans: tuple[str, ...] = ()
|
|
342
342
|
detail: str = ""
|
|
343
|
-
#: For
|
|
343
|
+
#: For a reasoning request that qualified for stage 3 but never reached it, the
|
|
344
344
|
#: priors that would have been analysed. Kept so a budget-limited or
|
|
345
|
-
#: `--
|
|
345
|
+
#: run without `--reasoning` still shows *why* the session was of interest.
|
|
346
346
|
reasons: tuple[str, ...] = field(default_factory=tuple)
|
|
347
347
|
#: Reasoning-stage only: whether this unknown was raised **after** the
|
|
348
348
|
#: session was actually sent to the analyzer. `run_detector` counts a
|
|
@@ -27,7 +27,7 @@ from data to instruction.
|
|
|
27
27
|
|
|
28
28
|
Finding the characters is not finding an attack, and the difference is
|
|
29
29
|
measurable. On 621 real sessions this rule fired three times — the only
|
|
30
|
-
evidence-backed
|
|
30
|
+
evidence-backed reasoning requests in the whole corpus — and all three were one file:
|
|
31
31
|
another agent-security tool's own detector, whose source reads
|
|
32
32
|
`_BIDI_OVERRIDE_CHARS = frozenset('\u202d\u202e')`. Security work names the
|
|
33
33
|
alphabet of the attacks it looks for, and so do specifications, test fixtures
|
|
@@ -14,14 +14,14 @@ bounds it:
|
|
|
14
14
|
- **An unreadable connection log is the same kind of bound.** Egress over MCP is
|
|
15
15
|
observed from `transport` alone, so a log that never applied is silent in
|
|
16
16
|
exactly the way a local call is — `_missing_mcp_transport` says which it was.
|
|
17
|
-
- **Nothing here
|
|
17
|
+
- **Nothing here requests reasoning.** The routes into stage three are stage-one facts;
|
|
18
18
|
an advisory is already settled by the record, and spending inference to
|
|
19
19
|
confirm it would buy nothing.
|
|
20
20
|
|
|
21
21
|
**Four rules have been removed from this stage over two decisions.** ADR-0013
|
|
22
22
|
withdrew `tool-shadowing`, which needs an `ambiguous` outcome the join has never
|
|
23
|
-
emitted, and `unpinned-invoked`, which could only
|
|
24
|
-
it; the combination
|
|
23
|
+
emitted, and `unpinned-invoked`, which could only request reasoning in combination
|
|
24
|
+
with it; the combination route went with them. ADR-0015 withdrew
|
|
25
25
|
`unsanctioned-mcp-tool-use` — and with it the `uninventoried_component` unknown
|
|
26
26
|
that reported the same fact for every other component type — and
|
|
27
27
|
`capability-crossing`, which ADR-0013 had deliberately kept on the argument that
|
|
@@ -58,7 +58,7 @@ _VERDICT = Verdict(stage="priors")
|
|
|
58
58
|
#: not read enough of it either to clear or to fault it.
|
|
59
59
|
_COVERAGE_FLOOR = 0.5
|
|
60
60
|
|
|
61
|
-
#: Stage-one rules that
|
|
61
|
+
#: Stage-one rules that request reasoning on their own. Each names a question only
|
|
62
62
|
#: `reasoning` can settle, and each would otherwise be unreachable:
|
|
63
63
|
#:
|
|
64
64
|
#: - a marker asks whether an injected instruction was *acted on*, which nothing
|
|
@@ -84,7 +84,7 @@ _SAMPLE_BUCKETS = 7
|
|
|
84
84
|
|
|
85
85
|
#: The one reason that names a session chosen without evidence — no prior fired,
|
|
86
86
|
#: so there is nothing a suppression verdict could withdraw. Named here, next to
|
|
87
|
-
#: `
|
|
87
|
+
#: `ReasoningRequest`, because `reasoning` needs it to recognise the case and `run`
|
|
88
88
|
#: needs it to build one.
|
|
89
89
|
SAMPLED = "sampled"
|
|
90
90
|
|
|
@@ -116,7 +116,7 @@ def in_sample_bucket(session_id: str, day_of_run: int) -> bool:
|
|
|
116
116
|
|
|
117
117
|
|
|
118
118
|
@dataclass(frozen=True)
|
|
119
|
-
class
|
|
119
|
+
class ReasoningRequest:
|
|
120
120
|
"""A session worth spending inference on, and what to look at.
|
|
121
121
|
|
|
122
122
|
Carries no budget: what remains of a run's allowance belongs to the
|
|
@@ -124,7 +124,7 @@ class Escalation:
|
|
|
124
124
|
"""
|
|
125
125
|
|
|
126
126
|
session: SessionRef
|
|
127
|
-
#: Which priors fired, by name, or the stage-one rule that
|
|
127
|
+
#: Which priors fired, by name, or the stage-one rule that requested alone.
|
|
128
128
|
reasons: tuple[str, ...]
|
|
129
129
|
#: Where a semantic question is worth asking.
|
|
130
130
|
spans: tuple[str, ...]
|
|
@@ -137,12 +137,12 @@ class Escalation:
|
|
|
137
137
|
|
|
138
138
|
def run_priors(
|
|
139
139
|
correlated: CorrelatedSession, stage_one: Sequence[Detection]
|
|
140
|
-
) -> tuple[tuple[Detection, ...], tuple[Unknown, ...],
|
|
140
|
+
) -> tuple[tuple[Detection, ...], tuple[Unknown, ...], ReasoningRequest | None]:
|
|
141
141
|
"""Score one correlated session against its composition.
|
|
142
142
|
|
|
143
143
|
Three values, not four. A `prior_recorded` bit used to travel alongside,
|
|
144
|
-
saying
|
|
145
|
-
threshold — a state that no longer exists, because neither the
|
|
144
|
+
saying a requesting prior had fired without crossing the combination
|
|
145
|
+
threshold — a state that no longer exists, because neither the requesting
|
|
146
146
|
priors nor the threshold do. A field that is now constantly `False` would
|
|
147
147
|
be a worse answer than no field.
|
|
148
148
|
"""
|
|
@@ -169,7 +169,7 @@ def run_priors(
|
|
|
169
169
|
),
|
|
170
170
|
),
|
|
171
171
|
),
|
|
172
|
-
|
|
172
|
+
_request(_solo_reasons(stage_one), ref, stage_one, coverage),
|
|
173
173
|
)
|
|
174
174
|
|
|
175
175
|
composition = correlated.composition
|
|
@@ -197,28 +197,28 @@ def run_priors(
|
|
|
197
197
|
else:
|
|
198
198
|
detections.extend(_advisory_reach(ordered, calls, composition, ref, coverage))
|
|
199
199
|
|
|
200
|
-
# One route. A stage-one fact this stage cannot settle
|
|
201
|
-
# nothing here combines, because nothing here
|
|
202
|
-
|
|
203
|
-
return tuple(detections), tuple(unknowns),
|
|
200
|
+
# One route. A stage-one fact this stage cannot settle requests reasoning on
|
|
201
|
+
# its own; nothing here combines, because nothing here requests it any more.
|
|
202
|
+
request = _request(_solo_reasons(stage_one), ref, stage_one, coverage, spans)
|
|
203
|
+
return tuple(detections), tuple(unknowns), request
|
|
204
204
|
|
|
205
205
|
|
|
206
206
|
def _solo_reasons(stage_one: Sequence[Detection]) -> list[str]:
|
|
207
|
-
"""The stage-one rules on this session that
|
|
207
|
+
"""The stage-one rules on this session that request reasoning on their own."""
|
|
208
208
|
return sorted({finding.rule_id for finding in stage_one} & _SOLO_ROUTES)
|
|
209
209
|
|
|
210
210
|
|
|
211
|
-
def
|
|
211
|
+
def _request(
|
|
212
212
|
reasons: Sequence[str],
|
|
213
213
|
ref: SessionRef,
|
|
214
214
|
stage_one: Sequence[Detection],
|
|
215
215
|
coverage: Coverage,
|
|
216
216
|
spans: Sequence[str] = (),
|
|
217
|
-
) ->
|
|
218
|
-
"""One
|
|
217
|
+
) -> ReasoningRequest | None:
|
|
218
|
+
"""One reasoning request naming every reason that fired, or nothing.
|
|
219
219
|
|
|
220
220
|
Spans from the priors that emitted are joined by those of every stage-one
|
|
221
|
-
finding whose rule
|
|
221
|
+
finding whose rule requested it, so the analyzer is pointed at both.
|
|
222
222
|
"""
|
|
223
223
|
if not reasons:
|
|
224
224
|
return None
|
|
@@ -230,7 +230,7 @@ def _escalation(
|
|
|
230
230
|
if finding.rule_id == reason
|
|
231
231
|
for evidence in finding.evidence
|
|
232
232
|
)
|
|
233
|
-
return
|
|
233
|
+
return ReasoningRequest(
|
|
234
234
|
session=ref,
|
|
235
235
|
reasons=tuple(dict.fromkeys(reasons)),
|
|
236
236
|
spans=tuple(dict.fromkeys(cited)),
|