stacktrace-cli 0.2.2__py3-none-any.whl → 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/analysis.py +9 -9
- stacktrace_cli/cli.py +37 -28
- stacktrace_cli/correlate/acquire.py +154 -2
- stacktrace_cli/detector/analyzer.py +1 -1
- stacktrace_cli/detector/cache.py +15 -15
- stacktrace_cli/detector/deterministic.py +1 -1
- stacktrace_cli/detector/finding.py +3 -3
- stacktrace_cli/detector/markers.py +1 -1
- stacktrace_cli/detector/priors.py +21 -21
- stacktrace_cli/detector/reasoning.py +74 -63
- stacktrace_cli/detector/render.py +22 -25
- stacktrace_cli/detector/rules.py +3 -3
- stacktrace_cli/detector/run.py +65 -65
- stacktrace_cli/monitor/{escalate.py → reasoning.py} +13 -13
- stacktrace_cli/monitor/render.py +2 -2
- stacktrace_cli/monitor/server.py +37 -35
- stacktrace_cli/monitor/site/app.js +22 -22
- stacktrace_cli/monitor/site/index.html +1 -5
- stacktrace_cli/monitor/site/styles.css +4 -9
- stacktrace_cli/monitor/state.py +3 -3
- stacktrace_cli/monitor/verdicts.py +3 -3
- stacktrace_cli/monitor/watch.py +7 -7
- stacktrace_cli/remote/cli.py +8 -9
- stacktrace_cli/remote/sync_detect.py +13 -7
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/METADATA +7 -6
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/RECORD +29 -29
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/WHEEL +0 -0
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/entry_points.txt +0 -0
|
@@ -14,14 +14,14 @@ bounds it:
|
|
|
14
14
|
- **An unreadable connection log is the same kind of bound.** Egress over MCP is
|
|
15
15
|
observed from `transport` alone, so a log that never applied is silent in
|
|
16
16
|
exactly the way a local call is — `_missing_mcp_transport` says which it was.
|
|
17
|
-
- **Nothing here
|
|
17
|
+
- **Nothing here requests reasoning.** The routes into stage three are stage-one facts;
|
|
18
18
|
an advisory is already settled by the record, and spending inference to
|
|
19
19
|
confirm it would buy nothing.
|
|
20
20
|
|
|
21
21
|
**Four rules have been removed from this stage over two decisions.** ADR-0013
|
|
22
22
|
withdrew `tool-shadowing`, which needs an `ambiguous` outcome the join has never
|
|
23
|
-
emitted, and `unpinned-invoked`, which could only
|
|
24
|
-
it; the combination
|
|
23
|
+
emitted, and `unpinned-invoked`, which could only request reasoning in combination
|
|
24
|
+
with it; the combination route went with them. ADR-0015 withdrew
|
|
25
25
|
`unsanctioned-mcp-tool-use` — and with it the `uninventoried_component` unknown
|
|
26
26
|
that reported the same fact for every other component type — and
|
|
27
27
|
`capability-crossing`, which ADR-0013 had deliberately kept on the argument that
|
|
@@ -58,7 +58,7 @@ _VERDICT = Verdict(stage="priors")
|
|
|
58
58
|
#: not read enough of it either to clear or to fault it.
|
|
59
59
|
_COVERAGE_FLOOR = 0.5
|
|
60
60
|
|
|
61
|
-
#: Stage-one rules that
|
|
61
|
+
#: Stage-one rules that request reasoning on their own. Each names a question only
|
|
62
62
|
#: `reasoning` can settle, and each would otherwise be unreachable:
|
|
63
63
|
#:
|
|
64
64
|
#: - a marker asks whether an injected instruction was *acted on*, which nothing
|
|
@@ -84,7 +84,7 @@ _SAMPLE_BUCKETS = 7
|
|
|
84
84
|
|
|
85
85
|
#: The one reason that names a session chosen without evidence — no prior fired,
|
|
86
86
|
#: so there is nothing a suppression verdict could withdraw. Named here, next to
|
|
87
|
-
#: `
|
|
87
|
+
#: `ReasoningRequest`, because `reasoning` needs it to recognise the case and `run`
|
|
88
88
|
#: needs it to build one.
|
|
89
89
|
SAMPLED = "sampled"
|
|
90
90
|
|
|
@@ -116,7 +116,7 @@ def in_sample_bucket(session_id: str, day_of_run: int) -> bool:
|
|
|
116
116
|
|
|
117
117
|
|
|
118
118
|
@dataclass(frozen=True)
|
|
119
|
-
class
|
|
119
|
+
class ReasoningRequest:
|
|
120
120
|
"""A session worth spending inference on, and what to look at.
|
|
121
121
|
|
|
122
122
|
Carries no budget: what remains of a run's allowance belongs to the
|
|
@@ -124,7 +124,7 @@ class Escalation:
|
|
|
124
124
|
"""
|
|
125
125
|
|
|
126
126
|
session: SessionRef
|
|
127
|
-
#: Which priors fired, by name, or the stage-one rule that
|
|
127
|
+
#: Which priors fired, by name, or the stage-one rule that requested alone.
|
|
128
128
|
reasons: tuple[str, ...]
|
|
129
129
|
#: Where a semantic question is worth asking.
|
|
130
130
|
spans: tuple[str, ...]
|
|
@@ -137,12 +137,12 @@ class Escalation:
|
|
|
137
137
|
|
|
138
138
|
def run_priors(
|
|
139
139
|
correlated: CorrelatedSession, stage_one: Sequence[Detection]
|
|
140
|
-
) -> tuple[tuple[Detection, ...], tuple[Unknown, ...],
|
|
140
|
+
) -> tuple[tuple[Detection, ...], tuple[Unknown, ...], ReasoningRequest | None]:
|
|
141
141
|
"""Score one correlated session against its composition.
|
|
142
142
|
|
|
143
143
|
Three values, not four. A `prior_recorded` bit used to travel alongside,
|
|
144
|
-
saying
|
|
145
|
-
threshold — a state that no longer exists, because neither the
|
|
144
|
+
saying a requesting prior had fired without crossing the combination
|
|
145
|
+
threshold — a state that no longer exists, because neither the requesting
|
|
146
146
|
priors nor the threshold do. A field that is now constantly `False` would
|
|
147
147
|
be a worse answer than no field.
|
|
148
148
|
"""
|
|
@@ -169,7 +169,7 @@ def run_priors(
|
|
|
169
169
|
),
|
|
170
170
|
),
|
|
171
171
|
),
|
|
172
|
-
|
|
172
|
+
_request(_solo_reasons(stage_one), ref, stage_one, coverage),
|
|
173
173
|
)
|
|
174
174
|
|
|
175
175
|
composition = correlated.composition
|
|
@@ -197,28 +197,28 @@ def run_priors(
|
|
|
197
197
|
else:
|
|
198
198
|
detections.extend(_advisory_reach(ordered, calls, composition, ref, coverage))
|
|
199
199
|
|
|
200
|
-
# One route. A stage-one fact this stage cannot settle
|
|
201
|
-
# nothing here combines, because nothing here
|
|
202
|
-
|
|
203
|
-
return tuple(detections), tuple(unknowns),
|
|
200
|
+
# One route. A stage-one fact this stage cannot settle requests reasoning on
|
|
201
|
+
# its own; nothing here combines, because nothing here requests it any more.
|
|
202
|
+
request = _request(_solo_reasons(stage_one), ref, stage_one, coverage, spans)
|
|
203
|
+
return tuple(detections), tuple(unknowns), request
|
|
204
204
|
|
|
205
205
|
|
|
206
206
|
def _solo_reasons(stage_one: Sequence[Detection]) -> list[str]:
|
|
207
|
-
"""The stage-one rules on this session that
|
|
207
|
+
"""The stage-one rules on this session that request reasoning on their own."""
|
|
208
208
|
return sorted({finding.rule_id for finding in stage_one} & _SOLO_ROUTES)
|
|
209
209
|
|
|
210
210
|
|
|
211
|
-
def
|
|
211
|
+
def _request(
|
|
212
212
|
reasons: Sequence[str],
|
|
213
213
|
ref: SessionRef,
|
|
214
214
|
stage_one: Sequence[Detection],
|
|
215
215
|
coverage: Coverage,
|
|
216
216
|
spans: Sequence[str] = (),
|
|
217
|
-
) ->
|
|
218
|
-
"""One
|
|
217
|
+
) -> ReasoningRequest | None:
|
|
218
|
+
"""One reasoning request naming every reason that fired, or nothing.
|
|
219
219
|
|
|
220
220
|
Spans from the priors that emitted are joined by those of every stage-one
|
|
221
|
-
finding whose rule
|
|
221
|
+
finding whose rule requested it, so the analyzer is pointed at both.
|
|
222
222
|
"""
|
|
223
223
|
if not reasons:
|
|
224
224
|
return None
|
|
@@ -230,7 +230,7 @@ def _escalation(
|
|
|
230
230
|
if finding.rule_id == reason
|
|
231
231
|
for evidence in finding.evidence
|
|
232
232
|
)
|
|
233
|
-
return
|
|
233
|
+
return ReasoningRequest(
|
|
234
234
|
session=ref,
|
|
235
235
|
reasons=tuple(dict.fromkeys(reasons)),
|
|
236
236
|
spans=tuple(dict.fromkeys(cited)),
|
|
@@ -31,15 +31,15 @@ completion* needs the closing claim and the checks it rests on.
|
|
|
31
31
|
## Suppression was removed, not disabled
|
|
32
32
|
|
|
33
33
|
A fourth question used to ask whether an innocent reading accounted for the
|
|
34
|
-
priors that
|
|
35
|
-
|
|
34
|
+
priors that requested reasoning on a session. It was never once asked: measured on 74
|
|
35
|
+
reasoning-eligible sessions, every route was either a stage-one *fact* — which
|
|
36
36
|
a transcript reader must never be able to talk the pipeline out of — or the
|
|
37
37
|
evidence-free `sampled` route, which carries no prior to withdraw. The two
|
|
38
|
-
|
|
38
|
+
requesting priors that could have produced a suppressible reason were withdrawn
|
|
39
39
|
for being unfireable, and the combination route with them.
|
|
40
40
|
|
|
41
41
|
Keeping the machinery would have been keeping a capability the pipeline cannot
|
|
42
|
-
reach. `docs/specs/detector.md` records what would bring it back:
|
|
42
|
+
reach. `docs/specs/detector.md` records what would bring it back: a requesting
|
|
43
43
|
prior that is a *reading* rather than a fact.
|
|
44
44
|
"""
|
|
45
45
|
|
|
@@ -52,7 +52,7 @@ from dataclasses import dataclass
|
|
|
52
52
|
from stacktrace_cli.correlate.observed import VALUE_TAKING_OPTIONS, command_segments
|
|
53
53
|
from stacktrace_cli.detector.analyzer import Analyzer, Answer, Unavailable
|
|
54
54
|
from stacktrace_cli.detector.finding import Detection, Evidence, SessionRef, Unknown, Verdict
|
|
55
|
-
from stacktrace_cli.detector.priors import
|
|
55
|
+
from stacktrace_cli.detector.priors import ReasoningRequest
|
|
56
56
|
from stacktrace_cli.detector.prompts import PROMPT_VERSION, load
|
|
57
57
|
from stacktrace_cli.detector.verdict import RuleVerdict, parse_verdict
|
|
58
58
|
from stacktrace_cli.sessions.outcome import denied
|
|
@@ -251,7 +251,9 @@ _MUST_HAVE_RUN = frozenset(
|
|
|
251
251
|
)
|
|
252
252
|
|
|
253
253
|
|
|
254
|
-
def admissible_spans(
|
|
254
|
+
def admissible_spans(
|
|
255
|
+
rule_id: str, session: SessionLike, request: ReasoningRequest
|
|
256
|
+
) -> frozenset[str]:
|
|
255
257
|
"""The calls a verdict on `rule_id` may cite.
|
|
256
258
|
|
|
257
259
|
Read by `run_reasoning` before a verdict is parsed, and by `run.py` before a
|
|
@@ -271,7 +273,7 @@ def admissible_spans(rule_id: str, session: SessionLike, escalation: Escalation)
|
|
|
271
273
|
"""
|
|
272
274
|
calls = [call for turn in session.turns for call in turn.tool_calls]
|
|
273
275
|
if rule_id == "stacktrace-injected-instruction-followed":
|
|
274
|
-
first = _first_cited_call(session,
|
|
276
|
+
first = _first_cited_call(session, request)
|
|
275
277
|
if first is not None:
|
|
276
278
|
# Strictly after: the call at `first` is the one whose *result*
|
|
277
279
|
# carried the instruction, and nothing it did preceded its own
|
|
@@ -284,7 +286,7 @@ def admissible_spans(rule_id: str, session: SessionLike, escalation: Escalation)
|
|
|
284
286
|
return frozenset(call.span for call in calls if not denied(call))
|
|
285
287
|
|
|
286
288
|
|
|
287
|
-
def _first_cited_call(session: SessionLike,
|
|
289
|
+
def _first_cited_call(session: SessionLike, request: ReasoningRequest) -> int | None:
|
|
288
290
|
"""Where in session order the earliest instruction-shaped *result* sits.
|
|
289
291
|
|
|
290
292
|
`None` where nothing bounds the front of the window: either no cited span
|
|
@@ -295,7 +297,7 @@ def _first_cited_call(session: SessionLike, escalation: Escalation) -> int | Non
|
|
|
295
297
|
function rather than repeating them, so what the analyzer is *sent* and
|
|
296
298
|
what it is allowed to *cite* cannot drift apart.
|
|
297
299
|
"""
|
|
298
|
-
cited = _injection_spans(
|
|
300
|
+
cited = _injection_spans(request)
|
|
299
301
|
calls = [call for turn in session.turns for call in turn.tool_calls]
|
|
300
302
|
if cited - {call.span for call in calls}:
|
|
301
303
|
return None
|
|
@@ -311,15 +313,15 @@ class ReasoningOutcome:
|
|
|
311
313
|
|
|
312
314
|
|
|
313
315
|
def run_reasoning(
|
|
314
|
-
|
|
316
|
+
request: ReasoningRequest, session: SessionLike, analyzer: Analyzer
|
|
315
317
|
) -> ReasoningOutcome:
|
|
316
|
-
"""Ask the analyzer about one
|
|
317
|
-
ref =
|
|
318
|
+
"""Ask the analyzer about one requested session."""
|
|
319
|
+
ref = request.session
|
|
318
320
|
verdict = _verdict_of(analyzer)
|
|
319
321
|
findings: list[Detection] = []
|
|
320
322
|
unknowns: list[Unknown] = []
|
|
321
323
|
|
|
322
|
-
for rule_id in applicable_rules(
|
|
324
|
+
for rule_id in applicable_rules(request):
|
|
323
325
|
if (
|
|
324
326
|
rule_id == "stacktrace-intent-drift"
|
|
325
327
|
and session.compactions
|
|
@@ -341,11 +343,11 @@ def run_reasoning(
|
|
|
341
343
|
),
|
|
342
344
|
),
|
|
343
345
|
ref,
|
|
344
|
-
|
|
346
|
+
request,
|
|
345
347
|
)
|
|
346
348
|
)
|
|
347
349
|
continue
|
|
348
|
-
truncated_span = _truncated_evidence(session, rule_id,
|
|
350
|
+
truncated_span = _truncated_evidence(session, rule_id, request)
|
|
349
351
|
if truncated_span is not None:
|
|
350
352
|
# The same reason `truncated` is `Unknown` in stage one: a body
|
|
351
353
|
# elided in the middle looks whole at both ends, and reasoning over
|
|
@@ -365,7 +367,7 @@ def run_reasoning(
|
|
|
365
367
|
),
|
|
366
368
|
),
|
|
367
369
|
ref,
|
|
368
|
-
|
|
370
|
+
request,
|
|
369
371
|
)
|
|
370
372
|
)
|
|
371
373
|
continue
|
|
@@ -373,7 +375,7 @@ def run_reasoning(
|
|
|
373
375
|
# tripwire rather than a window: a body elided in the middle looks
|
|
374
376
|
# whole at both ends, so a body that does not fit is `unknown` for
|
|
375
377
|
# *this* question rather than silently shortened for it.
|
|
376
|
-
transcript = _CONTEXT[rule_id](session,
|
|
378
|
+
transcript = _CONTEXT[rule_id](session, request)
|
|
377
379
|
if len(transcript) > CONTEXT_LIMIT:
|
|
378
380
|
unknowns.append(
|
|
379
381
|
_with(
|
|
@@ -387,7 +389,7 @@ def run_reasoning(
|
|
|
387
389
|
),
|
|
388
390
|
),
|
|
389
391
|
ref,
|
|
390
|
-
|
|
392
|
+
request,
|
|
391
393
|
)
|
|
392
394
|
)
|
|
393
395
|
continue
|
|
@@ -396,21 +398,21 @@ def run_reasoning(
|
|
|
396
398
|
# three report (`_MUST_HAVE_RUN`).
|
|
397
399
|
answer = _ask(
|
|
398
400
|
analyzer,
|
|
399
|
-
_prompt_for(rule_id,
|
|
401
|
+
_prompt_for(rule_id, request),
|
|
400
402
|
transcript,
|
|
401
403
|
rule_id,
|
|
402
|
-
admissible_spans(rule_id, session,
|
|
404
|
+
admissible_spans(rule_id, session, request),
|
|
403
405
|
)
|
|
404
406
|
if isinstance(answer, Unknown):
|
|
405
|
-
unknowns.append(_with(answer, ref,
|
|
407
|
+
unknowns.append(_with(answer, ref, request))
|
|
406
408
|
continue
|
|
407
409
|
if answer.fired:
|
|
408
|
-
findings.append(_finding(answer, ref,
|
|
410
|
+
findings.append(_finding(answer, ref, request, verdict))
|
|
409
411
|
|
|
410
412
|
return ReasoningOutcome(findings=tuple(findings), unknowns=tuple(unknowns))
|
|
411
413
|
|
|
412
414
|
|
|
413
|
-
def dispatchable(
|
|
415
|
+
def dispatchable(request: ReasoningRequest, session: SessionLike) -> bool:
|
|
414
416
|
"""Whether `run_reasoning` would ask the analyzer at least one question.
|
|
415
417
|
|
|
416
418
|
A preflight over the same per-rule bodies `run_reasoning` builds: a session
|
|
@@ -422,12 +424,11 @@ def dispatchable(escalation: Escalation, session: SessionLike) -> bool:
|
|
|
422
424
|
never calls the analyzer at all.
|
|
423
425
|
"""
|
|
424
426
|
return any(
|
|
425
|
-
_block_reason(rule_id,
|
|
426
|
-
for rule_id in applicable_rules(escalation)
|
|
427
|
+
_block_reason(rule_id, request, session) is None for rule_id in applicable_rules(request)
|
|
427
428
|
)
|
|
428
429
|
|
|
429
430
|
|
|
430
|
-
def undispatchable_reason(
|
|
431
|
+
def undispatchable_reason(request: ReasoningRequest, session: SessionLike) -> str:
|
|
431
432
|
"""Why `dispatchable()` returned False, for the caller to report accurately.
|
|
432
433
|
|
|
433
434
|
A session can be undispatchable for three distinct reasons — oversized,
|
|
@@ -438,8 +439,8 @@ def undispatchable_reason(escalation: Escalation, session: SessionLike) -> str:
|
|
|
438
439
|
first blocking reason found is the one reported: with none dispatchable,
|
|
439
440
|
every applicable rule is already blocked by one of these three.
|
|
440
441
|
"""
|
|
441
|
-
for rule_id in applicable_rules(
|
|
442
|
-
reason = _block_reason(rule_id,
|
|
442
|
+
for rule_id in applicable_rules(request):
|
|
443
|
+
reason = _block_reason(rule_id, request, session)
|
|
443
444
|
if reason is not None:
|
|
444
445
|
return reason
|
|
445
446
|
# Only meaningful once `dispatchable()` is False; unreachable otherwise
|
|
@@ -447,35 +448,37 @@ def undispatchable_reason(escalation: Escalation, session: SessionLike) -> str:
|
|
|
447
448
|
return "session_too_large"
|
|
448
449
|
|
|
449
450
|
|
|
450
|
-
def _block_reason(rule_id: str,
|
|
451
|
+
def _block_reason(rule_id: str, request: ReasoningRequest, session: SessionLike) -> str | None:
|
|
451
452
|
"""Why this one rule would not reach the analyzer, or None if it would."""
|
|
452
453
|
if rule_id == "stacktrace-intent-drift" and session.compactions and not session.initial_prompt:
|
|
453
454
|
return "insufficient_context"
|
|
454
|
-
if _truncated_evidence(session, rule_id,
|
|
455
|
+
if _truncated_evidence(session, rule_id, request) is not None:
|
|
455
456
|
return "truncated"
|
|
456
|
-
if len(_CONTEXT[rule_id](session,
|
|
457
|
+
if len(_CONTEXT[rule_id](session, request)) > CONTEXT_LIMIT:
|
|
457
458
|
return "session_too_large"
|
|
458
459
|
return None
|
|
459
460
|
|
|
460
461
|
|
|
461
|
-
def _injection_spans(
|
|
462
|
+
def _injection_spans(request: ReasoningRequest) -> set[str]:
|
|
462
463
|
"""Where stage one actually found instruction-shaped material.
|
|
463
464
|
|
|
464
|
-
`
|
|
465
|
-
one set, so a session that
|
|
465
|
+
`request.spans` is every solo-requesting reason's evidence merged into
|
|
466
|
+
one set, so a session that requested on both `stacktrace-credential-egress`
|
|
466
467
|
and `stacktrace-injection-marker` carries the credential call's span there
|
|
467
468
|
too. Filtering `stage_one` to the marker's own rule keeps that unrelated
|
|
468
469
|
span out of what `injected-instruction-followed` treats as cited.
|
|
469
470
|
"""
|
|
470
471
|
return {
|
|
471
472
|
evidence.span
|
|
472
|
-
for finding in
|
|
473
|
+
for finding in request.stage_one
|
|
473
474
|
if finding.rule_id == "stacktrace-injection-marker"
|
|
474
475
|
for evidence in finding.evidence
|
|
475
476
|
}
|
|
476
477
|
|
|
477
478
|
|
|
478
|
-
def _truncated_evidence(
|
|
479
|
+
def _truncated_evidence(
|
|
480
|
+
session: SessionLike, rule_id: str, request: ReasoningRequest
|
|
481
|
+
) -> str | None:
|
|
479
482
|
"""The span of a truncated call this rule would otherwise read as whole.
|
|
480
483
|
|
|
481
484
|
Truncation loses evidence in the middle: the retained fragment looks
|
|
@@ -486,7 +489,7 @@ def _truncated_evidence(session: SessionLike, rule_id: str, escalation: Escalati
|
|
|
486
489
|
and `deceptive-completion`'s verification output.
|
|
487
490
|
"""
|
|
488
491
|
if rule_id == "stacktrace-injected-instruction-followed":
|
|
489
|
-
cited = _injection_spans(
|
|
492
|
+
cited = _injection_spans(request)
|
|
490
493
|
for turn in session.turns:
|
|
491
494
|
for call in turn.tool_calls:
|
|
492
495
|
if call.span in cited and call.truncated:
|
|
@@ -499,14 +502,14 @@ def _truncated_evidence(session: SessionLike, rule_id: str, escalation: Escalati
|
|
|
499
502
|
return None
|
|
500
503
|
|
|
501
504
|
|
|
502
|
-
def applicable_rules(
|
|
503
|
-
"""The rules that could return a finding on this
|
|
505
|
+
def applicable_rules(request: ReasoningRequest) -> tuple[str, ...]:
|
|
506
|
+
"""The rules that could return a finding on this request.
|
|
504
507
|
|
|
505
508
|
Asking a rule whose precondition is absent spends an invocation to hear the
|
|
506
509
|
only answer it can give. `injected-instruction-followed` asks whether the
|
|
507
510
|
agent *carried out* instruction-shaped content stage 1 found in a result or
|
|
508
511
|
a context item — so with no such finding on this session there is nothing to
|
|
509
|
-
have carried out. Measured on 74 real
|
|
512
|
+
have carried out. Measured on 74 real reasoning-eligible sessions, 71 had
|
|
510
513
|
no marker, and skipping the rule there is not a narrowed catalogue: it is
|
|
511
514
|
declining to ask a question with no subject.
|
|
512
515
|
|
|
@@ -514,7 +517,7 @@ def applicable_rules(escalation: Escalation) -> tuple[str, ...]:
|
|
|
514
517
|
to any session that reached this stage.
|
|
515
518
|
"""
|
|
516
519
|
has_marker = any(
|
|
517
|
-
finding.rule_id == "stacktrace-injection-marker" for finding in
|
|
520
|
+
finding.rule_id == "stacktrace-injection-marker" for finding in request.stage_one
|
|
518
521
|
)
|
|
519
522
|
return tuple(
|
|
520
523
|
rule for rule in _RULES if rule != "stacktrace-injected-instruction-followed" or has_marker
|
|
@@ -681,7 +684,7 @@ def serialise(session: SessionLike) -> str:
|
|
|
681
684
|
return "\n".join(lines)
|
|
682
685
|
|
|
683
686
|
|
|
684
|
-
def serialise_scope(session: SessionLike,
|
|
687
|
+
def serialise_scope(session: SessionLike, request: ReasoningRequest | None = None) -> str:
|
|
685
688
|
"""Every request, every action, and no payloads — for `intent-drift`.
|
|
686
689
|
|
|
687
690
|
Drift is a question about *scope*: did the agent do work nobody asked for.
|
|
@@ -707,7 +710,7 @@ def serialise_scope(session: SessionLike, escalation: Escalation | None = None)
|
|
|
707
710
|
return "\n".join(lines)
|
|
708
711
|
|
|
709
712
|
|
|
710
|
-
def serialise_injection(session: SessionLike,
|
|
713
|
+
def serialise_injection(session: SessionLike, request: ReasoningRequest) -> str:
|
|
711
714
|
"""The instruction-shaped material, and everything the agent did after it.
|
|
712
715
|
|
|
713
716
|
The question is whether a *later call carried out* what the content asked
|
|
@@ -716,9 +719,9 @@ def serialise_injection(session: SessionLike, escalation: Escalation) -> str:
|
|
|
716
719
|
evidence — a call that acted on an injected instruction shows it in what it
|
|
717
720
|
was asked to do.
|
|
718
721
|
|
|
719
|
-
Cited from the `stacktrace-injection-marker` findings in `
|
|
720
|
-
alone — never `
|
|
721
|
-
other solo-
|
|
722
|
+
Cited from the `stacktrace-injection-marker` findings in `request.stage_one`
|
|
723
|
+
alone — never `request.spans`, which also carries the evidence of any
|
|
724
|
+
other solo-requesting reason (`stacktrace-credential-egress`) that fired on
|
|
722
725
|
this session. A credential call is not instruction-shaped material, and
|
|
723
726
|
treating its span as cited here would gate this question on that call's own
|
|
724
727
|
truncation and print its result as though it were the injected content.
|
|
@@ -729,12 +732,12 @@ def serialise_injection(session: SessionLike, escalation: Escalation) -> str:
|
|
|
729
732
|
call's: standing context precedes every turn, so a call obeying it can come
|
|
730
733
|
before that later call, and must not be starved of its own arguments.
|
|
731
734
|
"""
|
|
732
|
-
cited = _injection_spans(
|
|
735
|
+
cited = _injection_spans(request)
|
|
733
736
|
lines = _header(session)
|
|
734
737
|
# The same window `admissible_spans` refuses a citation outside of. Shared
|
|
735
738
|
# rather than restated: a call this body sends no arguments for is a call
|
|
736
739
|
# no verdict may cite, and two copies of that reading would drift.
|
|
737
|
-
reached = _first_cited_call(session,
|
|
740
|
+
reached = _first_cited_call(session, request) is None
|
|
738
741
|
for turn in session.turns:
|
|
739
742
|
lines.append(_turn_heading(turn))
|
|
740
743
|
if turn.text:
|
|
@@ -759,7 +762,7 @@ def serialise_injection(session: SessionLike, escalation: Escalation) -> str:
|
|
|
759
762
|
return "\n".join(lines)
|
|
760
763
|
|
|
761
764
|
|
|
762
|
-
def serialise_completion(session: SessionLike,
|
|
765
|
+
def serialise_completion(session: SessionLike, request: ReasoningRequest | None = None) -> str:
|
|
763
766
|
"""The closing claim, and the checks it rests on — for `deceptive-completion`.
|
|
764
767
|
|
|
765
768
|
Two things make this finding: an agent saying a task is done, and a
|
|
@@ -979,35 +982,43 @@ def _module_of(arguments: Sequence[str]) -> str | None:
|
|
|
979
982
|
#: Which body each question is sent. Keyed by rule so adding a rule without
|
|
980
983
|
#: deciding this is a `KeyError` at the call site rather than a silent
|
|
981
984
|
#: full-session send.
|
|
982
|
-
_CONTEXT: dict[str, Callable[[SessionLike,
|
|
985
|
+
_CONTEXT: dict[str, Callable[[SessionLike, ReasoningRequest], str]] = {
|
|
983
986
|
"stacktrace-intent-drift": serialise_scope,
|
|
984
987
|
"stacktrace-injected-instruction-followed": serialise_injection,
|
|
985
988
|
"stacktrace-deceptive-completion": serialise_completion,
|
|
986
989
|
}
|
|
987
990
|
|
|
988
991
|
|
|
989
|
-
def _prompt_for(name: str,
|
|
990
|
-
"""Framing, then the question, then why this
|
|
991
|
-
transcript — that is a separate argument, and it goes last."""
|
|
992
|
-
parts = [load("framing"), load(name), load("exclusions"), _why(
|
|
992
|
+
def _prompt_for(name: str, request: ReasoningRequest) -> str:
|
|
993
|
+
"""Framing, then the question, then why this session is being asked about.
|
|
994
|
+
Never the transcript — that is a separate argument, and it goes last."""
|
|
995
|
+
parts = [load("framing"), load(name), load("exclusions"), _why(request)]
|
|
993
996
|
return "\n\n".join(part for part in parts if part)
|
|
994
997
|
|
|
995
998
|
|
|
996
|
-
def _why(
|
|
999
|
+
def _why(request: ReasoningRequest) -> str:
|
|
997
1000
|
"""Which priors fired, by name and span. **Descriptors only.**
|
|
998
1001
|
|
|
999
1002
|
This position is read as pipeline testimony, so a decoded payload or a
|
|
1000
1003
|
secret quoted here would be attacker text promoted from data to
|
|
1001
1004
|
instruction. Rule names and spans carry everything the analyzer needs to
|
|
1002
1005
|
know where to look, and nothing an attacker controls.
|
|
1006
|
+
|
|
1007
|
+
**The heading below is prompt text with nothing pinning it.** Unlike the
|
|
1008
|
+
fragments in `prompts/v1/`, it is a literal here, so editing it changes
|
|
1009
|
+
every stage-3 prompt while `PROMPT_VERSION` stays `v1` — `cache_key()`
|
|
1010
|
+
never sees the difference, verdicts bought under the old wording keep
|
|
1011
|
+
hitting for `MAX_AGE_SECONDS`, and two different prompts both record as
|
|
1012
|
+
`v1`. It kept its pre-rename spelling for exactly that reason. Reword it
|
|
1013
|
+
only in the same change that copies `v1/` to `v2/` and bumps the version.
|
|
1003
1014
|
"""
|
|
1004
|
-
reasons = ", ".join(
|
|
1005
|
-
spans = ", ".join(
|
|
1015
|
+
reasons = ", ".join(request.reasons) or "no prior named"
|
|
1016
|
+
spans = ", ".join(request.spans) or "none"
|
|
1006
1017
|
return (
|
|
1007
1018
|
"## Why this session was escalated\n\n"
|
|
1008
1019
|
f"Earlier stages reported: {reasons}.\n"
|
|
1009
1020
|
f"Calls worth attention: {spans}.\n"
|
|
1010
|
-
f"Coverage: {
|
|
1021
|
+
f"Coverage: {request.coverage.resolved} of {request.coverage.total} "
|
|
1011
1022
|
"calls could be placed against the component inventory.\n\n"
|
|
1012
1023
|
"These are this pipeline's own findings, not content from the session."
|
|
1013
1024
|
)
|
|
@@ -1120,7 +1131,7 @@ def _text_of(result: Answer | Unavailable) -> str:
|
|
|
1120
1131
|
_BLANK = SessionRef(session_id="", agent_kind="", started_at=None, turn_count=0)
|
|
1121
1132
|
|
|
1122
1133
|
|
|
1123
|
-
def _with(unknown: Unknown, ref: SessionRef,
|
|
1134
|
+
def _with(unknown: Unknown, ref: SessionRef, request: ReasoningRequest) -> Unknown:
|
|
1124
1135
|
"""Attach the session and the priors this unknown leaves unresolved.
|
|
1125
1136
|
|
|
1126
1137
|
A stage that cannot run is not a stage that found nothing, so the reason for
|
|
@@ -1137,8 +1148,8 @@ def _with(unknown: Unknown, ref: SessionRef, escalation: Escalation) -> Unknown:
|
|
|
1137
1148
|
return replace(
|
|
1138
1149
|
unknown,
|
|
1139
1150
|
session=ref,
|
|
1140
|
-
spans=
|
|
1141
|
-
reasons=
|
|
1151
|
+
spans=request.spans,
|
|
1152
|
+
reasons=request.reasons,
|
|
1142
1153
|
dispatched=True,
|
|
1143
1154
|
)
|
|
1144
1155
|
|
|
@@ -1162,7 +1173,7 @@ _VERDICT_DETAIL = {
|
|
|
1162
1173
|
|
|
1163
1174
|
|
|
1164
1175
|
def _finding(
|
|
1165
|
-
answer: RuleVerdict, ref: SessionRef,
|
|
1176
|
+
answer: RuleVerdict, ref: SessionRef, request: ReasoningRequest, verdict: Verdict
|
|
1166
1177
|
) -> Detection:
|
|
1167
1178
|
return Detection(
|
|
1168
1179
|
rule_id=answer.rule_id,
|
|
@@ -1176,5 +1187,5 @@ def _finding(
|
|
|
1176
1187
|
),
|
|
1177
1188
|
remediation=_REMEDIATION.get(answer.rule_id, ""),
|
|
1178
1189
|
verdict=verdict,
|
|
1179
|
-
coverage=
|
|
1190
|
+
coverage=request.coverage,
|
|
1180
1191
|
)
|
|
@@ -31,7 +31,7 @@ def render_json(result: DetectorRun) -> str:
|
|
|
31
31
|
"reason": u.reason,
|
|
32
32
|
"spans": list(u.spans),
|
|
33
33
|
"detail": u.detail,
|
|
34
|
-
"
|
|
34
|
+
"reasoning_reasons": list(u.reasons),
|
|
35
35
|
# True for a reasoning-stage unknown raised after the session
|
|
36
36
|
# was already sent to the analyzer — one unanswered question
|
|
37
37
|
# inside a session `summary.analysed` already counts, not a
|
|
@@ -47,7 +47,7 @@ def render_json(result: DetectorRun) -> str:
|
|
|
47
47
|
"security": sum(1 for d in result.detections if d.family == "security"),
|
|
48
48
|
"reliability": sum(1 for d in result.detections if d.family == "reliability"),
|
|
49
49
|
"unknowns": len(result.unknowns),
|
|
50
|
-
"
|
|
50
|
+
"reasoning_requested": result.requested,
|
|
51
51
|
"analysed": result.analysed,
|
|
52
52
|
"cache_hits": result.cache_hits,
|
|
53
53
|
},
|
|
@@ -172,17 +172,17 @@ def render_text(result: DetectorRun, detail: bool = False) -> str:
|
|
|
172
172
|
"""
|
|
173
173
|
lines: list[str] = []
|
|
174
174
|
# Not `not result.detections and not result.unknowns` alone: a run that
|
|
175
|
-
#
|
|
175
|
+
# requested, analysed and cleared every session (nothing fired, nothing
|
|
176
176
|
# unanswered) has no detections and no unknowns either, and is the most
|
|
177
177
|
# common successful outcome — the one this shortcut must not hide the
|
|
178
|
-
# cost of. `
|
|
178
|
+
# cost of. `requested` is what distinguishes it from a run that never
|
|
179
179
|
# reached stage three at all. `collection_failures` is checked too: a
|
|
180
180
|
# reader that failed before correlation ever saw a session leaves every
|
|
181
181
|
# other count at zero, and that must never read as the clean-run shortcut.
|
|
182
182
|
if (
|
|
183
183
|
not result.detections
|
|
184
184
|
and not result.unknowns
|
|
185
|
-
and not result.
|
|
185
|
+
and not result.requested
|
|
186
186
|
and not result.collection_failures
|
|
187
187
|
):
|
|
188
188
|
return f"No detections across {result.sessions} sessions."
|
|
@@ -220,14 +220,11 @@ def render_text(result: DetectorRun, detail: bool = False) -> str:
|
|
|
220
220
|
|
|
221
221
|
lines.extend(_rollup_lines(result))
|
|
222
222
|
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
for reason, count in sorted(counts.items(), key=lambda kv: (-kv[1], kv[0])):
|
|
229
|
-
lines.append(f" {count:>4} {reason}")
|
|
230
|
-
lines.append("")
|
|
223
|
+
# The per-reason breakdown of what went unsettled is no longer printed by
|
|
224
|
+
# default. The count still travels in the summary line, `--json` still
|
|
225
|
+
# carries every `Unknown` whole, and the reasoning stage still names its
|
|
226
|
+
# own outcomes below, so nothing is dropped from the record -- only from
|
|
227
|
+
# the default terminal view.
|
|
231
228
|
|
|
232
229
|
lines.extend(_reasoning_lines(result))
|
|
233
230
|
|
|
@@ -240,7 +237,7 @@ def render_text(result: DetectorRun, detail: bool = False) -> str:
|
|
|
240
237
|
lines.append(
|
|
241
238
|
f"Summary — {result.sessions} sessions, {security} security, "
|
|
242
239
|
f"{reliability} reliability, {len(result.unknowns)} unsettled, "
|
|
243
|
-
f"{result.
|
|
240
|
+
f"{result.requested} for reasoning, {result.analysed} analysed {analysed_sessions}, "
|
|
244
241
|
f"{result.cache_hits} from cache"
|
|
245
242
|
)
|
|
246
243
|
return "\n".join(lines)
|
|
@@ -251,12 +248,12 @@ def render_text(result: DetectorRun, detail: bool = False) -> str:
|
|
|
251
248
|
#:
|
|
252
249
|
#: A gloss, not a gate: a reason with no entry still prints, under its bare name,
|
|
253
250
|
#: because the accounting below has to balance whether or not anyone wrote a
|
|
254
|
-
#: sentence for it. `
|
|
255
|
-
#: is what enforces that -- every
|
|
251
|
+
#: sentence for it. `test_the_reasoning_stage_block_accounts_for_every_request`
|
|
252
|
+
#: is what enforces that -- every requested session lands in exactly one
|
|
256
253
|
#: session-level row, whether or not it also has unanswered questions.
|
|
257
254
|
_REASONING_OUTCOMES: dict[str, str] = {
|
|
258
255
|
"budget_exhausted": "deferred: the run's --budget was reached",
|
|
259
|
-
"reasoning_not_requested": "not requested: --
|
|
256
|
+
"reasoning_not_requested": "not requested: re-run with --reasoning to analyse them",
|
|
260
257
|
"analyzer_unavailable": "no analyzer available for the agent kind",
|
|
261
258
|
"session_too_large": "too large for the analyzer's context window",
|
|
262
259
|
"insufficient_context": "compaction removed the turns a rule needed",
|
|
@@ -268,16 +265,16 @@ _REASONING_OUTCOMES: dict[str, str] = {
|
|
|
268
265
|
|
|
269
266
|
|
|
270
267
|
def _reasoning_lines(result: DetectorRun) -> list[str]:
|
|
271
|
-
"""What the third stage actually spent, and where every
|
|
268
|
+
"""What the third stage actually spent, and where every request went.
|
|
272
269
|
|
|
273
|
-
Stated as accounting rather than as one number
|
|
274
|
-
|
|
275
|
-
|
|
270
|
+
Stated as accounting rather than as one number. With the stage opt-in the
|
|
271
|
+
reader has both questions -- *did it run* and, when it did, *what did this
|
|
272
|
+
run cost me* -- and one number answers neither. Three outcomes are not the same
|
|
276
273
|
fact: a model call spends seconds and money, a cache hit settles the same
|
|
277
|
-
question for free, and a deferred
|
|
274
|
+
question for free, and a deferred request spends nothing and leaves the
|
|
278
275
|
concern open.
|
|
279
276
|
|
|
280
|
-
Every
|
|
277
|
+
Every requested session appears in exactly one *session-level* row —
|
|
281
278
|
analysed, a cache hit, or one of the reasons a session never reached the
|
|
282
279
|
analyzer at all — so those rows sum to the headline. `run_reasoning` asks
|
|
283
280
|
one to three rules per analysed session, and one rule's clean answer does
|
|
@@ -286,12 +283,12 @@ def _reasoning_lines(result: DetectorRun) -> list[str]:
|
|
|
286
283
|
since a session already counted in `analysed` must not be counted again
|
|
287
284
|
for a question it also failed to answer.
|
|
288
285
|
"""
|
|
289
|
-
if not result.
|
|
286
|
+
if not result.requested:
|
|
290
287
|
return []
|
|
291
288
|
lines = [
|
|
292
289
|
"",
|
|
293
290
|
(
|
|
294
|
-
f"Reasoning stage — {result.analysed} of {result.
|
|
291
|
+
f"Reasoning stage — {result.analysed} of {result.requested} requested "
|
|
295
292
|
f"session(s) reached the model"
|
|
296
293
|
),
|
|
297
294
|
]
|