stacktrace-cli 0.4.0__py3-none-any.whl → 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/analysis.py +40 -9
- stacktrace_cli/cli.py +39 -12
- stacktrace_cli/daemon/cli.py +2 -4
- stacktrace_cli/daemon/presentation.py +68 -0
- stacktrace_cli/daemon/store.py +15 -3
- stacktrace_cli/detector/deterministic.py +7 -3
- stacktrace_cli/detector/finding.py +15 -0
- stacktrace_cli/detector/reasoning.py +5 -2
- stacktrace_cli/detector/render.py +4 -0
- stacktrace_cli/detector/run.py +25 -17
- stacktrace_cli/monitor/render.py +20 -2
- stacktrace_cli/monitor/site/app.js +809 -30
- stacktrace_cli/monitor/site/index.html +28 -2
- stacktrace_cli/monitor/site/styles.css +73 -0
- stacktrace_cli/monitor/state.py +45 -0
- stacktrace_cli/monitor/watch.py +46 -14
- stacktrace_cli/sessions/protocols.py +8 -0
- {stacktrace_cli-0.4.0.dist-info → stacktrace_cli-0.5.0.dist-info}/METADATA +2 -2
- {stacktrace_cli-0.4.0.dist-info → stacktrace_cli-0.5.0.dist-info}/RECORD +22 -21
- {stacktrace_cli-0.4.0.dist-info → stacktrace_cli-0.5.0.dist-info}/WHEEL +0 -0
- {stacktrace_cli-0.4.0.dist-info → stacktrace_cli-0.5.0.dist-info}/entry_points.txt +0 -0
stacktrace_cli/__init__.py
CHANGED
stacktrace_cli/analysis.py
CHANGED
|
@@ -19,11 +19,12 @@ path's tests already pin.
|
|
|
19
19
|
|
|
20
20
|
from __future__ import annotations
|
|
21
21
|
|
|
22
|
+
import re
|
|
22
23
|
from collections.abc import Iterator
|
|
23
24
|
from dataclasses import dataclass, replace
|
|
24
25
|
from datetime import UTC, datetime, timedelta
|
|
25
26
|
from pathlib import Path
|
|
26
|
-
from typing import Any
|
|
27
|
+
from typing import Any, Final
|
|
27
28
|
|
|
28
29
|
from stacktrace_cli.correlate.orchestrate import Acquired, acquire_correlated_view
|
|
29
30
|
from stacktrace_cli.correlate.project_map import parse_mapping
|
|
@@ -63,25 +64,55 @@ class Analysis:
|
|
|
63
64
|
placed: bool = True
|
|
64
65
|
|
|
65
66
|
|
|
67
|
+
#: A count and an optional unit, anchored. Anchored rather than searched
|
|
68
|
+
#: because the tolerance this parser owes the upload path is about the *unit*
|
|
69
|
+
#: being omissible, not about the value being loosely shaped: `int` alone would
|
|
70
|
+
#: read `'5 m'` as five (it strips whitespace before parsing) and a signed
|
|
71
|
+
#: `'-1h'` as a negative window, so the shape is pinned here and the count is
|
|
72
|
+
#: parsed from digits already known to be digits.
|
|
73
|
+
_SINCE_SPELLING: Final = re.compile(r"^(\d+)([mhd]?)$")
|
|
74
|
+
|
|
75
|
+
#: The unit suffixes, and the `timedelta` keyword each one names. The empty
|
|
76
|
+
#: key is the bare count -- `'7'`, which the upload path's config value has
|
|
77
|
+
#: always spelled without a unit and which has always meant days.
|
|
78
|
+
_SINCE_UNITS: Final = {"m": "minutes", "h": "hours", "d": "days", "": "days"}
|
|
79
|
+
|
|
80
|
+
#: One message for every way the spelling can be wrong, because they are one
|
|
81
|
+
#: user mistake: a window that does not name a positive amount of time. Shared
|
|
82
|
+
#: with `cli.py`'s stricter parser so a reader meets the same vocabulary
|
|
83
|
+
#: whichever of the two refused them.
|
|
84
|
+
SINCE_HINT: Final = "like '5m', '2h' or '7d'"
|
|
85
|
+
|
|
86
|
+
|
|
66
87
|
def parse_since(value: str) -> datetime:
|
|
67
|
-
"""`7d` or `7` -> an aware UTC cut-off.
|
|
88
|
+
"""`5m`, `2h`, `7d` or `7` -> an aware UTC cut-off.
|
|
89
|
+
|
|
90
|
+
Sub-day units are here rather than only at the command line because the
|
|
91
|
+
window reaches this module as a *string* from every caller that has one --
|
|
92
|
+
`monitor` hands it over untouched and parses it on the watcher thread, so a
|
|
93
|
+
unit the command line accepted and this parser did not would raise where no
|
|
94
|
+
request is left to answer.
|
|
68
95
|
|
|
69
96
|
Raises `ValueError`, never a Click or transport exception: this module is
|
|
70
97
|
below both. The message is the upload path's, which its tests pin.
|
|
71
98
|
"""
|
|
72
|
-
|
|
99
|
+
match = _SINCE_SPELLING.match(value.strip().lower())
|
|
100
|
+
if not match:
|
|
101
|
+
raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}")
|
|
73
102
|
try:
|
|
74
|
-
|
|
103
|
+
count = int(match.group(1))
|
|
75
104
|
except ValueError:
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
raise ValueError(f"--since must be a positive
|
|
105
|
+
# `int` refuses a string past its default digit-conversion limit. The
|
|
106
|
+
# same user mistake as a non-numeric value, so the same diagnostic.
|
|
107
|
+
raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}") from None
|
|
108
|
+
if count <= 0:
|
|
109
|
+
raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}")
|
|
79
110
|
try:
|
|
80
|
-
return datetime.now(UTC) - timedelta(
|
|
111
|
+
return datetime.now(UTC) - timedelta(**{_SINCE_UNITS[match.group(2)]: count})
|
|
81
112
|
except OverflowError:
|
|
82
113
|
# `timedelta` takes any `int` but overflows past `timedelta.max.days`.
|
|
83
114
|
# The same user mistake as a non-numeric value, so the same diagnostic.
|
|
84
|
-
raise ValueError(f"--since must be a positive
|
|
115
|
+
raise ValueError(f"--since must be a positive window {SINCE_HINT}, not {value!r}") from None
|
|
85
116
|
|
|
86
117
|
|
|
87
118
|
def _only(
|
stacktrace_cli/cli.py
CHANGED
|
@@ -24,7 +24,7 @@ from typing import Final
|
|
|
24
24
|
import click
|
|
25
25
|
from openaca.cli import main as openaca_cli
|
|
26
26
|
|
|
27
|
-
from .analysis import analyse, parse_since
|
|
27
|
+
from .analysis import _SINCE_UNITS, SINCE_HINT, analyse, parse_since
|
|
28
28
|
from .build import version_label
|
|
29
29
|
from .daemon.cli import daemon as daemon_cmd
|
|
30
30
|
from .daemon.cli import findings
|
|
@@ -45,15 +45,33 @@ PASSTHROUGH: Final = ("scan", "bom", "policy")
|
|
|
45
45
|
|
|
46
46
|
HOMEPAGE = "https://stacktrace.ai"
|
|
47
47
|
|
|
48
|
-
|
|
48
|
+
#: The unit is required here and optional in `analysis.parse_since`. That is
|
|
49
|
+
#: the whole of the difference between the two spellings, and it is deliberate:
|
|
50
|
+
#: the forgiving parser also answers the upload path, whose value comes from a
|
|
51
|
+
#: config file rather than a shell, so a bare `7` has to keep meaning days
|
|
52
|
+
#: there while staying a usage error here.
|
|
53
|
+
_SINCE_PATTERN = re.compile(r"^(\d+)([mhd])$")
|
|
49
54
|
|
|
50
55
|
|
|
51
56
|
def _parse_since(value: str) -> datetime:
|
|
52
|
-
"""`14d` -> an aware UTC cut-off.
|
|
57
|
+
"""`5m`, `2h` or `14d` -> an aware UTC cut-off.
|
|
58
|
+
|
|
59
|
+
A zero count is a window that covers nothing, not a usage error: `sessions`
|
|
60
|
+
and `detect` have always read `0d` that way, and the units beside it answer
|
|
61
|
+
to the same rule. `monitor` is the one that refuses it, and it refuses it
|
|
62
|
+
through the second parser `_day_count` applies.
|
|
63
|
+
"""
|
|
53
64
|
match = _SINCE_PATTERN.match(value.strip())
|
|
54
65
|
if not match:
|
|
55
|
-
raise click.BadParameter(f"expected a
|
|
56
|
-
|
|
66
|
+
raise click.BadParameter(f"expected a window {SINCE_HINT}, got {value!r}")
|
|
67
|
+
try:
|
|
68
|
+
return datetime.now(UTC) - timedelta(**{_SINCE_UNITS[match.group(2)]: int(match.group(1))})
|
|
69
|
+
except (OverflowError, ValueError):
|
|
70
|
+
# `timedelta` takes any `int` but overflows past `timedelta.max.days`, and
|
|
71
|
+
# `int` itself refuses a string past its default digit-conversion limit.
|
|
72
|
+
# Both are the same user mistake as a non-numeric value, so the same
|
|
73
|
+
# diagnostic.
|
|
74
|
+
raise click.BadParameter(f"expected a window {SINCE_HINT}, got {value!r}") from None
|
|
57
75
|
|
|
58
76
|
|
|
59
77
|
def _window_start(ctx: click.Context, param: click.Parameter, value: str) -> datetime:
|
|
@@ -202,7 +220,12 @@ def main() -> None:
|
|
|
202
220
|
multiple=True,
|
|
203
221
|
help="Only this agent kind. Repeatable. Default: every kind.",
|
|
204
222
|
)
|
|
205
|
-
@click.option(
|
|
223
|
+
@click.option(
|
|
224
|
+
"--since",
|
|
225
|
+
default="14d",
|
|
226
|
+
show_default=True,
|
|
227
|
+
help="Window, as a count and a unit: 5m, 2h or 14d.",
|
|
228
|
+
)
|
|
206
229
|
@click.option(
|
|
207
230
|
"--include-content",
|
|
208
231
|
is_flag=True,
|
|
@@ -254,7 +277,7 @@ def sessions(
|
|
|
254
277
|
default="7d",
|
|
255
278
|
show_default=True,
|
|
256
279
|
callback=_window_start,
|
|
257
|
-
help="How far back to read.",
|
|
280
|
+
help="How far back to read, as a count and a unit: 5m, 2h or 7d.",
|
|
258
281
|
)
|
|
259
282
|
@click.option(
|
|
260
283
|
"--bom",
|
|
@@ -461,11 +484,15 @@ def detect(
|
|
|
461
484
|
default="3d",
|
|
462
485
|
show_default=True,
|
|
463
486
|
callback=_day_count,
|
|
464
|
-
help="How far back to read
|
|
465
|
-
"whole window is collected and
|
|
466
|
-
"anything — measured at ~2.5s for a
|
|
467
|
-
"machine — and a live view is about what
|
|
468
|
-
"happened. Widen it when you want the
|
|
487
|
+
help="How far back to read, as a count and a unit: 5m, 2h or 3d. Shorter "
|
|
488
|
+
"than `detect`'s week deliberately: the whole window is collected and "
|
|
489
|
+
"correlated before the page can render anything — measured at ~2.5s for a "
|
|
490
|
+
"day and ~6s for a week of a busy machine — and a live view is about what "
|
|
491
|
+
"is happening rather than what happened. Widen it when you want the "
|
|
492
|
+
"history; narrow it to a sub-day window when a finding from a session that "
|
|
493
|
+
"ended yesterday is noise rather than news. A window none of the page's "
|
|
494
|
+
"buttons names gets one of its own, so it stays reachable after you have "
|
|
495
|
+
"clicked away from it.",
|
|
469
496
|
)
|
|
470
497
|
@click.option(
|
|
471
498
|
"--bom",
|
stacktrace_cli/daemon/cli.py
CHANGED
|
@@ -9,6 +9,7 @@ import sys
|
|
|
9
9
|
import click
|
|
10
10
|
|
|
11
11
|
from .paths import RuntimePaths
|
|
12
|
+
from .presentation import render_finding
|
|
12
13
|
from .store import FindingStore, SessionKey
|
|
13
14
|
|
|
14
15
|
|
|
@@ -100,7 +101,4 @@ def findings(agent_kind: str, session_id: str, output_format: str) -> None:
|
|
|
100
101
|
click.echo("No Stacktrace findings for this session.")
|
|
101
102
|
return
|
|
102
103
|
for finding in stored:
|
|
103
|
-
click.echo(
|
|
104
|
-
f"{str(finding.get('severity', 'unknown')).upper()} "
|
|
105
|
-
f"{finding.get('rule_id', 'unknown-rule')}: {finding.get('title', 'Finding')}"
|
|
106
|
-
)
|
|
104
|
+
click.echo(render_finding(finding))
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Explain durable findings without reintroducing discarded transcript prose."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
from stacktrace_cli.detector.blocked import REASON_CODES, reason_for
|
|
8
|
+
|
|
9
|
+
# Notification prose is selected by rule ID only (ADR-0031). In particular,
|
|
10
|
+
# finding titles, component names and evidence never supply notification text.
|
|
11
|
+
_GUIDANCE = {
|
|
12
|
+
"stacktrace-credential-egress": (
|
|
13
|
+
"Credential-shaped material appeared in an outbound tool call",
|
|
14
|
+
(
|
|
15
|
+
"Treat the credential as potentially exposed: revoke or rotate it, "
|
|
16
|
+
"then inspect why the outbound tool received it."
|
|
17
|
+
),
|
|
18
|
+
),
|
|
19
|
+
"stacktrace-deceptive-completion": (
|
|
20
|
+
"A success claim conflicts with verification evidence",
|
|
21
|
+
"Inspect the cited verification results before relying on the completion claim.",
|
|
22
|
+
),
|
|
23
|
+
"stacktrace-agent-blocked": (
|
|
24
|
+
"The agent encountered a block while working",
|
|
25
|
+
"Inspect the recorded block reason before retrying the task.",
|
|
26
|
+
),
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def rule_guidance(rule_id: str) -> tuple[str, str]:
|
|
31
|
+
return _GUIDANCE.get(
|
|
32
|
+
rule_id,
|
|
33
|
+
("A finding was recorded", "Inspect the retained evidence in the local session."),
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def render_finding(finding: dict[str, object]) -> str:
|
|
38
|
+
"""Render the old, title-less ledger as well as newly recorded findings."""
|
|
39
|
+
rule_id = str(finding.get("rule_id", "unknown-rule"))
|
|
40
|
+
title, remedy = rule_guidance(rule_id)
|
|
41
|
+
evidence = finding.get("evidence", [])
|
|
42
|
+
items = (
|
|
43
|
+
[item for item in evidence if isinstance(item, dict)] if isinstance(evidence, list) else []
|
|
44
|
+
)
|
|
45
|
+
if rule_id == "stacktrace-agent-blocked" and items:
|
|
46
|
+
code = items[0].get("kind")
|
|
47
|
+
if isinstance(code, str) and code in REASON_CODES:
|
|
48
|
+
reason = reason_for(code)
|
|
49
|
+
title, remedy = reason.title, reason.remedy
|
|
50
|
+
|
|
51
|
+
lines = [
|
|
52
|
+
f"{str(finding.get('severity', 'unknown')).upper()} {rule_id}: {title}",
|
|
53
|
+
f" Confidence: {finding.get('confidence', 'unknown')}",
|
|
54
|
+
]
|
|
55
|
+
for item in items:
|
|
56
|
+
# Quote stored descriptors as data, including escaped terminal controls.
|
|
57
|
+
kind = json.dumps(item.get("kind"), ensure_ascii=True)
|
|
58
|
+
detail = json.dumps(item.get("detail"), ensure_ascii=True)
|
|
59
|
+
span = json.dumps(item.get("span"), ensure_ascii=True)
|
|
60
|
+
lines.extend((f" Evidence ({kind}): {detail}", f" Location: {span}"))
|
|
61
|
+
if not items:
|
|
62
|
+
lines.append(" Evidence: no descriptors were retained.")
|
|
63
|
+
lines.append(f" Next step: {remedy}")
|
|
64
|
+
lines.append(
|
|
65
|
+
" Tool/component names and raw call content are not retained in this ledger; "
|
|
66
|
+
"inspect the cited locations in the local session for attribution."
|
|
67
|
+
)
|
|
68
|
+
return "\n".join(lines)
|
stacktrace_cli/daemon/store.py
CHANGED
|
@@ -17,6 +17,8 @@ from stacktrace_cli.detector.rules import scope_of
|
|
|
17
17
|
from stacktrace_cli.private_state import ensure_private_directory
|
|
18
18
|
from stacktrace_cli.telemetry import emit_derived
|
|
19
19
|
|
|
20
|
+
from .presentation import rule_guidance
|
|
21
|
+
|
|
20
22
|
|
|
21
23
|
@dataclass(frozen=True)
|
|
22
24
|
class SessionKey:
|
|
@@ -266,10 +268,16 @@ class FindingStore:
|
|
|
266
268
|
identity = f"{session.agent_kind}\0{session.session_id}\0{rule_id}\0{anchor}".encode()
|
|
267
269
|
return f"stn_{hashlib.sha256(identity).hexdigest()[:24]}"
|
|
268
270
|
|
|
271
|
+
#: Severities that earn a notification. `Severity` is `low | medium | high`;
|
|
272
|
+
#: `critical` is carried because the gate has always tested for it, though
|
|
273
|
+
#: no rule emits it. `low` stays silent.
|
|
274
|
+
_NOTIFIABLE_SEVERITIES = frozenset({"medium", "high", "critical"})
|
|
275
|
+
|
|
269
276
|
@staticmethod
|
|
270
277
|
def _eligible(finding: dict[str, object]) -> bool:
|
|
271
278
|
return (
|
|
272
|
-
finding.get("severity") in
|
|
279
|
+
finding.get("severity") in FindingStore._NOTIFIABLE_SEVERITIES
|
|
280
|
+
and finding.get("confidence") == "high"
|
|
273
281
|
)
|
|
274
282
|
|
|
275
283
|
@staticmethod
|
|
@@ -302,13 +310,17 @@ class FindingStore:
|
|
|
302
310
|
severity = finding["severity"]
|
|
303
311
|
assert isinstance(rule_id, str)
|
|
304
312
|
assert isinstance(severity, str)
|
|
313
|
+
explanation, _ = rule_guidance(rule_id)
|
|
305
314
|
return Notification(
|
|
306
315
|
event_id=event_id,
|
|
307
316
|
session=session,
|
|
308
317
|
rule_id=rule_id,
|
|
309
318
|
severity=severity,
|
|
310
|
-
title=f"Stacktrace
|
|
311
|
-
body=
|
|
319
|
+
title=f"Stacktrace: {explanation}",
|
|
320
|
+
body=(
|
|
321
|
+
f"Severity: {severity}. Rule: {rule_id}. "
|
|
322
|
+
"Run /stacktrace:findings for evidence and next steps."
|
|
323
|
+
),
|
|
312
324
|
)
|
|
313
325
|
|
|
314
326
|
@staticmethod
|
|
@@ -276,9 +276,13 @@ def _blocked_remediation(
|
|
|
276
276
|
if reason.terminal:
|
|
277
277
|
parts.append("It holds for the rest of the session, so retrying only spends the run.")
|
|
278
278
|
if repeated:
|
|
279
|
+
# How many and for how long, never between which turns. The ordinal is
|
|
280
|
+
# the detector's cursor and reads here as an address a reader could
|
|
281
|
+
# follow; the run's extent is already on the card as a clock, and the
|
|
282
|
+
# positions themselves now travel per citation as `Evidence.turn`.
|
|
279
283
|
parts.append(
|
|
280
284
|
f"The agent sat on it, saying the same thing {len(repeated)} times over, "
|
|
281
|
-
f"
|
|
285
|
+
f"doing no work in between."
|
|
282
286
|
)
|
|
283
287
|
scale = "once" if events == 1 else f"{events} times"
|
|
284
288
|
parts.append(
|
|
@@ -384,7 +388,7 @@ def _agent_blocked(session: SessionLike, ref: SessionRef) -> tuple[Detection, ..
|
|
|
384
388
|
span=_block_span(session, turn),
|
|
385
389
|
kind=reason.code,
|
|
386
390
|
detail=_blocked_detail(
|
|
387
|
-
|
|
391
|
+
"in a reply that made no call",
|
|
388
392
|
reason.specific(turn.text),
|
|
389
393
|
),
|
|
390
394
|
)
|
|
@@ -441,7 +445,7 @@ def _agent_blocked(session: SessionLike, ref: SessionRef) -> tuple[Detection, ..
|
|
|
441
445
|
span=call.span,
|
|
442
446
|
kind=reason.code,
|
|
443
447
|
detail=_blocked_detail(
|
|
444
|
-
|
|
448
|
+
tool_name_for(call),
|
|
445
449
|
reason.specific(_block_body(call)),
|
|
446
450
|
),
|
|
447
451
|
)
|
|
@@ -193,6 +193,21 @@ class Evidence:
|
|
|
193
193
|
#: none -- or where the transcript dated nothing. Attached at the same seam
|
|
194
194
|
#: as the detection's range, never by a rule.
|
|
195
195
|
turn_occurred_at: datetime | None = None
|
|
196
|
+
#: Which turn issued this span, as the session's own ordinal.
|
|
197
|
+
#:
|
|
198
|
+
#: **A field rather than a phrase, and that is the whole point of it.** Two
|
|
199
|
+
#: rules used to write `turn {position}` into `detail`, where it read as an
|
|
200
|
+
#: address a reader could follow. It is not one -- nothing a consumer draws
|
|
201
|
+
#: numbers turns -- and on a measured corpus the ordinals ran past a
|
|
202
|
+
#: thousand while the commonest citation on the page spent a third of its
|
|
203
|
+
#: length on one. Here it is what it actually is: the session's ordering of
|
|
204
|
+
#: its own work, for a consumer that wants to sort or to say how far in
|
|
205
|
+
#: something happened, and never prose.
|
|
206
|
+
#:
|
|
207
|
+
#: `None` where the citation resolves to no turn, on the same terms as
|
|
208
|
+
#: `turn_occurred_at` -- and set by a rule rather than at that seam, because
|
|
209
|
+
#: unlike the clock it is already in the rule's hand when it cites.
|
|
210
|
+
turn: int | None = None
|
|
196
211
|
|
|
197
212
|
|
|
198
213
|
def _is_unit(value: object) -> bool:
|
|
@@ -1698,16 +1698,19 @@ def _cited_call(span: str, session: SessionLike) -> tuple[str, int] | None:
|
|
|
1698
1698
|
def _verdict_evidence(
|
|
1699
1699
|
span: str, confidence: str, session: SessionLike, scores: Mapping[str, float]
|
|
1700
1700
|
) -> str:
|
|
1701
|
-
"""One row: which call,
|
|
1701
|
+
"""One row: which call, and how sure the analyzer was.
|
|
1702
1702
|
|
|
1703
1703
|
The score is included where there is one. A windowed analyzer answers with
|
|
1704
1704
|
a probability per span and nothing else, so it is the only quantity behind
|
|
1705
1705
|
the grade — and a reader comparing three cited calls wants to know which of
|
|
1706
1706
|
them the analyzer was least comfortable with.
|
|
1707
|
+
|
|
1708
|
+
The turn's position used to be in this sentence and is now `Evidence.turn`.
|
|
1709
|
+
It reads as an address and is not one; see that field.
|
|
1707
1710
|
"""
|
|
1708
1711
|
grade = _VERDICT_DETAIL[confidence]
|
|
1709
1712
|
named = _cited_call(span, session)
|
|
1710
|
-
where =
|
|
1713
|
+
where = named[0] if named else "a call this session made"
|
|
1711
1714
|
score = scores.get(span)
|
|
1712
1715
|
return f"{where} — {grade}" + (f" (scored {score:.2f})" if score is not None else "")
|
|
1713
1716
|
|
|
@@ -139,6 +139,10 @@ def detection_json(detection) -> dict[str, object]:
|
|
|
139
139
|
"kind": e.kind,
|
|
140
140
|
"detail": e.detail,
|
|
141
141
|
"occurred_at": e.turn_occurred_at.isoformat() if e.turn_occurred_at else None,
|
|
142
|
+
# The position, beside the sentence rather than inside it. A
|
|
143
|
+
# consumer that wants to say how far into a session something
|
|
144
|
+
# happened reads this; `detail` stays a sentence.
|
|
145
|
+
"turn": e.turn,
|
|
142
146
|
}
|
|
143
147
|
for e in detection.evidence
|
|
144
148
|
],
|
stacktrace_cli/detector/run.py
CHANGED
|
@@ -84,7 +84,7 @@ from stacktrace_cli.detector.reasoning import (
|
|
|
84
84
|
)
|
|
85
85
|
from stacktrace_cli.detector.secrets import find_secrets
|
|
86
86
|
from stacktrace_cli.sessions.index import turns_by_span
|
|
87
|
-
from stacktrace_cli.sessions.protocols import ToolCallLike
|
|
87
|
+
from stacktrace_cli.sessions.protocols import ToolCallLike, TurnLike
|
|
88
88
|
|
|
89
89
|
#: The analyzers a run may be asked for, by the name `--analyzer` takes.
|
|
90
90
|
#: `jev` is TypeSafe's Jev, a boundary the flag must name (ADR-0033).
|
|
@@ -264,11 +264,18 @@ class DetectorRun:
|
|
|
264
264
|
analyzer: str = NO_ANALYZER
|
|
265
265
|
|
|
266
266
|
|
|
267
|
-
#: A finding paired with its producing session's own span -> turn
|
|
267
|
+
#: A finding paired with its producing session's own span -> turn map.
|
|
268
268
|
#: The pairing is the invariant: a producer cannot add a finding to the run
|
|
269
|
-
#: without saying which session's
|
|
269
|
+
#: without saying which session's turns place it, because the list holds no
|
|
270
270
|
#: other shape and `pyright` refuses a bare `Detection`.
|
|
271
|
-
|
|
271
|
+
#:
|
|
272
|
+
#: The turn itself rather than its time, because a citation carries two facts
|
|
273
|
+
#: derived from it -- when it ran and where in the session it sat -- and two
|
|
274
|
+
#: maps would be two chances for a producer to supply one and not the other.
|
|
275
|
+
#: That is exactly what happened while the position was set per rule: two of
|
|
276
|
+
#: the five producers never set it, and `None` stopped meaning "resolves to no
|
|
277
|
+
#: turn" and started also meaning "nobody wired this site up".
|
|
278
|
+
Produced = tuple[Detection, Mapping[str, TurnLike]]
|
|
272
279
|
|
|
273
280
|
|
|
274
281
|
def _finish(produced: Sequence[Produced]) -> tuple[Detection, ...]:
|
|
@@ -298,9 +305,14 @@ def _finish(produced: Sequence[Produced]) -> tuple[Detection, ...]:
|
|
|
298
305
|
range at all, which is why the two bounds are always null together.
|
|
299
306
|
"""
|
|
300
307
|
finished: list[Detection] = []
|
|
301
|
-
for detection,
|
|
308
|
+
for detection, turns in produced:
|
|
302
309
|
evidence = tuple(
|
|
303
|
-
replace(
|
|
310
|
+
replace(
|
|
311
|
+
item,
|
|
312
|
+
turn_occurred_at=(found := turns.get(item.span)) and found.occurred_at,
|
|
313
|
+
turn=found.position if found else None,
|
|
314
|
+
)
|
|
315
|
+
for item in detection.evidence
|
|
304
316
|
)
|
|
305
317
|
moments = sorted(item.turn_occurred_at for item in evidence if item.turn_occurred_at)
|
|
306
318
|
finished.append(
|
|
@@ -359,7 +371,7 @@ def run_detector(
|
|
|
359
371
|
#: does; this dict is built and drained within one `run_detector` call,
|
|
360
372
|
#: over sessions `correlated.sessions` holds live throughout, so nothing
|
|
361
373
|
#: here is freed and reused before the lookups that need it are done.
|
|
362
|
-
|
|
374
|
+
turns_by_session: dict[int, dict[str, TurnLike]] = {}
|
|
363
375
|
|
|
364
376
|
for session in correlated.sessions:
|
|
365
377
|
stage_one, stage_one_unknowns = run_deterministic(session.session)
|
|
@@ -370,16 +382,12 @@ def run_detector(
|
|
|
370
382
|
# rule that never sees the record cannot say.
|
|
371
383
|
coverage = Coverage(total=session.coverage.total, resolved=session.coverage.resolved)
|
|
372
384
|
index = _SessionIndex.of(session)
|
|
373
|
-
|
|
374
|
-
span: turn.occurred_at for span, turn in turns_by_span(session.session).items()
|
|
375
|
-
}
|
|
385
|
+
session_turns = dict(turns_by_span(session.session))
|
|
376
386
|
# A narration-only block cites a span no call ever carries
|
|
377
387
|
# (`block_spans_by_turn`), so it resolves to no turn here unless it is
|
|
378
388
|
# added the same way a call-backed one already is.
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
)
|
|
382
|
-
times_by_session[id(session.session)] = session_times
|
|
389
|
+
session_turns.update(block_spans_by_turn(session.session))
|
|
390
|
+
turns_by_session[id(session.session)] = session_turns
|
|
383
391
|
stage_one = tuple(
|
|
384
392
|
replace(
|
|
385
393
|
d,
|
|
@@ -390,7 +398,7 @@ def run_detector(
|
|
|
390
398
|
)
|
|
391
399
|
for d in stage_one
|
|
392
400
|
)
|
|
393
|
-
produced.extend((d,
|
|
401
|
+
produced.extend((d, session_turns) for d in stage_one)
|
|
394
402
|
unknowns.extend(stage_one_unknowns)
|
|
395
403
|
|
|
396
404
|
stage_two_unknowns, request = run_priors(session, stage_one)
|
|
@@ -554,7 +562,7 @@ def run_detector(
|
|
|
554
562
|
),
|
|
555
563
|
):
|
|
556
564
|
produced.extend(
|
|
557
|
-
(d,
|
|
565
|
+
(d, turns_by_session.get(id(session), {})) for d in _from_cache(cached, request)
|
|
558
566
|
)
|
|
559
567
|
maps.extend(_maps_from_cache(cached, request))
|
|
560
568
|
cache_hits += 1
|
|
@@ -660,7 +668,7 @@ def run_detector(
|
|
|
660
668
|
for position, outcome in sorted(outcomes):
|
|
661
669
|
request, session, chosen = runnable[position]
|
|
662
670
|
produced.extend(
|
|
663
|
-
(d,
|
|
671
|
+
(d, turns_by_session.get(id(session), {}))
|
|
664
672
|
for d in outcome.findings # type: ignore[attr-defined]
|
|
665
673
|
)
|
|
666
674
|
unknowns.extend(outcome.unknowns) # type: ignore[attr-defined]
|
stacktrace_cli/monitor/render.py
CHANGED
|
@@ -173,8 +173,26 @@ _TOPIC_CHARS = 90
|
|
|
173
173
|
|
|
174
174
|
|
|
175
175
|
def _topic(session: Any) -> str | None:
|
|
176
|
-
"""
|
|
177
|
-
|
|
176
|
+
"""What to call a session: its own title, else its opening request.
|
|
177
|
+
|
|
178
|
+
**The title first, because it was chosen to identify the session** -- by
|
|
179
|
+
the person or by the agent -- while an opening request merely happens to be
|
|
180
|
+
first. Measured on a 76-session corpus the prompt is frequently the worse
|
|
181
|
+
label by a wide margin: the orchestrator opened with the word "hello", and
|
|
182
|
+
its 75 sub-agents all opened with the same harness template about fetching
|
|
183
|
+
origin/main, so the prompt told two of them apart in neither direction. The
|
|
184
|
+
one session carrying a title was the one whose prompt was "hello".
|
|
185
|
+
|
|
186
|
+
Both are `LOCAL_ONLY` upstream and stay on this page only, the same
|
|
187
|
+
asymmetry `working_directory` has (ADR-0006).
|
|
188
|
+
|
|
189
|
+
Read through `getattr`, like `working_directory` below it: `title` arrives
|
|
190
|
+
with a version of OpenAIDR newer than the pinned one, and a session without
|
|
191
|
+
it falls through to the prompt exactly as before rather than raising.
|
|
192
|
+
"""
|
|
193
|
+
text = getattr(session, "title", None)
|
|
194
|
+
if not text:
|
|
195
|
+
text = getattr(session, "initial_prompt", None)
|
|
178
196
|
if not text:
|
|
179
197
|
text = next(
|
|
180
198
|
(
|