stacktrace-cli 0.2.3__py3-none-any.whl → 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,3 @@
1
1
  """The `stacktrace` command-line interface — detection and response for AI agents."""
2
2
 
3
- __version__ = "0.2.3"
3
+ __version__ = "0.3.0"
@@ -106,7 +106,7 @@ def analyse(
106
106
  project_map: tuple[str, ...] = (),
107
107
  root: Path | None = None,
108
108
  session_ids: tuple[str, ...] = (),
109
- escalate: bool = False,
109
+ reasoning: bool = False,
110
110
  budget: int = DEFAULT_BUDGET,
111
111
  sample_budget: int = DEFAULT_SAMPLE_BUDGET,
112
112
  cache: VerdictCache | None = None,
@@ -123,7 +123,7 @@ def analyse(
123
123
  pass; a one-shot command passes nothing and gets the default.
124
124
 
125
125
  `session_ids` narrows the window to named sessions. Empty means the whole
126
- window, which is what every command passes; monitor's escalate button is
126
+ window, which is what every command passes; monitor's reasoning button is
127
127
  what needs the narrowing, and needs it to be structural.
128
128
 
129
129
  `since` is the window's spelling — `parse_since` reads it — or the cutoff
@@ -142,7 +142,7 @@ def analyse(
142
142
  )
143
143
  run = run_detector(
144
144
  acquired.view,
145
- escalate=escalate,
145
+ reasoning=reasoning,
146
146
  budget=budget,
147
147
  sample_budget=sample_budget,
148
148
  cache=cache,
@@ -287,11 +287,11 @@ def analyse_progressively(
287
287
  exists to prevent.
288
288
 
289
289
  `cache` is read and never written. A session graded by an earlier
290
- `detect --escalate` shows that grade here; this path never commissions one.
290
+ `detect --reasoning` shows that grade here; this path never commissions one.
291
291
 
292
- **No `escalate` parameter, deliberately.** Stage 3's budget is per *run*, so
292
+ **No `reasoning` parameter, deliberately.** Stage 3's budget is per *run*, so
293
293
  judging in batches would give each batch its own budget and spend a multiple
294
- of what was authorised. A streaming escalation needs a budget shared across
294
+ of what was authorised. A streaming reasoning run needs a budget shared across
295
295
  batches; until it has one, this path does not offer the option rather than
296
296
  offering it wrongly.
297
297
  """
@@ -358,8 +358,8 @@ def analyse_progressively(
358
358
  for index in range(0, len(ordered), step):
359
359
  judged = run_detector(
360
360
  CorrelatedView(sessions=tuple(ordered[index : index + step])),
361
- # Read, never written, and never escalating: a session an earlier
362
- # `detect --escalate` graded renders with that grade here, and this
361
+ # Read, never written, and never requesting: a session an earlier
362
+ # `detect --reasoning` graded renders with that grade here, and this
363
363
  # path commissions nothing (ADR-0026 clause 2).
364
364
  cache=cache,
365
365
  )
@@ -367,7 +367,7 @@ def analyse_progressively(
367
367
  detections=accumulated.detections + judged.detections,
368
368
  unknowns=accumulated.unknowns + judged.unknowns,
369
369
  sessions=accumulated.sessions + judged.sessions,
370
- escalated=accumulated.escalated + judged.escalated,
370
+ requested=accumulated.requested + judged.requested,
371
371
  analysed=accumulated.analysed + judged.analysed,
372
372
  cache_hits=accumulated.cache_hits + judged.cache_hits,
373
373
  collection_failures=failures,
stacktrace_cli/cli.py CHANGED
@@ -223,13 +223,14 @@ def sessions(
223
223
  help="Show every detection and all of its evidence. Default: the summary alone.",
224
224
  )
225
225
  @click.option(
226
- "--escalate/--no-escalate",
226
+ "--reasoning",
227
+ is_flag=True,
227
228
  default=False,
228
- show_default=True,
229
- help="Send flagged sessions to the agent's own CLI for semantic analysis. Off "
230
- "by default: it is the only stage that sends anything session-derived off "
231
- "this machine, and it spends the developer's own provider quota, so a run "
232
- "that does neither is the one to reach for first. --escalate turns it on "
229
+ help="Analyse flagged sessions with a reasoning model, run through the "
230
+ "agent's own CLI. Off by default: it is the only stage that sends anything "
231
+ "session-derived off this machine, and it spends the developer's own "
232
+ "provider quota, so a run that does neither is the one to reach for first. "
233
+ "--reasoning turns it on "
233
234
  "and --budget caps a noisy day. Leaving it off does not make the run "
234
235
  "offline: correlation runs first either way and matches advisories by "
235
236
  "sending the package coordinates of the components a session invoked to "
@@ -281,7 +282,7 @@ def detect(
281
282
  bom_paths: tuple[Path, ...],
282
283
  output_format: str,
283
284
  detail: bool,
284
- escalate: bool,
285
+ reasoning: bool,
285
286
  budget: int,
286
287
  sample_budget: int,
287
288
  cache: bool,
@@ -299,7 +300,7 @@ def detect(
299
300
 
300
301
  Two stages need no model or credential, and they are the two that run by
301
302
  default. The third sends flagged sessions to the agent's own CLI:
302
- --escalate turns it on, capped by --budget. Sessions that qualified for it
303
+ --reasoning turns it on, capped by --budget. Sessions that qualified for it
303
304
  are counted either way, and named in --format json, so a run without it
304
305
  says so rather than reading as a clean one.
305
306
 
@@ -307,10 +308,10 @@ def detect(
307
308
  the agent's own CLI hands the prompts, arguments and results a rule needs
308
309
  to the provider it is already authenticated against (ADR-0004 -- provider
309
310
  affinity, so a transcript goes back to the vendor that produced it, and
310
- there is no fallback to any other). Without --escalate the run is the two
311
+ there is no fallback to any other). Without --reasoning the run is the two
311
312
  stages that send nothing.
312
313
 
313
- One network call happens before any of them, with or without --escalate:
314
+ One network call happens before any of them, with or without --reasoning:
314
315
  correlation matches advisories by sending the package coordinates of the
315
316
  components a session invoked to osv.dev. Coordinates only -- never a
316
317
  prompt, an argument or a result.
@@ -325,11 +326,11 @@ def detect(
325
326
  bom_paths=bom_paths,
326
327
  project_map=project_map,
327
328
  root=root,
328
- escalate=escalate,
329
+ reasoning=reasoning,
329
330
  budget=budget,
330
331
  sample_budget=sample_budget,
331
- # Not gated on `escalate`. A verdict already paid for is an
332
- # answer this run has, and `escalate` governs whether new ones may
332
+ # Not gated on `reasoning`. A verdict already paid for is an
333
+ # answer this run has, and `reasoning` governs whether new ones may
333
334
  # be commissioned, not whether old ones may be read -- the rule
334
335
  # `run_detector` states at its cache pass. Gating it here made a
335
336
  # session read as graded or ungraded depending on a flag that
@@ -368,11 +369,11 @@ def detect(
368
369
  "entirely when nothing on disk has changed.",
369
370
  )
370
371
  @click.option(
371
- "--escalate/--no-escalate",
372
+ "--reasoning",
373
+ is_flag=True,
372
374
  default=False,
373
- show_default=True,
374
- help="Let the page send one session to the agent's own CLI for semantic "
375
- "analysis. Off by default: stage 3 spends your provider quota, and a page "
375
+ help="Let the page analyse one session with a reasoning model, run through "
376
+ "the agent's own CLI. Off by default: stage 3 spends your provider quota, and a page "
376
377
  "left open is the wrong place for that to be implicit. Capped by --budget.",
377
378
  )
378
379
  @click.option(
@@ -380,7 +381,7 @@ def detect(
380
381
  type=click.IntRange(min=1),
381
382
  default=DEFAULT_BUDGET,
382
383
  show_default=True,
383
- help="Escalations this monitor may spend in total, if --escalate is on.",
384
+ help="Reasoning runs this monitor may spend in total, if --reasoning is on.",
384
385
  )
385
386
  @click.option(
386
387
  "--no-open", is_flag=True, default=False, help="Print the URL, do not open a browser."
@@ -421,7 +422,7 @@ def monitor(
421
422
  port: int,
422
423
  host: str,
423
424
  interval: float,
424
- escalate: bool,
425
+ reasoning: bool,
425
426
  budget: int,
426
427
  no_open: bool,
427
428
  agent_kinds: tuple[str, ...],
@@ -439,7 +440,7 @@ def monitor(
439
440
 
440
441
  Loopback only, and free to leave open: a pass is skipped when nothing has
441
442
  changed, advisory lookups are asked once per component, and the two stages
442
- that need no model are the only ones that run. --escalate adds a button
443
+ that need no model are the only ones that run. --reasoning adds a button
443
444
  that spends your provider quota, off unless asked for.
444
445
  """
445
446
  try:
@@ -447,7 +448,7 @@ def monitor(
447
448
  host=host,
448
449
  port=port,
449
450
  interval=interval,
450
- escalate=escalate,
451
+ reasoning=reasoning,
451
452
  budget=budget,
452
453
  open_browser=not no_open,
453
454
  echo=click.echo,
@@ -64,7 +64,7 @@ _TIMEOUT = 180
64
64
  #: run /login"* and exits 1, on a machine whose CLI is authenticated and working.
65
65
  #: That is the same premise `--bare` is avoided for (ADR-0004: the developer's
66
66
  #: own CLI, already authenticated), broken through the environment instead of
67
- #: through a flag. The failure mode is the dangerous kind: every escalation
67
+ #: through a flag. The failure mode is the dangerous kind: every reasoning request
68
68
  #: returns `analyzer_error`, the pipeline records `Unknown`, and stage 3 looks
69
69
  #: like it ran. `LOGNAME` accompanies it as the same identity on systems that
70
70
  #: set that instead.
@@ -3,7 +3,7 @@
3
3
  Authorized by ADR-0011, which amends ADR-0006's persistence clause. The argument
4
4
  is narrow: a reasoning verdict is an **expensive pure function of an immutable,
5
5
  ended session**. `stacktrace detect` is a command a person runs repeatedly, each
6
- escalation costs roughly $0.09 and 9 seconds, and the answer cannot change once
6
+ reasoning request costs roughly $0.09 and 9 seconds, and the answer cannot change once
7
7
  the session has stopped growing.
8
8
 
9
9
  Three properties carry the whole design, and each exists because its absence was
@@ -116,7 +116,7 @@ class CachedVerdict:
116
116
 
117
117
  @dataclass(frozen=True)
118
118
  class CachedOutcome:
119
- """One escalation's stored answer: provenance, plus whichever rules fired.
119
+ """One reasoning request's stored answer: provenance, plus whichever rules fired.
120
120
 
121
121
  Provenance sits here rather than on each verdict because ADR-0011
122
122
  constraint 1 lists it as part of what an entry *carries*, not only what it is
@@ -132,7 +132,7 @@ class CachedOutcome:
132
132
 
133
133
  def cache_key(
134
134
  session: SessionLike,
135
- escalation: object,
135
+ request: object,
136
136
  rule_ids: Sequence[str],
137
137
  prompt_version: str,
138
138
  identity: tuple[str, str],
@@ -143,17 +143,17 @@ def cache_key(
143
143
  ADR-0011 constraint 3 states them apart, and neither is a field `serialise()`
144
144
  shows the analyzer — so a fingerprint built only from analyzer-visible
145
145
  content would let two distinct sessions with byte-identical rendered turns
146
- and escalation context share one verdict.
146
+ and reasoning-request context share one verdict.
147
147
 
148
148
  **Reasons and spans are hashed in their given order, never sorted.**
149
149
  `_why()` joins each collection in order and puts the result straight into
150
150
  the prompt, so the same members in a different order are two different
151
151
  prompts. Sorting first would merge them.
152
152
 
153
- **Which rule cited which span, not only the merged set.** `escalation.spans`
154
- is a deduplicated union over every escalating reason, so it cannot say
153
+ **Which rule cited which span, not only the merged set.** `request.spans`
154
+ is a deduplicated union over every requesting reason, so it cannot say
155
155
  whether a span was the injection marker's evidence or the credential rule's
156
- — and `serialise_injection()` reads exactly that, from `escalation.stage_one`,
156
+ — and `serialise_injection()` reads exactly that, from `request.stage_one`,
157
157
  to decide which results and arguments the transcript carries. Two builds
158
158
  that attribute overlapping spans differently produce the same union and
159
159
  wholly different transcripts, so without this a verdict computed from one
@@ -181,12 +181,12 @@ def cache_key(
181
181
  {
182
182
  "session_id": session.session_id,
183
183
  "turn_count": session.turn_count,
184
- "reasons": list(getattr(escalation, "reasons", ()) or ()),
185
- "spans": list(getattr(escalation, "spans", ()) or ()),
186
- "stage_one": _cited_per_rule(escalation),
184
+ "reasons": list(getattr(request, "reasons", ()) or ()),
185
+ "spans": list(getattr(request, "spans", ()) or ()),
186
+ "stage_one": _cited_per_rule(request),
187
187
  "coverage": [
188
- getattr(getattr(escalation, "coverage", None), "total", None),
189
- getattr(getattr(escalation, "coverage", None), "resolved", None),
188
+ getattr(getattr(request, "coverage", None), "total", None),
189
+ getattr(getattr(request, "coverage", None), "resolved", None),
190
190
  ],
191
191
  "prompt_version": prompt_version,
192
192
  "verdict_logic_version": _VERDICT_LOGIC_VERSION,
@@ -201,7 +201,7 @@ def cache_key(
201
201
  return digest.hexdigest()
202
202
 
203
203
 
204
- def _cited_per_rule(escalation: object) -> list[list[object]]:
204
+ def _cited_per_rule(request: object) -> list[list[object]]:
205
205
  """Which stage-one rule cited which spans, in the order they were found.
206
206
 
207
207
  Descriptors only, and only the two a prompt reads: `serialise_injection()`
@@ -212,13 +212,13 @@ def _cited_per_rule(escalation: object) -> list[list[object]]:
212
212
 
213
213
  Order is preserved rather than sorted. Nothing downstream reads it — the
214
214
  spans become a set — so the worst a preserved order can do is separate two
215
- escalations that would have produced one prompt, which costs a miss. The
215
+ requests that would have produced one prompt, which costs a miss. The
216
216
  opposite mistake merges two prompts under one key, and that is the one
217
217
  ADR-0011 constraint 5 forbids.
218
218
  """
219
219
  return [
220
220
  [finding.rule_id, [evidence.span for evidence in finding.evidence]]
221
- for finding in getattr(escalation, "stage_one", ()) or ()
221
+ for finding in getattr(request, "stage_one", ()) or ()
222
222
  ]
223
223
 
224
224
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  Every rule here is one a responder can verify by looking at the session. That is
4
4
  the bar for this stage, and it is why the stage emits directly rather than
5
- escalating: nothing a slower stage adds changes *this string looks like a
5
+ requesting reasoning: nothing a slower stage adds changes *this string looks like a
6
6
  credential and it went to an outbound call*.
7
7
 
8
8
  Two asymmetries carry most of the precision:
@@ -150,7 +150,7 @@ class Evidence:
150
150
  class Verdict:
151
151
  """What produced a finding, so two machines' answers are comparable.
152
152
 
153
- Recorded for every stage, not only escalated ones: knowing a finding came
153
+ Recorded for every stage, not only requested ones: knowing a finding came
154
154
  from `priors` rather than `reasoning` is what tells a reader whether a model
155
155
  was involved at all.
156
156
  """
@@ -340,9 +340,9 @@ class Unknown:
340
340
  reason: UnknownReason
341
341
  spans: tuple[str, ...] = ()
342
342
  detail: str = ""
343
- #: For an escalation that qualified for stage 3 but never reached it, the
343
+ #: For a reasoning request that qualified for stage 3 but never reached it, the
344
344
  #: priors that would have been analysed. Kept so a budget-limited or
345
- #: `--no-escalate` run still shows *why* the session was of interest.
345
+ #: run without `--reasoning` still shows *why* the session was of interest.
346
346
  reasons: tuple[str, ...] = field(default_factory=tuple)
347
347
  #: Reasoning-stage only: whether this unknown was raised **after** the
348
348
  #: session was actually sent to the analyzer. `run_detector` counts a
@@ -27,7 +27,7 @@ from data to instruction.
27
27
 
28
28
  Finding the characters is not finding an attack, and the difference is
29
29
  measurable. On 621 real sessions this rule fired three times — the only
30
- evidence-backed escalations in the whole corpus — and all three were one file:
30
+ evidence-backed reasoning requests in the whole corpus — and all three were one file:
31
31
  another agent-security tool's own detector, whose source reads
32
32
  `_BIDI_OVERRIDE_CHARS = frozenset('\u202d\u202e')`. Security work names the
33
33
  alphabet of the attacks it looks for, and so do specifications, test fixtures
@@ -14,14 +14,14 @@ bounds it:
14
14
  - **An unreadable connection log is the same kind of bound.** Egress over MCP is
15
15
  observed from `transport` alone, so a log that never applied is silent in
16
16
  exactly the way a local call is — `_missing_mcp_transport` says which it was.
17
- - **Nothing here escalates.** The routes into stage three are stage-one facts;
17
+ - **Nothing here requests reasoning.** The routes into stage three are stage-one facts;
18
18
  an advisory is already settled by the record, and spending inference to
19
19
  confirm it would buy nothing.
20
20
 
21
21
  **Four rules have been removed from this stage over two decisions.** ADR-0013
22
22
  withdrew `tool-shadowing`, which needs an `ambiguous` outcome the join has never
23
- emitted, and `unpinned-invoked`, which could only escalate in combination with
24
- it; the combination escalation route went with them. ADR-0015 withdrew
23
+ emitted, and `unpinned-invoked`, which could only request reasoning in combination
24
+ with it; the combination route went with them. ADR-0015 withdrew
25
25
  `unsanctioned-mcp-tool-use` — and with it the `uninventoried_component` unknown
26
26
  that reported the same fact for every other component type — and
27
27
  `capability-crossing`, which ADR-0013 had deliberately kept on the argument that
@@ -58,7 +58,7 @@ _VERDICT = Verdict(stage="priors")
58
58
  #: not read enough of it either to clear or to fault it.
59
59
  _COVERAGE_FLOOR = 0.5
60
60
 
61
- #: Stage-one rules that escalate on their own. Each names a question only
61
+ #: Stage-one rules that request reasoning on their own. Each names a question only
62
62
  #: `reasoning` can settle, and each would otherwise be unreachable:
63
63
  #:
64
64
  #: - a marker asks whether an injected instruction was *acted on*, which nothing
@@ -84,7 +84,7 @@ _SAMPLE_BUCKETS = 7
84
84
 
85
85
  #: The one reason that names a session chosen without evidence — no prior fired,
86
86
  #: so there is nothing a suppression verdict could withdraw. Named here, next to
87
- #: `Escalation`, because `reasoning` needs it to recognise the case and `run`
87
+ #: `ReasoningRequest`, because `reasoning` needs it to recognise the case and `run`
88
88
  #: needs it to build one.
89
89
  SAMPLED = "sampled"
90
90
 
@@ -116,7 +116,7 @@ def in_sample_bucket(session_id: str, day_of_run: int) -> bool:
116
116
 
117
117
 
118
118
  @dataclass(frozen=True)
119
- class Escalation:
119
+ class ReasoningRequest:
120
120
  """A session worth spending inference on, and what to look at.
121
121
 
122
122
  Carries no budget: what remains of a run's allowance belongs to the
@@ -124,7 +124,7 @@ class Escalation:
124
124
  """
125
125
 
126
126
  session: SessionRef
127
- #: Which priors fired, by name, or the stage-one rule that escalated alone.
127
+ #: Which priors fired, by name, or the stage-one rule that requested alone.
128
128
  reasons: tuple[str, ...]
129
129
  #: Where a semantic question is worth asking.
130
130
  spans: tuple[str, ...]
@@ -137,12 +137,12 @@ class Escalation:
137
137
 
138
138
  def run_priors(
139
139
  correlated: CorrelatedSession, stage_one: Sequence[Detection]
140
- ) -> tuple[tuple[Detection, ...], tuple[Unknown, ...], Escalation | None]:
140
+ ) -> tuple[tuple[Detection, ...], tuple[Unknown, ...], ReasoningRequest | None]:
141
141
  """Score one correlated session against its composition.
142
142
 
143
143
  Three values, not four. A `prior_recorded` bit used to travel alongside,
144
- saying an escalating prior had fired without crossing the combination
145
- threshold — a state that no longer exists, because neither the escalating
144
+ saying a requesting prior had fired without crossing the combination
145
+ threshold — a state that no longer exists, because neither the requesting
146
146
  priors nor the threshold do. A field that is now constantly `False` would
147
147
  be a worse answer than no field.
148
148
  """
@@ -169,7 +169,7 @@ def run_priors(
169
169
  ),
170
170
  ),
171
171
  ),
172
- _escalation(_solo_reasons(stage_one), ref, stage_one, coverage),
172
+ _request(_solo_reasons(stage_one), ref, stage_one, coverage),
173
173
  )
174
174
 
175
175
  composition = correlated.composition
@@ -197,28 +197,28 @@ def run_priors(
197
197
  else:
198
198
  detections.extend(_advisory_reach(ordered, calls, composition, ref, coverage))
199
199
 
200
- # One route. A stage-one fact this stage cannot settle escalates on its own;
201
- # nothing here combines, because nothing here escalates any more.
202
- escalation = _escalation(_solo_reasons(stage_one), ref, stage_one, coverage, spans)
203
- return tuple(detections), tuple(unknowns), escalation
200
+ # One route. A stage-one fact this stage cannot settle requests reasoning on
201
+ # its own; nothing here combines, because nothing here requests it any more.
202
+ request = _request(_solo_reasons(stage_one), ref, stage_one, coverage, spans)
203
+ return tuple(detections), tuple(unknowns), request
204
204
 
205
205
 
206
206
  def _solo_reasons(stage_one: Sequence[Detection]) -> list[str]:
207
- """The stage-one rules on this session that escalate on their own."""
207
+ """The stage-one rules on this session that request reasoning on their own."""
208
208
  return sorted({finding.rule_id for finding in stage_one} & _SOLO_ROUTES)
209
209
 
210
210
 
211
- def _escalation(
211
+ def _request(
212
212
  reasons: Sequence[str],
213
213
  ref: SessionRef,
214
214
  stage_one: Sequence[Detection],
215
215
  coverage: Coverage,
216
216
  spans: Sequence[str] = (),
217
- ) -> Escalation | None:
218
- """One escalation naming every reason that fired, or nothing.
217
+ ) -> ReasoningRequest | None:
218
+ """One reasoning request naming every reason that fired, or nothing.
219
219
 
220
220
  Spans from the priors that emitted are joined by those of every stage-one
221
- finding whose rule escalated, so the analyzer is pointed at both.
221
+ finding whose rule requested it, so the analyzer is pointed at both.
222
222
  """
223
223
  if not reasons:
224
224
  return None
@@ -230,7 +230,7 @@ def _escalation(
230
230
  if finding.rule_id == reason
231
231
  for evidence in finding.evidence
232
232
  )
233
- return Escalation(
233
+ return ReasoningRequest(
234
234
  session=ref,
235
235
  reasons=tuple(dict.fromkeys(reasons)),
236
236
  spans=tuple(dict.fromkeys(cited)),