stacktrace-cli 0.2.2__py3-none-any.whl → 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/analysis.py +9 -9
- stacktrace_cli/cli.py +37 -28
- stacktrace_cli/correlate/acquire.py +154 -2
- stacktrace_cli/detector/analyzer.py +1 -1
- stacktrace_cli/detector/cache.py +15 -15
- stacktrace_cli/detector/deterministic.py +1 -1
- stacktrace_cli/detector/finding.py +3 -3
- stacktrace_cli/detector/markers.py +1 -1
- stacktrace_cli/detector/priors.py +21 -21
- stacktrace_cli/detector/reasoning.py +74 -63
- stacktrace_cli/detector/render.py +22 -25
- stacktrace_cli/detector/rules.py +3 -3
- stacktrace_cli/detector/run.py +65 -65
- stacktrace_cli/monitor/{escalate.py → reasoning.py} +13 -13
- stacktrace_cli/monitor/render.py +2 -2
- stacktrace_cli/monitor/server.py +37 -35
- stacktrace_cli/monitor/site/app.js +22 -22
- stacktrace_cli/monitor/site/index.html +1 -5
- stacktrace_cli/monitor/site/styles.css +4 -9
- stacktrace_cli/monitor/state.py +3 -3
- stacktrace_cli/monitor/verdicts.py +3 -3
- stacktrace_cli/monitor/watch.py +7 -7
- stacktrace_cli/remote/cli.py +8 -9
- stacktrace_cli/remote/sync_detect.py +13 -7
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/METADATA +7 -6
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/RECORD +29 -29
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/WHEEL +0 -0
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/entry_points.txt +0 -0
stacktrace_cli/detector/rules.py
CHANGED
|
@@ -14,8 +14,8 @@ These are not advisory identifiers, they never enter the overlay corpus
|
|
|
14
14
|
`guardrail-modification`, `destructive-action`, `tool-shadowing` and
|
|
15
15
|
`unpinned-invoked` — were removed rather than kept as aspirations (ADR-0013).
|
|
16
16
|
Two produced *noise*: output whose majority a reader learns to skip. Two had *no
|
|
17
|
-
input*: a match outcome the join has never emitted, and
|
|
18
|
-
that could not be met.
|
|
17
|
+
input*: a match outcome the join has never emitted, and a reasoning-request
|
|
18
|
+
threshold that could not be met.
|
|
19
19
|
|
|
20
20
|
**Five more were withdrawn by ADR-0015, and not for either of those reasons.**
|
|
21
21
|
`unsanctioned-mcp-tool-use`, `unattended-privileged-action`, `provider-refusal`,
|
|
@@ -57,7 +57,7 @@ SECURITY_RULES = frozenset(
|
|
|
57
57
|
#: A stalled loop and a leaked credential competed for one reader's attention
|
|
58
58
|
#: while sharing one severity ladder, and the ladder cannot express the
|
|
59
59
|
#: difference. ADR-0010 gives each finding family its own grades; this is that
|
|
60
|
-
#: argument one level down. These rules never
|
|
60
|
+
#: argument one level down. These rules never request reasoning: no amount of inference
|
|
61
61
|
#: makes a repeated call more or less repeated.
|
|
62
62
|
RELIABILITY_RULES = frozenset(
|
|
63
63
|
{
|
stacktrace_cli/detector/run.py
CHANGED
|
@@ -3,19 +3,19 @@
|
|
|
3
3
|
Microseconds, then milliseconds, then — only for the residue neither could
|
|
4
4
|
resolve — seconds of inference. The per-session cost of the ordinary case is
|
|
5
5
|
effectively zero, so session volume does not scale the model bill, and
|
|
6
|
-
|
|
6
|
+
reasoning is capped for the same reason: an unbounded reasoning path would
|
|
7
7
|
make cost a function of how noisy a day was.
|
|
8
8
|
|
|
9
9
|
Three things the ordering does not settle on its own, and this module does:
|
|
10
10
|
|
|
11
|
-
- **A qualifying
|
|
11
|
+
- **A qualifying request that never runs is recorded, not dropped.** A
|
|
12
12
|
default run, a budget exhaustion and an absent analyzer all leave a session
|
|
13
13
|
*pending*, never indistinguishable from one with no prior at all.
|
|
14
|
-
- **The budget is spent in priority order.** A marker-route
|
|
14
|
+
- **The budget is spent in priority order.** A marker-route request asks
|
|
15
15
|
whether an injection was acted on and nothing else can answer it; a
|
|
16
16
|
combination-route one is two supply-chain priors that happened to coincide.
|
|
17
17
|
Ties resolve in session order, so a run is reproducible.
|
|
18
|
-
- **Reliability findings never
|
|
18
|
+
- **Reliability findings never request reasoning.** A repeated call is not more or less
|
|
19
19
|
repeated for having been reasoned about, and a session whose only stage-one
|
|
20
20
|
finding is a stall is still evidence-free of security concern — so it stays
|
|
21
21
|
eligible for the quiet-session sample.
|
|
@@ -53,7 +53,7 @@ from stacktrace_cli.detector.finding import (
|
|
|
53
53
|
from stacktrace_cli.detector.priors import (
|
|
54
54
|
_SAMPLE_BUCKETS,
|
|
55
55
|
SAMPLED,
|
|
56
|
-
|
|
56
|
+
ReasoningRequest,
|
|
57
57
|
_session_ref,
|
|
58
58
|
in_sample_bucket,
|
|
59
59
|
run_priors,
|
|
@@ -90,9 +90,9 @@ _WORKERS = 4
|
|
|
90
90
|
#: default. This is where "a sample never crowds out evidence" is actually
|
|
91
91
|
#: enforced: excluding evidence-backed sessions from candidacy keeps the
|
|
92
92
|
#: candidate list clean, but says nothing about the order the *shared* budget is
|
|
93
|
-
#: spent in once sampled and evidence-backed
|
|
93
|
+
#: spent in once sampled and evidence-backed requests sit in one pending list.
|
|
94
94
|
#: With this rank, a sample can consume a slot only when no evidence-backed
|
|
95
|
-
#:
|
|
95
|
+
#: request is waiting for one — regardless of session order.
|
|
96
96
|
_PRIORITY = {"stacktrace-injection-marker": 0, SAMPLED: 9}
|
|
97
97
|
|
|
98
98
|
#: Quiet sessions analysed per run with no evidence behind them. **Zero by
|
|
@@ -109,10 +109,10 @@ _PRIORITY = {"stacktrace-injection-marker": 0, SAMPLED: 9}
|
|
|
109
109
|
#:
|
|
110
110
|
#: Not additional to `DEFAULT_BUDGET`, despite an earlier comment here saying
|
|
111
111
|
#: so. A selected sample joins the same `pending` list as every evidence-backed
|
|
112
|
-
#:
|
|
112
|
+
#: request, and the run-level budget caps that list as a whole -- so
|
|
113
113
|
#: `sample_budget` bounds how many samples become *candidates for* a slot, not
|
|
114
114
|
#: how many extra invocations a run may make. `SAMPLED`'s rank above is what
|
|
115
|
-
#: keeps one from taking a slot an evidence-backed
|
|
115
|
+
#: keeps one from taking a slot an evidence-backed request is waiting for.
|
|
116
116
|
DEFAULT_SAMPLE_BUDGET = 0
|
|
117
117
|
|
|
118
118
|
|
|
@@ -123,9 +123,9 @@ class DetectorRun:
|
|
|
123
123
|
detections: tuple[Detection, ...] = ()
|
|
124
124
|
unknowns: tuple[Unknown, ...] = ()
|
|
125
125
|
sessions: int = 0
|
|
126
|
-
|
|
126
|
+
requested: int = 0
|
|
127
127
|
analysed: int = 0
|
|
128
|
-
#:
|
|
128
|
+
#: Reasoning requests answered from the cache. Reported because a run that spent
|
|
129
129
|
#: nothing and a run that had nothing to do are not the same run.
|
|
130
130
|
cache_hits: int = 0
|
|
131
131
|
#: Session readers that failed before correlation ever saw their sessions —
|
|
@@ -140,7 +140,7 @@ class DetectorRun:
|
|
|
140
140
|
def run_detector(
|
|
141
141
|
correlated: CorrelatedView,
|
|
142
142
|
*,
|
|
143
|
-
|
|
143
|
+
reasoning: bool = False,
|
|
144
144
|
budget: int = DEFAULT_BUDGET,
|
|
145
145
|
analyzers: dict[str, Analyzer] | None = None,
|
|
146
146
|
cache: VerdictCache | None = None,
|
|
@@ -152,8 +152,8 @@ def run_detector(
|
|
|
152
152
|
|
|
153
153
|
detections: list[Detection] = []
|
|
154
154
|
unknowns: list[Unknown] = []
|
|
155
|
-
pending: list[tuple[
|
|
156
|
-
candidates: list[tuple[
|
|
155
|
+
pending: list[tuple[ReasoningRequest, object]] = []
|
|
156
|
+
candidates: list[tuple[ReasoningRequest, object]] = []
|
|
157
157
|
cache_hits = 0
|
|
158
158
|
# The run's own UTC calendar day. Never `date.today()`, which is local time
|
|
159
159
|
# and would make eligibility depend on the machine's timezone rather than
|
|
@@ -182,23 +182,23 @@ def run_detector(
|
|
|
182
182
|
detections.extend(stage_one)
|
|
183
183
|
unknowns.extend(stage_one_unknowns)
|
|
184
184
|
|
|
185
|
-
stage_two, stage_two_unknowns,
|
|
185
|
+
stage_two, stage_two_unknowns, request = run_priors(session, stage_one)
|
|
186
186
|
detections.extend(stage_two)
|
|
187
187
|
unknowns.extend(stage_two_unknowns)
|
|
188
|
-
if
|
|
189
|
-
pending.append((
|
|
188
|
+
if request is not None:
|
|
189
|
+
pending.append((request, session.session))
|
|
190
190
|
continue
|
|
191
191
|
# A sample *candidate* is evidence-free by every route already built.
|
|
192
192
|
# **Of security concern**, which is narrower than "found nothing": a
|
|
193
193
|
# session that stalled or hung has a reliability finding and no
|
|
194
194
|
# security evidence at all, and excluding it would make the sample
|
|
195
195
|
# skip exactly the sessions where something went wrong in a way no
|
|
196
|
-
# security rule reads. Reliability rules never
|
|
196
|
+
# security rule reads. Reliability rules never request reasoning, so nothing
|
|
197
197
|
# else in this path distinguishes them.
|
|
198
198
|
#
|
|
199
199
|
# Both stages are checked: a stage-two rule such as
|
|
200
200
|
# `stacktrace-advisory-reach`
|
|
201
|
-
# is security-family evidence that does not itself
|
|
201
|
+
# is security-family evidence that does not itself request reasoning (it is not
|
|
202
202
|
# in `_SOLO_ROUTES`), so a stage-one-only check would still route an
|
|
203
203
|
# evidence-backed session into the evidence-free sample.
|
|
204
204
|
if not any(d.family == "security" for d in (*stage_one, *stage_two)) and in_sample_bucket(
|
|
@@ -206,7 +206,7 @@ def run_detector(
|
|
|
206
206
|
):
|
|
207
207
|
candidates.append(
|
|
208
208
|
(
|
|
209
|
-
|
|
209
|
+
ReasoningRequest(
|
|
210
210
|
session=_session_ref(session),
|
|
211
211
|
reasons=(SAMPLED,),
|
|
212
212
|
spans=(),
|
|
@@ -235,28 +235,28 @@ def run_detector(
|
|
|
235
235
|
# Three passes, not one index-gated walk, because budget must be spent on
|
|
236
236
|
# dispatched work rather than on list position. A cache hit or an
|
|
237
237
|
# unavailable analyzer anywhere in priority order must never reduce how many
|
|
238
|
-
# real invocations a later, lower-priority
|
|
238
|
+
# real invocations a later, lower-priority request can still consume.
|
|
239
239
|
#
|
|
240
240
|
# Pass one: cache hits. No subprocess, no thread, no budget — and no
|
|
241
|
-
# `
|
|
242
|
-
# *has*, and `
|
|
241
|
+
# `reasoning` either. A verdict already paid for is an answer this run
|
|
242
|
+
# *has*, and `reasoning` governs whether new ones may be commissioned, not
|
|
243
243
|
# whether old ones may be read. Withholding it would make one session read
|
|
244
244
|
# as graded or ungraded depending on a flag that spent nothing either way,
|
|
245
245
|
# which is how a reader ends up paying twice for the same answer.
|
|
246
246
|
keys: dict[int, str] = {}
|
|
247
|
-
misses: list[tuple[
|
|
248
|
-
not_requested: list[
|
|
249
|
-
for position, (
|
|
250
|
-
analyzer = analyzers.get(
|
|
247
|
+
misses: list[tuple[ReasoningRequest, object, Analyzer]] = []
|
|
248
|
+
not_requested: list[ReasoningRequest] = []
|
|
249
|
+
for position, (request, session) in enumerate(ordered):
|
|
250
|
+
analyzer = analyzers.get(request.session.agent_kind)
|
|
251
251
|
if cache is not None and analyzer is not None:
|
|
252
252
|
identity = analyzer.identity()
|
|
253
253
|
key = cache_key(
|
|
254
254
|
session, # type: ignore[arg-type]
|
|
255
|
-
|
|
255
|
+
request,
|
|
256
256
|
# The rules actually asked, not the whole shipped set: a session
|
|
257
257
|
# with no marker is never asked the marker follow-up, so its
|
|
258
258
|
# entry answers for a smaller set and must key on that.
|
|
259
|
-
applicable_rules(
|
|
259
|
+
applicable_rules(request),
|
|
260
260
|
PROMPT_VERSION,
|
|
261
261
|
identity,
|
|
262
262
|
)
|
|
@@ -267,7 +267,7 @@ def run_detector(
|
|
|
267
267
|
# session, or a rule id to be one this outcome actually asked.
|
|
268
268
|
if cached is not None and validate(
|
|
269
269
|
cached,
|
|
270
|
-
rule_ids=applicable_rules(
|
|
270
|
+
rule_ids=applicable_rules(request),
|
|
271
271
|
prompt_version=PROMPT_VERSION,
|
|
272
272
|
identity=identity,
|
|
273
273
|
# Per rule, and the same function the live path parses against:
|
|
@@ -276,47 +276,45 @@ def run_detector(
|
|
|
276
276
|
admissible_spans=partial(
|
|
277
277
|
admissible_spans,
|
|
278
278
|
session=session, # type: ignore[arg-type]
|
|
279
|
-
|
|
279
|
+
request=request,
|
|
280
280
|
),
|
|
281
281
|
):
|
|
282
|
-
detections.extend(_from_cache(cached,
|
|
282
|
+
detections.extend(_from_cache(cached, request))
|
|
283
283
|
cache_hits += 1
|
|
284
284
|
continue
|
|
285
|
-
if not
|
|
285
|
+
if not reasoning:
|
|
286
286
|
# Not silence: a session that qualified and was never analysed is
|
|
287
287
|
# pending, and the priors that qualified it travel with it.
|
|
288
|
-
not_requested.append(
|
|
288
|
+
not_requested.append(request)
|
|
289
289
|
continue
|
|
290
290
|
# Pass two: an analyzer that cannot run. Also no budget — a session
|
|
291
291
|
# analysed by another vendor's CLI is a trust boundary created for
|
|
292
292
|
# convenience, and declining costs nothing to declare.
|
|
293
293
|
if analyzer is None or not analyzer.available():
|
|
294
|
-
unknowns.append(_pending(
|
|
294
|
+
unknowns.append(_pending(request, "analyzer_unavailable"))
|
|
295
295
|
continue
|
|
296
296
|
# Also no budget — a session whose every applicable rule is oversized
|
|
297
297
|
# or otherwise unanswerable would submit to `run_reasoning` and consume
|
|
298
298
|
# a slot without ever calling `analyzer.analyze`, silently starving a
|
|
299
|
-
# dispatchable
|
|
300
|
-
if not dispatchable(
|
|
301
|
-
unknowns.append(_pending(
|
|
299
|
+
# dispatchable request ranked behind it.
|
|
300
|
+
if not dispatchable(request, session): # type: ignore[arg-type]
|
|
301
|
+
unknowns.append(_pending(request, undispatchable_reason(request, session))) # type: ignore[arg-type]
|
|
302
302
|
continue
|
|
303
|
-
misses.append((
|
|
303
|
+
misses.append((request, session, analyzer))
|
|
304
304
|
|
|
305
|
-
if not
|
|
306
|
-
unknowns.extend(
|
|
307
|
-
_pending(escalation, "reasoning_not_requested") for escalation in not_requested
|
|
308
|
-
)
|
|
305
|
+
if not reasoning:
|
|
306
|
+
unknowns.extend(_pending(request, "reasoning_not_requested") for request in not_requested)
|
|
309
307
|
return DetectorRun(
|
|
310
308
|
detections=tuple(detections),
|
|
311
309
|
unknowns=tuple(unknowns),
|
|
312
310
|
sessions=len(correlated.sessions),
|
|
313
|
-
|
|
311
|
+
requested=len(pending),
|
|
314
312
|
cache_hits=cache_hits,
|
|
315
313
|
collection_failures=_collection_failures(correlated.unavailable),
|
|
316
314
|
)
|
|
317
315
|
|
|
318
316
|
# Pass three: the budget, on what is actually left to dispatch.
|
|
319
|
-
runnable: list[tuple[
|
|
317
|
+
runnable: list[tuple[ReasoningRequest, object, Analyzer]] = []
|
|
320
318
|
for index, entry in enumerate(misses):
|
|
321
319
|
if index >= budget:
|
|
322
320
|
unknowns.append(_pending(entry[0], "budget_exhausted"))
|
|
@@ -333,24 +331,24 @@ def run_detector(
|
|
|
333
331
|
if runnable:
|
|
334
332
|
with ThreadPoolExecutor(max_workers=min(_WORKERS, len(runnable))) as pool:
|
|
335
333
|
futures = {
|
|
336
|
-
pool.submit(run_reasoning,
|
|
337
|
-
for position, (
|
|
334
|
+
pool.submit(run_reasoning, request, session, analyzer): position # type: ignore[arg-type]
|
|
335
|
+
for position, (request, session, analyzer) in enumerate(runnable)
|
|
338
336
|
}
|
|
339
337
|
for future in as_completed(futures):
|
|
340
338
|
outcomes.append((futures[future], future.result()))
|
|
341
339
|
|
|
342
340
|
for position, outcome in sorted(outcomes):
|
|
343
|
-
|
|
341
|
+
request, session, analyzer = runnable[position]
|
|
344
342
|
detections.extend(outcome.findings) # type: ignore[attr-defined]
|
|
345
343
|
unknowns.extend(outcome.unknowns) # type: ignore[attr-defined]
|
|
346
344
|
analysed += 1
|
|
347
345
|
# Written from this loop and nowhere else. `run_reasoning` runs on a
|
|
348
346
|
# worker thread; a write from inside it would happen on whichever worker
|
|
349
|
-
# ran that
|
|
347
|
+
# ran that request, and two workers writing at once could interleave.
|
|
350
348
|
# Every `get` and `put` is therefore made from this one thread by
|
|
351
349
|
# construction, needing no lock within one process.
|
|
352
350
|
if cache is not None and cacheable(outcome):
|
|
353
|
-
key = keys.get(_position_of(ordered,
|
|
351
|
+
key = keys.get(_position_of(ordered, request))
|
|
354
352
|
if key is not None:
|
|
355
353
|
cache.put(key, _to_cache(outcome, analyzer)) # type: ignore[arg-type]
|
|
356
354
|
|
|
@@ -358,7 +356,7 @@ def run_detector(
|
|
|
358
356
|
detections=tuple(detections),
|
|
359
357
|
unknowns=tuple(unknowns),
|
|
360
358
|
sessions=len(correlated.sessions),
|
|
361
|
-
|
|
359
|
+
requested=len(pending),
|
|
362
360
|
analysed=analysed,
|
|
363
361
|
cache_hits=cache_hits,
|
|
364
362
|
collection_failures=_collection_failures(correlated.unavailable),
|
|
@@ -719,7 +717,7 @@ def _collection_failures(unavailable: Sequence[str]) -> tuple[str, ...]:
|
|
|
719
717
|
)
|
|
720
718
|
|
|
721
719
|
|
|
722
|
-
def _in_priority_order(pending: Sequence[tuple[
|
|
720
|
+
def _in_priority_order(pending: Sequence[tuple[ReasoningRequest, object]]):
|
|
723
721
|
"""Marker route first, then session order.
|
|
724
722
|
|
|
725
723
|
Ties resolve by position, so two runs over the same corpus spend the budget
|
|
@@ -727,9 +725,9 @@ def _in_priority_order(pending: Sequence[tuple[Escalation, object]]):
|
|
|
727
725
|
"analysed" a property of the day rather than of the corpus.
|
|
728
726
|
"""
|
|
729
727
|
|
|
730
|
-
def rank(item: tuple[int, tuple[
|
|
731
|
-
index, (
|
|
732
|
-
return (min((_PRIORITY.get(r, 1) for r in
|
|
728
|
+
def rank(item: tuple[int, tuple[ReasoningRequest, object]]) -> tuple[int, int]:
|
|
729
|
+
index, (request, _) = item
|
|
730
|
+
return (min((_PRIORITY.get(r, 1) for r in request.reasons), default=1), index)
|
|
733
731
|
|
|
734
732
|
return [pair for _, pair in sorted(enumerate(pending), key=rank)]
|
|
735
733
|
|
|
@@ -740,9 +738,11 @@ def _in_priority_order(pending: Sequence[tuple[Escalation, object]]):
|
|
|
740
738
|
_CACHED_DETAIL = "confirmed by a cached stage-3 verdict"
|
|
741
739
|
|
|
742
740
|
|
|
743
|
-
def _position_of(
|
|
741
|
+
def _position_of(
|
|
742
|
+
ordered: Sequence[tuple[ReasoningRequest, object]], request: ReasoningRequest
|
|
743
|
+
) -> int:
|
|
744
744
|
for position, (candidate, _) in enumerate(ordered):
|
|
745
|
-
if candidate is
|
|
745
|
+
if candidate is request:
|
|
746
746
|
return position
|
|
747
747
|
return -1
|
|
748
748
|
|
|
@@ -774,7 +774,7 @@ def _to_cache(outcome: ReasoningOutcome, analyzer: Analyzer) -> CachedOutcome:
|
|
|
774
774
|
)
|
|
775
775
|
|
|
776
776
|
|
|
777
|
-
def _from_cache(cached: CachedOutcome,
|
|
777
|
+
def _from_cache(cached: CachedOutcome, request: ReasoningRequest) -> tuple[Detection, ...]:
|
|
778
778
|
"""Rebuild the findings a stored outcome represents.
|
|
779
779
|
|
|
780
780
|
Titles and remediations come from `reasoning.py`'s tables, already keyed by
|
|
@@ -797,25 +797,25 @@ def _from_cache(cached: CachedOutcome, escalation: Escalation) -> tuple[Detectio
|
|
|
797
797
|
severity=entry.severity,
|
|
798
798
|
confidence=entry.confidence,
|
|
799
799
|
title=_TITLES.get(entry.rule_id, entry.rule_id),
|
|
800
|
-
session=
|
|
800
|
+
session=request.session,
|
|
801
801
|
evidence=tuple(
|
|
802
802
|
Evidence(span=span, kind="cached_verdict", detail=_CACHED_DETAIL)
|
|
803
803
|
for span in entry.spans
|
|
804
804
|
),
|
|
805
805
|
remediation=_REMEDIATION.get(entry.rule_id, ""),
|
|
806
806
|
verdict=verdict,
|
|
807
|
-
coverage=
|
|
807
|
+
coverage=request.coverage,
|
|
808
808
|
)
|
|
809
809
|
for entry in cached.verdicts
|
|
810
810
|
)
|
|
811
811
|
|
|
812
812
|
|
|
813
|
-
def _pending(
|
|
813
|
+
def _pending(request: ReasoningRequest, reason: str) -> Unknown:
|
|
814
814
|
return Unknown(
|
|
815
|
-
session=
|
|
815
|
+
session=request.session,
|
|
816
816
|
stage="reasoning",
|
|
817
817
|
reason=reason, # type: ignore[arg-type]
|
|
818
|
-
spans=
|
|
819
|
-
detail=f"
|
|
820
|
-
reasons=
|
|
818
|
+
spans=request.spans,
|
|
819
|
+
detail=f"requested on {', '.join(request.reasons)}; not analysed",
|
|
820
|
+
reasons=request.reasons,
|
|
821
821
|
)
|
|
@@ -22,8 +22,8 @@ from stacktrace_cli.monitor.state import Snapshot, StateStore
|
|
|
22
22
|
JsonObject = dict[str, Any]
|
|
23
23
|
|
|
24
24
|
|
|
25
|
-
class
|
|
26
|
-
"""The
|
|
25
|
+
class ReasoningRunner:
|
|
26
|
+
"""The reasoning budget, and the sessions currently spending it."""
|
|
27
27
|
|
|
28
28
|
def __init__(
|
|
29
29
|
self,
|
|
@@ -61,7 +61,7 @@ class Escalator:
|
|
|
61
61
|
}
|
|
62
62
|
|
|
63
63
|
def request(self, session_id: str) -> tuple[bool, str | None]:
|
|
64
|
-
"""Accept or refuse one
|
|
64
|
+
"""Accept or refuse one reasoning run, and say which it was.
|
|
65
65
|
|
|
66
66
|
The decision and the decrement happen under one lock: two clicks
|
|
67
67
|
arriving together must not both see the last unit of budget, or the
|
|
@@ -82,7 +82,7 @@ class Escalator:
|
|
|
82
82
|
)
|
|
83
83
|
if self._remaining <= 0:
|
|
84
84
|
return False, (
|
|
85
|
-
"refused: this monitor's
|
|
85
|
+
"refused: this monitor's reasoning budget is spent. Restart "
|
|
86
86
|
"monitor, or raise --budget, to allow more"
|
|
87
87
|
)
|
|
88
88
|
self._remaining -= 1
|
|
@@ -91,7 +91,7 @@ class Escalator:
|
|
|
91
91
|
# Before the thread starts, so the page cannot see a click that spent a
|
|
92
92
|
# unit and left no trace. The page renders on a changed revision and
|
|
93
93
|
# nothing else moves the revision on an idle machine, so an unpublished
|
|
94
|
-
# acceptance is an accepted
|
|
94
|
+
# acceptance is an accepted run nobody is told about.
|
|
95
95
|
self._publish()
|
|
96
96
|
thread = threading.Thread(target=self._work, args=(session_id,), daemon=True)
|
|
97
97
|
thread.start()
|
|
@@ -107,7 +107,7 @@ class Escalator:
|
|
|
107
107
|
and so does any local client that mistypes an id.
|
|
108
108
|
|
|
109
109
|
Read outside this class's lock and through the store's, which is the
|
|
110
|
-
watcher's order — store first,
|
|
110
|
+
watcher's order — store first, runner second — so no pair of
|
|
111
111
|
publishers can take the two the other way round.
|
|
112
112
|
"""
|
|
113
113
|
return any(
|
|
@@ -129,8 +129,8 @@ class Escalator:
|
|
|
129
129
|
banner: str | None = None
|
|
130
130
|
try:
|
|
131
131
|
banner = self._run(session_id)
|
|
132
|
-
except Exception as error: # noqa: BLE001 - a failed
|
|
133
|
-
banner = f"
|
|
132
|
+
except Exception as error: # noqa: BLE001 - a failed reasoning run must not kill the thread
|
|
133
|
+
banner = f"reasoning about {session_id} failed: {error}"
|
|
134
134
|
finally:
|
|
135
135
|
with self._lock:
|
|
136
136
|
if session_id in self._running:
|
|
@@ -138,9 +138,9 @@ class Escalator:
|
|
|
138
138
|
self._publish(banner=banner)
|
|
139
139
|
|
|
140
140
|
def _publish(self, banner: str | None = None) -> None:
|
|
141
|
-
"""Republish the current view with this
|
|
141
|
+
"""Republish the current view with this reasoning run's state in it.
|
|
142
142
|
|
|
143
|
-
Only the
|
|
143
|
+
Only the reasoning fields and, on a failure, one banner: the sessions
|
|
144
144
|
and alerts on the page belong to the watcher's last pass and are not
|
|
145
145
|
this thread's to rebuild.
|
|
146
146
|
|
|
@@ -149,7 +149,7 @@ class Escalator:
|
|
|
149
149
|
publish them in the other, and the newer snapshot is then overwritten
|
|
150
150
|
by the older: a spent unit of budget back on the page, or a finished
|
|
151
151
|
session left listed as running with no later revision to correct it.
|
|
152
|
-
The lock order is the watcher's — the store first, this
|
|
152
|
+
The lock order is the watcher's — the store first, this runner's
|
|
153
153
|
second — so the two publishers cannot deadlock against each other.
|
|
154
154
|
|
|
155
155
|
A banner is handed to `note` inside the same builder, for the same
|
|
@@ -160,12 +160,12 @@ class Escalator:
|
|
|
160
160
|
|
|
161
161
|
def build(previous: Snapshot, revision: int) -> Snapshot:
|
|
162
162
|
if banner is None:
|
|
163
|
-
return replace(previous, revision=revision,
|
|
163
|
+
return replace(previous, revision=revision, reasoning=self.state())
|
|
164
164
|
self._note(banner)
|
|
165
165
|
return replace(
|
|
166
166
|
previous,
|
|
167
167
|
revision=revision,
|
|
168
|
-
|
|
168
|
+
reasoning=self.state(),
|
|
169
169
|
banners=[*previous.banners, banner],
|
|
170
170
|
)
|
|
171
171
|
|
stacktrace_cli/monitor/render.py
CHANGED
|
@@ -196,7 +196,7 @@ def snapshot_from(
|
|
|
196
196
|
analysis: Analysis,
|
|
197
197
|
*,
|
|
198
198
|
revision: int,
|
|
199
|
-
|
|
199
|
+
reasoning: JsonObject,
|
|
200
200
|
last_active: dict[str, float] | None = None,
|
|
201
201
|
) -> Snapshot:
|
|
202
202
|
"""Build the page's whole state from one pipeline run.
|
|
@@ -251,6 +251,6 @@ def snapshot_from(
|
|
|
251
251
|
},
|
|
252
252
|
sessions=sessions,
|
|
253
253
|
alerts=alerts,
|
|
254
|
-
|
|
254
|
+
reasoning=reasoning,
|
|
255
255
|
banners=_banners(analysis),
|
|
256
256
|
)
|