stacktrace-cli 0.2.2__py3-none-any.whl → 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/analysis.py +9 -9
- stacktrace_cli/cli.py +37 -28
- stacktrace_cli/correlate/acquire.py +154 -2
- stacktrace_cli/detector/analyzer.py +1 -1
- stacktrace_cli/detector/cache.py +15 -15
- stacktrace_cli/detector/deterministic.py +1 -1
- stacktrace_cli/detector/finding.py +3 -3
- stacktrace_cli/detector/markers.py +1 -1
- stacktrace_cli/detector/priors.py +21 -21
- stacktrace_cli/detector/reasoning.py +74 -63
- stacktrace_cli/detector/render.py +22 -25
- stacktrace_cli/detector/rules.py +3 -3
- stacktrace_cli/detector/run.py +65 -65
- stacktrace_cli/monitor/{escalate.py → reasoning.py} +13 -13
- stacktrace_cli/monitor/render.py +2 -2
- stacktrace_cli/monitor/server.py +37 -35
- stacktrace_cli/monitor/site/app.js +22 -22
- stacktrace_cli/monitor/site/index.html +1 -5
- stacktrace_cli/monitor/site/styles.css +4 -9
- stacktrace_cli/monitor/state.py +3 -3
- stacktrace_cli/monitor/verdicts.py +3 -3
- stacktrace_cli/monitor/watch.py +7 -7
- stacktrace_cli/remote/cli.py +8 -9
- stacktrace_cli/remote/sync_detect.py +13 -7
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/METADATA +7 -6
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/RECORD +29 -29
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/WHEEL +0 -0
- {stacktrace_cli-0.2.2.dist-info → stacktrace_cli-0.3.0.dist-info}/entry_points.txt +0 -0
stacktrace_cli/monitor/server.py
CHANGED
|
@@ -76,7 +76,7 @@ class MonitorConfig:
|
|
|
76
76
|
#: Minted per run. The page receives it in the URL monitor opens and echoes
|
|
77
77
|
#: it on the one request that spends money.
|
|
78
78
|
token: str = field(default_factory=lambda: secrets.token_urlsafe(16))
|
|
79
|
-
|
|
79
|
+
reasoning: bool = False
|
|
80
80
|
budget: int = 10
|
|
81
81
|
|
|
82
82
|
|
|
@@ -87,7 +87,7 @@ def _asset(name: str) -> bytes:
|
|
|
87
87
|
def build_server(
|
|
88
88
|
store: StateStore,
|
|
89
89
|
config: MonitorConfig,
|
|
90
|
-
|
|
90
|
+
runner: Any = None,
|
|
91
91
|
set_window: Callable[[str], None] | None = None,
|
|
92
92
|
) -> ThreadingHTTPServer:
|
|
93
93
|
"""Bind and return the server, or raise saying why it could not.
|
|
@@ -150,25 +150,25 @@ def build_server(
|
|
|
150
150
|
if path == "/api/window":
|
|
151
151
|
self._window()
|
|
152
152
|
return
|
|
153
|
-
if path != "/api/
|
|
153
|
+
if path != "/api/reasoning":
|
|
154
154
|
# One POST route exists. Without this check a typo — `POST /`,
|
|
155
|
-
# `POST /api/
|
|
155
|
+
# `POST /api/reasonin` — reaches the runner and spends a unit
|
|
156
156
|
# of budget on a route that was never registered.
|
|
157
157
|
self.send_error(404)
|
|
158
158
|
return
|
|
159
|
-
if not config.
|
|
159
|
+
if not config.reasoning or runner is None:
|
|
160
160
|
# Not 403: without the flag this endpoint does not exist, and
|
|
161
161
|
# saying "forbidden" would advertise a feature that is off.
|
|
162
162
|
self.send_error(404)
|
|
163
163
|
return
|
|
164
|
-
refusal =
|
|
164
|
+
refusal = _refuse_reasoning(self, config)
|
|
165
165
|
if refusal is not None:
|
|
166
166
|
# 403 and no decrement: a refused request never spends.
|
|
167
167
|
self._json(
|
|
168
168
|
{
|
|
169
169
|
"accepted": False,
|
|
170
170
|
"reason": refusal,
|
|
171
|
-
"remaining":
|
|
171
|
+
"remaining": runner.remaining(),
|
|
172
172
|
},
|
|
173
173
|
status=403,
|
|
174
174
|
)
|
|
@@ -182,30 +182,30 @@ def build_server(
|
|
|
182
182
|
{
|
|
183
183
|
"accepted": False,
|
|
184
184
|
"reason": "malformed request body",
|
|
185
|
-
"remaining":
|
|
185
|
+
"remaining": runner.remaining(),
|
|
186
186
|
},
|
|
187
187
|
status=400,
|
|
188
188
|
)
|
|
189
189
|
return
|
|
190
|
-
accepted, reason =
|
|
191
|
-
self._json({"accepted": accepted, "reason": reason, "remaining":
|
|
190
|
+
accepted, reason = runner.request(session_id)
|
|
191
|
+
self._json({"accepted": accepted, "reason": reason, "remaining": runner.remaining()})
|
|
192
192
|
|
|
193
193
|
def _window(self) -> None:
|
|
194
194
|
"""Re-run the pipeline over a different window.
|
|
195
195
|
|
|
196
|
-
Gated exactly as `/api/
|
|
196
|
+
Gated exactly as `/api/reasoning` is. It spends no budget, but it
|
|
197
197
|
makes this process walk every transcript on the machine again, and
|
|
198
198
|
the origin check is what stops any page in this browser from
|
|
199
199
|
driving that -- a remote page cannot forge `Origin` though it could
|
|
200
200
|
guess the port.
|
|
201
201
|
"""
|
|
202
202
|
if set_window is None:
|
|
203
|
-
# 404 rather than 403, for `/api/
|
|
203
|
+
# 404 rather than 403, for `/api/reasoning`'s reason: without a
|
|
204
204
|
# watcher to re-run there is no such route, and "forbidden"
|
|
205
205
|
# would advertise a feature that is not there.
|
|
206
206
|
self.send_error(404)
|
|
207
207
|
return
|
|
208
|
-
refusal =
|
|
208
|
+
refusal = _refuse_reasoning(self, config)
|
|
209
209
|
if refusal is not None:
|
|
210
210
|
self._json({"accepted": False, "reason": refusal}, status=403)
|
|
211
211
|
return
|
|
@@ -271,7 +271,7 @@ def _since(query: str) -> int:
|
|
|
271
271
|
return 0
|
|
272
272
|
|
|
273
273
|
|
|
274
|
-
def
|
|
274
|
+
def _refuse_reasoning(handler: BaseHTTPRequestHandler, config: MonitorConfig) -> str | None:
|
|
275
275
|
"""Why this POST may not spend money, or None if it may.
|
|
276
276
|
|
|
277
277
|
Origin first, because it is the gate that stops the attacker the token
|
|
@@ -292,7 +292,7 @@ def serve(
|
|
|
292
292
|
host: str = "127.0.0.1",
|
|
293
293
|
port: int = 0,
|
|
294
294
|
interval: float = 4.0,
|
|
295
|
-
|
|
295
|
+
reasoning: bool = False,
|
|
296
296
|
budget: int = 10,
|
|
297
297
|
open_browser: bool = True,
|
|
298
298
|
echo: Any = print,
|
|
@@ -305,12 +305,12 @@ def serve(
|
|
|
305
305
|
exists means there is nothing to unwind.
|
|
306
306
|
"""
|
|
307
307
|
from stacktrace_cli.detector.cache import default_directory
|
|
308
|
-
from stacktrace_cli.monitor.
|
|
308
|
+
from stacktrace_cli.monitor.reasoning import ReasoningRunner
|
|
309
309
|
from stacktrace_cli.monitor.state import Snapshot, StateStore
|
|
310
310
|
from stacktrace_cli.monitor.verdicts import RetainedVerdicts
|
|
311
311
|
from stacktrace_cli.monitor.watch import Watcher
|
|
312
312
|
|
|
313
|
-
config = MonitorConfig(host=host, port=port,
|
|
313
|
+
config = MonitorConfig(host=host, port=port, reasoning=reasoning, budget=budget)
|
|
314
314
|
store = StateStore(
|
|
315
315
|
Snapshot(
|
|
316
316
|
revision=0,
|
|
@@ -319,8 +319,8 @@ def serve(
|
|
|
319
319
|
)
|
|
320
320
|
)
|
|
321
321
|
|
|
322
|
-
|
|
323
|
-
# One cache for both, not one each:
|
|
322
|
+
runner: ReasoningRunner | None = None
|
|
323
|
+
# One cache for both, not one each: a reasoning run's verdict has to be
|
|
324
324
|
# readable by the pass that renders it, and what this process paid for is
|
|
325
325
|
# held in memory rather than on the disk two instances would share.
|
|
326
326
|
verdicts = RetainedVerdicts(default_directory())
|
|
@@ -328,13 +328,15 @@ def serve(
|
|
|
328
328
|
store,
|
|
329
329
|
options=_watcher_options(config, options, verdicts),
|
|
330
330
|
interval=interval,
|
|
331
|
-
|
|
331
|
+
reasoning_state=lambda: runner.state() if runner else {"enabled": False},
|
|
332
332
|
)
|
|
333
|
-
if
|
|
334
|
-
|
|
333
|
+
if reasoning:
|
|
334
|
+
runner = ReasoningRunner(
|
|
335
335
|
store=store,
|
|
336
336
|
budget=budget,
|
|
337
|
-
run=lambda session_id:
|
|
337
|
+
run=lambda session_id: _reason_about_one(
|
|
338
|
+
session_id, options, verdicts, watcher.invalidate
|
|
339
|
+
),
|
|
338
340
|
note=watcher.note,
|
|
339
341
|
)
|
|
340
342
|
|
|
@@ -342,15 +344,15 @@ def serve(
|
|
|
342
344
|
# here rather than passed into `Watcher`: the server drives the watcher,
|
|
343
345
|
# never the reverse, and a watcher that could serve would be two owners of
|
|
344
346
|
# one loop.
|
|
345
|
-
httpd = build_server(store, config,
|
|
347
|
+
httpd = build_server(store, config, runner=runner, set_window=watcher.set_window)
|
|
346
348
|
_thread, stop = watcher.start()
|
|
347
349
|
|
|
348
350
|
url = f"http://{_authority(config.host, httpd.server_address[1])}/?token={config.token}"
|
|
349
351
|
echo(f"stacktrace monitor on {url}")
|
|
350
352
|
echo("Press Ctrl-C to stop.")
|
|
351
|
-
if
|
|
353
|
+
if reasoning:
|
|
352
354
|
echo(
|
|
353
|
-
f"--
|
|
355
|
+
f"--reasoning is on: this page can spend your provider quota, "
|
|
354
356
|
f"{budget} session(s) at most."
|
|
355
357
|
)
|
|
356
358
|
if open_browser:
|
|
@@ -403,14 +405,14 @@ def _watcher_options(
|
|
|
403
405
|
return {
|
|
404
406
|
**options,
|
|
405
407
|
# Read on every pass and written on none: a verdict an earlier
|
|
406
|
-
# `detect --
|
|
408
|
+
# `detect --reasoning` paid for is an answer this machine already has,
|
|
407
409
|
# and a page that made you buy it again would be the opposite of
|
|
408
410
|
# "free to leave open".
|
|
409
411
|
"cache": verdicts,
|
|
410
412
|
}
|
|
411
413
|
|
|
412
414
|
|
|
413
|
-
def
|
|
415
|
+
def _reason_about_one(
|
|
414
416
|
session_id: str,
|
|
415
417
|
options: JsonObject,
|
|
416
418
|
cache: VerdictCache,
|
|
@@ -419,7 +421,7 @@ def _escalate_one(
|
|
|
419
421
|
"""Run stage 3 for the named session, and say what went unanswered.
|
|
420
422
|
|
|
421
423
|
The id is a *selector*, not a label. `budget=1` alone would spend the unit
|
|
422
|
-
on whichever
|
|
424
|
+
on whichever request the detector ranks first, which on a busy machine
|
|
423
425
|
is routinely not the one the person confirmed — so the window is narrowed
|
|
424
426
|
to this session before stage 3 can choose.
|
|
425
427
|
|
|
@@ -434,16 +436,16 @@ def _escalate_one(
|
|
|
434
436
|
ordinary `Unknown` *values* rather than exceptions, and an outcome holding
|
|
435
437
|
one is deliberately never cached — so the next pass reconstructs the
|
|
436
438
|
session as merely not-requested and the reason the paid run failed is gone.
|
|
437
|
-
It is returned instead, and the
|
|
439
|
+
It is returned instead, and the runner says it on the page.
|
|
438
440
|
"""
|
|
439
441
|
from stacktrace_cli.analysis import analyse
|
|
440
442
|
|
|
441
443
|
analysis = analyse(
|
|
442
444
|
**options,
|
|
443
445
|
session_ids=(session_id,),
|
|
444
|
-
|
|
446
|
+
reasoning=True,
|
|
445
447
|
# One session's worth. The page's own budget is what bounds the run;
|
|
446
|
-
# this bounds the single
|
|
448
|
+
# this bounds the single reasoning run it just authorised.
|
|
447
449
|
budget=1,
|
|
448
450
|
cache=cache,
|
|
449
451
|
)
|
|
@@ -453,7 +455,7 @@ def _escalate_one(
|
|
|
453
455
|
# written, so there is nothing to re-read — but the unit is spent, and a
|
|
454
456
|
# click that leaves neither an answer nor a reason reads as a dead page.
|
|
455
457
|
return (
|
|
456
|
-
f"
|
|
458
|
+
f"reasoning about {session_id} found nothing to analyse: the session is no "
|
|
457
459
|
"longer on this machine"
|
|
458
460
|
)
|
|
459
461
|
invalidate()
|
|
@@ -465,7 +467,7 @@ def _unanswered(session_id: str, unknowns: Sequence[Any]) -> str | None:
|
|
|
465
467
|
|
|
466
468
|
The stage matters: a session can carry a `low_coverage` or `truncated`
|
|
467
469
|
unknown from stages one and two on every pass, and repeating those on a
|
|
468
|
-
click would report a standing condition as this
|
|
470
|
+
click would report a standing condition as this run's outcome.
|
|
469
471
|
|
|
470
472
|
Reasons, never an unknown's `detail` — the closed vocabulary says which
|
|
471
473
|
kind of failure this was, which is what a banner is for, and the detail
|
|
@@ -474,4 +476,4 @@ def _unanswered(session_id: str, unknowns: Sequence[Any]) -> str | None:
|
|
|
474
476
|
reasons = sorted({u.reason for u in unknowns if u.stage == "reasoning"})
|
|
475
477
|
if not reasons:
|
|
476
478
|
return None
|
|
477
|
-
return f"
|
|
479
|
+
return f"reasoning about {session_id} left a question unanswered: {', '.join(reasons)}"
|
|
@@ -88,7 +88,7 @@ function windowWidth(summary) {
|
|
|
88
88
|
* that has been shown can be absent from what comes next. Choosing a narrower
|
|
89
89
|
* range breaks that. Kept additively, the rows and cards of the wider range
|
|
90
90
|
* would sit under a header and summaries that had already shrunk — and their
|
|
91
|
-
*
|
|
91
|
+
* reasoning buttons would name sessions the server no longer offers. */
|
|
92
92
|
function clearFeeds() {
|
|
93
93
|
rowBySignature.clear();
|
|
94
94
|
feedOrder.length = 0;
|
|
@@ -118,7 +118,7 @@ function render(state) {
|
|
|
118
118
|
renderAlerts(state.alerts || []);
|
|
119
119
|
/* Last, because it repaints the buttons on cards the line above may have
|
|
120
120
|
* just added. */
|
|
121
|
-
|
|
121
|
+
renderReasoning(state.reasoning || {});
|
|
122
122
|
}
|
|
123
123
|
|
|
124
124
|
/* Write only when the markup actually differs.
|
|
@@ -844,42 +844,42 @@ function addAlert(alert, signature) {
|
|
|
844
844
|
key: alertKeyOf(alert, signature),
|
|
845
845
|
};
|
|
846
846
|
placeAlert(entry);
|
|
847
|
-
|
|
847
|
+
paintReasoning(entry);
|
|
848
848
|
return entry;
|
|
849
849
|
}
|
|
850
850
|
|
|
851
|
-
/* ---
|
|
851
|
+
/* --- reasoning ----------------------------------------------------------- */
|
|
852
852
|
|
|
853
|
-
let
|
|
853
|
+
let reasoningState = { enabled: false, remaining: 0, running: [] };
|
|
854
854
|
let sessionsById = new Map();
|
|
855
855
|
|
|
856
|
-
function
|
|
857
|
-
|
|
858
|
-
for (const entry of alertBySignature.values())
|
|
856
|
+
function renderReasoning(state) {
|
|
857
|
+
reasoningState = Object.assign({ enabled: false, remaining: 0, running: [] }, state);
|
|
858
|
+
for (const entry of alertBySignature.values()) paintReasoning(entry);
|
|
859
859
|
}
|
|
860
860
|
|
|
861
|
-
function
|
|
862
|
-
entry.host.innerHTML =
|
|
861
|
+
function paintReasoning(entry) {
|
|
862
|
+
entry.host.innerHTML = reasoningButton(entry.sessionId);
|
|
863
863
|
}
|
|
864
864
|
|
|
865
|
-
function
|
|
866
|
-
if (!
|
|
867
|
-
if ((
|
|
865
|
+
function reasoningButton(sessionId) {
|
|
866
|
+
if (!reasoningState.enabled || !sessionId) return "";
|
|
867
|
+
if ((reasoningState.running || []).indexOf(sessionId) !== -1) {
|
|
868
868
|
return '<div class="al-lbl">analysing with a model...</div>';
|
|
869
869
|
}
|
|
870
|
-
if (!
|
|
870
|
+
if (!reasoningState.remaining) {
|
|
871
871
|
/* The ceiling is visible before it is hit, not only on the click that
|
|
872
872
|
* finds it. */
|
|
873
|
-
return '<div class="al-lbl">
|
|
873
|
+
return '<div class="al-lbl">reasoning budget spent</div>';
|
|
874
874
|
}
|
|
875
|
-
return '<button class="
|
|
875
|
+
return '<button class="reasoning-btn" data-session="' + esc(sessionId) +
|
|
876
876
|
'">Analyse with a model</button>';
|
|
877
877
|
}
|
|
878
878
|
|
|
879
879
|
/* The consent is the prompt, so it is spelled out rather than summarised: which
|
|
880
880
|
* session, where the transcript goes, what it costs in time and money, what the
|
|
881
881
|
* reader gets for it, and how much budget is left. */
|
|
882
|
-
function
|
|
882
|
+
function reasoningPrompt(sessionId) {
|
|
883
883
|
return [
|
|
884
884
|
"Analyse this session with a model?",
|
|
885
885
|
"",
|
|
@@ -892,7 +892,7 @@ function escalatePrompt(sessionId) {
|
|
|
892
892
|
"- Answers the three questions the other stages cannot: whether an injected",
|
|
893
893
|
" instruction was acted on, whether the agent drifted from the request, and",
|
|
894
894
|
" whether a completion claims work it did not do.",
|
|
895
|
-
"- " +
|
|
895
|
+
"- " + reasoningState.remaining + " reasoning run(s) left for this monitor session.",
|
|
896
896
|
"",
|
|
897
897
|
"A session that has not changed since it was last analysed is answered from",
|
|
898
898
|
"the local cache and costs nothing.",
|
|
@@ -916,13 +916,13 @@ function sessionIdentity(sessionId) {
|
|
|
916
916
|
}
|
|
917
917
|
|
|
918
918
|
document.addEventListener("click", async function (event) {
|
|
919
|
-
const button = event.target.closest(".
|
|
919
|
+
const button = event.target.closest(".reasoning-btn");
|
|
920
920
|
if (!button) return;
|
|
921
921
|
const sessionId = button.dataset.session;
|
|
922
|
-
if (!window.confirm(
|
|
922
|
+
if (!window.confirm(reasoningPrompt(sessionId))) return;
|
|
923
923
|
button.disabled = true;
|
|
924
924
|
try {
|
|
925
|
-
const response = await fetch("/api/
|
|
925
|
+
const response = await fetch("/api/reasoning", {
|
|
926
926
|
method: "POST",
|
|
927
927
|
headers: { "Content-Type": "application/json", "X-Stacktrace-Token": TOKEN },
|
|
928
928
|
body: JSON.stringify({ session_id: sessionId }),
|
|
@@ -931,7 +931,7 @@ document.addEventListener("click", async function (event) {
|
|
|
931
931
|
if (!body.accepted) {
|
|
932
932
|
/* Say which gate refused, rather than leaving a dead button: a silent
|
|
933
933
|
* no-op is indistinguishable from a broken page. */
|
|
934
|
-
window.alert(body.reason || "
|
|
934
|
+
window.alert(body.reason || "Reasoning refused (HTTP " + response.status + ").");
|
|
935
935
|
button.disabled = false;
|
|
936
936
|
}
|
|
937
937
|
} catch (error) {
|
|
@@ -72,15 +72,11 @@
|
|
|
72
72
|
<span class="panel-note">read from this machine's own transcripts</span>
|
|
73
73
|
</div>
|
|
74
74
|
|
|
75
|
-
<div class="legend" id="live-legend">
|
|
76
|
-
<span><b class="lg-conn">MCP</b> = a link to an outside service</span>
|
|
77
|
-
</div>
|
|
78
|
-
|
|
79
75
|
<div class="logs">
|
|
80
76
|
<section class="logcol">
|
|
81
77
|
<header class="logcol-head">
|
|
82
78
|
<span class="logcol-dot"></span>
|
|
83
|
-
<h3>Agent Session
|
|
79
|
+
<h3>Agent Session Trace</h3>
|
|
84
80
|
<span class="logcol-sub">every tool, skill & connector your sessions used</span>
|
|
85
81
|
</header>
|
|
86
82
|
<div class="feed" id="activity"><div class="feed-empty">Watching for agent activity…</div></div>
|
|
@@ -294,11 +294,6 @@ body.watch-mode .hero, body.watch-mode .band, body.watch-mode .status, body.watc
|
|
|
294
294
|
.comp-chip.hot { border-color:var(--coral); color:var(--coral-d); background:color-mix(in srgb,var(--coral) 6%,var(--paper)); }
|
|
295
295
|
.comp-chip.hot b { color:var(--coral-d); }
|
|
296
296
|
|
|
297
|
-
.legend { display:flex; gap:1.4rem; flex-wrap:wrap; margin:1rem 0 1.2rem; padding:0.6rem 0.9rem;
|
|
298
|
-
background:var(--panel); border:1px solid var(--line); border-radius:4px; font-size:0.74rem; color:var(--ink-dim); }
|
|
299
|
-
.legend b { font-weight:700; }
|
|
300
|
-
.lg-conn { color:var(--coral-d); } .lg-ok { color:var(--ok); } .lg-warn { color:var(--coral-d); }
|
|
301
|
-
|
|
302
297
|
/* two continuous logs */
|
|
303
298
|
.logs { display:grid; grid-template-columns:minmax(0,1.5fr) minmax(0,1fr); gap:1.2rem; align-items:start; }
|
|
304
299
|
@media (max-width:860px){ .logs { grid-template-columns:1fr; } }
|
|
@@ -552,7 +547,7 @@ body.watch-mode .hero, body.watch-mode .band, body.watch-mode .status, body.watc
|
|
|
552
547
|
|
|
553
548
|
Everything above this line is the console stylesheet as it stands; these are
|
|
554
549
|
the few rules for elements only `stacktrace monitor` has — a banner for what
|
|
555
|
-
could not be read, the
|
|
550
|
+
could not be read, the reasoning button, and an empty overview row. They use
|
|
556
551
|
the tokens declared at the top rather than new colours, so the additions
|
|
557
552
|
cannot drift from the palette they sit in.
|
|
558
553
|
--------------------------------------------------------------------------- */
|
|
@@ -604,12 +599,12 @@ body.watch-mode .hero, body.watch-mode .band, body.watch-mode .status, body.watc
|
|
|
604
599
|
.banner-warn { border-left:3px solid var(--coral); color:var(--ink); }
|
|
605
600
|
.banner code { font-family:var(--mono); font-size:0.9em; }
|
|
606
601
|
|
|
607
|
-
.
|
|
602
|
+
.reasoning-btn { font-family:var(--mono); font-size:0.6rem; font-weight:700;
|
|
608
603
|
letter-spacing:0.06em; text-transform:uppercase; cursor:pointer;
|
|
609
604
|
margin-top:0.45rem; padding:0.3rem 0.6rem; border-radius:2px;
|
|
610
605
|
border:1px solid var(--line-2); background:var(--panel); color:var(--ink-dim); }
|
|
611
|
-
.
|
|
612
|
-
.
|
|
606
|
+
.reasoning-btn:hover { border-color:var(--coral); color:var(--coral-d); }
|
|
607
|
+
.reasoning-btn:disabled { opacity:0.5; cursor:progress; }
|
|
613
608
|
|
|
614
609
|
/* Spans, session ids and component coordinates are identifiers, not prose:
|
|
615
610
|
they are compared character by character and wrap badly in a proportional
|
stacktrace_cli/monitor/state.py
CHANGED
|
@@ -31,7 +31,7 @@ class Snapshot:
|
|
|
31
31
|
summary: JsonObject = field(default_factory=dict)
|
|
32
32
|
sessions: list[JsonObject] = field(default_factory=list)
|
|
33
33
|
alerts: list[JsonObject] = field(default_factory=list)
|
|
34
|
-
|
|
34
|
+
reasoning: JsonObject = field(default_factory=dict)
|
|
35
35
|
#: What the page must say out loud rather than leave to an empty column: a
|
|
36
36
|
#: reader that failed, a session with no agent kind, a tick that raised.
|
|
37
37
|
banners: list[str] = field(default_factory=list)
|
|
@@ -43,7 +43,7 @@ class Snapshot:
|
|
|
43
43
|
"summary": self.summary,
|
|
44
44
|
"sessions": self.sessions,
|
|
45
45
|
"alerts": self.alerts,
|
|
46
|
-
"
|
|
46
|
+
"reasoning": self.reasoning,
|
|
47
47
|
"banners": self.banners,
|
|
48
48
|
}
|
|
49
49
|
|
|
@@ -65,7 +65,7 @@ class StateStore:
|
|
|
65
65
|
as "nothing changed".
|
|
66
66
|
|
|
67
67
|
It takes the previous snapshot for the same reason. A publisher that
|
|
68
|
-
changes one field —
|
|
68
|
+
changes one field — a reasoning run starting, a banner arriving — has to
|
|
69
69
|
read the rest of the view from somewhere, and reading it with
|
|
70
70
|
`current()` before calling this would let a tick land in between and be
|
|
71
71
|
overwritten with its own predecessor.
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
"""The verdict cache a page reads, holding on to what this process paid for.
|
|
2
2
|
|
|
3
3
|
Monitor reads the verdict cache and never writes it (ADR-0026, clause 2): a
|
|
4
|
-
verdict on disk is one `detect --
|
|
4
|
+
verdict on disk is one `detect --reasoning` paid for, and a page left open must
|
|
5
5
|
not be able to touch it.
|
|
6
6
|
|
|
7
7
|
That alone loses the answer a click just bought, because a page never sees the
|
|
8
|
-
run's return value.
|
|
8
|
+
run's return value. A reasoning run's finding reaches the reader by being read
|
|
9
9
|
back out of the cache on the next pass — so with nothing between the two but a
|
|
10
10
|
cache this side may not write, the finding is gone and a person spent a budget
|
|
11
11
|
unit and provider quota for nothing.
|
|
@@ -31,7 +31,7 @@ class RetainedVerdicts(VerdictCache):
|
|
|
31
31
|
|
|
32
32
|
def __init__(self, directory: Path, max_age_seconds: int = MAX_AGE_SECONDS) -> None:
|
|
33
33
|
super().__init__(directory, max_age_seconds)
|
|
34
|
-
# Written by the
|
|
34
|
+
# Written by the reasoning worker, read by the watcher thread: two
|
|
35
35
|
# threads by construction, unlike the single-threaded `detect` run the
|
|
36
36
|
# base class was written for.
|
|
37
37
|
self._lock = threading.Lock()
|
stacktrace_cli/monitor/watch.py
CHANGED
|
@@ -122,7 +122,7 @@ class Watcher:
|
|
|
122
122
|
analyse_fn: Callable[..., Analysis] | None = None,
|
|
123
123
|
stream_fn: Callable[..., Iterable[Analysis]] = analyse_progressively,
|
|
124
124
|
advisory_hook: Callable[..., Any] | None = None,
|
|
125
|
-
|
|
125
|
+
reasoning_state: Callable[[], JsonObject] | None = None,
|
|
126
126
|
) -> None:
|
|
127
127
|
self._store = store
|
|
128
128
|
self._options = dict(options)
|
|
@@ -148,7 +148,7 @@ class Watcher:
|
|
|
148
148
|
#: Held on the instance, not built per tick: the whole point is that it
|
|
149
149
|
#: outlives one pass. A test asserts the identity is stable.
|
|
150
150
|
self.advisory_hook = advisory_hook or advisory_attacher(_CachingAdvisoryLookup())
|
|
151
|
-
self.
|
|
151
|
+
self._reasoning_state = reasoning_state or (dict)
|
|
152
152
|
self._fingerprint: _Fingerprint | None = None
|
|
153
153
|
#: Whether anything has reached the page yet. Progressive publication
|
|
154
154
|
#: exists for a page with nothing on it; once there is something, an
|
|
@@ -156,7 +156,7 @@ class Watcher:
|
|
|
156
156
|
self._painted = False
|
|
157
157
|
#: Said once, kept on every later pass. A pass rebuilds the page from
|
|
158
158
|
#: its own analysis, and something that happened *between* passes — an
|
|
159
|
-
#:
|
|
159
|
+
#: reasoning run that answered nothing — has no analysis to be rebuilt
|
|
160
160
|
#: from, so a message published on its own would be erased by the next
|
|
161
161
|
#: tick rather than read.
|
|
162
162
|
self._notices: list[str] = []
|
|
@@ -233,9 +233,9 @@ class Watcher:
|
|
|
233
233
|
def invalidate(self) -> None:
|
|
234
234
|
"""Make the next tick re-run even though nothing on disk moved.
|
|
235
235
|
|
|
236
|
-
|
|
236
|
+
A reasoning run leaves a verdict in the cache, which is an input to the
|
|
237
237
|
pipeline the transcript fingerprint cannot see. Without this the page
|
|
238
|
-
would keep its pre-
|
|
238
|
+
would keep its pre-reasoning snapshot until an agent happened to write
|
|
239
239
|
something — so the answer a person just paid for would arrive whenever,
|
|
240
240
|
or never.
|
|
241
241
|
"""
|
|
@@ -286,7 +286,7 @@ class Watcher:
|
|
|
286
286
|
snapshot_from(
|
|
287
287
|
found,
|
|
288
288
|
revision=revision,
|
|
289
|
-
|
|
289
|
+
reasoning=self._reasoning_state(),
|
|
290
290
|
last_active=_last_active(found, by_stem),
|
|
291
291
|
)
|
|
292
292
|
)
|
|
@@ -326,7 +326,7 @@ class Watcher:
|
|
|
326
326
|
lambda previous, revision: replace(
|
|
327
327
|
previous,
|
|
328
328
|
revision=revision,
|
|
329
|
-
|
|
329
|
+
reasoning=self._reasoning_state(),
|
|
330
330
|
banners=[*previous.banners, message],
|
|
331
331
|
)
|
|
332
332
|
)
|
stacktrace_cli/remote/cli.py
CHANGED
|
@@ -350,14 +350,13 @@ def _mask_token(token: str) -> str:
|
|
|
350
350
|
help="Use this Agent BOM instead of scanning for its kind.",
|
|
351
351
|
)
|
|
352
352
|
@click.option(
|
|
353
|
-
"--
|
|
353
|
+
"--reasoning",
|
|
354
|
+
is_flag=True,
|
|
354
355
|
default=False,
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
"
|
|
358
|
-
"
|
|
359
|
-
"session content off this machine, and an unattended run is the wrong "
|
|
360
|
-
"place for either to be implicit.",
|
|
356
|
+
help="Analyse flagged sessions with a reasoning model, run through the "
|
|
357
|
+
"agent's own CLI. Off by default: stage 3 spends the developer's own "
|
|
358
|
+
"provider quota and is the one stage that sends session content off this "
|
|
359
|
+
"machine, and an unattended run is the wrong place for either to be implicit.",
|
|
361
360
|
)
|
|
362
361
|
@click.option(
|
|
363
362
|
"--budget",
|
|
@@ -409,7 +408,7 @@ def detect(
|
|
|
409
408
|
agent_kinds: tuple[str, ...],
|
|
410
409
|
since: str,
|
|
411
410
|
bom_paths: tuple[Path, ...],
|
|
412
|
-
|
|
411
|
+
reasoning: bool,
|
|
413
412
|
budget: int,
|
|
414
413
|
sample_budget: int,
|
|
415
414
|
cache: bool,
|
|
@@ -437,7 +436,7 @@ def detect(
|
|
|
437
436
|
"bom_paths": bom_paths,
|
|
438
437
|
"project_map": project_map,
|
|
439
438
|
"root": root,
|
|
440
|
-
"
|
|
439
|
+
"reasoning": reasoning,
|
|
441
440
|
"budget": budget,
|
|
442
441
|
"sample_budget": sample_budget,
|
|
443
442
|
"cache": cache,
|
|
@@ -131,7 +131,7 @@ def _collect_detect_run(
|
|
|
131
131
|
bom_paths: tuple[Path, ...],
|
|
132
132
|
project_map: tuple[str, ...],
|
|
133
133
|
root: Path | None,
|
|
134
|
-
|
|
134
|
+
reasoning: bool,
|
|
135
135
|
budget: int,
|
|
136
136
|
sample_budget: int,
|
|
137
137
|
cache: bool,
|
|
@@ -151,10 +151,16 @@ def _collect_detect_run(
|
|
|
151
151
|
bom_paths=bom_paths,
|
|
152
152
|
project_map=project_map,
|
|
153
153
|
root=root,
|
|
154
|
-
|
|
154
|
+
reasoning=reasoning,
|
|
155
155
|
budget=budget,
|
|
156
156
|
sample_budget=sample_budget,
|
|
157
|
-
|
|
157
|
+
# Not gated on `reasoning`. A verdict already paid for is an answer
|
|
158
|
+
# this run has, and `reasoning` governs whether new ones may be
|
|
159
|
+
# commissioned, not whether old ones may be read -- the rule
|
|
160
|
+
# `run_detector` states at its cache pass. Gating it here made a
|
|
161
|
+
# scheduled run call a session ungraded that `stacktrace detect`
|
|
162
|
+
# calls graded, off the same cache and with neither one spending.
|
|
163
|
+
cache=VerdictCache(default_directory()) if cache else None,
|
|
158
164
|
)
|
|
159
165
|
except ValueError as error:
|
|
160
166
|
raise SyncError(str(error)) from error
|
|
@@ -314,7 +320,7 @@ def sync_detect(
|
|
|
314
320
|
bom_paths: tuple[Path, ...] = (),
|
|
315
321
|
project_map: tuple[str, ...] = (),
|
|
316
322
|
root: Path | None = None,
|
|
317
|
-
|
|
323
|
+
reasoning: bool = False,
|
|
318
324
|
budget: int = 10,
|
|
319
325
|
sample_budget: int = 0,
|
|
320
326
|
cache: bool = True,
|
|
@@ -338,7 +344,7 @@ def sync_detect(
|
|
|
338
344
|
bom_paths=bom_paths,
|
|
339
345
|
project_map=project_map,
|
|
340
346
|
root=root,
|
|
341
|
-
|
|
347
|
+
reasoning=reasoning,
|
|
342
348
|
budget=budget,
|
|
343
349
|
sample_budget=sample_budget,
|
|
344
350
|
cache=cache,
|
|
@@ -414,7 +420,7 @@ def build_detect_dry_run_payloads(
|
|
|
414
420
|
bom_paths: tuple[Path, ...] = (),
|
|
415
421
|
project_map: tuple[str, ...] = (),
|
|
416
422
|
root: Path | None = None,
|
|
417
|
-
|
|
423
|
+
reasoning: bool = False,
|
|
418
424
|
budget: int = 10,
|
|
419
425
|
sample_budget: int = 0,
|
|
420
426
|
cache: bool = True,
|
|
@@ -433,7 +439,7 @@ def build_detect_dry_run_payloads(
|
|
|
433
439
|
bom_paths=bom_paths,
|
|
434
440
|
project_map=project_map,
|
|
435
441
|
root=root,
|
|
436
|
-
|
|
442
|
+
reasoning=reasoning,
|
|
437
443
|
budget=budget,
|
|
438
444
|
sample_budget=sample_budget,
|
|
439
445
|
cache=cache,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: stacktrace-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: CLI for Stacktrace — Detection and Response platform for AI Agents.
|
|
5
5
|
Project-URL: Homepage, https://stacktrace.ai
|
|
6
6
|
Author-email: "Stacktrace AI, Inc" <founders@stacktrace.ai>
|
|
@@ -126,14 +126,15 @@ because it is installed.
|
|
|
126
126
|
## What leaves your machine
|
|
127
127
|
|
|
128
128
|
Two of `detect`'s three stages run entirely locally and need no model or
|
|
129
|
-
credential
|
|
130
|
-
|
|
131
|
-
|
|
129
|
+
credential, and they are the two a bare `detect` runs. The third sends flagged
|
|
130
|
+
sessions to the agent's *own* CLI — the provider that produced the transcript,
|
|
131
|
+
never a different one — and runs only when you pass `--reasoning`, capped by
|
|
132
|
+
`--budget`.
|
|
132
133
|
|
|
133
134
|
`sessions` omits prompts, tool arguments and results unless you pass
|
|
134
135
|
`--include-content`. `monitor` binds to loopback only, refuses a non-loopback
|
|
135
|
-
address rather than warning about it, and
|
|
136
|
-
is given.
|
|
136
|
+
address rather than warning about it, and analyses nothing with a model unless
|
|
137
|
+
`--reasoning` is given.
|
|
137
138
|
|
|
138
139
|
## Status
|
|
139
140
|
|