holt-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- holt/__init__.py +0 -0
- holt/agent/__init__.py +0 -0
- holt/agent/entry.py +86 -0
- holt/agent/findings.py +49 -0
- holt/agent/landing.py +154 -0
- holt/agent/pipeline.py +244 -0
- holt/agent/progression.py +408 -0
- holt/agent/signals.py +220 -0
- holt/agent/stages.py +533 -0
- holt/agent/verdict.py +226 -0
- holt/agent/verify.py +140 -0
- holt/baseline.py +89 -0
- holt/baseline_matched.py +116 -0
- holt/cli.py +616 -0
- holt/discover.py +497 -0
- holt/evidence/__init__.py +3 -0
- holt/evidence/fixtures.py +154 -0
- holt/evidence/github_graphql.py +538 -0
- holt/evidence/provider.py +77 -0
- holt/evidence/redact.py +79 -0
- holt/issues.py +41 -0
- holt/model.py +516 -0
- holt/profile.py +126 -0
- holt/report.py +157 -0
- holt/tui/__init__.py +0 -0
- holt/tui/animation.py +84 -0
- holt/tui/app.py +294 -0
- holt/tui/clipboard.py +89 -0
- holt/tui/commands.py +134 -0
- holt/tui/discovery.py +305 -0
- holt/tui/env.py +49 -0
- holt/tui/events.py +245 -0
- holt/tui/mascot.py +121 -0
- holt/tui/models.py +590 -0
- holt/tui/observe.py +297 -0
- holt/tui/screens/__init__.py +35 -0
- holt/tui/screens/assessment.py +337 -0
- holt/tui/screens/confirm.py +62 -0
- holt/tui/screens/discover.py +444 -0
- holt/tui/screens/home.py +519 -0
- holt/tui/screens/inspector.py +106 -0
- holt/tui/screens/live.py +335 -0
- holt/tui/screens/models.py +393 -0
- holt/tui/screens/next_steps.py +425 -0
- holt/tui/screens/profile.py +129 -0
- holt/tui/session.py +711 -0
- holt/tui/store.py +458 -0
- holt/tui/theme.py +479 -0
- holt/tui/visual.py +33 -0
- holt/tui/widgets/__init__.py +0 -0
- holt/tui/widgets/candidates.py +78 -0
- holt/tui/widgets/claims.py +59 -0
- holt/tui/widgets/disclosure.py +121 -0
- holt/tui/widgets/evidence.py +121 -0
- holt/tui/widgets/masthead.py +122 -0
- holt/tui/widgets/recent.py +167 -0
- holt/tui/widgets/scrolling.py +38 -0
- holt/tui/widgets/stages.py +232 -0
- holt/types.py +48 -0
- holt_cli-0.1.0.dist-info/METADATA +198 -0
- holt_cli-0.1.0.dist-info/RECORD +65 -0
- holt_cli-0.1.0.dist-info/WHEEL +4 -0
- holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
- holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
- holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/tui/screens/live.py
ADDED
|
@@ -0,0 +1,335 @@
|
|
|
1
|
+
"""Watching the pipeline work.
|
|
2
|
+
|
|
3
|
+
The screen owns no engine knowledge. It drains a `Session`'s event queue and
|
|
4
|
+
renders what arrives, dispatching on event type through `HANDLERS` — a dict, not
|
|
5
|
+
a chain of branches. Adding an event to the schema means adding an entry here;
|
|
6
|
+
an event with no entry still renders, as a dim line, because a screen that
|
|
7
|
+
raises on an unfamiliar event is a screen that breaks every time a stage learns
|
|
8
|
+
something new.
|
|
9
|
+
|
|
10
|
+
The screen watches; it does not own. The events are drained by the app, which
|
|
11
|
+
ticks whether or not this screen exists, and rendered here from the session's
|
|
12
|
+
log. Two things follow. Leaving does not stop the run — escape means "stop
|
|
13
|
+
looking", and stopping has its own key and its own confirmation. And coming
|
|
14
|
+
back replays the log from the start, so a run rejoined half way through shows
|
|
15
|
+
everything it did while nobody was watching.
|
|
16
|
+
|
|
17
|
+
When a run finishes with this screen up, it moves to the report. A finished run
|
|
18
|
+
has nothing left to watch, and making someone press a key to leave a screen that
|
|
19
|
+
is done is a small insult repeated every time. Storing the result is the app's
|
|
20
|
+
job, not this screen's: a screen that has been popped cannot store anything, and
|
|
21
|
+
that is exactly how a completed assessment used to get lost.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from typing import Callable
|
|
27
|
+
|
|
28
|
+
from rich.text import Text
|
|
29
|
+
from textual.app import ComposeResult
|
|
30
|
+
from textual.containers import Horizontal, VerticalScroll
|
|
31
|
+
from textual.screen import Screen
|
|
32
|
+
from textual.widgets import Footer
|
|
33
|
+
|
|
34
|
+
from holt.tui import animation, events, mascot, theme
|
|
35
|
+
from holt.tui.visual import Line
|
|
36
|
+
from holt.types import T_CUTOFF
|
|
37
|
+
from holt.tui.widgets.masthead import Cat
|
|
38
|
+
from holt.tui.widgets.stages import DroppedFinding, EmittedFinding, StageList, unknown
|
|
39
|
+
|
|
40
|
+
POLL_SECONDS = 0.05
|
|
41
|
+
|
|
42
|
+
#: How long the finished screen stays up before the report replaces it. Long
|
|
43
|
+
#: enough to see the last stage land, short enough not to feel like waiting.
|
|
44
|
+
HANDOFF_SECONDS = 0.45
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def evidence_line(event: events.EvidenceLoaded) -> str:
|
|
48
|
+
"""What the run read, and the boundary it actually stopped at.
|
|
49
|
+
|
|
50
|
+
The date is the provider's, never a constant written here. T = 2026-06-01
|
|
51
|
+
is an evaluation device: a live run reads through today, and a screen that
|
|
52
|
+
printed T anyway would claim a holdout the run did not apply. Naming the
|
|
53
|
+
holdout only when the cutoff really is T keeps the benchmark's own framing
|
|
54
|
+
where it is true and out of the way where it is not.
|
|
55
|
+
"""
|
|
56
|
+
if event.cutoff is None:
|
|
57
|
+
return f"{event.count} evidence records"
|
|
58
|
+
day = event.cutoff.date().isoformat()
|
|
59
|
+
if event.cutoff == T_CUTOFF:
|
|
60
|
+
return f"{event.count} evidence records · holdout window, ≤ {day}"
|
|
61
|
+
return f"{event.count} evidence records · read through {day}"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class LiveScreen(Screen):
|
|
65
|
+
BINDINGS = [
|
|
66
|
+
# Named "leave running" rather than "home" because that is the fact
|
|
67
|
+
# someone needs at the moment they press it.
|
|
68
|
+
("escape", "home", "leave running"),
|
|
69
|
+
("a", "assessment", "report"),
|
|
70
|
+
("ctrl+x", "stop", "stop"),
|
|
71
|
+
("q", "quit", "quit"),
|
|
72
|
+
]
|
|
73
|
+
|
|
74
|
+
#: What the chrome calls this. A class attribute so `TraceScreen` differs
|
|
75
|
+
#: by the one word that is actually different between them.
|
|
76
|
+
chrome_title = "assessing"
|
|
77
|
+
_spend = ""
|
|
78
|
+
_handed_off = False
|
|
79
|
+
#: How far through the session's log this screen has rendered. Rendering
|
|
80
|
+
#: from the log rather than the queue is what lets a second visit replay a
|
|
81
|
+
#: run from its beginning: the cursor starts at zero on a fresh screen.
|
|
82
|
+
_cursor = 0
|
|
83
|
+
|
|
84
|
+
def compose(self) -> ComposeResult:
|
|
85
|
+
with Horizontal(id="chrome"):
|
|
86
|
+
yield Cat("working", id="cat", classes="chrome-cat")
|
|
87
|
+
yield Line(self.chrome_title, id="chrome-left")
|
|
88
|
+
yield Line("", id="chrome-right")
|
|
89
|
+
yield Line("─" * 240, classes="rule")
|
|
90
|
+
yield StageList(id="stages")
|
|
91
|
+
yield VerticalScroll(id="stream")
|
|
92
|
+
yield Line("─" * 240, classes="rule")
|
|
93
|
+
yield Footer()
|
|
94
|
+
|
|
95
|
+
def on_mount(self) -> None:
|
|
96
|
+
self._paint_chrome()
|
|
97
|
+
self.set_interval(POLL_SECONDS, self._pump)
|
|
98
|
+
|
|
99
|
+
def _paint_chrome(self) -> None:
|
|
100
|
+
options = self.app.session.options
|
|
101
|
+
parts = [options.repo, options.mode]
|
|
102
|
+
if self._spend:
|
|
103
|
+
parts.append(self._spend)
|
|
104
|
+
self.query_one("#chrome-right", Line).update(
|
|
105
|
+
Text(" ".join(parts), style=theme.FAINT)
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
# ─── event pump ─────────────────────────────────────────────────────────
|
|
109
|
+
|
|
110
|
+
def _pump(self) -> None:
|
|
111
|
+
"""Render whatever the app has absorbed since the last tick.
|
|
112
|
+
|
|
113
|
+
Never drains the session itself. The app does that for every run at
|
|
114
|
+
once, so this screen can be absent for a minute and still catch up.
|
|
115
|
+
"""
|
|
116
|
+
log = self.app.session.log
|
|
117
|
+
pending = log[self._cursor :]
|
|
118
|
+
self._cursor = len(log)
|
|
119
|
+
for event in pending:
|
|
120
|
+
self._handle(event)
|
|
121
|
+
|
|
122
|
+
def _handle(self, event: events.Event) -> None:
|
|
123
|
+
handler = HANDLERS.get(type(event))
|
|
124
|
+
if handler is None:
|
|
125
|
+
self._append(unknown(event))
|
|
126
|
+
return
|
|
127
|
+
handler(self, event)
|
|
128
|
+
|
|
129
|
+
def _append(self, widget) -> None:
|
|
130
|
+
stream = self.query_one("#stream", VerticalScroll)
|
|
131
|
+
stream.mount(widget)
|
|
132
|
+
stream.scroll_end(animate=False)
|
|
133
|
+
|
|
134
|
+
def _line(self, text: Text) -> None:
|
|
135
|
+
line = Line(text, classes="finding")
|
|
136
|
+
self._append(line)
|
|
137
|
+
animation.reveal(line)
|
|
138
|
+
|
|
139
|
+
# ─── per-event rendering ────────────────────────────────────────────────
|
|
140
|
+
|
|
141
|
+
def _on_run_started(self, event: events.RunStarted) -> None:
|
|
142
|
+
return
|
|
143
|
+
|
|
144
|
+
def _on_evidence_loaded(self, event: events.EvidenceLoaded) -> None:
|
|
145
|
+
self._line(Text(evidence_line(event), style=theme.FAINT))
|
|
146
|
+
|
|
147
|
+
def _on_stage_started(self, event: events.StageStarted) -> None:
|
|
148
|
+
self.query_one("#stages", StageList).ensure(event.stage).started(event.model)
|
|
149
|
+
|
|
150
|
+
def _on_stage_finished(self, event: events.StageFinished) -> None:
|
|
151
|
+
self.query_one("#stages", StageList).ensure(event.stage).finished(
|
|
152
|
+
event.summary, event.seconds
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
def _on_finding(self, event: events.FindingEmitted) -> None:
|
|
156
|
+
self._append(EmittedFinding(event, repo=self.app.session.options.repo))
|
|
157
|
+
|
|
158
|
+
def _on_dropped(self, event: events.FindingDropped) -> None:
|
|
159
|
+
# The cat notices. This is the moment the pipeline exists to make, and
|
|
160
|
+
# the mascot reports state or it would not be here at all.
|
|
161
|
+
self._mood("claim_dropped")
|
|
162
|
+
self._append(DroppedFinding(event))
|
|
163
|
+
|
|
164
|
+
def _on_resolved(self, event: events.EvidenceResolved) -> None:
|
|
165
|
+
# A lookup that succeeds is not news; only a failure is shown, and the
|
|
166
|
+
# drop that follows says which claim it cost. Labelled as a lookup
|
|
167
|
+
# because that is the whole point: no model decided this. Stage D asked
|
|
168
|
+
# the provider for the record and there was not one.
|
|
169
|
+
if event.resolved:
|
|
170
|
+
return
|
|
171
|
+
text = Text()
|
|
172
|
+
text.append("lookup ", style=theme.DIM)
|
|
173
|
+
text.append(event.evidence_id, style=f"{theme.DROP} strike")
|
|
174
|
+
text.append(" not found", style=theme.DROP)
|
|
175
|
+
self._line(text)
|
|
176
|
+
|
|
177
|
+
def _on_usage(self, event: events.UsageUpdated) -> None:
|
|
178
|
+
# Spend belongs in the chrome, not the stream: it is a running total,
|
|
179
|
+
# not something that happened.
|
|
180
|
+
#
|
|
181
|
+
# Never shown during a replay. `ReplayModel` reports the token counts
|
|
182
|
+
# the *original* run recorded, which price out to a real number, and a
|
|
183
|
+
# dollar figure on a run that called nothing would say you had just
|
|
184
|
+
# spent money you did not spend.
|
|
185
|
+
if self.app.session.options.replay:
|
|
186
|
+
return
|
|
187
|
+
if not event.input_tokens and not event.output_tokens:
|
|
188
|
+
return
|
|
189
|
+
self._spend = (
|
|
190
|
+
f"{event.input_tokens:,} in / {event.output_tokens:,} out "
|
|
191
|
+
f"${event.cost_usd:.4f}"
|
|
192
|
+
)
|
|
193
|
+
self._paint_chrome()
|
|
194
|
+
|
|
195
|
+
def _on_retry(self, event: events.Retry) -> None:
|
|
196
|
+
self._line(
|
|
197
|
+
Text(
|
|
198
|
+
f"retrying {event.stage} (attempt {event.attempt}) {event.reason}",
|
|
199
|
+
style=theme.DIM,
|
|
200
|
+
)
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
def _mood(self, mood: str) -> None:
|
|
204
|
+
try:
|
|
205
|
+
self.query_one("#cat", Cat).set_mood(mood)
|
|
206
|
+
except Exception: # noqa: BLE001 - the cat is never load-bearing
|
|
207
|
+
pass
|
|
208
|
+
|
|
209
|
+
def _on_failed(self, event: events.RunFailed) -> None:
|
|
210
|
+
# Every spinner stops. A stage left turning after a failure reads as
|
|
211
|
+
# still working, which is the one thing it is definitely not doing.
|
|
212
|
+
self.query_one("#stages", StageList).settle()
|
|
213
|
+
self._line(Text(event.error, style=theme.DROP))
|
|
214
|
+
self._line(Text("escape to go back", style=theme.FAINT))
|
|
215
|
+
|
|
216
|
+
def _on_cancelled(self, event: events.RunCancelled) -> None:
|
|
217
|
+
# Not styled as a failure. Nothing went wrong: the run did what it was
|
|
218
|
+
# told. Naming the stages that had finished is the one useful thing to
|
|
219
|
+
# say, because on live those were paid for.
|
|
220
|
+
self.query_one("#stages", StageList).settle()
|
|
221
|
+
self._line(Text("stopped", style=theme.DIM))
|
|
222
|
+
if event.completed_stages:
|
|
223
|
+
done = ", ".join(dict.fromkeys(event.completed_stages))
|
|
224
|
+
self._line(Text(f"completed before stopping: {done}", style=theme.FAINT))
|
|
225
|
+
self._line(Text("escape to go back", style=theme.FAINT))
|
|
226
|
+
|
|
227
|
+
def _on_finished(self, event: events.RunFinished) -> None:
|
|
228
|
+
verdict = getattr(event.assessment, "verdict", None)
|
|
229
|
+
self._mood(mascot.mood_for_verdict(getattr(verdict, "value", "")))
|
|
230
|
+
self.query_one("#stages", StageList).settle()
|
|
231
|
+
if self._handed_off:
|
|
232
|
+
return
|
|
233
|
+
self._handed_off = True
|
|
234
|
+
self.set_timer(HANDOFF_SECONDS, self._to_report)
|
|
235
|
+
|
|
236
|
+
def _to_report(self) -> None:
|
|
237
|
+
# Only if nobody has navigated away in the meantime.
|
|
238
|
+
if self.app.screen is self:
|
|
239
|
+
self.app.show_assessment()
|
|
240
|
+
|
|
241
|
+
# ─── actions ────────────────────────────────────────────────────────────
|
|
242
|
+
|
|
243
|
+
def action_assessment(self) -> None:
|
|
244
|
+
self.app.show_assessment()
|
|
245
|
+
|
|
246
|
+
def action_home(self) -> None:
|
|
247
|
+
self.app.go_home()
|
|
248
|
+
|
|
249
|
+
def action_stop(self) -> None:
|
|
250
|
+
self.app.confirm_stop(self.app.session)
|
|
251
|
+
|
|
252
|
+
def action_quit(self) -> None:
|
|
253
|
+
# Through the app, so a run still in flight is named before it dies.
|
|
254
|
+
self.app.action_quit()
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
class TraceScreen(LiveScreen):
|
|
258
|
+
"""The same stream, for a run that is already over.
|
|
259
|
+
|
|
260
|
+
An assessment reopened out of the store carries the run's own events, so
|
|
261
|
+
the trace behind it is not lost with the process that produced it — it is
|
|
262
|
+
rendered here, by exactly the code that rendered it live. Only three things
|
|
263
|
+
differ, and all three are about the run being finished: the chrome says so,
|
|
264
|
+
escape goes back to the report rather than leaving a run going, and there
|
|
265
|
+
is nothing to stop.
|
|
266
|
+
"""
|
|
267
|
+
|
|
268
|
+
BINDINGS = [
|
|
269
|
+
("escape", "back", "back"),
|
|
270
|
+
("a", "assessment", "report"),
|
|
271
|
+
("q", "quit", "quit"),
|
|
272
|
+
]
|
|
273
|
+
|
|
274
|
+
chrome_title = "trace"
|
|
275
|
+
_settled = False
|
|
276
|
+
|
|
277
|
+
def check_action(self, action: str, parameters) -> bool | None:
|
|
278
|
+
"""Hide what a finished run cannot do.
|
|
279
|
+
|
|
280
|
+
Textual collects `BINDINGS` up the class hierarchy, so this screen
|
|
281
|
+
inherits the live view's `stop` and its escape-to-leave-running. Both
|
|
282
|
+
are about a run in flight. Returning False takes them off the footer
|
|
283
|
+
rather than leaving keys advertised that would do nothing.
|
|
284
|
+
"""
|
|
285
|
+
if action in ("stop", "home"):
|
|
286
|
+
return False
|
|
287
|
+
return True
|
|
288
|
+
|
|
289
|
+
def _pump(self) -> None:
|
|
290
|
+
super()._pump()
|
|
291
|
+
if self._settled:
|
|
292
|
+
return
|
|
293
|
+
self._settled = True
|
|
294
|
+
# A stored stream ends where the run ended and carries no `RunFinished`
|
|
295
|
+
# to settle the spinners on — this is a run that is over, and a stage
|
|
296
|
+
# left turning would say it was still working. The cat lands on the
|
|
297
|
+
# verdict for the same reason: the last thing it saw in the stream was
|
|
298
|
+
# a claim being dropped, which was true a minute into the run and is
|
|
299
|
+
# not the state of the report you pressed `t` on.
|
|
300
|
+
self.query_one("#stages", StageList).settle()
|
|
301
|
+
verdict = getattr(self.app.session.assessment, "verdict", None)
|
|
302
|
+
self._mood(mascot.mood_for_verdict(getattr(verdict, "value", "")))
|
|
303
|
+
|
|
304
|
+
def action_back(self) -> None:
|
|
305
|
+
self.app.pop_screen()
|
|
306
|
+
|
|
307
|
+
def action_assessment(self) -> None:
|
|
308
|
+
# The report is the screen underneath this one, so `a` goes back to it
|
|
309
|
+
# rather than pushing a second copy of it on top.
|
|
310
|
+
self.app.pop_screen()
|
|
311
|
+
|
|
312
|
+
def action_quit(self) -> None:
|
|
313
|
+
self.app.exit()
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
#: Event type → renderer. The registry is the extension point: a new event class
|
|
317
|
+
#: is a new entry, and nothing here needs restructuring to accept it.
|
|
318
|
+
HANDLERS: dict[type, Callable] = {
|
|
319
|
+
events.RunStarted: LiveScreen._on_run_started,
|
|
320
|
+
events.EvidenceLoaded: LiveScreen._on_evidence_loaded,
|
|
321
|
+
events.StageStarted: LiveScreen._on_stage_started,
|
|
322
|
+
events.StageFinished: LiveScreen._on_stage_finished,
|
|
323
|
+
events.FindingEmitted: LiveScreen._on_finding,
|
|
324
|
+
events.FindingDropped: LiveScreen._on_dropped,
|
|
325
|
+
events.EvidenceResolved: LiveScreen._on_resolved,
|
|
326
|
+
events.UsageUpdated: LiveScreen._on_usage,
|
|
327
|
+
events.Retry: LiveScreen._on_retry,
|
|
328
|
+
events.RunFailed: LiveScreen._on_failed,
|
|
329
|
+
events.RunCancelled: LiveScreen._on_cancelled,
|
|
330
|
+
events.RunFinished: LiveScreen._on_finished,
|
|
331
|
+
# `ToolResponse` carries the raw payload for future use and is deliberately
|
|
332
|
+
# not rendered: the findings read off it are shown instead, and printing
|
|
333
|
+
# both would be the log spew this view exists to avoid.
|
|
334
|
+
events.ToolResponse: lambda self, event: None,
|
|
335
|
+
}
|