holt-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. holt/__init__.py +0 -0
  2. holt/agent/__init__.py +0 -0
  3. holt/agent/entry.py +86 -0
  4. holt/agent/findings.py +49 -0
  5. holt/agent/landing.py +154 -0
  6. holt/agent/pipeline.py +244 -0
  7. holt/agent/progression.py +408 -0
  8. holt/agent/signals.py +220 -0
  9. holt/agent/stages.py +533 -0
  10. holt/agent/verdict.py +226 -0
  11. holt/agent/verify.py +140 -0
  12. holt/baseline.py +89 -0
  13. holt/baseline_matched.py +116 -0
  14. holt/cli.py +616 -0
  15. holt/discover.py +497 -0
  16. holt/evidence/__init__.py +3 -0
  17. holt/evidence/fixtures.py +154 -0
  18. holt/evidence/github_graphql.py +538 -0
  19. holt/evidence/provider.py +77 -0
  20. holt/evidence/redact.py +79 -0
  21. holt/issues.py +41 -0
  22. holt/model.py +516 -0
  23. holt/profile.py +126 -0
  24. holt/report.py +157 -0
  25. holt/tui/__init__.py +0 -0
  26. holt/tui/animation.py +84 -0
  27. holt/tui/app.py +294 -0
  28. holt/tui/clipboard.py +89 -0
  29. holt/tui/commands.py +134 -0
  30. holt/tui/discovery.py +305 -0
  31. holt/tui/env.py +49 -0
  32. holt/tui/events.py +245 -0
  33. holt/tui/mascot.py +121 -0
  34. holt/tui/models.py +590 -0
  35. holt/tui/observe.py +297 -0
  36. holt/tui/screens/__init__.py +35 -0
  37. holt/tui/screens/assessment.py +337 -0
  38. holt/tui/screens/confirm.py +62 -0
  39. holt/tui/screens/discover.py +444 -0
  40. holt/tui/screens/home.py +519 -0
  41. holt/tui/screens/inspector.py +106 -0
  42. holt/tui/screens/live.py +335 -0
  43. holt/tui/screens/models.py +393 -0
  44. holt/tui/screens/next_steps.py +425 -0
  45. holt/tui/screens/profile.py +129 -0
  46. holt/tui/session.py +711 -0
  47. holt/tui/store.py +458 -0
  48. holt/tui/theme.py +479 -0
  49. holt/tui/visual.py +33 -0
  50. holt/tui/widgets/__init__.py +0 -0
  51. holt/tui/widgets/candidates.py +78 -0
  52. holt/tui/widgets/claims.py +59 -0
  53. holt/tui/widgets/disclosure.py +121 -0
  54. holt/tui/widgets/evidence.py +121 -0
  55. holt/tui/widgets/masthead.py +122 -0
  56. holt/tui/widgets/recent.py +167 -0
  57. holt/tui/widgets/scrolling.py +38 -0
  58. holt/tui/widgets/stages.py +232 -0
  59. holt/types.py +48 -0
  60. holt_cli-0.1.0.dist-info/METADATA +198 -0
  61. holt_cli-0.1.0.dist-info/RECORD +65 -0
  62. holt_cli-0.1.0.dist-info/WHEEL +4 -0
  63. holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
  64. holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
  65. holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
@@ -0,0 +1,335 @@
1
+ """Watching the pipeline work.
2
+
3
+ The screen owns no engine knowledge. It drains a `Session`'s event queue and
4
+ renders what arrives, dispatching on event type through `HANDLERS` — a dict, not
5
+ a chain of branches. Adding an event to the schema means adding an entry here;
6
+ an event with no entry still renders, as a dim line, because a screen that
7
+ raises on an unfamiliar event is a screen that breaks every time a stage learns
8
+ something new.
9
+
10
+ The screen watches; it does not own. The events are drained by the app, which
11
+ ticks whether or not this screen exists, and rendered here from the session's
12
+ log. Two things follow. Leaving does not stop the run — escape means "stop
13
+ looking", and stopping has its own key and its own confirmation. And coming
14
+ back replays the log from the start, so a run rejoined half way through shows
15
+ everything it did while nobody was watching.
16
+
17
+ When a run finishes with this screen up, it moves to the report. A finished run
18
+ has nothing left to watch, and making someone press a key to leave a screen that
19
+ is done is a small insult repeated every time. Storing the result is the app's
20
+ job, not this screen's: a screen that has been popped cannot store anything, and
21
+ that is exactly how a completed assessment used to get lost.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from typing import Callable
27
+
28
+ from rich.text import Text
29
+ from textual.app import ComposeResult
30
+ from textual.containers import Horizontal, VerticalScroll
31
+ from textual.screen import Screen
32
+ from textual.widgets import Footer
33
+
34
+ from holt.tui import animation, events, mascot, theme
35
+ from holt.tui.visual import Line
36
+ from holt.types import T_CUTOFF
37
+ from holt.tui.widgets.masthead import Cat
38
+ from holt.tui.widgets.stages import DroppedFinding, EmittedFinding, StageList, unknown
39
+
40
+ POLL_SECONDS = 0.05
41
+
42
+ #: How long the finished screen stays up before the report replaces it. Long
43
+ #: enough to see the last stage land, short enough not to feel like waiting.
44
+ HANDOFF_SECONDS = 0.45
45
+
46
+
47
+ def evidence_line(event: events.EvidenceLoaded) -> str:
48
+ """What the run read, and the boundary it actually stopped at.
49
+
50
+ The date is the provider's, never a constant written here. T = 2026-06-01
51
+ is an evaluation device: a live run reads through today, and a screen that
52
+ printed T anyway would claim a holdout the run did not apply. Naming the
53
+ holdout only when the cutoff really is T keeps the benchmark's own framing
54
+ where it is true and out of the way where it is not.
55
+ """
56
+ if event.cutoff is None:
57
+ return f"{event.count} evidence records"
58
+ day = event.cutoff.date().isoformat()
59
+ if event.cutoff == T_CUTOFF:
60
+ return f"{event.count} evidence records · holdout window, ≤ {day}"
61
+ return f"{event.count} evidence records · read through {day}"
62
+
63
+
64
+ class LiveScreen(Screen):
65
+ BINDINGS = [
66
+ # Named "leave running" rather than "home" because that is the fact
67
+ # someone needs at the moment they press it.
68
+ ("escape", "home", "leave running"),
69
+ ("a", "assessment", "report"),
70
+ ("ctrl+x", "stop", "stop"),
71
+ ("q", "quit", "quit"),
72
+ ]
73
+
74
+ #: What the chrome calls this. A class attribute so `TraceScreen` differs
75
+ #: by the one word that is actually different between them.
76
+ chrome_title = "assessing"
77
+ _spend = ""
78
+ _handed_off = False
79
+ #: How far through the session's log this screen has rendered. Rendering
80
+ #: from the log rather than the queue is what lets a second visit replay a
81
+ #: run from its beginning: the cursor starts at zero on a fresh screen.
82
+ _cursor = 0
83
+
84
+ def compose(self) -> ComposeResult:
85
+ with Horizontal(id="chrome"):
86
+ yield Cat("working", id="cat", classes="chrome-cat")
87
+ yield Line(self.chrome_title, id="chrome-left")
88
+ yield Line("", id="chrome-right")
89
+ yield Line("─" * 240, classes="rule")
90
+ yield StageList(id="stages")
91
+ yield VerticalScroll(id="stream")
92
+ yield Line("─" * 240, classes="rule")
93
+ yield Footer()
94
+
95
+ def on_mount(self) -> None:
96
+ self._paint_chrome()
97
+ self.set_interval(POLL_SECONDS, self._pump)
98
+
99
+ def _paint_chrome(self) -> None:
100
+ options = self.app.session.options
101
+ parts = [options.repo, options.mode]
102
+ if self._spend:
103
+ parts.append(self._spend)
104
+ self.query_one("#chrome-right", Line).update(
105
+ Text(" ".join(parts), style=theme.FAINT)
106
+ )
107
+
108
+ # ─── event pump ─────────────────────────────────────────────────────────
109
+
110
+ def _pump(self) -> None:
111
+ """Render whatever the app has absorbed since the last tick.
112
+
113
+ Never drains the session itself. The app does that for every run at
114
+ once, so this screen can be absent for a minute and still catch up.
115
+ """
116
+ log = self.app.session.log
117
+ pending = log[self._cursor :]
118
+ self._cursor = len(log)
119
+ for event in pending:
120
+ self._handle(event)
121
+
122
+ def _handle(self, event: events.Event) -> None:
123
+ handler = HANDLERS.get(type(event))
124
+ if handler is None:
125
+ self._append(unknown(event))
126
+ return
127
+ handler(self, event)
128
+
129
+ def _append(self, widget) -> None:
130
+ stream = self.query_one("#stream", VerticalScroll)
131
+ stream.mount(widget)
132
+ stream.scroll_end(animate=False)
133
+
134
+ def _line(self, text: Text) -> None:
135
+ line = Line(text, classes="finding")
136
+ self._append(line)
137
+ animation.reveal(line)
138
+
139
+ # ─── per-event rendering ────────────────────────────────────────────────
140
+
141
+ def _on_run_started(self, event: events.RunStarted) -> None:
142
+ return
143
+
144
+ def _on_evidence_loaded(self, event: events.EvidenceLoaded) -> None:
145
+ self._line(Text(evidence_line(event), style=theme.FAINT))
146
+
147
+ def _on_stage_started(self, event: events.StageStarted) -> None:
148
+ self.query_one("#stages", StageList).ensure(event.stage).started(event.model)
149
+
150
+ def _on_stage_finished(self, event: events.StageFinished) -> None:
151
+ self.query_one("#stages", StageList).ensure(event.stage).finished(
152
+ event.summary, event.seconds
153
+ )
154
+
155
+ def _on_finding(self, event: events.FindingEmitted) -> None:
156
+ self._append(EmittedFinding(event, repo=self.app.session.options.repo))
157
+
158
+ def _on_dropped(self, event: events.FindingDropped) -> None:
159
+ # The cat notices. This is the moment the pipeline exists to make, and
160
+ # the mascot reports state or it would not be here at all.
161
+ self._mood("claim_dropped")
162
+ self._append(DroppedFinding(event))
163
+
164
+ def _on_resolved(self, event: events.EvidenceResolved) -> None:
165
+ # A lookup that succeeds is not news; only a failure is shown, and the
166
+ # drop that follows says which claim it cost. Labelled as a lookup
167
+ # because that is the whole point: no model decided this. Stage D asked
168
+ # the provider for the record and there was not one.
169
+ if event.resolved:
170
+ return
171
+ text = Text()
172
+ text.append("lookup ", style=theme.DIM)
173
+ text.append(event.evidence_id, style=f"{theme.DROP} strike")
174
+ text.append(" not found", style=theme.DROP)
175
+ self._line(text)
176
+
177
+ def _on_usage(self, event: events.UsageUpdated) -> None:
178
+ # Spend belongs in the chrome, not the stream: it is a running total,
179
+ # not something that happened.
180
+ #
181
+ # Never shown during a replay. `ReplayModel` reports the token counts
182
+ # the *original* run recorded, which price out to a real number, and a
183
+ # dollar figure on a run that called nothing would say you had just
184
+ # spent money you did not spend.
185
+ if self.app.session.options.replay:
186
+ return
187
+ if not event.input_tokens and not event.output_tokens:
188
+ return
189
+ self._spend = (
190
+ f"{event.input_tokens:,} in / {event.output_tokens:,} out "
191
+ f"${event.cost_usd:.4f}"
192
+ )
193
+ self._paint_chrome()
194
+
195
+ def _on_retry(self, event: events.Retry) -> None:
196
+ self._line(
197
+ Text(
198
+ f"retrying {event.stage} (attempt {event.attempt}) {event.reason}",
199
+ style=theme.DIM,
200
+ )
201
+ )
202
+
203
+ def _mood(self, mood: str) -> None:
204
+ try:
205
+ self.query_one("#cat", Cat).set_mood(mood)
206
+ except Exception: # noqa: BLE001 - the cat is never load-bearing
207
+ pass
208
+
209
+ def _on_failed(self, event: events.RunFailed) -> None:
210
+ # Every spinner stops. A stage left turning after a failure reads as
211
+ # still working, which is the one thing it is definitely not doing.
212
+ self.query_one("#stages", StageList).settle()
213
+ self._line(Text(event.error, style=theme.DROP))
214
+ self._line(Text("escape to go back", style=theme.FAINT))
215
+
216
+ def _on_cancelled(self, event: events.RunCancelled) -> None:
217
+ # Not styled as a failure. Nothing went wrong: the run did what it was
218
+ # told. Naming the stages that had finished is the one useful thing to
219
+ # say, because on live those were paid for.
220
+ self.query_one("#stages", StageList).settle()
221
+ self._line(Text("stopped", style=theme.DIM))
222
+ if event.completed_stages:
223
+ done = ", ".join(dict.fromkeys(event.completed_stages))
224
+ self._line(Text(f"completed before stopping: {done}", style=theme.FAINT))
225
+ self._line(Text("escape to go back", style=theme.FAINT))
226
+
227
+ def _on_finished(self, event: events.RunFinished) -> None:
228
+ verdict = getattr(event.assessment, "verdict", None)
229
+ self._mood(mascot.mood_for_verdict(getattr(verdict, "value", "")))
230
+ self.query_one("#stages", StageList).settle()
231
+ if self._handed_off:
232
+ return
233
+ self._handed_off = True
234
+ self.set_timer(HANDOFF_SECONDS, self._to_report)
235
+
236
+ def _to_report(self) -> None:
237
+ # Only if nobody has navigated away in the meantime.
238
+ if self.app.screen is self:
239
+ self.app.show_assessment()
240
+
241
+ # ─── actions ────────────────────────────────────────────────────────────
242
+
243
+ def action_assessment(self) -> None:
244
+ self.app.show_assessment()
245
+
246
+ def action_home(self) -> None:
247
+ self.app.go_home()
248
+
249
+ def action_stop(self) -> None:
250
+ self.app.confirm_stop(self.app.session)
251
+
252
+ def action_quit(self) -> None:
253
+ # Through the app, so a run still in flight is named before it dies.
254
+ self.app.action_quit()
255
+
256
+
257
+ class TraceScreen(LiveScreen):
258
+ """The same stream, for a run that is already over.
259
+
260
+ An assessment reopened out of the store carries the run's own events, so
261
+ the trace behind it is not lost with the process that produced it — it is
262
+ rendered here, by exactly the code that rendered it live. Only three things
263
+ differ, and all three are about the run being finished: the chrome says so,
264
+ escape goes back to the report rather than leaving a run going, and there
265
+ is nothing to stop.
266
+ """
267
+
268
+ BINDINGS = [
269
+ ("escape", "back", "back"),
270
+ ("a", "assessment", "report"),
271
+ ("q", "quit", "quit"),
272
+ ]
273
+
274
+ chrome_title = "trace"
275
+ _settled = False
276
+
277
+ def check_action(self, action: str, parameters) -> bool | None:
278
+ """Hide what a finished run cannot do.
279
+
280
+ Textual collects `BINDINGS` up the class hierarchy, so this screen
281
+ inherits the live view's `stop` and its escape-to-leave-running. Both
282
+ are about a run in flight. Returning False takes them off the footer
283
+ rather than leaving keys advertised that would do nothing.
284
+ """
285
+ if action in ("stop", "home"):
286
+ return False
287
+ return True
288
+
289
+ def _pump(self) -> None:
290
+ super()._pump()
291
+ if self._settled:
292
+ return
293
+ self._settled = True
294
+ # A stored stream ends where the run ended and carries no `RunFinished`
295
+ # to settle the spinners on — this is a run that is over, and a stage
296
+ # left turning would say it was still working. The cat lands on the
297
+ # verdict for the same reason: the last thing it saw in the stream was
298
+ # a claim being dropped, which was true a minute into the run and is
299
+ # not the state of the report you pressed `t` on.
300
+ self.query_one("#stages", StageList).settle()
301
+ verdict = getattr(self.app.session.assessment, "verdict", None)
302
+ self._mood(mascot.mood_for_verdict(getattr(verdict, "value", "")))
303
+
304
+ def action_back(self) -> None:
305
+ self.app.pop_screen()
306
+
307
+ def action_assessment(self) -> None:
308
+ # The report is the screen underneath this one, so `a` goes back to it
309
+ # rather than pushing a second copy of it on top.
310
+ self.app.pop_screen()
311
+
312
+ def action_quit(self) -> None:
313
+ self.app.exit()
314
+
315
+
316
+ #: Event type → renderer. The registry is the extension point: a new event class
317
+ #: is a new entry, and nothing here needs restructuring to accept it.
318
+ HANDLERS: dict[type, Callable] = {
319
+ events.RunStarted: LiveScreen._on_run_started,
320
+ events.EvidenceLoaded: LiveScreen._on_evidence_loaded,
321
+ events.StageStarted: LiveScreen._on_stage_started,
322
+ events.StageFinished: LiveScreen._on_stage_finished,
323
+ events.FindingEmitted: LiveScreen._on_finding,
324
+ events.FindingDropped: LiveScreen._on_dropped,
325
+ events.EvidenceResolved: LiveScreen._on_resolved,
326
+ events.UsageUpdated: LiveScreen._on_usage,
327
+ events.Retry: LiveScreen._on_retry,
328
+ events.RunFailed: LiveScreen._on_failed,
329
+ events.RunCancelled: LiveScreen._on_cancelled,
330
+ events.RunFinished: LiveScreen._on_finished,
331
+ # `ToolResponse` carries the raw payload for future use and is deliberately
332
+ # not rendered: the findings read off it are shown instead, and printing
333
+ # both would be the log spew this view exists to avoid.
334
+ events.ToolResponse: lambda self, event: None,
335
+ }