holt-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- holt/__init__.py +0 -0
- holt/agent/__init__.py +0 -0
- holt/agent/entry.py +86 -0
- holt/agent/findings.py +49 -0
- holt/agent/landing.py +154 -0
- holt/agent/pipeline.py +244 -0
- holt/agent/progression.py +408 -0
- holt/agent/signals.py +220 -0
- holt/agent/stages.py +533 -0
- holt/agent/verdict.py +226 -0
- holt/agent/verify.py +140 -0
- holt/baseline.py +89 -0
- holt/baseline_matched.py +116 -0
- holt/cli.py +616 -0
- holt/discover.py +497 -0
- holt/evidence/__init__.py +3 -0
- holt/evidence/fixtures.py +154 -0
- holt/evidence/github_graphql.py +538 -0
- holt/evidence/provider.py +77 -0
- holt/evidence/redact.py +79 -0
- holt/issues.py +41 -0
- holt/model.py +516 -0
- holt/profile.py +126 -0
- holt/report.py +157 -0
- holt/tui/__init__.py +0 -0
- holt/tui/animation.py +84 -0
- holt/tui/app.py +294 -0
- holt/tui/clipboard.py +89 -0
- holt/tui/commands.py +134 -0
- holt/tui/discovery.py +305 -0
- holt/tui/env.py +49 -0
- holt/tui/events.py +245 -0
- holt/tui/mascot.py +121 -0
- holt/tui/models.py +590 -0
- holt/tui/observe.py +297 -0
- holt/tui/screens/__init__.py +35 -0
- holt/tui/screens/assessment.py +337 -0
- holt/tui/screens/confirm.py +62 -0
- holt/tui/screens/discover.py +444 -0
- holt/tui/screens/home.py +519 -0
- holt/tui/screens/inspector.py +106 -0
- holt/tui/screens/live.py +335 -0
- holt/tui/screens/models.py +393 -0
- holt/tui/screens/next_steps.py +425 -0
- holt/tui/screens/profile.py +129 -0
- holt/tui/session.py +711 -0
- holt/tui/store.py +458 -0
- holt/tui/theme.py +479 -0
- holt/tui/visual.py +33 -0
- holt/tui/widgets/__init__.py +0 -0
- holt/tui/widgets/candidates.py +78 -0
- holt/tui/widgets/claims.py +59 -0
- holt/tui/widgets/disclosure.py +121 -0
- holt/tui/widgets/evidence.py +121 -0
- holt/tui/widgets/masthead.py +122 -0
- holt/tui/widgets/recent.py +167 -0
- holt/tui/widgets/scrolling.py +38 -0
- holt/tui/widgets/stages.py +232 -0
- holt/types.py +48 -0
- holt_cli-0.1.0.dist-info/METADATA +198 -0
- holt_cli-0.1.0.dist-info/RECORD +65 -0
- holt_cli-0.1.0.dist-info/WHEEL +4 -0
- holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
- holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
- holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/cli.py
ADDED
|
@@ -0,0 +1,616 @@
|
|
|
1
|
+
"""holt — is this repository worth your time?"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from datetime import UTC, datetime
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from holt import baseline, model
|
|
11
|
+
from holt.agent import entry, pipeline
|
|
12
|
+
from holt.evidence.fixtures import FixtureProvider
|
|
13
|
+
from holt.evidence.provider import EvidenceProvider
|
|
14
|
+
from holt.report import EntryPoint
|
|
15
|
+
from holt.types import T_CUTOFF, Window
|
|
16
|
+
|
|
17
|
+
# Issue evidence is captured and replayed separately from pull-request evidence,
|
|
18
|
+
# so that adding the ranker did not invalidate a single recorded verdict
|
|
19
|
+
# trajectory. The evaluation of the verdict and the evaluation of the ranking stay
|
|
20
|
+
# independent of each other.
|
|
21
|
+
ISSUE_ROOT = "fixtures/issues"
|
|
22
|
+
PATHFINDER_TRAJECTORIES = "pathfinder"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def normalise(repo: str) -> str:
|
|
26
|
+
repo = repo.strip().rstrip("/")
|
|
27
|
+
if "github.com" in repo:
|
|
28
|
+
repo = repo.split("github.com", 1)[1].lstrip("/:")
|
|
29
|
+
return "/".join(repo.split("/")[:2])
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def as_of_from(args: argparse.Namespace) -> datetime:
|
|
33
|
+
"""How recent the evidence may be.
|
|
34
|
+
|
|
35
|
+
**T = 2026-06-01 is an evaluation device, not a product setting.** It exists so
|
|
36
|
+
labels can be computed from records the agent was never shown. A person asking
|
|
37
|
+
about a repository today wants everything up to today, and cutting them off in
|
|
38
|
+
June throws away the three most relevant months -- badly enough that an active
|
|
39
|
+
repository created in July reports "no outsider activity" and looks dead.
|
|
40
|
+
|
|
41
|
+
So: fixtures answer as of T, because that is what they contain; live runs
|
|
42
|
+
answer as of now, unless `--as-of` says otherwise. `--as-of 2026-06-01` on a
|
|
43
|
+
live run reproduces the benchmark's view.
|
|
44
|
+
"""
|
|
45
|
+
if getattr(args, "as_of", None):
|
|
46
|
+
return datetime.fromisoformat(args.as_of).replace(tzinfo=UTC)
|
|
47
|
+
return datetime.now(UTC) if args.live else T_CUTOFF
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def make_issue_provider(live: bool, as_of: datetime) -> EvidenceProvider:
|
|
51
|
+
if not live:
|
|
52
|
+
return FixtureProvider(Window.PRE_T, root=Path(ISSUE_ROOT))
|
|
53
|
+
from holt.evidence.github_graphql import LiveGitHubIssueProvider
|
|
54
|
+
|
|
55
|
+
return LiveGitHubIssueProvider(Window.PRE_T, cutoff=as_of)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def make_provider(live: bool, as_of: datetime) -> EvidenceProvider:
|
|
59
|
+
if not live:
|
|
60
|
+
return FixtureProvider(Window.PRE_T)
|
|
61
|
+
from holt.evidence.github_graphql import LiveGitHubProvider
|
|
62
|
+
|
|
63
|
+
return LiveGitHubProvider(Window.PRE_T, cutoff=as_of)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def add_entry_points(assessment, repo: str, provider, args) -> None:
|
|
67
|
+
"""Attach a ranked reading order, or say nothing at all.
|
|
68
|
+
|
|
69
|
+
Silent on failure by design: a missing issue fixture means we cannot rank,
|
|
70
|
+
and a tool that invents an entry point when it has no issues is worse than
|
|
71
|
+
one that omits the section.
|
|
72
|
+
"""
|
|
73
|
+
try:
|
|
74
|
+
issues = make_issue_provider(args.live, as_of_from(args)).fetch(repo)
|
|
75
|
+
except FileNotFoundError:
|
|
76
|
+
return
|
|
77
|
+
# Recorded under its own directory, so the ranker's calls and the verdict's
|
|
78
|
+
# calls replay independently and adding one never invalidated the other.
|
|
79
|
+
path = model.TRAJECTORY_DIR / PATHFINDER_TRAJECTORIES / (repo.replace("/", "__") + ".jsonl")
|
|
80
|
+
client = model.ReplayModel(path) if args.replay else model.live_client(path)
|
|
81
|
+
ranked = entry.rank(repo, list(issues), list(provider.fetch(repo)), client)
|
|
82
|
+
assessment.entry_points = [
|
|
83
|
+
EntryPoint(r["evidence_id"], r["first_step"], r.get("why", "")) for r in ranked
|
|
84
|
+
]
|
|
85
|
+
# The ranker runs on its own client, and `--stage pathfinder=` can point it
|
|
86
|
+
# at a different model. The footer names every model that wrote something on
|
|
87
|
+
# the page, so it has to hear about this one too.
|
|
88
|
+
for name in client.usage.models:
|
|
89
|
+
if name not in assessment.models:
|
|
90
|
+
assessment.models.append(name)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def cmd_analyze(args: argparse.Namespace) -> int:
|
|
94
|
+
repo = normalise(args.repo)
|
|
95
|
+
as_of = as_of_from(args)
|
|
96
|
+
provider = make_provider(args.live, as_of)
|
|
97
|
+
# Not built at all under --no-model: the mode's whole claim is that it needs
|
|
98
|
+
# no key and spends nothing, and constructing a live client would demand one.
|
|
99
|
+
client = None if args.no_model else model.build(repo, replay=args.replay)
|
|
100
|
+
|
|
101
|
+
if args.baseline:
|
|
102
|
+
assessment = baseline.assess(repo, provider, client)
|
|
103
|
+
else:
|
|
104
|
+
assessment, trace = pipeline.analyze(
|
|
105
|
+
repo, provider, None if args.no_model else client,
|
|
106
|
+
contributor_days=args.days, as_of=as_of,
|
|
107
|
+
)
|
|
108
|
+
if args.show_verification:
|
|
109
|
+
print(
|
|
110
|
+
f"<!-- findings before verification: {trace.before_verification}, "
|
|
111
|
+
f"after: {trace.after_verification}, dropped: {len(trace.dropped)}, "
|
|
112
|
+
f"unquoted: {len(trace.invented)} -->",
|
|
113
|
+
file=sys.stderr,
|
|
114
|
+
)
|
|
115
|
+
for d in trace.dropped:
|
|
116
|
+
print(f"<!-- DROPPED {d.field}={d.value!r} cited {list(d.evidence_ids)} -->",
|
|
117
|
+
file=sys.stderr)
|
|
118
|
+
for d in trace.invented:
|
|
119
|
+
print(f"<!-- UNQUOTED {d.field}={d.value!r} cited {list(d.evidence_ids)}: "
|
|
120
|
+
"the thread resolves and does not say this -->", file=sys.stderr)
|
|
121
|
+
if not args.baseline and args.entry_points:
|
|
122
|
+
add_entry_points(assessment, repo, provider, args)
|
|
123
|
+
print(assessment.render())
|
|
124
|
+
if not args.replay and client is not None:
|
|
125
|
+
u = client.usage
|
|
126
|
+
print(
|
|
127
|
+
f"<!-- {u.input_tokens} in / {u.output_tokens} out tokens, "
|
|
128
|
+
f"${u.cost_usd:.4f} -->",
|
|
129
|
+
file=sys.stderr,
|
|
130
|
+
)
|
|
131
|
+
return 0
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# A shortlist is the real situation. Nobody has one repository they are deciding
|
|
135
|
+
# about; they have five tabs open. This orders nothing -- the rows come out in the
|
|
136
|
+
# order they were asked for -- because ordering is a claim, and five capabilities
|
|
137
|
+
# have been cut here for making one that a cheap signal already made.
|
|
138
|
+
COMPARE_HEADERS = ("repository", "verdict", "outsiders in", "first reply", "why")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def cmd_compare(args: argparse.Namespace) -> int:
|
|
142
|
+
as_of = as_of_from(args)
|
|
143
|
+
provider = make_provider(args.live, as_of)
|
|
144
|
+
rows = []
|
|
145
|
+
for raw in args.repos:
|
|
146
|
+
repo = normalise(raw)
|
|
147
|
+
client = model.build(repo, replay=args.replay)
|
|
148
|
+
assessment, trace = pipeline.analyze(
|
|
149
|
+
repo, provider, client, contributor_days=args.days, as_of=as_of
|
|
150
|
+
)
|
|
151
|
+
signals = trace.signals
|
|
152
|
+
landed = f"{signals.outsider_merged}/{signals.outsider_threads}"
|
|
153
|
+
reply = (f"{signals.median_first_response_hours:.1f}h"
|
|
154
|
+
if signals.median_first_response_hours is not None else "never")
|
|
155
|
+
# The rule that fired, not a summary of the prose. If nothing fired the
|
|
156
|
+
# verdict came from the default path and saying so is more honest than
|
|
157
|
+
# inventing a reason.
|
|
158
|
+
why = assessment.rules[0] if assessment.rules else "no rule fired"
|
|
159
|
+
why = why if len(why) <= 58 else why[:57].rstrip(" ,;:") + "…"
|
|
160
|
+
rows.append((repo, assessment.verdict.value, landed, reply, why))
|
|
161
|
+
|
|
162
|
+
widths = [max(len(str(r[i])) for r in (*rows, COMPARE_HEADERS)) for i in range(5)]
|
|
163
|
+
widths[4] = min(widths[4], 60)
|
|
164
|
+
|
|
165
|
+
def line(cells) -> str:
|
|
166
|
+
return "| " + " | ".join(
|
|
167
|
+
str(c)[:widths[i]].ljust(widths[i]) for i, c in enumerate(cells)
|
|
168
|
+
) + " |"
|
|
169
|
+
|
|
170
|
+
print(f"# Comparison — for a contributor with {args.days} "
|
|
171
|
+
f"day{'' if args.days == 1 else 's'}\n")
|
|
172
|
+
print(line(COMPARE_HEADERS))
|
|
173
|
+
print("|" + "|".join("-" * (w + 2) for w in widths) + "|") # matches "| cell " padding
|
|
174
|
+
for row in rows:
|
|
175
|
+
print(line(row))
|
|
176
|
+
print("\n`outsiders in` counts pull requests merged from people with no prior "
|
|
177
|
+
"merge, over the number who tried.")
|
|
178
|
+
print("Run `holt analyze <repo>` for the evidence behind any row.")
|
|
179
|
+
# Declared `-> int` and every sibling returns one; falling off the end made
|
|
180
|
+
# `compare` the one command whose exit status was an accident.
|
|
181
|
+
return 0
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def cmd_tui(args: argparse.Namespace) -> int:
|
|
185
|
+
"""Open the terminal interface.
|
|
186
|
+
|
|
187
|
+
Textual is imported here and nowhere else. It ships in the default product
|
|
188
|
+
install because bare `holt` opens this interface, while the lazy boundary
|
|
189
|
+
keeps non-interface commands independent from UI startup.
|
|
190
|
+
|
|
191
|
+
The interface runs the same `pipeline.analyze` the CLI runs. It is a way of
|
|
192
|
+
watching a stage, never the only way of running one.
|
|
193
|
+
"""
|
|
194
|
+
try:
|
|
195
|
+
from holt.tui import env
|
|
196
|
+
from holt.tui.app import run
|
|
197
|
+
from holt.tui.session import RunOptions, missing_credentials
|
|
198
|
+
except ImportError as exc:
|
|
199
|
+
print(
|
|
200
|
+
f"The terminal interface dependency is unavailable ({exc}).\n"
|
|
201
|
+
"Reinstall Holt with: uv tool install --force holt-cli",
|
|
202
|
+
file=sys.stderr,
|
|
203
|
+
)
|
|
204
|
+
return 2
|
|
205
|
+
|
|
206
|
+
# Names only. A value read from `.env` is never printed.
|
|
207
|
+
from_env_file = env.load()
|
|
208
|
+
|
|
209
|
+
# The interface honours the user's chosen model, the same deliberate opt-in
|
|
210
|
+
# `main` makes for the command line. The library still never reads it, so
|
|
211
|
+
# the eval harness and the committed recordings stay on the pinned ids.
|
|
212
|
+
model.enable_user_models_config()
|
|
213
|
+
if from_env_file:
|
|
214
|
+
print(f"Read {', '.join(from_env_file)} from .env", file=sys.stderr)
|
|
215
|
+
|
|
216
|
+
# No repository: open on the list of what has already been assessed. This is
|
|
217
|
+
# the ordinary way in, which is why it is what bare `holt` does.
|
|
218
|
+
repo = getattr(args, "repo", None)
|
|
219
|
+
if not repo:
|
|
220
|
+
run(None)
|
|
221
|
+
return 0
|
|
222
|
+
|
|
223
|
+
options = RunOptions(
|
|
224
|
+
repo=normalise(repo),
|
|
225
|
+
replay=args.replay,
|
|
226
|
+
live=args.live,
|
|
227
|
+
entry_points=args.entry_points,
|
|
228
|
+
contributor_days=args.days,
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
# Checked before the screen is taken over, so a missing key reads as a
|
|
232
|
+
# sentence in the terminal rather than a traceback behind a full-screen app.
|
|
233
|
+
missing = missing_credentials(options)
|
|
234
|
+
if missing:
|
|
235
|
+
print("This run needs:", file=sys.stderr)
|
|
236
|
+
for item in missing:
|
|
237
|
+
print(f" {item}", file=sys.stderr)
|
|
238
|
+
return 2
|
|
239
|
+
|
|
240
|
+
if not options.replay:
|
|
241
|
+
print(
|
|
242
|
+
f"Recording this run to {options.recording(options.repo, 'verdict').parent}"
|
|
243
|
+
" — the committed fixtures are not written to.",
|
|
244
|
+
file=sys.stderr,
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
run(options)
|
|
248
|
+
return 0
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def cmd_next(args: argparse.Namespace) -> int:
|
|
252
|
+
"""Where this contributor might look next. One deterministic rule, no model."""
|
|
253
|
+
from holt.agent import progression
|
|
254
|
+
from holt.agent.signals import build_threads
|
|
255
|
+
from holt.issues import open_at_cutoff
|
|
256
|
+
|
|
257
|
+
repo = normalise(args.repo)
|
|
258
|
+
as_of = as_of_from(args)
|
|
259
|
+
records = make_provider(args.live, as_of).fetch(repo)
|
|
260
|
+
contributor = progression.history_for(args.as_login, build_threads(records))
|
|
261
|
+
if not contributor.merged_count:
|
|
262
|
+
print(f"`{args.as_login}` has no merged pull request in {repo} in this "
|
|
263
|
+
"evidence, so path overlap has nothing to work from. "
|
|
264
|
+
"`holt analyze` answers the question that comes before this one.",
|
|
265
|
+
file=sys.stderr)
|
|
266
|
+
return 1
|
|
267
|
+
try:
|
|
268
|
+
issues = make_issue_provider(args.live, as_of).fetch(repo)
|
|
269
|
+
except FileNotFoundError:
|
|
270
|
+
print(f"No issue evidence for {repo}; nothing to rank.", file=sys.stderr)
|
|
271
|
+
return 1
|
|
272
|
+
candidates = open_at_cutoff(issues)
|
|
273
|
+
if not candidates:
|
|
274
|
+
print(f"No issue in the evidence was open at {as_of.date().isoformat()}; "
|
|
275
|
+
"nothing to rank.", file=sys.stderr)
|
|
276
|
+
return 1
|
|
277
|
+
ranked = progression.path_overlap_rank(contributor.files, candidates)
|
|
278
|
+
print(progression.render_next(repo, contributor, ranked, candidates, top=args.top))
|
|
279
|
+
return 0
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def cmd_profile(args: argparse.Namespace) -> int:
|
|
283
|
+
from holt import profile as profile_mod
|
|
284
|
+
|
|
285
|
+
stored = profile_mod.load()
|
|
286
|
+
if any(getattr(args, f, None) for f in ("lang", "topic", "contribution", "days_flag")):
|
|
287
|
+
args.days = args.days_flag
|
|
288
|
+
updated = profile_mod.from_args(args, stored)
|
|
289
|
+
else:
|
|
290
|
+
updated = profile_mod.ask(stored)
|
|
291
|
+
path = profile_mod.save(updated)
|
|
292
|
+
print(f"Saved to {path}")
|
|
293
|
+
return 0
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def cmd_discover(args: argparse.Namespace) -> int:
|
|
297
|
+
from holt import discover, profile as profile_mod
|
|
298
|
+
|
|
299
|
+
args.days = args.days_flag
|
|
300
|
+
if args.live or args.record:
|
|
301
|
+
stated = profile_mod.from_args(args, profile_mod.load())
|
|
302
|
+
if not (stated.languages or stated.topics):
|
|
303
|
+
print("Nothing to search for. Run `holt profile` once, or pass "
|
|
304
|
+
"--lang/--topic.", file=sys.stderr)
|
|
305
|
+
return 2
|
|
306
|
+
out = discover.run_live(
|
|
307
|
+
stated, limit=args.limit, max_analyze=args.max_analyze,
|
|
308
|
+
record=args.record, progress=lambda s: print(s, file=sys.stderr),
|
|
309
|
+
)
|
|
310
|
+
else:
|
|
311
|
+
# The free path: replay a recorded session. The default session ships
|
|
312
|
+
# in the repository so the demo needs no token and no key.
|
|
313
|
+
try:
|
|
314
|
+
out = discover.run_replay(args.session, days=args.days_flag,
|
|
315
|
+
max_analyze=args.max_analyze)
|
|
316
|
+
except FileNotFoundError:
|
|
317
|
+
print(f"No recorded discovery session named {args.session!r}. "
|
|
318
|
+
"Run with --live for a fresh search (needs GITHUB_TOKEN and "
|
|
319
|
+
"OPENAI_API_KEY).", file=sys.stderr)
|
|
320
|
+
return 2
|
|
321
|
+
print(out)
|
|
322
|
+
return 0
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def cmd_models(args: argparse.Namespace) -> int:
|
|
326
|
+
"""Show or change which model answers, per provider. The default stays the
|
|
327
|
+
pinned OpenAI ids the benchmark was measured on."""
|
|
328
|
+
if args.reset:
|
|
329
|
+
path = model.models_config_path()
|
|
330
|
+
if path.exists():
|
|
331
|
+
path.unlink()
|
|
332
|
+
model.enable_user_models_config(model.ModelsConfig())
|
|
333
|
+
print("Model configuration reset to the defaults.")
|
|
334
|
+
return 0
|
|
335
|
+
|
|
336
|
+
if args.provider or args.model_id or args.base_url or args.api_key_env or args.stage:
|
|
337
|
+
current = model.load_models_config()
|
|
338
|
+
if args.provider:
|
|
339
|
+
if args.provider not in model.PROVIDER_PRESETS:
|
|
340
|
+
print(f"Unknown provider {args.provider!r}. One of: "
|
|
341
|
+
f"{', '.join(sorted(model.PROVIDER_PRESETS))}", file=sys.stderr)
|
|
342
|
+
return 2
|
|
343
|
+
current.provider = args.provider
|
|
344
|
+
if args.model_id:
|
|
345
|
+
current.model = args.model_id
|
|
346
|
+
if args.base_url:
|
|
347
|
+
current.base_url = args.base_url
|
|
348
|
+
if args.api_key_env:
|
|
349
|
+
current.api_key_env = args.api_key_env
|
|
350
|
+
for spec in args.stage or []:
|
|
351
|
+
stage, _, model_id = spec.partition("=")
|
|
352
|
+
if not model_id or stage not in model.STAGE_MODELS:
|
|
353
|
+
print(f"--stage wants <stage>=<model> with stage one of: "
|
|
354
|
+
f"{', '.join(sorted(model.STAGE_MODELS))}", file=sys.stderr)
|
|
355
|
+
return 2
|
|
356
|
+
current.stages[stage] = model_id
|
|
357
|
+
path = model.save_models_config(current)
|
|
358
|
+
model.enable_user_models_config(current)
|
|
359
|
+
print(f"Saved to {path}\n")
|
|
360
|
+
|
|
361
|
+
config = model.active_config()
|
|
362
|
+
print(f"provider {config.provider}")
|
|
363
|
+
if config.resolved_base_url():
|
|
364
|
+
print(f"base_url {config.resolved_base_url()}")
|
|
365
|
+
print(f"api key env {config.resolved_key_env()}")
|
|
366
|
+
print()
|
|
367
|
+
print(f"{'stage':<18}{'model':<28}{'pricing':<24}")
|
|
368
|
+
for stage in model.STAGE_MODELS:
|
|
369
|
+
resolved = model.model_for(stage)
|
|
370
|
+
# The rate itself, not the word "known". Someone reading this is about
|
|
371
|
+
# to spend money and the number is the thing they came for.
|
|
372
|
+
rates, exact = model.resolve_price(resolved)
|
|
373
|
+
if rates is None:
|
|
374
|
+
priced = "unknown ($0 recorded)"
|
|
375
|
+
else:
|
|
376
|
+
priced = f"${rates[0]:g} / ${rates[1]:g} per M"
|
|
377
|
+
if not exact:
|
|
378
|
+
# Priced through a floating alias, which can be repointed.
|
|
379
|
+
priced += " (alias)"
|
|
380
|
+
print(f"{stage:<18}{resolved:<28}{priced:<24}")
|
|
381
|
+
if not config.is_default():
|
|
382
|
+
print(
|
|
383
|
+
"\nNot the defaults. Committed trajectories and benchmark results "
|
|
384
|
+
"were recorded under the default models; `--replay` of those "
|
|
385
|
+
"recordings will fail loudly under this configuration rather than "
|
|
386
|
+
"serve another model's answers. `holt models --reset` restores the "
|
|
387
|
+
"defaults. Recordings you make now will replay under this "
|
|
388
|
+
"configuration."
|
|
389
|
+
)
|
|
390
|
+
return 0
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def main(argv: list[str] | None = None) -> int:
|
|
394
|
+
# The one place the user's model configuration takes effect. Library and
|
|
395
|
+
# eval code resolve against the pinned defaults, always.
|
|
396
|
+
model.enable_user_models_config()
|
|
397
|
+
parser = argparse.ArgumentParser(prog="holt", description=__doc__)
|
|
398
|
+
# Not required: bare `holt` opens the interface. Every existing invocation
|
|
399
|
+
# keeps working unchanged, and the eval harness calls `holt analyze`
|
|
400
|
+
# explicitly, so the reproduction path is unaffected either way.
|
|
401
|
+
sub = parser.add_subparsers(dest="command")
|
|
402
|
+
|
|
403
|
+
analyze = sub.add_parser("analyze", help="assess one repository")
|
|
404
|
+
analyze.add_argument("repo", help="owner/name or a github.com URL")
|
|
405
|
+
analyze.add_argument(
|
|
406
|
+
"--baseline",
|
|
407
|
+
action="store_true",
|
|
408
|
+
help="run the baseline solution: a single prompt over README and metadata",
|
|
409
|
+
)
|
|
410
|
+
analyze.add_argument(
|
|
411
|
+
"--replay",
|
|
412
|
+
action="store_true",
|
|
413
|
+
help="replay recorded model output; no API key, no spend",
|
|
414
|
+
)
|
|
415
|
+
analyze.add_argument(
|
|
416
|
+
"--days",
|
|
417
|
+
type=int,
|
|
418
|
+
default=7,
|
|
419
|
+
help="how many days you actually have; everything time-shaped scales from it",
|
|
420
|
+
)
|
|
421
|
+
analyze.add_argument(
|
|
422
|
+
"--entry-points",
|
|
423
|
+
action="store_true",
|
|
424
|
+
help="append the prototype issue ranking. Off by default: it does not beat "
|
|
425
|
+
"GitHub's `good first issue` label, and the reason is that it never sees "
|
|
426
|
+
"who is asking. See `eval/PATHFINDER-DESIGN.md`",
|
|
427
|
+
)
|
|
428
|
+
analyze.add_argument(
|
|
429
|
+
"--no-model",
|
|
430
|
+
action="store_true",
|
|
431
|
+
help="decide with the rules alone: no API key, no spend, no model call. "
|
|
432
|
+
"MCC +0.60 against the pipeline's +0.61 in sample and +0.55 against "
|
|
433
|
+
"+0.63 out of sample; the report loses every citable statement",
|
|
434
|
+
)
|
|
435
|
+
analyze.add_argument(
|
|
436
|
+
"--show-verification",
|
|
437
|
+
action="store_true",
|
|
438
|
+
help="print the findings Stage D dropped and why",
|
|
439
|
+
)
|
|
440
|
+
analyze.add_argument(
|
|
441
|
+
"--live",
|
|
442
|
+
action="store_true",
|
|
443
|
+
help="read GitHub directly instead of committed fixtures (needs GITHUB_TOKEN)",
|
|
444
|
+
)
|
|
445
|
+
analyze.add_argument(
|
|
446
|
+
"--as-of",
|
|
447
|
+
help="only use evidence up to this date (YYYY-MM-DD). Defaults to today "
|
|
448
|
+
"for --live and to the benchmark cutoff for fixtures",
|
|
449
|
+
)
|
|
450
|
+
analyze.set_defaults(func=cmd_analyze)
|
|
451
|
+
|
|
452
|
+
compare = sub.add_parser(
|
|
453
|
+
"compare", help="assess several repositories and show them side by side"
|
|
454
|
+
)
|
|
455
|
+
compare.add_argument("repos", nargs="+", help="owner/name or github.com URLs")
|
|
456
|
+
compare.add_argument("--replay", action="store_true",
|
|
457
|
+
help="replay recorded model output; no API key, no spend")
|
|
458
|
+
compare.add_argument("--days", type=int, default=7,
|
|
459
|
+
help="how many days you actually have")
|
|
460
|
+
compare.add_argument("--live", action="store_true",
|
|
461
|
+
help="read GitHub directly instead of committed fixtures")
|
|
462
|
+
compare.add_argument(
|
|
463
|
+
"--as-of",
|
|
464
|
+
help="only use evidence up to this date (YYYY-MM-DD). Defaults to today "
|
|
465
|
+
"for --live and to the benchmark cutoff for fixtures",
|
|
466
|
+
)
|
|
467
|
+
compare.set_defaults(func=cmd_compare)
|
|
468
|
+
|
|
469
|
+
next_p = sub.add_parser(
|
|
470
|
+
"next",
|
|
471
|
+
help="rank a repository's open issues for someone who has merged work "
|
|
472
|
+
"there. Deterministic, no model call; the measured numbers print "
|
|
473
|
+
"with the ranking",
|
|
474
|
+
)
|
|
475
|
+
next_p.add_argument("repo", help="owner/name or a github.com URL")
|
|
476
|
+
next_p.add_argument("--as", dest="as_login", required=True,
|
|
477
|
+
help="the contributor's GitHub login")
|
|
478
|
+
next_p.add_argument("--top", type=int, default=10,
|
|
479
|
+
help="how many issues to show")
|
|
480
|
+
next_p.add_argument("--live", action="store_true",
|
|
481
|
+
help="read GitHub directly instead of committed fixtures")
|
|
482
|
+
next_p.add_argument(
|
|
483
|
+
"--as-of",
|
|
484
|
+
help="only use evidence up to this date (YYYY-MM-DD). Defaults to today "
|
|
485
|
+
"for --live and to the benchmark cutoff for fixtures",
|
|
486
|
+
)
|
|
487
|
+
next_p.set_defaults(func=cmd_next)
|
|
488
|
+
|
|
489
|
+
profile_p = sub.add_parser(
|
|
490
|
+
"profile",
|
|
491
|
+
help="say once what you want to work on; `holt discover` reads it",
|
|
492
|
+
)
|
|
493
|
+
profile_p.add_argument("--lang", help="comma-separated languages")
|
|
494
|
+
profile_p.add_argument("--topic", help="comma-separated GitHub topics")
|
|
495
|
+
profile_p.add_argument("--contribution",
|
|
496
|
+
help="comma-separated: docs, tests, ci, code")
|
|
497
|
+
profile_p.add_argument("--days", dest="days_flag", type=int,
|
|
498
|
+
help="how many days you actually have")
|
|
499
|
+
profile_p.set_defaults(func=cmd_profile)
|
|
500
|
+
|
|
501
|
+
discover_p = sub.add_parser(
|
|
502
|
+
"discover",
|
|
503
|
+
help="search GitHub for candidates, screen them for free, analyse the "
|
|
504
|
+
"survivors",
|
|
505
|
+
)
|
|
506
|
+
discover_p.add_argument("--lang", help="comma-separated languages")
|
|
507
|
+
discover_p.add_argument("--topic", help="comma-separated GitHub topics")
|
|
508
|
+
discover_p.add_argument("--contribution",
|
|
509
|
+
help="comma-separated: docs, tests, ci, code")
|
|
510
|
+
discover_p.add_argument("--days", dest="days_flag", type=int,
|
|
511
|
+
help="how many days you actually have")
|
|
512
|
+
discover_p.add_argument("--limit", type=int, default=25,
|
|
513
|
+
help="how many candidates to source")
|
|
514
|
+
discover_p.add_argument("--max-analyze", type=int, default=8,
|
|
515
|
+
help="full analyses to run at most; survivors past "
|
|
516
|
+
"the cap are listed, not silently dropped")
|
|
517
|
+
discover_p.add_argument("--live", action="store_true",
|
|
518
|
+
help="search GitHub now (needs GITHUB_TOKEN and "
|
|
519
|
+
"OPENAI_API_KEY)")
|
|
520
|
+
discover_p.add_argument("--record",
|
|
521
|
+
help="record this live session under a name so it "
|
|
522
|
+
"replays with no credentials (implies --live)")
|
|
523
|
+
discover_p.add_argument("--session", default="demo",
|
|
524
|
+
help="which recorded session to replay "
|
|
525
|
+
"(default: demo)")
|
|
526
|
+
discover_p.set_defaults(func=cmd_discover)
|
|
527
|
+
|
|
528
|
+
models_p = sub.add_parser(
|
|
529
|
+
"models",
|
|
530
|
+
help="show or change which model answers; defaults to the pinned ids "
|
|
531
|
+
"the benchmark was measured on",
|
|
532
|
+
)
|
|
533
|
+
models_p.add_argument("--provider",
|
|
534
|
+
help="openai, anthropic, ollama, gemini, or "
|
|
535
|
+
"openai-compatible")
|
|
536
|
+
models_p.add_argument("--model", dest="model_id",
|
|
537
|
+
help="model id for every stage (e.g. claude-opus-5, "
|
|
538
|
+
"llama3.3)")
|
|
539
|
+
models_p.add_argument("--base-url",
|
|
540
|
+
help="endpoint for an openai-compatible server")
|
|
541
|
+
models_p.add_argument("--api-key-env",
|
|
542
|
+
help="environment variable holding the API key")
|
|
543
|
+
models_p.add_argument("--stage", action="append",
|
|
544
|
+
help="per-stage override, <stage>=<model>; repeatable")
|
|
545
|
+
models_p.add_argument("--reset", action="store_true",
|
|
546
|
+
help="delete the configuration and restore defaults")
|
|
547
|
+
models_p.set_defaults(func=cmd_models)
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
tui = sub.add_parser(
|
|
553
|
+
"tui",
|
|
554
|
+
help="open the terminal interface; also what bare `holt` does",
|
|
555
|
+
)
|
|
556
|
+
tui.add_argument(
|
|
557
|
+
"repo",
|
|
558
|
+
nargs="?",
|
|
559
|
+
help="owner/name or a github.com URL. Omit to open on what you have "
|
|
560
|
+
"already assessed",
|
|
561
|
+
)
|
|
562
|
+
tui.add_argument(
|
|
563
|
+
"--replay",
|
|
564
|
+
action="store_true",
|
|
565
|
+
help="replay recorded model output; no API key, no spend",
|
|
566
|
+
)
|
|
567
|
+
tui.add_argument(
|
|
568
|
+
"--live",
|
|
569
|
+
action="store_true",
|
|
570
|
+
help="read GitHub directly instead of committed fixtures (needs GITHUB_TOKEN)",
|
|
571
|
+
)
|
|
572
|
+
tui.add_argument(
|
|
573
|
+
"--days",
|
|
574
|
+
type=int,
|
|
575
|
+
default=7,
|
|
576
|
+
help="how many days you actually have; everything time-shaped scales from it",
|
|
577
|
+
)
|
|
578
|
+
tui.add_argument(
|
|
579
|
+
"--entry-points",
|
|
580
|
+
action="store_true",
|
|
581
|
+
help="include the prototype issue ranking. Off by default, matching "
|
|
582
|
+
"`holt analyze`: it does not beat GitHub's `good first issue` label",
|
|
583
|
+
)
|
|
584
|
+
tui.set_defaults(func=cmd_tui)
|
|
585
|
+
|
|
586
|
+
args = parser.parse_args(argv)
|
|
587
|
+
if getattr(args, "func", None) is None:
|
|
588
|
+
# No subcommand. Open the interface on what has already been assessed,
|
|
589
|
+
# which is what someone typing `holt` almost always wants.
|
|
590
|
+
args = parser.parse_args(["tui"])
|
|
591
|
+
|
|
592
|
+
# Replay reproduces a recording, so it resolves against the ids that made
|
|
593
|
+
# the recording -- the pinned defaults -- and not against whatever the
|
|
594
|
+
# reader has since chosen with `holt models`. Without this, selecting a
|
|
595
|
+
# model turned every documented `--replay` command into a replay miss,
|
|
596
|
+
# because the chosen id is part of a call's identity. The user's choice
|
|
597
|
+
# governs calls that actually reach a provider, which is where it means
|
|
598
|
+
# something.
|
|
599
|
+
replaying = getattr(args, "replay", False) or (
|
|
600
|
+
args.func is cmd_discover and not getattr(args, "live", False)
|
|
601
|
+
)
|
|
602
|
+
if replaying:
|
|
603
|
+
model.enable_user_models_config(model.ModelsConfig())
|
|
604
|
+
|
|
605
|
+
# A missing fixture or trajectory is a thing the reader can act on -- capture
|
|
606
|
+
# it, or use `--replay` -- and the message already says so. A traceback buries
|
|
607
|
+
# that under twenty frames of our internals, so print the message and stop.
|
|
608
|
+
try:
|
|
609
|
+
return args.func(args)
|
|
610
|
+
except (FileNotFoundError, RuntimeError) as exc:
|
|
611
|
+
print(f"holt: {exc}", file=sys.stderr)
|
|
612
|
+
return 1
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
if __name__ == "__main__":
|
|
616
|
+
raise SystemExit(main())
|