holt-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- holt/__init__.py +0 -0
- holt/agent/__init__.py +0 -0
- holt/agent/entry.py +86 -0
- holt/agent/findings.py +49 -0
- holt/agent/landing.py +154 -0
- holt/agent/pipeline.py +244 -0
- holt/agent/progression.py +408 -0
- holt/agent/signals.py +220 -0
- holt/agent/stages.py +533 -0
- holt/agent/verdict.py +226 -0
- holt/agent/verify.py +140 -0
- holt/baseline.py +89 -0
- holt/baseline_matched.py +116 -0
- holt/cli.py +616 -0
- holt/discover.py +497 -0
- holt/evidence/__init__.py +3 -0
- holt/evidence/fixtures.py +154 -0
- holt/evidence/github_graphql.py +538 -0
- holt/evidence/provider.py +77 -0
- holt/evidence/redact.py +79 -0
- holt/issues.py +41 -0
- holt/model.py +516 -0
- holt/profile.py +126 -0
- holt/report.py +157 -0
- holt/tui/__init__.py +0 -0
- holt/tui/animation.py +84 -0
- holt/tui/app.py +294 -0
- holt/tui/clipboard.py +89 -0
- holt/tui/commands.py +134 -0
- holt/tui/discovery.py +305 -0
- holt/tui/env.py +49 -0
- holt/tui/events.py +245 -0
- holt/tui/mascot.py +121 -0
- holt/tui/models.py +590 -0
- holt/tui/observe.py +297 -0
- holt/tui/screens/__init__.py +35 -0
- holt/tui/screens/assessment.py +337 -0
- holt/tui/screens/confirm.py +62 -0
- holt/tui/screens/discover.py +444 -0
- holt/tui/screens/home.py +519 -0
- holt/tui/screens/inspector.py +106 -0
- holt/tui/screens/live.py +335 -0
- holt/tui/screens/models.py +393 -0
- holt/tui/screens/next_steps.py +425 -0
- holt/tui/screens/profile.py +129 -0
- holt/tui/session.py +711 -0
- holt/tui/store.py +458 -0
- holt/tui/theme.py +479 -0
- holt/tui/visual.py +33 -0
- holt/tui/widgets/__init__.py +0 -0
- holt/tui/widgets/candidates.py +78 -0
- holt/tui/widgets/claims.py +59 -0
- holt/tui/widgets/disclosure.py +121 -0
- holt/tui/widgets/evidence.py +121 -0
- holt/tui/widgets/masthead.py +122 -0
- holt/tui/widgets/recent.py +167 -0
- holt/tui/widgets/scrolling.py +38 -0
- holt/tui/widgets/stages.py +232 -0
- holt/types.py +48 -0
- holt_cli-0.1.0.dist-info/METADATA +198 -0
- holt_cli-0.1.0.dist-info/RECORD +65 -0
- holt_cli-0.1.0.dist-info/WHEEL +4 -0
- holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
- holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
- holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/issues.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Which issues were open at the cutoff.
|
|
2
|
+
|
|
3
|
+
This lives outside both `holt.agent` and `eval.labels` on purpose. The label
|
|
4
|
+
modules must not import from the agent, so a definition they both need cannot
|
|
5
|
+
sit in either one. Duplicating it would let the ranked set and the scored set
|
|
6
|
+
drift apart without a test noticing, which is the one failure that would make a
|
|
7
|
+
Path Finder number meaningless.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from collections.abc import Iterable
|
|
13
|
+
|
|
14
|
+
from holt.types import EvidenceRecord
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def issue_key(evidence_id: str) -> str:
|
|
18
|
+
"""`issue:owner/name#12:closed` -> `issue:owner/name#12`."""
|
|
19
|
+
return ":".join(evidence_id.split(":")[:2])
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def open_at_cutoff(pre_t: Iterable[EvidenceRecord]) -> dict[str, EvidenceRecord]:
|
|
23
|
+
"""Issues open at T, keyed by issue.
|
|
24
|
+
|
|
25
|
+
A record only reaches here if the provider already asserted its timestamp is
|
|
26
|
+
at or before the cutoff, so "opened before T" needs no separate check. An
|
|
27
|
+
issue closed *before* T never produces a pre-cutoff `:closed` record either,
|
|
28
|
+
so anything with an `:opened` record and no pre-cutoff closure was open.
|
|
29
|
+
"""
|
|
30
|
+
records = list(pre_t)
|
|
31
|
+
opened = {
|
|
32
|
+
issue_key(r.evidence_id): r
|
|
33
|
+
for r in records
|
|
34
|
+
if r.evidence_id.startswith("issue:") and r.evidence_id.endswith(":opened")
|
|
35
|
+
}
|
|
36
|
+
closed_before = {
|
|
37
|
+
issue_key(r.evidence_id)
|
|
38
|
+
for r in records
|
|
39
|
+
if r.evidence_id.startswith("issue:") and r.evidence_id.endswith(":closed")
|
|
40
|
+
}
|
|
41
|
+
return {k: v for k, v in opened.items() if k not in closed_before}
|
holt/model.py
ADDED
|
@@ -0,0 +1,516 @@
|
|
|
1
|
+
"""Every model call goes through here.
|
|
2
|
+
|
|
3
|
+
One seam, four jobs. It records trajectories, which are a required deliverable
|
|
4
|
+
and a qualification-gate item. It makes replay possible, so a judge reproduces
|
|
5
|
+
the headline result with no key and no spend. It pins a model per *stage* rather
|
|
6
|
+
than globally. And it is the only file that touches an LLM at all, which is why
|
|
7
|
+
swapping provider took one rewrite instead of a refactor.
|
|
8
|
+
|
|
9
|
+
Model choice is per stage and evidence-led. Counting and verification run no
|
|
10
|
+
model; classification, opportunity and prose run the small one; thread
|
|
11
|
+
interpretation is the stage that may need the larger one, and whether it does is
|
|
12
|
+
measured rather than assumed.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import hashlib
|
|
18
|
+
import json
|
|
19
|
+
import os
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any, Protocol
|
|
23
|
+
|
|
24
|
+
# Dated ids, not floating aliases: `gpt-5-mini` can be repointed underneath a
|
|
25
|
+
# recorded run, and a reproduction claim that drifts is not a claim.
|
|
26
|
+
SMALL = "gpt-5-mini-2025-08-07"
|
|
27
|
+
LARGE = "gpt-5-2025-08-07"
|
|
28
|
+
|
|
29
|
+
# USD per million tokens, (input, output). A model not listed here is charged
|
|
30
|
+
# at zero and `holt models` says so, rather than inventing a price.
|
|
31
|
+
PRICES = {
|
|
32
|
+
SMALL: (0.25, 2.00),
|
|
33
|
+
LARGE: (1.25, 10.00),
|
|
34
|
+
"claude-opus-5": (5.00, 25.00),
|
|
35
|
+
"claude-sonnet-5": (3.00, 15.00),
|
|
36
|
+
"claude-haiku-4-5": (1.00, 5.00),
|
|
37
|
+
"claude-fable-5": (10.00, 50.00),
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
# Floating aliases, and the dated snapshot each one currently points at.
|
|
41
|
+
#
|
|
42
|
+
# Kept separate from `PRICES` because the two facts have different lifetimes. A
|
|
43
|
+
# snapshot's price is fixed for as long as that snapshot exists; where an alias
|
|
44
|
+
# points is only true until the provider repoints it. Merging them would state
|
|
45
|
+
# the second with the confidence of the first.
|
|
46
|
+
#
|
|
47
|
+
# This exists for *reporting a rate*, never for pinning a run — `STAGE_MODELS`
|
|
48
|
+
# names dated ids precisely so a reproduction cannot drift. Before it, selecting
|
|
49
|
+
# `gpt-5` in the interface showed "unpriced — cost recorded as 0" next to
|
|
50
|
+
# `gpt-5-2025-08-07` showing "priced", which reads as two different models
|
|
51
|
+
# rather than one name for the other.
|
|
52
|
+
MODEL_ALIASES: dict[str, str] = {
|
|
53
|
+
"gpt-5": LARGE,
|
|
54
|
+
"gpt-5-mini": SMALL,
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def resolve_price(model_id: str) -> tuple[tuple[float, float] | None, bool]:
|
|
59
|
+
"""`((input, output) per million, exact)`, or `(None, False)` if unknown.
|
|
60
|
+
|
|
61
|
+
`exact` is False when the rate came from the snapshot an alias points at.
|
|
62
|
+
Callers say "approximately" in that case rather than asserting a price for
|
|
63
|
+
an id that can be repointed underneath them — the rate is right today and
|
|
64
|
+
nobody can promise it is right tomorrow.
|
|
65
|
+
"""
|
|
66
|
+
if model_id in PRICES:
|
|
67
|
+
return PRICES[model_id], True
|
|
68
|
+
target = MODEL_ALIASES.get(model_id)
|
|
69
|
+
if target is not None and target in PRICES:
|
|
70
|
+
return PRICES[target], False
|
|
71
|
+
return None, False
|
|
72
|
+
|
|
73
|
+
# The per-stage assignment under test. Everything starts on the small model; a
|
|
74
|
+
# stage is promoted only if the pilot shows it needs to be.
|
|
75
|
+
STAGE_MODELS: dict[str, str] = {
|
|
76
|
+
"baseline": SMALL,
|
|
77
|
+
"baseline_matched": SMALL,
|
|
78
|
+
"classify": SMALL,
|
|
79
|
+
"opportunity": SMALL,
|
|
80
|
+
"outcomes": SMALL,
|
|
81
|
+
"narrate": SMALL,
|
|
82
|
+
"pathfinder": SMALL,
|
|
83
|
+
"profile": SMALL,
|
|
84
|
+
"describe": SMALL,
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
TRAJECTORY_DIR = Path("fixtures/trajectories")
|
|
88
|
+
|
|
89
|
+
# Long enough for a large reasoning response, short enough that a dead connection
|
|
90
|
+
# surfaces as an error in the same session rather than as an unexplained silence.
|
|
91
|
+
REQUEST_TIMEOUT_S = 300.0
|
|
92
|
+
MAX_RETRIES = 4
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# --- provider configuration -------------------------------------------------
|
|
96
|
+
#
|
|
97
|
+
# The default is the pinned OpenAI models above, and the *library* never reads
|
|
98
|
+
# the user's model configuration on its own: every benchmark number, committed
|
|
99
|
+
# trajectory and replay was produced under the defaults, and an eval script
|
|
100
|
+
# silently inheriting somebody's Ollama config would be the exact reproducibility
|
|
101
|
+
# failure this file exists to prevent. Only the CLI (and any front end that
|
|
102
|
+
# makes the same deliberate call) opts in, via `enable_user_models_config()`.
|
|
103
|
+
|
|
104
|
+
PROVIDER_PRESETS: dict[str, dict[str, str]] = {
|
|
105
|
+
# provider -> filled-in defaults; anything the user sets explicitly wins.
|
|
106
|
+
"openai": {"api_key_env": "OPENAI_API_KEY"},
|
|
107
|
+
"anthropic": {"api_key_env": "ANTHROPIC_API_KEY", "model": "claude-opus-5"},
|
|
108
|
+
"ollama": {"base_url": "http://localhost:11434/v1", "api_key_env": "OLLAMA_API_KEY"},
|
|
109
|
+
"gemini": {
|
|
110
|
+
"base_url": "https://generativelanguage.googleapis.com/v1beta/openai/",
|
|
111
|
+
"api_key_env": "GEMINI_API_KEY",
|
|
112
|
+
},
|
|
113
|
+
"openai-compatible": {"api_key_env": "OPENAI_API_KEY"},
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
# Providers that speak the OpenAI wire protocol; everything except anthropic.
|
|
117
|
+
_OPENAI_WIRE = {"openai", "ollama", "gemini", "openai-compatible"}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@dataclass(slots=True)
|
|
121
|
+
class ModelsConfig:
|
|
122
|
+
provider: str = "openai"
|
|
123
|
+
model: str = "" # applied to every stage when set; stage overrides win
|
|
124
|
+
base_url: str = ""
|
|
125
|
+
api_key_env: str = ""
|
|
126
|
+
stages: dict[str, str] = field(default_factory=dict)
|
|
127
|
+
|
|
128
|
+
def resolved_key_env(self) -> str:
|
|
129
|
+
return self.api_key_env or PROVIDER_PRESETS.get(self.provider, {}).get(
|
|
130
|
+
"api_key_env", "OPENAI_API_KEY"
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
def resolved_base_url(self) -> str:
|
|
134
|
+
return self.base_url or PROVIDER_PRESETS.get(self.provider, {}).get("base_url", "")
|
|
135
|
+
|
|
136
|
+
def is_default(self) -> bool:
|
|
137
|
+
return self == ModelsConfig()
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def models_config_path() -> Path:
|
|
141
|
+
base = os.environ.get("XDG_CONFIG_HOME") or str(Path.home() / ".config")
|
|
142
|
+
return Path(base) / "holt" / "models.toml"
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def load_models_config(path: Path | None = None) -> ModelsConfig:
|
|
146
|
+
import tomllib
|
|
147
|
+
|
|
148
|
+
path = path or models_config_path()
|
|
149
|
+
if not path.exists():
|
|
150
|
+
return ModelsConfig()
|
|
151
|
+
data = tomllib.loads(path.read_text())
|
|
152
|
+
return ModelsConfig(
|
|
153
|
+
provider=data.get("provider", "openai"),
|
|
154
|
+
model=data.get("model", ""),
|
|
155
|
+
base_url=data.get("base_url", ""),
|
|
156
|
+
api_key_env=data.get("api_key_env", ""),
|
|
157
|
+
stages={k: str(v) for k, v in (data.get("stages") or {}).items()},
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def save_models_config(config: ModelsConfig, path: Path | None = None) -> Path:
|
|
162
|
+
path = path or models_config_path()
|
|
163
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
164
|
+
lines = [
|
|
165
|
+
f'provider = "{config.provider}"',
|
|
166
|
+
f'model = "{config.model}"',
|
|
167
|
+
f'base_url = "{config.base_url}"',
|
|
168
|
+
f'api_key_env = "{config.api_key_env}"',
|
|
169
|
+
]
|
|
170
|
+
if config.stages:
|
|
171
|
+
lines.append("")
|
|
172
|
+
lines.append("[stages]")
|
|
173
|
+
lines += [f'{k} = "{v}"' for k, v in sorted(config.stages.items())]
|
|
174
|
+
path.write_text("\n".join(lines) + "\n")
|
|
175
|
+
return path
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
_user_config: ModelsConfig | None = None
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def enable_user_models_config(config: ModelsConfig | None = None) -> ModelsConfig:
|
|
182
|
+
"""Opt this process into the user's model configuration.
|
|
183
|
+
|
|
184
|
+
Called by the CLI entry point (and deliberately by any front end that wants
|
|
185
|
+
the same behavior). Library and eval code never call it, so recorded runs
|
|
186
|
+
and replays always resolve against the pinned defaults.
|
|
187
|
+
"""
|
|
188
|
+
global _user_config
|
|
189
|
+
_user_config = config if config is not None else load_models_config()
|
|
190
|
+
return _user_config
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def active_config() -> ModelsConfig:
|
|
194
|
+
return _user_config if _user_config is not None else ModelsConfig()
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def model_for(label: str) -> str:
|
|
198
|
+
config = active_config()
|
|
199
|
+
if label in config.stages:
|
|
200
|
+
return config.stages[label]
|
|
201
|
+
if config.model:
|
|
202
|
+
return config.model
|
|
203
|
+
preset_model = PROVIDER_PRESETS.get(config.provider, {}).get("model")
|
|
204
|
+
if preset_model and not config.is_default():
|
|
205
|
+
return preset_model
|
|
206
|
+
return STAGE_MODELS.get(label, SMALL)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def call_key(label: str, system: str, prompt: str) -> str:
|
|
210
|
+
"""Stable identity for a call, so a replay matches its recording.
|
|
211
|
+
|
|
212
|
+
Covers the prompt text and the model. An edited prompt or a swapped model
|
|
213
|
+
fails loudly instead of quietly serving an answer to a question nobody asked.
|
|
214
|
+
"""
|
|
215
|
+
blob = json.dumps(
|
|
216
|
+
[label, model_for(label), system, prompt], sort_keys=True, separators=(",", ":")
|
|
217
|
+
)
|
|
218
|
+
return hashlib.sha256(blob.encode()).hexdigest()[:24]
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
@dataclass(slots=True)
|
|
222
|
+
class Usage:
|
|
223
|
+
input_tokens: int = 0
|
|
224
|
+
output_tokens: int = 0
|
|
225
|
+
cost_usd: float = 0.0
|
|
226
|
+
# Which models actually answered, in first-use order. Recorded here rather
|
|
227
|
+
# than read back off the configuration because this is the one place that
|
|
228
|
+
# sees every call: on replay it is the ids from the recording, not whatever
|
|
229
|
+
# the reader happens to have configured today.
|
|
230
|
+
models: list[str] = field(default_factory=list)
|
|
231
|
+
|
|
232
|
+
def add(self, model: str, inp: int, out: int) -> None:
|
|
233
|
+
# Through the alias table: a run on `gpt-5` spends real money, and
|
|
234
|
+
# recording zero for it because the id carries no date would understate
|
|
235
|
+
# the bill rather than decline to guess at it.
|
|
236
|
+
rates, _exact = resolve_price(model)
|
|
237
|
+
rate_in, rate_out = rates or (0.0, 0.0)
|
|
238
|
+
self.input_tokens += inp
|
|
239
|
+
self.output_tokens += out
|
|
240
|
+
self.cost_usd += inp / 1e6 * rate_in + out / 1e6 * rate_out
|
|
241
|
+
if model not in self.models:
|
|
242
|
+
self.models.append(model)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
class ModelClient(Protocol):
|
|
246
|
+
replayed: bool
|
|
247
|
+
usage: Usage
|
|
248
|
+
|
|
249
|
+
def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict: ...
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
@dataclass
|
|
253
|
+
class OpenAIModel:
|
|
254
|
+
"""Live calls, recorded as they go."""
|
|
255
|
+
|
|
256
|
+
trajectory_path: Path
|
|
257
|
+
replayed: bool = False
|
|
258
|
+
usage: Usage = field(default_factory=Usage)
|
|
259
|
+
_client: Any = None
|
|
260
|
+
|
|
261
|
+
def __post_init__(self) -> None:
|
|
262
|
+
from openai import OpenAI
|
|
263
|
+
|
|
264
|
+
config = active_config()
|
|
265
|
+
key_env = config.resolved_key_env()
|
|
266
|
+
base_url = config.resolved_base_url()
|
|
267
|
+
api_key = os.environ.get(key_env)
|
|
268
|
+
if not api_key:
|
|
269
|
+
if base_url:
|
|
270
|
+
# Local OpenAI-compatible servers (Ollama, vLLM, LM Studio)
|
|
271
|
+
# accept any key; a missing variable must not block them.
|
|
272
|
+
api_key = "unused"
|
|
273
|
+
else:
|
|
274
|
+
raise RuntimeError(
|
|
275
|
+
f"{key_env} is not set. Use --replay to reproduce recorded "
|
|
276
|
+
"results with no key and no spend."
|
|
277
|
+
)
|
|
278
|
+
# A request with no timeout can hang for hours on a half-open socket, and
|
|
279
|
+
# a recording run that stalls silently is worse than one that fails: the
|
|
280
|
+
# log simply stops and nothing says why. Bounded and retried instead.
|
|
281
|
+
self._client = OpenAI(
|
|
282
|
+
timeout=REQUEST_TIMEOUT_S,
|
|
283
|
+
max_retries=MAX_RETRIES,
|
|
284
|
+
api_key=api_key,
|
|
285
|
+
base_url=base_url or None,
|
|
286
|
+
)
|
|
287
|
+
self.trajectory_path.parent.mkdir(parents=True, exist_ok=True)
|
|
288
|
+
|
|
289
|
+
def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict:
|
|
290
|
+
model = model_for(label)
|
|
291
|
+
response = self._client.chat.completions.create(
|
|
292
|
+
model=model,
|
|
293
|
+
messages=[
|
|
294
|
+
{"role": "system", "content": system},
|
|
295
|
+
{"role": "user", "content": prompt},
|
|
296
|
+
],
|
|
297
|
+
response_format={
|
|
298
|
+
"type": "json_schema",
|
|
299
|
+
"json_schema": {"name": label, "schema": schema, "strict": True},
|
|
300
|
+
},
|
|
301
|
+
)
|
|
302
|
+
parsed = json.loads(response.choices[0].message.content)
|
|
303
|
+
u = response.usage
|
|
304
|
+
self.usage.add(model, u.prompt_tokens, u.completion_tokens)
|
|
305
|
+
|
|
306
|
+
with self.trajectory_path.open("a") as handle:
|
|
307
|
+
handle.write(
|
|
308
|
+
json.dumps(
|
|
309
|
+
{
|
|
310
|
+
"key": call_key(label, system, prompt),
|
|
311
|
+
"label": label,
|
|
312
|
+
"model": model,
|
|
313
|
+
"system": system,
|
|
314
|
+
"prompt": prompt,
|
|
315
|
+
"response": parsed,
|
|
316
|
+
"usage": {
|
|
317
|
+
"input_tokens": u.prompt_tokens,
|
|
318
|
+
"output_tokens": u.completion_tokens,
|
|
319
|
+
},
|
|
320
|
+
}
|
|
321
|
+
)
|
|
322
|
+
+ "\n"
|
|
323
|
+
)
|
|
324
|
+
return parsed
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
@dataclass
|
|
328
|
+
class AnthropicModel:
|
|
329
|
+
"""Live calls against Claude, recorded exactly like the OpenAI ones.
|
|
330
|
+
|
|
331
|
+
Structured output goes through `output_config.format` with the same JSON
|
|
332
|
+
schema every stage already declares, so the `complete()` contract -- a dict
|
|
333
|
+
matching the schema -- holds regardless of provider. Safety classifiers can
|
|
334
|
+
decline a request with `stop_reason: "refusal"`; that surfaces as a loud
|
|
335
|
+
error rather than an empty finding.
|
|
336
|
+
"""
|
|
337
|
+
|
|
338
|
+
trajectory_path: Path
|
|
339
|
+
replayed: bool = False
|
|
340
|
+
usage: Usage = field(default_factory=Usage)
|
|
341
|
+
_client: Any = None
|
|
342
|
+
|
|
343
|
+
MAX_TOKENS = 16000 # thinking counts toward this on current Claude models
|
|
344
|
+
|
|
345
|
+
def __post_init__(self) -> None:
|
|
346
|
+
if self._client is None:
|
|
347
|
+
import anthropic
|
|
348
|
+
|
|
349
|
+
key_env = active_config().resolved_key_env()
|
|
350
|
+
if not os.environ.get(key_env):
|
|
351
|
+
raise RuntimeError(
|
|
352
|
+
f"{key_env} is not set. Use --replay to reproduce recorded "
|
|
353
|
+
"results with no key and no spend."
|
|
354
|
+
)
|
|
355
|
+
self._client = anthropic.Anthropic(
|
|
356
|
+
timeout=REQUEST_TIMEOUT_S, max_retries=MAX_RETRIES
|
|
357
|
+
)
|
|
358
|
+
self.trajectory_path.parent.mkdir(parents=True, exist_ok=True)
|
|
359
|
+
|
|
360
|
+
def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict:
|
|
361
|
+
model = model_for(label)
|
|
362
|
+
response = self._client.messages.create(
|
|
363
|
+
model=model,
|
|
364
|
+
max_tokens=self.MAX_TOKENS,
|
|
365
|
+
system=system,
|
|
366
|
+
messages=[{"role": "user", "content": prompt}],
|
|
367
|
+
output_config={"format": {"type": "json_schema", "schema": schema}},
|
|
368
|
+
)
|
|
369
|
+
if response.stop_reason == "refusal":
|
|
370
|
+
raise RuntimeError(
|
|
371
|
+
f"{model} declined the {label} request (stop_reason=refusal); "
|
|
372
|
+
"nothing was recorded for it"
|
|
373
|
+
)
|
|
374
|
+
text = next(b.text for b in response.content if b.type == "text")
|
|
375
|
+
parsed = json.loads(text)
|
|
376
|
+
u = response.usage
|
|
377
|
+
self.usage.add(model, u.input_tokens, u.output_tokens)
|
|
378
|
+
|
|
379
|
+
with self.trajectory_path.open("a") as handle:
|
|
380
|
+
handle.write(
|
|
381
|
+
json.dumps(
|
|
382
|
+
{
|
|
383
|
+
"key": call_key(label, system, prompt),
|
|
384
|
+
"label": label,
|
|
385
|
+
"model": model,
|
|
386
|
+
"system": system,
|
|
387
|
+
"prompt": prompt,
|
|
388
|
+
"response": parsed,
|
|
389
|
+
"usage": {
|
|
390
|
+
"input_tokens": u.input_tokens,
|
|
391
|
+
"output_tokens": u.output_tokens,
|
|
392
|
+
},
|
|
393
|
+
}
|
|
394
|
+
)
|
|
395
|
+
+ "\n"
|
|
396
|
+
)
|
|
397
|
+
return parsed
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
@dataclass
|
|
401
|
+
class ReplayModel:
|
|
402
|
+
"""Serves recorded responses. No network, no key, no spend.
|
|
403
|
+
|
|
404
|
+
A miss raises rather than falling back to a live call: replay that silently
|
|
405
|
+
bills a judge who asked for the free path is not reproduction.
|
|
406
|
+
"""
|
|
407
|
+
|
|
408
|
+
trajectory_path: Path
|
|
409
|
+
replayed: bool = True
|
|
410
|
+
usage: Usage = field(default_factory=Usage)
|
|
411
|
+
_recorded: dict[str, dict] = field(default_factory=dict)
|
|
412
|
+
|
|
413
|
+
def __post_init__(self) -> None:
|
|
414
|
+
if not self.trajectory_path.exists():
|
|
415
|
+
raise FileNotFoundError(
|
|
416
|
+
f"No trajectory at {self.trajectory_path}. Replay needs a recorded run."
|
|
417
|
+
)
|
|
418
|
+
for line in self.trajectory_path.read_text().splitlines():
|
|
419
|
+
if line.strip():
|
|
420
|
+
entry = json.loads(line)
|
|
421
|
+
self._recorded[entry["key"]] = entry
|
|
422
|
+
|
|
423
|
+
def _miss(self, label: str, system: str, prompt: str, key: str) -> str:
|
|
424
|
+
"""Say which of the three things that identify a call actually differs.
|
|
425
|
+
|
|
426
|
+
A miss used to blame "the prompt or the stage's model", naming both and
|
|
427
|
+
diagnosing neither. The recording carries the text it was made with, so
|
|
428
|
+
the answer is available: if some entry holds this exact prompt, the
|
|
429
|
+
model id is what moved, and the usual reason is a model chosen with
|
|
430
|
+
`holt models` after the recording was committed.
|
|
431
|
+
"""
|
|
432
|
+
same_text = [
|
|
433
|
+
e for e in self._recorded.values()
|
|
434
|
+
if e["label"] == label and e["system"] == system and e["prompt"] == prompt
|
|
435
|
+
]
|
|
436
|
+
if same_text:
|
|
437
|
+
recorded = ", ".join(sorted({e["model"] for e in same_text}))
|
|
438
|
+
return (
|
|
439
|
+
f"No recorded response for {label} (key {key}). The prompt is "
|
|
440
|
+
f"unchanged; the model is not. This run resolves {label} to "
|
|
441
|
+
f"{model_for(label)!r}, and the recording was made with "
|
|
442
|
+
f"{recorded!r}. Committed recordings replay only under the pinned "
|
|
443
|
+
f"defaults — clear the choice with `holt models --reset`, or "
|
|
444
|
+
f"re-record with a key."
|
|
445
|
+
)
|
|
446
|
+
if any(e["label"] == label for e in self._recorded.values()):
|
|
447
|
+
return (
|
|
448
|
+
f"No recorded response for {label} (key {key}). A {label} call is "
|
|
449
|
+
"recorded here but was made with different prompt text, so "
|
|
450
|
+
"replaying it would answer a question that is no longer being "
|
|
451
|
+
"asked. Re-record with a key, or check out the committed state."
|
|
452
|
+
)
|
|
453
|
+
return (
|
|
454
|
+
f"No recorded response for {label} (key {key}), and no {label} call "
|
|
455
|
+
f"is recorded in {self.trajectory_path} at all. Re-record with a key."
|
|
456
|
+
)
|
|
457
|
+
|
|
458
|
+
def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict:
|
|
459
|
+
key = call_key(label, system, prompt)
|
|
460
|
+
entry = self._recorded.get(key)
|
|
461
|
+
if entry is None:
|
|
462
|
+
raise KeyError(self._miss(label, system, prompt, key))
|
|
463
|
+
u = entry.get("usage", {})
|
|
464
|
+
self.usage.add(entry["model"], u.get("input_tokens", 0), u.get("output_tokens", 0))
|
|
465
|
+
return entry["response"]
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
@dataclass
|
|
469
|
+
class PatchModel:
|
|
470
|
+
"""Replay where the recording still matches; record live where it does not.
|
|
471
|
+
|
|
472
|
+
The cheap way to heal recordings after a change that reaches only some
|
|
473
|
+
prompts — a reworded rule trace, say, which enters the narration prompt for
|
|
474
|
+
some repositories and not others. Unchanged calls replay free; changed
|
|
475
|
+
calls re-record at the pinned models and are appended, so the file becomes
|
|
476
|
+
fully replayable again without paying for a complete re-record. Entries
|
|
477
|
+
orphaned by the change stay in the file and are inert: replay looks up by
|
|
478
|
+
key and never serves them.
|
|
479
|
+
"""
|
|
480
|
+
|
|
481
|
+
trajectory_path: Path
|
|
482
|
+
replayed: bool = False
|
|
483
|
+
usage: Usage = field(default_factory=Usage)
|
|
484
|
+
patched: int = 0
|
|
485
|
+
_recorded: dict[str, dict] = field(default_factory=dict)
|
|
486
|
+
_live: Any = None
|
|
487
|
+
|
|
488
|
+
def __post_init__(self) -> None:
|
|
489
|
+
if self.trajectory_path.exists():
|
|
490
|
+
for line in self.trajectory_path.read_text().splitlines():
|
|
491
|
+
if line.strip():
|
|
492
|
+
entry = json.loads(line)
|
|
493
|
+
self._recorded[entry["key"]] = entry
|
|
494
|
+
if self._live is None:
|
|
495
|
+
self._live = OpenAIModel(self.trajectory_path)
|
|
496
|
+
# One usage object: only the live calls cost anything.
|
|
497
|
+
self.usage = self._live.usage
|
|
498
|
+
|
|
499
|
+
def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict:
|
|
500
|
+
entry = self._recorded.get(call_key(label, system, prompt))
|
|
501
|
+
if entry is not None:
|
|
502
|
+
return entry["response"]
|
|
503
|
+
self.patched += 1
|
|
504
|
+
return self._live.complete(label=label, system=system, prompt=prompt, schema=schema)
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def live_client(path: Path) -> ModelClient:
|
|
508
|
+
"""The provider dispatch. One place, so nothing else needs to know."""
|
|
509
|
+
if active_config().provider == "anthropic":
|
|
510
|
+
return AnthropicModel(path)
|
|
511
|
+
return OpenAIModel(path)
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def build(repo_slug: str, replay: bool) -> ModelClient:
|
|
515
|
+
path = TRAJECTORY_DIR / (repo_slug.replace("/", "__") + ".jsonl")
|
|
516
|
+
return ReplayModel(path) if replay else live_client(path)
|
holt/profile.py
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""What the contributor tells us, stored so they say it once.
|
|
2
|
+
|
|
3
|
+
The profile is *stated*, not inferred. An earlier version tried to infer it from
|
|
4
|
+
the person's GitHub history and was cut on data: the median contributor in our
|
|
5
|
+
pool has one merged pull request and five touched files, and 98% of
|
|
6
|
+
cross-repository area overlap was generic-path collisions (`src`, `docs`,
|
|
7
|
+
`tests`). Inferring "you work on Python developer tooling" from one pull request
|
|
8
|
+
would be invention. Asking is more honest and more accurate.
|
|
9
|
+
|
|
10
|
+
Every question here maps to something that changes the output, or it is not
|
|
11
|
+
asked. Languages and topics are search qualifiers — sourcing only, no claim.
|
|
12
|
+
Days feeds `verdict.py`, where the slow-response threshold is `days * 24`.
|
|
13
|
+
Contribution type is matched against where outsider work actually landed.
|
|
14
|
+
Experience level is deliberately absent: nothing downstream could map it to a
|
|
15
|
+
threshold, so asking it would be decoration.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
import tomllib
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
from holt.agent.verdict import DEFAULT_CONTRIBUTOR_DAYS
|
|
26
|
+
|
|
27
|
+
# Contribution types we can actually check, each against the directories where
|
|
28
|
+
# outsider work merged. Anything else would be a question that changes nothing.
|
|
29
|
+
CONTRIBUTION_AREAS: dict[str, tuple[str, ...]] = {
|
|
30
|
+
"docs": ("doc", "docs", "documentation", "wiki", "website"),
|
|
31
|
+
"tests": ("test", "tests", "testing", "spec", "specs"),
|
|
32
|
+
"ci": (".github", ".gitlab", "ci", ".ci", "workflows"),
|
|
33
|
+
"code": (), # the default kind of contribution; every non-docs area counts
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def config_path() -> Path:
|
|
38
|
+
base = os.environ.get("XDG_CONFIG_HOME") or str(Path.home() / ".config")
|
|
39
|
+
return Path(base) / "holt" / "profile.toml"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(slots=True)
|
|
43
|
+
class Profile:
|
|
44
|
+
languages: list[str] = field(default_factory=list)
|
|
45
|
+
topics: list[str] = field(default_factory=list)
|
|
46
|
+
contributions: list[str] = field(default_factory=list)
|
|
47
|
+
days: int = DEFAULT_CONTRIBUTOR_DAYS
|
|
48
|
+
|
|
49
|
+
def describe(self) -> str:
|
|
50
|
+
parts = [" + ".join(self.languages + self.topics) or "any repository"]
|
|
51
|
+
parts.append(f"{self.days} day{'s' if self.days != 1 else ''}")
|
|
52
|
+
if self.contributions:
|
|
53
|
+
parts.append(", ".join(self.contributions))
|
|
54
|
+
return ", ".join(parts)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _csv(value: str | list[str] | None) -> list[str]:
|
|
58
|
+
if not value:
|
|
59
|
+
return []
|
|
60
|
+
if isinstance(value, str):
|
|
61
|
+
value = value.split(",")
|
|
62
|
+
return [v.strip().lower() for v in value if v.strip()]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def load(path: Path | None = None) -> Profile | None:
|
|
66
|
+
path = path or config_path()
|
|
67
|
+
if not path.exists():
|
|
68
|
+
return None
|
|
69
|
+
data = tomllib.loads(path.read_text())
|
|
70
|
+
return Profile(
|
|
71
|
+
languages=_csv(data.get("languages")),
|
|
72
|
+
topics=_csv(data.get("topics")),
|
|
73
|
+
contributions=_csv(data.get("contributions")),
|
|
74
|
+
days=int(data.get("days", DEFAULT_CONTRIBUTOR_DAYS)),
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def save(profile: Profile, path: Path | None = None) -> Path:
|
|
79
|
+
path = path or config_path()
|
|
80
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
81
|
+
|
|
82
|
+
def toml_list(values: list[str]) -> str:
|
|
83
|
+
return "[" + ", ".join(f'"{v}"' for v in values) + "]"
|
|
84
|
+
|
|
85
|
+
path.write_text(
|
|
86
|
+
f"languages = {toml_list(profile.languages)}\n"
|
|
87
|
+
f"topics = {toml_list(profile.topics)}\n"
|
|
88
|
+
f"contributions = {toml_list(profile.contributions)}\n"
|
|
89
|
+
f"days = {profile.days}\n"
|
|
90
|
+
)
|
|
91
|
+
return path
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def from_args(args, stored: Profile | None = None) -> Profile:
|
|
95
|
+
"""Flags override the stored profile field by field; absent flags fall back."""
|
|
96
|
+
stored = stored or Profile()
|
|
97
|
+
return Profile(
|
|
98
|
+
languages=_csv(getattr(args, "lang", None)) or stored.languages,
|
|
99
|
+
topics=_csv(getattr(args, "topic", None)) or stored.topics,
|
|
100
|
+
contributions=_csv(getattr(args, "contribution", None)) or stored.contributions,
|
|
101
|
+
days=getattr(args, "days", None) or stored.days,
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def ask(existing: Profile | None = None) -> Profile:
|
|
106
|
+
"""The interactive form. Four questions, each of which changes the output."""
|
|
107
|
+
current = existing or Profile()
|
|
108
|
+
|
|
109
|
+
def prompt(question: str, default: str) -> str:
|
|
110
|
+
suffix = f" [{default}]" if default else ""
|
|
111
|
+
answer = input(f" {question}{suffix} > ").strip()
|
|
112
|
+
return answer or default
|
|
113
|
+
|
|
114
|
+
languages = _csv(prompt("Languages you want to work in", ", ".join(current.languages)))
|
|
115
|
+
topics = _csv(prompt("What kind of project (topics)", ", ".join(current.topics)))
|
|
116
|
+
contributions = _csv(prompt(
|
|
117
|
+
f"What you want to contribute ({'/'.join(CONTRIBUTION_AREAS)})",
|
|
118
|
+
", ".join(current.contributions),
|
|
119
|
+
))
|
|
120
|
+
days_raw = prompt("How many days you actually have", str(current.days))
|
|
121
|
+
try:
|
|
122
|
+
days = max(1, int(days_raw))
|
|
123
|
+
except ValueError:
|
|
124
|
+
days = current.days
|
|
125
|
+
return Profile(languages=languages, topics=topics,
|
|
126
|
+
contributions=contributions, days=days)
|