holt-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. holt/__init__.py +0 -0
  2. holt/agent/__init__.py +0 -0
  3. holt/agent/entry.py +86 -0
  4. holt/agent/findings.py +49 -0
  5. holt/agent/landing.py +154 -0
  6. holt/agent/pipeline.py +244 -0
  7. holt/agent/progression.py +408 -0
  8. holt/agent/signals.py +220 -0
  9. holt/agent/stages.py +533 -0
  10. holt/agent/verdict.py +226 -0
  11. holt/agent/verify.py +140 -0
  12. holt/baseline.py +89 -0
  13. holt/baseline_matched.py +116 -0
  14. holt/cli.py +616 -0
  15. holt/discover.py +497 -0
  16. holt/evidence/__init__.py +3 -0
  17. holt/evidence/fixtures.py +154 -0
  18. holt/evidence/github_graphql.py +538 -0
  19. holt/evidence/provider.py +77 -0
  20. holt/evidence/redact.py +79 -0
  21. holt/issues.py +41 -0
  22. holt/model.py +516 -0
  23. holt/profile.py +126 -0
  24. holt/report.py +157 -0
  25. holt/tui/__init__.py +0 -0
  26. holt/tui/animation.py +84 -0
  27. holt/tui/app.py +294 -0
  28. holt/tui/clipboard.py +89 -0
  29. holt/tui/commands.py +134 -0
  30. holt/tui/discovery.py +305 -0
  31. holt/tui/env.py +49 -0
  32. holt/tui/events.py +245 -0
  33. holt/tui/mascot.py +121 -0
  34. holt/tui/models.py +590 -0
  35. holt/tui/observe.py +297 -0
  36. holt/tui/screens/__init__.py +35 -0
  37. holt/tui/screens/assessment.py +337 -0
  38. holt/tui/screens/confirm.py +62 -0
  39. holt/tui/screens/discover.py +444 -0
  40. holt/tui/screens/home.py +519 -0
  41. holt/tui/screens/inspector.py +106 -0
  42. holt/tui/screens/live.py +335 -0
  43. holt/tui/screens/models.py +393 -0
  44. holt/tui/screens/next_steps.py +425 -0
  45. holt/tui/screens/profile.py +129 -0
  46. holt/tui/session.py +711 -0
  47. holt/tui/store.py +458 -0
  48. holt/tui/theme.py +479 -0
  49. holt/tui/visual.py +33 -0
  50. holt/tui/widgets/__init__.py +0 -0
  51. holt/tui/widgets/candidates.py +78 -0
  52. holt/tui/widgets/claims.py +59 -0
  53. holt/tui/widgets/disclosure.py +121 -0
  54. holt/tui/widgets/evidence.py +121 -0
  55. holt/tui/widgets/masthead.py +122 -0
  56. holt/tui/widgets/recent.py +167 -0
  57. holt/tui/widgets/scrolling.py +38 -0
  58. holt/tui/widgets/stages.py +232 -0
  59. holt/types.py +48 -0
  60. holt_cli-0.1.0.dist-info/METADATA +198 -0
  61. holt_cli-0.1.0.dist-info/RECORD +65 -0
  62. holt_cli-0.1.0.dist-info/WHEEL +4 -0
  63. holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
  64. holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
  65. holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/issues.py ADDED
@@ -0,0 +1,41 @@
1
+ """Which issues were open at the cutoff.
2
+
3
+ This lives outside both `holt.agent` and `eval.labels` on purpose. The label
4
+ modules must not import from the agent, so a definition they both need cannot
5
+ sit in either one. Duplicating it would let the ranked set and the scored set
6
+ drift apart without a test noticing, which is the one failure that would make a
7
+ Path Finder number meaningless.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections.abc import Iterable
13
+
14
+ from holt.types import EvidenceRecord
15
+
16
+
17
+ def issue_key(evidence_id: str) -> str:
18
+ """`issue:owner/name#12:closed` -> `issue:owner/name#12`."""
19
+ return ":".join(evidence_id.split(":")[:2])
20
+
21
+
22
+ def open_at_cutoff(pre_t: Iterable[EvidenceRecord]) -> dict[str, EvidenceRecord]:
23
+ """Issues open at T, keyed by issue.
24
+
25
+ A record only reaches here if the provider already asserted its timestamp is
26
+ at or before the cutoff, so "opened before T" needs no separate check. An
27
+ issue closed *before* T never produces a pre-cutoff `:closed` record either,
28
+ so anything with an `:opened` record and no pre-cutoff closure was open.
29
+ """
30
+ records = list(pre_t)
31
+ opened = {
32
+ issue_key(r.evidence_id): r
33
+ for r in records
34
+ if r.evidence_id.startswith("issue:") and r.evidence_id.endswith(":opened")
35
+ }
36
+ closed_before = {
37
+ issue_key(r.evidence_id)
38
+ for r in records
39
+ if r.evidence_id.startswith("issue:") and r.evidence_id.endswith(":closed")
40
+ }
41
+ return {k: v for k, v in opened.items() if k not in closed_before}
holt/model.py ADDED
@@ -0,0 +1,516 @@
1
+ """Every model call goes through here.
2
+
3
+ One seam, four jobs. It records trajectories, which are a required deliverable
4
+ and a qualification-gate item. It makes replay possible, so a judge reproduces
5
+ the headline result with no key and no spend. It pins a model per *stage* rather
6
+ than globally. And it is the only file that touches an LLM at all, which is why
7
+ swapping provider took one rewrite instead of a refactor.
8
+
9
+ Model choice is per stage and evidence-led. Counting and verification run no
10
+ model; classification, opportunity and prose run the small one; thread
11
+ interpretation is the stage that may need the larger one, and whether it does is
12
+ measured rather than assumed.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import hashlib
18
+ import json
19
+ import os
20
+ from dataclasses import dataclass, field
21
+ from pathlib import Path
22
+ from typing import Any, Protocol
23
+
24
+ # Dated ids, not floating aliases: `gpt-5-mini` can be repointed underneath a
25
+ # recorded run, and a reproduction claim that drifts is not a claim.
26
+ SMALL = "gpt-5-mini-2025-08-07"
27
+ LARGE = "gpt-5-2025-08-07"
28
+
29
+ # USD per million tokens, (input, output). A model not listed here is charged
30
+ # at zero and `holt models` says so, rather than inventing a price.
31
+ PRICES = {
32
+ SMALL: (0.25, 2.00),
33
+ LARGE: (1.25, 10.00),
34
+ "claude-opus-5": (5.00, 25.00),
35
+ "claude-sonnet-5": (3.00, 15.00),
36
+ "claude-haiku-4-5": (1.00, 5.00),
37
+ "claude-fable-5": (10.00, 50.00),
38
+ }
39
+
40
+ # Floating aliases, and the dated snapshot each one currently points at.
41
+ #
42
+ # Kept separate from `PRICES` because the two facts have different lifetimes. A
43
+ # snapshot's price is fixed for as long as that snapshot exists; where an alias
44
+ # points is only true until the provider repoints it. Merging them would state
45
+ # the second with the confidence of the first.
46
+ #
47
+ # This exists for *reporting a rate*, never for pinning a run — `STAGE_MODELS`
48
+ # names dated ids precisely so a reproduction cannot drift. Before it, selecting
49
+ # `gpt-5` in the interface showed "unpriced — cost recorded as 0" next to
50
+ # `gpt-5-2025-08-07` showing "priced", which reads as two different models
51
+ # rather than one name for the other.
52
+ MODEL_ALIASES: dict[str, str] = {
53
+ "gpt-5": LARGE,
54
+ "gpt-5-mini": SMALL,
55
+ }
56
+
57
+
58
+ def resolve_price(model_id: str) -> tuple[tuple[float, float] | None, bool]:
59
+ """`((input, output) per million, exact)`, or `(None, False)` if unknown.
60
+
61
+ `exact` is False when the rate came from the snapshot an alias points at.
62
+ Callers say "approximately" in that case rather than asserting a price for
63
+ an id that can be repointed underneath them — the rate is right today and
64
+ nobody can promise it is right tomorrow.
65
+ """
66
+ if model_id in PRICES:
67
+ return PRICES[model_id], True
68
+ target = MODEL_ALIASES.get(model_id)
69
+ if target is not None and target in PRICES:
70
+ return PRICES[target], False
71
+ return None, False
72
+
73
+ # The per-stage assignment under test. Everything starts on the small model; a
74
+ # stage is promoted only if the pilot shows it needs to be.
75
+ STAGE_MODELS: dict[str, str] = {
76
+ "baseline": SMALL,
77
+ "baseline_matched": SMALL,
78
+ "classify": SMALL,
79
+ "opportunity": SMALL,
80
+ "outcomes": SMALL,
81
+ "narrate": SMALL,
82
+ "pathfinder": SMALL,
83
+ "profile": SMALL,
84
+ "describe": SMALL,
85
+ }
86
+
87
+ TRAJECTORY_DIR = Path("fixtures/trajectories")
88
+
89
+ # Long enough for a large reasoning response, short enough that a dead connection
90
+ # surfaces as an error in the same session rather than as an unexplained silence.
91
+ REQUEST_TIMEOUT_S = 300.0
92
+ MAX_RETRIES = 4
93
+
94
+
95
+ # --- provider configuration -------------------------------------------------
96
+ #
97
+ # The default is the pinned OpenAI models above, and the *library* never reads
98
+ # the user's model configuration on its own: every benchmark number, committed
99
+ # trajectory and replay was produced under the defaults, and an eval script
100
+ # silently inheriting somebody's Ollama config would be the exact reproducibility
101
+ # failure this file exists to prevent. Only the CLI (and any front end that
102
+ # makes the same deliberate call) opts in, via `enable_user_models_config()`.
103
+
104
+ PROVIDER_PRESETS: dict[str, dict[str, str]] = {
105
+ # provider -> filled-in defaults; anything the user sets explicitly wins.
106
+ "openai": {"api_key_env": "OPENAI_API_KEY"},
107
+ "anthropic": {"api_key_env": "ANTHROPIC_API_KEY", "model": "claude-opus-5"},
108
+ "ollama": {"base_url": "http://localhost:11434/v1", "api_key_env": "OLLAMA_API_KEY"},
109
+ "gemini": {
110
+ "base_url": "https://generativelanguage.googleapis.com/v1beta/openai/",
111
+ "api_key_env": "GEMINI_API_KEY",
112
+ },
113
+ "openai-compatible": {"api_key_env": "OPENAI_API_KEY"},
114
+ }
115
+
116
+ # Providers that speak the OpenAI wire protocol; everything except anthropic.
117
+ _OPENAI_WIRE = {"openai", "ollama", "gemini", "openai-compatible"}
118
+
119
+
120
+ @dataclass(slots=True)
121
+ class ModelsConfig:
122
+ provider: str = "openai"
123
+ model: str = "" # applied to every stage when set; stage overrides win
124
+ base_url: str = ""
125
+ api_key_env: str = ""
126
+ stages: dict[str, str] = field(default_factory=dict)
127
+
128
+ def resolved_key_env(self) -> str:
129
+ return self.api_key_env or PROVIDER_PRESETS.get(self.provider, {}).get(
130
+ "api_key_env", "OPENAI_API_KEY"
131
+ )
132
+
133
+ def resolved_base_url(self) -> str:
134
+ return self.base_url or PROVIDER_PRESETS.get(self.provider, {}).get("base_url", "")
135
+
136
+ def is_default(self) -> bool:
137
+ return self == ModelsConfig()
138
+
139
+
140
+ def models_config_path() -> Path:
141
+ base = os.environ.get("XDG_CONFIG_HOME") or str(Path.home() / ".config")
142
+ return Path(base) / "holt" / "models.toml"
143
+
144
+
145
+ def load_models_config(path: Path | None = None) -> ModelsConfig:
146
+ import tomllib
147
+
148
+ path = path or models_config_path()
149
+ if not path.exists():
150
+ return ModelsConfig()
151
+ data = tomllib.loads(path.read_text())
152
+ return ModelsConfig(
153
+ provider=data.get("provider", "openai"),
154
+ model=data.get("model", ""),
155
+ base_url=data.get("base_url", ""),
156
+ api_key_env=data.get("api_key_env", ""),
157
+ stages={k: str(v) for k, v in (data.get("stages") or {}).items()},
158
+ )
159
+
160
+
161
+ def save_models_config(config: ModelsConfig, path: Path | None = None) -> Path:
162
+ path = path or models_config_path()
163
+ path.parent.mkdir(parents=True, exist_ok=True)
164
+ lines = [
165
+ f'provider = "{config.provider}"',
166
+ f'model = "{config.model}"',
167
+ f'base_url = "{config.base_url}"',
168
+ f'api_key_env = "{config.api_key_env}"',
169
+ ]
170
+ if config.stages:
171
+ lines.append("")
172
+ lines.append("[stages]")
173
+ lines += [f'{k} = "{v}"' for k, v in sorted(config.stages.items())]
174
+ path.write_text("\n".join(lines) + "\n")
175
+ return path
176
+
177
+
178
+ _user_config: ModelsConfig | None = None
179
+
180
+
181
+ def enable_user_models_config(config: ModelsConfig | None = None) -> ModelsConfig:
182
+ """Opt this process into the user's model configuration.
183
+
184
+ Called by the CLI entry point (and deliberately by any front end that wants
185
+ the same behavior). Library and eval code never call it, so recorded runs
186
+ and replays always resolve against the pinned defaults.
187
+ """
188
+ global _user_config
189
+ _user_config = config if config is not None else load_models_config()
190
+ return _user_config
191
+
192
+
193
+ def active_config() -> ModelsConfig:
194
+ return _user_config if _user_config is not None else ModelsConfig()
195
+
196
+
197
+ def model_for(label: str) -> str:
198
+ config = active_config()
199
+ if label in config.stages:
200
+ return config.stages[label]
201
+ if config.model:
202
+ return config.model
203
+ preset_model = PROVIDER_PRESETS.get(config.provider, {}).get("model")
204
+ if preset_model and not config.is_default():
205
+ return preset_model
206
+ return STAGE_MODELS.get(label, SMALL)
207
+
208
+
209
+ def call_key(label: str, system: str, prompt: str) -> str:
210
+ """Stable identity for a call, so a replay matches its recording.
211
+
212
+ Covers the prompt text and the model. An edited prompt or a swapped model
213
+ fails loudly instead of quietly serving an answer to a question nobody asked.
214
+ """
215
+ blob = json.dumps(
216
+ [label, model_for(label), system, prompt], sort_keys=True, separators=(",", ":")
217
+ )
218
+ return hashlib.sha256(blob.encode()).hexdigest()[:24]
219
+
220
+
221
+ @dataclass(slots=True)
222
+ class Usage:
223
+ input_tokens: int = 0
224
+ output_tokens: int = 0
225
+ cost_usd: float = 0.0
226
+ # Which models actually answered, in first-use order. Recorded here rather
227
+ # than read back off the configuration because this is the one place that
228
+ # sees every call: on replay it is the ids from the recording, not whatever
229
+ # the reader happens to have configured today.
230
+ models: list[str] = field(default_factory=list)
231
+
232
+ def add(self, model: str, inp: int, out: int) -> None:
233
+ # Through the alias table: a run on `gpt-5` spends real money, and
234
+ # recording zero for it because the id carries no date would understate
235
+ # the bill rather than decline to guess at it.
236
+ rates, _exact = resolve_price(model)
237
+ rate_in, rate_out = rates or (0.0, 0.0)
238
+ self.input_tokens += inp
239
+ self.output_tokens += out
240
+ self.cost_usd += inp / 1e6 * rate_in + out / 1e6 * rate_out
241
+ if model not in self.models:
242
+ self.models.append(model)
243
+
244
+
245
+ class ModelClient(Protocol):
246
+ replayed: bool
247
+ usage: Usage
248
+
249
+ def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict: ...
250
+
251
+
252
+ @dataclass
253
+ class OpenAIModel:
254
+ """Live calls, recorded as they go."""
255
+
256
+ trajectory_path: Path
257
+ replayed: bool = False
258
+ usage: Usage = field(default_factory=Usage)
259
+ _client: Any = None
260
+
261
+ def __post_init__(self) -> None:
262
+ from openai import OpenAI
263
+
264
+ config = active_config()
265
+ key_env = config.resolved_key_env()
266
+ base_url = config.resolved_base_url()
267
+ api_key = os.environ.get(key_env)
268
+ if not api_key:
269
+ if base_url:
270
+ # Local OpenAI-compatible servers (Ollama, vLLM, LM Studio)
271
+ # accept any key; a missing variable must not block them.
272
+ api_key = "unused"
273
+ else:
274
+ raise RuntimeError(
275
+ f"{key_env} is not set. Use --replay to reproduce recorded "
276
+ "results with no key and no spend."
277
+ )
278
+ # A request with no timeout can hang for hours on a half-open socket, and
279
+ # a recording run that stalls silently is worse than one that fails: the
280
+ # log simply stops and nothing says why. Bounded and retried instead.
281
+ self._client = OpenAI(
282
+ timeout=REQUEST_TIMEOUT_S,
283
+ max_retries=MAX_RETRIES,
284
+ api_key=api_key,
285
+ base_url=base_url or None,
286
+ )
287
+ self.trajectory_path.parent.mkdir(parents=True, exist_ok=True)
288
+
289
+ def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict:
290
+ model = model_for(label)
291
+ response = self._client.chat.completions.create(
292
+ model=model,
293
+ messages=[
294
+ {"role": "system", "content": system},
295
+ {"role": "user", "content": prompt},
296
+ ],
297
+ response_format={
298
+ "type": "json_schema",
299
+ "json_schema": {"name": label, "schema": schema, "strict": True},
300
+ },
301
+ )
302
+ parsed = json.loads(response.choices[0].message.content)
303
+ u = response.usage
304
+ self.usage.add(model, u.prompt_tokens, u.completion_tokens)
305
+
306
+ with self.trajectory_path.open("a") as handle:
307
+ handle.write(
308
+ json.dumps(
309
+ {
310
+ "key": call_key(label, system, prompt),
311
+ "label": label,
312
+ "model": model,
313
+ "system": system,
314
+ "prompt": prompt,
315
+ "response": parsed,
316
+ "usage": {
317
+ "input_tokens": u.prompt_tokens,
318
+ "output_tokens": u.completion_tokens,
319
+ },
320
+ }
321
+ )
322
+ + "\n"
323
+ )
324
+ return parsed
325
+
326
+
327
+ @dataclass
328
+ class AnthropicModel:
329
+ """Live calls against Claude, recorded exactly like the OpenAI ones.
330
+
331
+ Structured output goes through `output_config.format` with the same JSON
332
+ schema every stage already declares, so the `complete()` contract -- a dict
333
+ matching the schema -- holds regardless of provider. Safety classifiers can
334
+ decline a request with `stop_reason: "refusal"`; that surfaces as a loud
335
+ error rather than an empty finding.
336
+ """
337
+
338
+ trajectory_path: Path
339
+ replayed: bool = False
340
+ usage: Usage = field(default_factory=Usage)
341
+ _client: Any = None
342
+
343
+ MAX_TOKENS = 16000 # thinking counts toward this on current Claude models
344
+
345
+ def __post_init__(self) -> None:
346
+ if self._client is None:
347
+ import anthropic
348
+
349
+ key_env = active_config().resolved_key_env()
350
+ if not os.environ.get(key_env):
351
+ raise RuntimeError(
352
+ f"{key_env} is not set. Use --replay to reproduce recorded "
353
+ "results with no key and no spend."
354
+ )
355
+ self._client = anthropic.Anthropic(
356
+ timeout=REQUEST_TIMEOUT_S, max_retries=MAX_RETRIES
357
+ )
358
+ self.trajectory_path.parent.mkdir(parents=True, exist_ok=True)
359
+
360
+ def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict:
361
+ model = model_for(label)
362
+ response = self._client.messages.create(
363
+ model=model,
364
+ max_tokens=self.MAX_TOKENS,
365
+ system=system,
366
+ messages=[{"role": "user", "content": prompt}],
367
+ output_config={"format": {"type": "json_schema", "schema": schema}},
368
+ )
369
+ if response.stop_reason == "refusal":
370
+ raise RuntimeError(
371
+ f"{model} declined the {label} request (stop_reason=refusal); "
372
+ "nothing was recorded for it"
373
+ )
374
+ text = next(b.text for b in response.content if b.type == "text")
375
+ parsed = json.loads(text)
376
+ u = response.usage
377
+ self.usage.add(model, u.input_tokens, u.output_tokens)
378
+
379
+ with self.trajectory_path.open("a") as handle:
380
+ handle.write(
381
+ json.dumps(
382
+ {
383
+ "key": call_key(label, system, prompt),
384
+ "label": label,
385
+ "model": model,
386
+ "system": system,
387
+ "prompt": prompt,
388
+ "response": parsed,
389
+ "usage": {
390
+ "input_tokens": u.input_tokens,
391
+ "output_tokens": u.output_tokens,
392
+ },
393
+ }
394
+ )
395
+ + "\n"
396
+ )
397
+ return parsed
398
+
399
+
400
+ @dataclass
401
+ class ReplayModel:
402
+ """Serves recorded responses. No network, no key, no spend.
403
+
404
+ A miss raises rather than falling back to a live call: replay that silently
405
+ bills a judge who asked for the free path is not reproduction.
406
+ """
407
+
408
+ trajectory_path: Path
409
+ replayed: bool = True
410
+ usage: Usage = field(default_factory=Usage)
411
+ _recorded: dict[str, dict] = field(default_factory=dict)
412
+
413
+ def __post_init__(self) -> None:
414
+ if not self.trajectory_path.exists():
415
+ raise FileNotFoundError(
416
+ f"No trajectory at {self.trajectory_path}. Replay needs a recorded run."
417
+ )
418
+ for line in self.trajectory_path.read_text().splitlines():
419
+ if line.strip():
420
+ entry = json.loads(line)
421
+ self._recorded[entry["key"]] = entry
422
+
423
+ def _miss(self, label: str, system: str, prompt: str, key: str) -> str:
424
+ """Say which of the three things that identify a call actually differs.
425
+
426
+ A miss used to blame "the prompt or the stage's model", naming both and
427
+ diagnosing neither. The recording carries the text it was made with, so
428
+ the answer is available: if some entry holds this exact prompt, the
429
+ model id is what moved, and the usual reason is a model chosen with
430
+ `holt models` after the recording was committed.
431
+ """
432
+ same_text = [
433
+ e for e in self._recorded.values()
434
+ if e["label"] == label and e["system"] == system and e["prompt"] == prompt
435
+ ]
436
+ if same_text:
437
+ recorded = ", ".join(sorted({e["model"] for e in same_text}))
438
+ return (
439
+ f"No recorded response for {label} (key {key}). The prompt is "
440
+ f"unchanged; the model is not. This run resolves {label} to "
441
+ f"{model_for(label)!r}, and the recording was made with "
442
+ f"{recorded!r}. Committed recordings replay only under the pinned "
443
+ f"defaults — clear the choice with `holt models --reset`, or "
444
+ f"re-record with a key."
445
+ )
446
+ if any(e["label"] == label for e in self._recorded.values()):
447
+ return (
448
+ f"No recorded response for {label} (key {key}). A {label} call is "
449
+ "recorded here but was made with different prompt text, so "
450
+ "replaying it would answer a question that is no longer being "
451
+ "asked. Re-record with a key, or check out the committed state."
452
+ )
453
+ return (
454
+ f"No recorded response for {label} (key {key}), and no {label} call "
455
+ f"is recorded in {self.trajectory_path} at all. Re-record with a key."
456
+ )
457
+
458
+ def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict:
459
+ key = call_key(label, system, prompt)
460
+ entry = self._recorded.get(key)
461
+ if entry is None:
462
+ raise KeyError(self._miss(label, system, prompt, key))
463
+ u = entry.get("usage", {})
464
+ self.usage.add(entry["model"], u.get("input_tokens", 0), u.get("output_tokens", 0))
465
+ return entry["response"]
466
+
467
+
468
+ @dataclass
469
+ class PatchModel:
470
+ """Replay where the recording still matches; record live where it does not.
471
+
472
+ The cheap way to heal recordings after a change that reaches only some
473
+ prompts — a reworded rule trace, say, which enters the narration prompt for
474
+ some repositories and not others. Unchanged calls replay free; changed
475
+ calls re-record at the pinned models and are appended, so the file becomes
476
+ fully replayable again without paying for a complete re-record. Entries
477
+ orphaned by the change stay in the file and are inert: replay looks up by
478
+ key and never serves them.
479
+ """
480
+
481
+ trajectory_path: Path
482
+ replayed: bool = False
483
+ usage: Usage = field(default_factory=Usage)
484
+ patched: int = 0
485
+ _recorded: dict[str, dict] = field(default_factory=dict)
486
+ _live: Any = None
487
+
488
+ def __post_init__(self) -> None:
489
+ if self.trajectory_path.exists():
490
+ for line in self.trajectory_path.read_text().splitlines():
491
+ if line.strip():
492
+ entry = json.loads(line)
493
+ self._recorded[entry["key"]] = entry
494
+ if self._live is None:
495
+ self._live = OpenAIModel(self.trajectory_path)
496
+ # One usage object: only the live calls cost anything.
497
+ self.usage = self._live.usage
498
+
499
+ def complete(self, *, label: str, system: str, prompt: str, schema: dict) -> dict:
500
+ entry = self._recorded.get(call_key(label, system, prompt))
501
+ if entry is not None:
502
+ return entry["response"]
503
+ self.patched += 1
504
+ return self._live.complete(label=label, system=system, prompt=prompt, schema=schema)
505
+
506
+
507
+ def live_client(path: Path) -> ModelClient:
508
+ """The provider dispatch. One place, so nothing else needs to know."""
509
+ if active_config().provider == "anthropic":
510
+ return AnthropicModel(path)
511
+ return OpenAIModel(path)
512
+
513
+
514
+ def build(repo_slug: str, replay: bool) -> ModelClient:
515
+ path = TRAJECTORY_DIR / (repo_slug.replace("/", "__") + ".jsonl")
516
+ return ReplayModel(path) if replay else live_client(path)
holt/profile.py ADDED
@@ -0,0 +1,126 @@
1
+ """What the contributor tells us, stored so they say it once.
2
+
3
+ The profile is *stated*, not inferred. An earlier version tried to infer it from
4
+ the person's GitHub history and was cut on data: the median contributor in our
5
+ pool has one merged pull request and five touched files, and 98% of
6
+ cross-repository area overlap was generic-path collisions (`src`, `docs`,
7
+ `tests`). Inferring "you work on Python developer tooling" from one pull request
8
+ would be invention. Asking is more honest and more accurate.
9
+
10
+ Every question here maps to something that changes the output, or it is not
11
+ asked. Languages and topics are search qualifiers — sourcing only, no claim.
12
+ Days feeds `verdict.py`, where the slow-response threshold is `days * 24`.
13
+ Contribution type is matched against where outsider work actually landed.
14
+ Experience level is deliberately absent: nothing downstream could map it to a
15
+ threshold, so asking it would be decoration.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import os
21
+ import tomllib
22
+ from dataclasses import dataclass, field
23
+ from pathlib import Path
24
+
25
+ from holt.agent.verdict import DEFAULT_CONTRIBUTOR_DAYS
26
+
27
+ # Contribution types we can actually check, each against the directories where
28
+ # outsider work merged. Anything else would be a question that changes nothing.
29
+ CONTRIBUTION_AREAS: dict[str, tuple[str, ...]] = {
30
+ "docs": ("doc", "docs", "documentation", "wiki", "website"),
31
+ "tests": ("test", "tests", "testing", "spec", "specs"),
32
+ "ci": (".github", ".gitlab", "ci", ".ci", "workflows"),
33
+ "code": (), # the default kind of contribution; every non-docs area counts
34
+ }
35
+
36
+
37
+ def config_path() -> Path:
38
+ base = os.environ.get("XDG_CONFIG_HOME") or str(Path.home() / ".config")
39
+ return Path(base) / "holt" / "profile.toml"
40
+
41
+
42
+ @dataclass(slots=True)
43
+ class Profile:
44
+ languages: list[str] = field(default_factory=list)
45
+ topics: list[str] = field(default_factory=list)
46
+ contributions: list[str] = field(default_factory=list)
47
+ days: int = DEFAULT_CONTRIBUTOR_DAYS
48
+
49
+ def describe(self) -> str:
50
+ parts = [" + ".join(self.languages + self.topics) or "any repository"]
51
+ parts.append(f"{self.days} day{'s' if self.days != 1 else ''}")
52
+ if self.contributions:
53
+ parts.append(", ".join(self.contributions))
54
+ return ", ".join(parts)
55
+
56
+
57
+ def _csv(value: str | list[str] | None) -> list[str]:
58
+ if not value:
59
+ return []
60
+ if isinstance(value, str):
61
+ value = value.split(",")
62
+ return [v.strip().lower() for v in value if v.strip()]
63
+
64
+
65
+ def load(path: Path | None = None) -> Profile | None:
66
+ path = path or config_path()
67
+ if not path.exists():
68
+ return None
69
+ data = tomllib.loads(path.read_text())
70
+ return Profile(
71
+ languages=_csv(data.get("languages")),
72
+ topics=_csv(data.get("topics")),
73
+ contributions=_csv(data.get("contributions")),
74
+ days=int(data.get("days", DEFAULT_CONTRIBUTOR_DAYS)),
75
+ )
76
+
77
+
78
+ def save(profile: Profile, path: Path | None = None) -> Path:
79
+ path = path or config_path()
80
+ path.parent.mkdir(parents=True, exist_ok=True)
81
+
82
+ def toml_list(values: list[str]) -> str:
83
+ return "[" + ", ".join(f'"{v}"' for v in values) + "]"
84
+
85
+ path.write_text(
86
+ f"languages = {toml_list(profile.languages)}\n"
87
+ f"topics = {toml_list(profile.topics)}\n"
88
+ f"contributions = {toml_list(profile.contributions)}\n"
89
+ f"days = {profile.days}\n"
90
+ )
91
+ return path
92
+
93
+
94
+ def from_args(args, stored: Profile | None = None) -> Profile:
95
+ """Flags override the stored profile field by field; absent flags fall back."""
96
+ stored = stored or Profile()
97
+ return Profile(
98
+ languages=_csv(getattr(args, "lang", None)) or stored.languages,
99
+ topics=_csv(getattr(args, "topic", None)) or stored.topics,
100
+ contributions=_csv(getattr(args, "contribution", None)) or stored.contributions,
101
+ days=getattr(args, "days", None) or stored.days,
102
+ )
103
+
104
+
105
+ def ask(existing: Profile | None = None) -> Profile:
106
+ """The interactive form. Four questions, each of which changes the output."""
107
+ current = existing or Profile()
108
+
109
+ def prompt(question: str, default: str) -> str:
110
+ suffix = f" [{default}]" if default else ""
111
+ answer = input(f" {question}{suffix} > ").strip()
112
+ return answer or default
113
+
114
+ languages = _csv(prompt("Languages you want to work in", ", ".join(current.languages)))
115
+ topics = _csv(prompt("What kind of project (topics)", ", ".join(current.topics)))
116
+ contributions = _csv(prompt(
117
+ f"What you want to contribute ({'/'.join(CONTRIBUTION_AREAS)})",
118
+ ", ".join(current.contributions),
119
+ ))
120
+ days_raw = prompt("How many days you actually have", str(current.days))
121
+ try:
122
+ days = max(1, int(days_raw))
123
+ except ValueError:
124
+ days = current.days
125
+ return Profile(languages=languages, topics=topics,
126
+ contributions=contributions, days=days)