holt-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. holt/__init__.py +0 -0
  2. holt/agent/__init__.py +0 -0
  3. holt/agent/entry.py +86 -0
  4. holt/agent/findings.py +49 -0
  5. holt/agent/landing.py +154 -0
  6. holt/agent/pipeline.py +244 -0
  7. holt/agent/progression.py +408 -0
  8. holt/agent/signals.py +220 -0
  9. holt/agent/stages.py +533 -0
  10. holt/agent/verdict.py +226 -0
  11. holt/agent/verify.py +140 -0
  12. holt/baseline.py +89 -0
  13. holt/baseline_matched.py +116 -0
  14. holt/cli.py +616 -0
  15. holt/discover.py +497 -0
  16. holt/evidence/__init__.py +3 -0
  17. holt/evidence/fixtures.py +154 -0
  18. holt/evidence/github_graphql.py +538 -0
  19. holt/evidence/provider.py +77 -0
  20. holt/evidence/redact.py +79 -0
  21. holt/issues.py +41 -0
  22. holt/model.py +516 -0
  23. holt/profile.py +126 -0
  24. holt/report.py +157 -0
  25. holt/tui/__init__.py +0 -0
  26. holt/tui/animation.py +84 -0
  27. holt/tui/app.py +294 -0
  28. holt/tui/clipboard.py +89 -0
  29. holt/tui/commands.py +134 -0
  30. holt/tui/discovery.py +305 -0
  31. holt/tui/env.py +49 -0
  32. holt/tui/events.py +245 -0
  33. holt/tui/mascot.py +121 -0
  34. holt/tui/models.py +590 -0
  35. holt/tui/observe.py +297 -0
  36. holt/tui/screens/__init__.py +35 -0
  37. holt/tui/screens/assessment.py +337 -0
  38. holt/tui/screens/confirm.py +62 -0
  39. holt/tui/screens/discover.py +444 -0
  40. holt/tui/screens/home.py +519 -0
  41. holt/tui/screens/inspector.py +106 -0
  42. holt/tui/screens/live.py +335 -0
  43. holt/tui/screens/models.py +393 -0
  44. holt/tui/screens/next_steps.py +425 -0
  45. holt/tui/screens/profile.py +129 -0
  46. holt/tui/session.py +711 -0
  47. holt/tui/store.py +458 -0
  48. holt/tui/theme.py +479 -0
  49. holt/tui/visual.py +33 -0
  50. holt/tui/widgets/__init__.py +0 -0
  51. holt/tui/widgets/candidates.py +78 -0
  52. holt/tui/widgets/claims.py +59 -0
  53. holt/tui/widgets/disclosure.py +121 -0
  54. holt/tui/widgets/evidence.py +121 -0
  55. holt/tui/widgets/masthead.py +122 -0
  56. holt/tui/widgets/recent.py +167 -0
  57. holt/tui/widgets/scrolling.py +38 -0
  58. holt/tui/widgets/stages.py +232 -0
  59. holt/types.py +48 -0
  60. holt_cli-0.1.0.dist-info/METADATA +198 -0
  61. holt_cli-0.1.0.dist-info/RECORD +65 -0
  62. holt_cli-0.1.0.dist-info/WHEEL +4 -0
  63. holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
  64. holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
  65. holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
@@ -0,0 +1,538 @@
1
+ """Live GitHub evidence, via GraphQL.
2
+
3
+ REST needs roughly four calls per pull request (the PR, its reviews, its
4
+ comments, its files). At 5,000 requests/hour that exhausts the budget well
5
+ before the pool is crawled. One GraphQL query returns a page of PRs with all
6
+ four, so the same crawl costs a couple of hundred points instead.
7
+
8
+ A pull request is decomposed into *events*, not stored whole. A PR opened in
9
+ April and merged in July is two facts with two timestamps: the agent may see
10
+ the first and must not see the second. Storing the PR as a single record with
11
+ a single timestamp would force a choice between leaking the merge and hiding
12
+ the thread. Event decomposition removes the choice.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import os
18
+ from collections.abc import Iterable, Iterator
19
+ from datetime import UTC, datetime
20
+ from typing import Any
21
+
22
+ import httpx
23
+
24
+ from holt.evidence.provider import EvidenceProvider
25
+ from holt.types import EvidenceRecord, Window
26
+
27
+ API = "https://api.github.com/graphql"
28
+
29
+ # Comment and review bodies carry the signal Holt actually reads: tone, intent,
30
+ # whether a maintainer engaged. Four thousand characters is far more than any of
31
+ # that needs. What blows past it is log dumps and stack traces -- one observed
32
+ # comment ran to 74,000 characters -- which cost a judge download size and cost
33
+ # the model context without changing a single judgement. Truncation is recorded
34
+ # on the record so a reader is never silently shown a partial quote.
35
+ MAX_BODY_CHARS = 4000
36
+
37
+ REPO_META = """
38
+ query($owner:String!, $name:String!, $until:GitTimestamp!) {
39
+ rateLimit { remaining resetAt }
40
+ repository(owner:$owner, name:$name) {
41
+ createdAt pushedAt isArchived isMirror isFork stargazerCount
42
+ description homepageUrl primaryLanguage { name }
43
+ defaultBranchRef {
44
+ name
45
+ target {
46
+ ... on Commit { history(until:$until, first:1) { nodes { oid committedDate } } }
47
+ }
48
+ }
49
+ }
50
+ }
51
+ """
52
+
53
+ # The README as it stood at the cutoff, not as it stands today. Reading HEAD
54
+ # would hand the agent a document rewritten months after the window it is
55
+ # supposed to be reasoning about -- a leak that would never announce itself.
56
+ REPO_DOCS = """
57
+ query($owner:String!, $name:String!, $readme:String!, $contributing:String!) {
58
+ rateLimit { remaining resetAt }
59
+ repository(owner:$owner, name:$name) {
60
+ readme: object(expression:$readme) { ... on Blob { text } }
61
+ contributing: object(expression:$contributing) { ... on Blob { text } }
62
+ }
63
+ }
64
+ """
65
+
66
+ # Date filtering happens server-side. Ordering by newest and paging until the
67
+ # timestamps fall past the cutoff would burn most of the rate-limit budget on
68
+ # records the window filter then discards.
69
+ PR_SEARCH = """
70
+ query($q:String!, $cursor:String) {
71
+ rateLimit { remaining resetAt }
72
+ search(query:$q, type:ISSUE, first:25, after:$cursor) {
73
+ issueCount
74
+ pageInfo { hasNextPage endCursor }
75
+ nodes {
76
+ ... on PullRequest {
77
+ number title createdAt mergedAt closedAt merged
78
+ additions deletions changedFiles
79
+ author { login __typename }
80
+ files(first:20) { nodes { path additions deletions } }
81
+ reviews(first:20) { nodes { createdAt state body author { login __typename } } }
82
+ comments(first:30) { nodes { createdAt body author { login __typename } } }
83
+ }
84
+ }
85
+ }
86
+ }
87
+ """
88
+
89
+
90
+ # Issues, for Path Finder. Decomposed the same way pull requests are: an issue
91
+ # opening is a pre-cutoff fact, and an issue being closed by somebody's merged
92
+ # pull request is a post-cutoff one. The two must not travel together.
93
+ ISSUE_SEARCH = """
94
+ query($q:String!, $cursor:String) {
95
+ rateLimit { remaining resetAt }
96
+ search(query:$q, type:ISSUE, first:50, after:$cursor) {
97
+ issueCount
98
+ pageInfo { hasNextPage endCursor }
99
+ nodes {
100
+ ... on Issue {
101
+ number title body createdAt closedAt lastEditedAt
102
+ author { login __typename }
103
+ labels(first:12) { nodes { name } }
104
+ comments { totalCount }
105
+ closedByPullRequestsReferences(first:5, includeClosedPrs:true) {
106
+ nodes { number mergedAt author { login __typename } }
107
+ }
108
+ }
109
+ }
110
+ }
111
+ }
112
+ """
113
+
114
+ MAX_ISSUE_BODY = 4000
115
+
116
+ # Repository search, for `holt discover`. Sourcing only: these results are where
117
+ # candidates come from, and the output says so. Nothing downstream treats search
118
+ # rank as a signal — the screening pass re-derives everything it uses from the
119
+ # contribution history.
120
+ REPO_SEARCH = """
121
+ query($q:String!, $cursor:String) {
122
+ rateLimit { remaining resetAt }
123
+ search(query:$q, type:REPOSITORY, first:25, after:$cursor) {
124
+ repositoryCount
125
+ pageInfo { hasNextPage endCursor }
126
+ nodes {
127
+ ... on Repository {
128
+ nameWithOwner description stargazerCount pushedAt isArchived isFork
129
+ primaryLanguage { name }
130
+ }
131
+ }
132
+ }
133
+ }
134
+ """
135
+
136
+
137
+ def _ts(value: str | None) -> datetime | None:
138
+ return datetime.fromisoformat(value.replace("Z", "+00:00")) if value else None
139
+
140
+
141
+ def _body(text: str | None) -> tuple[str | None, bool, int]:
142
+ """Return (possibly truncated body, was_truncated, original_length)."""
143
+ if not text:
144
+ return text, False, 0
145
+ if len(text) <= MAX_BODY_CHARS:
146
+ return text, False, len(text)
147
+ return text[:MAX_BODY_CHARS], True, len(text)
148
+
149
+
150
+ def _login(actor: dict[str, Any] | None) -> str:
151
+ """Deleted accounts come back as null; bots carry a distinct __typename."""
152
+ if not actor:
153
+ return "(ghost)"
154
+ return actor.get("login") or "(ghost)"
155
+
156
+
157
+ def _is_bot(actor: dict[str, Any] | None) -> bool:
158
+ if not actor:
159
+ return False
160
+ if actor.get("__typename") == "Bot":
161
+ return True
162
+ login = (actor.get("login") or "").lower()
163
+ return login.endswith("[bot]") or login in {"dependabot", "renovate", "greenkeeper"}
164
+
165
+
166
+ class GitHubGraphQL:
167
+ """Thin transport. Knows about auth, pagination and the rate-limit budget."""
168
+
169
+ def __init__(self, token: str | None = None, client: httpx.Client | None = None) -> None:
170
+ self.token = token or os.environ.get("GITHUB_TOKEN")
171
+ if not self.token:
172
+ raise RuntimeError(
173
+ "GITHUB_TOKEN is not set. Live mode needs a token; "
174
+ "use fixture or replay mode to run without one."
175
+ )
176
+ self._client = client or httpx.Client(timeout=30.0)
177
+ self.remaining: int | None = None
178
+
179
+ def query(self, document: str, **variables: object) -> dict[str, Any]:
180
+ response = self._client.post(
181
+ API,
182
+ headers={"Authorization": f"bearer {self.token}"},
183
+ json={"query": document, "variables": variables},
184
+ )
185
+ response.raise_for_status()
186
+ body = response.json()
187
+ if "errors" in body:
188
+ raise RuntimeError(f"GraphQL error: {body['errors']}")
189
+ data = body["data"]
190
+ if limit := data.get("rateLimit"):
191
+ self.remaining = limit["remaining"]
192
+ return data
193
+
194
+ def repo_meta(self, owner: str, name: str, until: datetime) -> dict[str, Any]:
195
+ repo = self.query(
196
+ REPO_META, owner=owner, name=name, until=until.isoformat()
197
+ )["repository"]
198
+ if repo is None:
199
+ raise RuntimeError(f"{owner}/{name} not found or not public")
200
+ return repo
201
+
202
+ def docs_at(self, owner: str, name: str, oid: str) -> dict[str, Any]:
203
+ """README and CONTRIBUTING at a specific commit."""
204
+ return self.query(
205
+ REPO_DOCS,
206
+ owner=owner,
207
+ name=name,
208
+ readme=f"{oid}:README.md",
209
+ contributing=f"{oid}:CONTRIBUTING.md",
210
+ )["repository"]
211
+
212
+ def search_issues(self, q: str, max_pages: int = 6) -> Iterator[dict[str, Any]]:
213
+ cursor: str | None = None
214
+ for _ in range(max_pages):
215
+ search = self.query(ISSUE_SEARCH, q=q, cursor=cursor)["search"]
216
+ yield from (n for n in search["nodes"] if n)
217
+ page = search["pageInfo"]
218
+ if not page["hasNextPage"]:
219
+ return
220
+ cursor = page["endCursor"]
221
+
222
+ def search_repositories(self, q: str, max_pages: int = 2) -> Iterator[dict[str, Any]]:
223
+ cursor: str | None = None
224
+ for _ in range(max_pages):
225
+ search = self.query(REPO_SEARCH, q=q, cursor=cursor)["search"]
226
+ yield from (n for n in search["nodes"] if n)
227
+ page = search["pageInfo"]
228
+ if not page["hasNextPage"]:
229
+ return
230
+ cursor = page["endCursor"]
231
+
232
+ def search_pull_requests(self, q: str, max_pages: int = 8) -> Iterator[dict[str, Any]]:
233
+ cursor: str | None = None
234
+ for _ in range(max_pages):
235
+ search = self.query(PR_SEARCH, q=q, cursor=cursor)["search"]
236
+ yield from (n for n in search["nodes"] if n)
237
+ page = search["pageInfo"]
238
+ if not page["hasNextPage"]:
239
+ return
240
+ cursor = page["endCursor"]
241
+
242
+
243
+ def search_query(repo_slug: str, window: Window, cutoff: datetime) -> str:
244
+ """Bound the crawl by date server-side, on the side of the holdout we are on."""
245
+ day = cutoff.date().isoformat()
246
+ bound = f"created:<{day}" if window is Window.PRE_T else f"created:>={day}"
247
+ return f"repo:{repo_slug} is:pr {bound} sort:created-desc"
248
+
249
+
250
+ def _nodes(connection: dict[str, Any] | None) -> list[dict[str, Any]]:
251
+ """A connection with nothing in it can come back as null, not as an empty list.
252
+
253
+ Observed on a pull request that changed no files: `files` was null while
254
+ `changedFiles` was 0. Treating null and empty as the same thing here keeps a
255
+ single odd pull request from aborting a repository's whole capture.
256
+ """
257
+ if not connection:
258
+ return []
259
+ return [n for n in (connection.get("nodes") or []) if n]
260
+
261
+
262
+ def _with_body(payload: dict[str, Any], raw: str | None) -> dict[str, Any]:
263
+ body, truncated, original = _body(raw)
264
+ payload["body"] = body
265
+ if truncated:
266
+ payload["body_truncated"] = True
267
+ payload["body_original_chars"] = original
268
+ return payload
269
+
270
+
271
+ def project(repo_slug: str, nodes: Iterable[dict[str, Any]]) -> Iterator[EvidenceRecord]:
272
+ """Turn pull requests into timestamped, individually-addressable evidence."""
273
+ for pr in nodes:
274
+ number = pr["number"]
275
+ base = f"pr:{repo_slug}#{number}"
276
+ url = f"https://github.com/{repo_slug}/pull/{number}"
277
+ shared = {"author": _login(pr["author"]), "author_is_bot": _is_bot(pr["author"])}
278
+
279
+ yield EvidenceRecord(
280
+ evidence_id=f"{base}:opened",
281
+ source="github",
282
+ url=url,
283
+ timestamp=_ts(pr["createdAt"]),
284
+ payload={
285
+ **shared,
286
+ "title": pr["title"],
287
+ "additions": pr["additions"],
288
+ "deletions": pr["deletions"],
289
+ "changed_files": pr["changedFiles"],
290
+ "files": [f["path"] for f in _nodes(pr["files"])],
291
+ },
292
+ )
293
+
294
+ if merged_at := _ts(pr["mergedAt"]):
295
+ yield EvidenceRecord(
296
+ evidence_id=f"{base}:merged",
297
+ source="github",
298
+ url=url,
299
+ timestamp=merged_at,
300
+ payload={**shared, "merged": True},
301
+ )
302
+ elif (closed_at := _ts(pr["closedAt"])) and not pr["merged"]:
303
+ yield EvidenceRecord(
304
+ evidence_id=f"{base}:closed",
305
+ source="github",
306
+ url=url,
307
+ timestamp=closed_at,
308
+ payload={**shared, "merged": False},
309
+ )
310
+
311
+ for i, review in enumerate(_nodes(pr["reviews"])):
312
+ yield EvidenceRecord(
313
+ evidence_id=f"{base}:review:{i}",
314
+ source="github",
315
+ url=url,
316
+ timestamp=_ts(review["createdAt"]),
317
+ payload=_with_body(
318
+ {
319
+ "author": _login(review["author"]),
320
+ "author_is_bot": _is_bot(review["author"]),
321
+ "state": review["state"],
322
+ },
323
+ review["body"],
324
+ ),
325
+ )
326
+
327
+ for i, comment in enumerate(_nodes(pr["comments"])):
328
+ yield EvidenceRecord(
329
+ evidence_id=f"{base}:comment:{i}",
330
+ source="github",
331
+ url=url,
332
+ timestamp=_ts(comment["createdAt"]),
333
+ payload=_with_body(
334
+ {
335
+ "author": _login(comment["author"]),
336
+ "author_is_bot": _is_bot(comment["author"]),
337
+ },
338
+ comment["body"],
339
+ ),
340
+ )
341
+
342
+
343
+ def project_repo_meta(repo_slug: str, repo: dict[str, Any]) -> EvidenceRecord:
344
+ """Repository-level facts.
345
+
346
+ Mutable counters (stars) are as-of-fetch, not as-of-T: GitHub does not expose
347
+ a historical star count, so they cannot be reconstructed at the cutoff. The
348
+ payload says so. Holt's own reasoning must not lean on them; the popularity
349
+ diagnostic does, and that limitation is published rather than hidden.
350
+ """
351
+ return EvidenceRecord(
352
+ evidence_id=f"repo:{repo_slug}:meta",
353
+ source="github",
354
+ url=f"https://github.com/{repo_slug}",
355
+ timestamp=_ts(repo["createdAt"]),
356
+ payload={
357
+ "pushed_at": repo["pushedAt"],
358
+ "is_archived": repo["isArchived"],
359
+ "is_mirror": repo["isMirror"],
360
+ "is_fork": repo["isFork"],
361
+ "description": repo["description"],
362
+ "homepage_url": repo["homepageUrl"],
363
+ "primary_language": (repo["primaryLanguage"] or {}).get("name"),
364
+ "stargazer_count": repo["stargazerCount"],
365
+ "_counters_are_as_of_fetch_not_cutoff": True,
366
+ },
367
+ )
368
+
369
+
370
+ MAX_DOC_CHARS = 12000
371
+
372
+
373
+ def project_docs(repo_slug: str, docs: dict[str, Any], commit: dict[str, Any]) -> Iterator[EvidenceRecord]:
374
+ """README and CONTRIBUTING as they stood at the cutoff commit."""
375
+ when = _ts(commit["committedDate"])
376
+ for kind in ("readme", "contributing"):
377
+ blob = docs.get(kind)
378
+ text = (blob or {}).get("text")
379
+ if not text:
380
+ continue
381
+ truncated = len(text) > MAX_DOC_CHARS
382
+ payload: dict[str, Any] = {
383
+ "kind": kind,
384
+ "text": text[:MAX_DOC_CHARS],
385
+ "commit_oid": commit["oid"],
386
+ }
387
+ if truncated:
388
+ payload["text_truncated"] = True
389
+ payload["text_original_chars"] = len(text)
390
+ yield EvidenceRecord(
391
+ evidence_id=f"repo:{repo_slug}:{kind}",
392
+ source="github",
393
+ url=f"https://github.com/{repo_slug}/blob/{commit['oid']}/{kind.upper()}.md",
394
+ timestamp=when,
395
+ payload=payload,
396
+ )
397
+
398
+
399
+ def project_issues(repo_slug: str, nodes: Iterable[dict[str, Any]]) -> Iterator[EvidenceRecord]:
400
+ """Issue events. Opening is pre-cutoff evidence; being resolved is the label."""
401
+ for issue in nodes:
402
+ number = issue["number"]
403
+ base = f"issue:{repo_slug}#{number}"
404
+ url = f"https://github.com/{repo_slug}/issues/{number}"
405
+ body = issue.get("body") or ""
406
+
407
+ yield EvidenceRecord(
408
+ evidence_id=f"{base}:opened",
409
+ source="github",
410
+ url=url,
411
+ timestamp=_ts(issue["createdAt"]),
412
+ payload={
413
+ "title": issue.get("title"),
414
+ "body": body[:MAX_ISSUE_BODY],
415
+ "body_truncated": len(body) > MAX_ISSUE_BODY,
416
+ "labels": [n["name"] for n in (issue.get("labels") or {}).get("nodes", [])],
417
+ "comments": (issue.get("comments") or {}).get("totalCount", 0),
418
+ "author": _login(issue.get("author")),
419
+ # The body GitHub returns is the current one, not the one that
420
+ # existed at the cutoff. `lastEditedAt` is null unless the body
421
+ # itself was edited; `updatedAt` bumps on any comment or label
422
+ # change, and using it measured "had activity" rather than "was
423
+ # edited" -- reporting a 100% leak that was not real.
424
+ "last_edited_at": issue.get("lastEditedAt"),
425
+ },
426
+ )
427
+
428
+ closed_at = _ts(issue.get("closedAt"))
429
+ if not closed_at:
430
+ continue
431
+ merged = [
432
+ p for p in (issue.get("closedByPullRequestsReferences") or {}).get("nodes", [])
433
+ if p and p.get("mergedAt")
434
+ ]
435
+ yield EvidenceRecord(
436
+ evidence_id=f"{base}:closed",
437
+ source="github",
438
+ url=url,
439
+ timestamp=closed_at,
440
+ payload={
441
+ "resolved_by_merged_pr": bool(merged),
442
+ "closing_prs": [
443
+ {"number": p["number"], "author": _login(p.get("author")),
444
+ "author_is_bot": _is_bot(p.get("author"))}
445
+ for p in merged
446
+ ],
447
+ },
448
+ )
449
+
450
+
451
+ class LiveGitHubProvider(EvidenceProvider):
452
+ """Crawls GitHub, then hands every record to the base-class window check.
453
+
454
+ The default cutoff is **now**, not the benchmark's T. T = 2026-06-01 is an
455
+ evaluation device; a live reader wants everything up to today, and a caller
456
+ that inherited T by default reported an active repository created in July as
457
+ having no history at all. The evaluation and the fixture capture pass their
458
+ cutoff explicitly, which is the correct place for that decision to be
459
+ visible.
460
+ """
461
+
462
+ def __init__(
463
+ self,
464
+ window: Window,
465
+ cutoff: datetime | None = None,
466
+ transport: GitHubGraphQL | None = None,
467
+ max_pages: int = 8,
468
+ ) -> None:
469
+ super().__init__(window, cutoff or datetime.now(UTC))
470
+ self.transport = transport or GitHubGraphQL()
471
+ self.max_pages = max_pages
472
+ self._seen: dict[str, EvidenceRecord] = {}
473
+
474
+ def _fetch_raw(self, request: str, /, **params: object) -> Iterable[EvidenceRecord]:
475
+ owner, _, name = request.partition("/")
476
+ meta = self.transport.repo_meta(owner, name, self.cutoff)
477
+ records: list[EvidenceRecord] = [project_repo_meta(request, meta)]
478
+
479
+ branch = meta.get("defaultBranchRef") or {}
480
+ history = ((branch.get("target") or {}).get("history") or {}).get("nodes") or []
481
+ if history:
482
+ docs = self.transport.docs_at(owner, name, history[0]["oid"])
483
+ records.extend(project_docs(request, docs, history[0]))
484
+ nodes = self.transport.search_pull_requests(
485
+ search_query(request, self.window, self.cutoff), self.max_pages
486
+ )
487
+ records.extend(project(request, nodes))
488
+
489
+ # Slice at the source; the base-class assertion is the safety net, not
490
+ # the filter. A PR created before T can still carry a merge after it.
491
+ kept = [r for r in records if self._in_window(r)]
492
+ self._seen.update({r.evidence_id: r for r in kept})
493
+ return kept
494
+
495
+ def _in_window(self, record: EvidenceRecord) -> bool:
496
+ if self.window is Window.PRE_T:
497
+ return record.timestamp <= self.cutoff
498
+ return record.timestamp > self.cutoff
499
+
500
+ def _resolve_raw(self, evidence_id: str) -> EvidenceRecord | None:
501
+ return self._seen.get(evidence_id)
502
+
503
+
504
+ class LiveGitHubIssueProvider(EvidenceProvider):
505
+ """Issues, through the same chokepoint as everything else.
506
+
507
+ Separate from `LiveGitHubProvider` rather than a flag on it because the two
508
+ answer different questions and are captured into different fixture roots. A
509
+ provider that returned issues or pull requests depending on a constructor
510
+ argument would make every window assertion harder to read for no gain.
511
+ """
512
+
513
+ def __init__(
514
+ self,
515
+ window: Window,
516
+ cutoff: datetime | None = None,
517
+ transport: GitHubGraphQL | None = None,
518
+ max_pages: int = 6,
519
+ ) -> None:
520
+ # Same default as LiveGitHubProvider, for the same reason: live means now.
521
+ super().__init__(window, cutoff or datetime.now(UTC))
522
+ self.transport = transport or GitHubGraphQL()
523
+ self.max_pages = max_pages
524
+ self._seen: dict[str, EvidenceRecord] = {}
525
+
526
+ def _fetch_raw(self, request: str, /, **params: object) -> Iterable[EvidenceRecord]:
527
+ # Slice at the source. `created:<T` is a server-side qualifier, so the
528
+ # newest-first ordering cannot fill the page with issues we must not see.
529
+ day = self.cutoff.date().isoformat()
530
+ nodes = self.transport.search_issues(
531
+ f"repo:{request} is:issue created:<{day}", self.max_pages
532
+ )
533
+ kept = [r for r in project_issues(request, nodes) if r.timestamp <= self.cutoff]
534
+ self._seen.update({r.evidence_id: r for r in kept})
535
+ return kept
536
+
537
+ def _resolve_raw(self, evidence_id: str) -> EvidenceRecord | None:
538
+ return self._seen.get(evidence_id)
@@ -0,0 +1,77 @@
1
+ """The single chokepoint every fact passes through.
2
+
3
+ Both GitHub and web results resolve through this interface, in one of two
4
+ implementations: live (real network) or fixture (committed JSON). That buys
5
+ three things at once:
6
+
7
+ * fixture mode makes the eval reproducible with no token and no rate limits
8
+ * live mode is the real product
9
+ * one place asserts the holdout boundary, so contamination is structurally
10
+ impossible rather than a matter of discipline
11
+
12
+ Subclasses implement ``_fetch_raw`` / ``_resolve_raw``. They cannot skip the
13
+ boundary check: the public methods own it.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from abc import ABC, abstractmethod
19
+ from collections.abc import Iterable
20
+ from datetime import datetime
21
+
22
+ from holt.types import T_CUTOFF, EvidenceRecord, Window
23
+
24
+
25
+ class ContaminationError(AssertionError):
26
+ """A provider returned a record from the wrong side of the holdout.
27
+
28
+ This is a bug, not a condition to handle. It means the agent was about to
29
+ see post-cutoff data, or a label was about to be computed from pre-cutoff
30
+ data. Either invalidates the measured claim.
31
+ """
32
+
33
+
34
+ class EvidenceProvider(ABC):
35
+ def __init__(self, window: Window, cutoff: datetime = T_CUTOFF) -> None:
36
+ self.window = window
37
+ self.cutoff = cutoff
38
+ # Every access, in order. The trajectory deliverable asks what the agent
39
+ # did and how its tools responded; the model calls are only half of that.
40
+ self.call_log: list[tuple[str, str, int]] = []
41
+
42
+ def fetch(self, request: str, /, **params: object) -> list[EvidenceRecord]:
43
+ records = list(self._fetch_raw(request, **params))
44
+ for record in records:
45
+ self._assert_in_window(record)
46
+ self.call_log.append(("fetch", request, len(records)))
47
+ return records
48
+
49
+ def resolve(self, evidence_id: str) -> EvidenceRecord | None:
50
+ """Return the record behind an id, or None if it does not resolve.
51
+
52
+ Stage D uses the None case to drop findings rather than soften them.
53
+ """
54
+ record = self._resolve_raw(evidence_id)
55
+ if record is not None:
56
+ self._assert_in_window(record)
57
+ self.call_log.append(("resolve", evidence_id, 1 if record else 0))
58
+ return record
59
+
60
+ def _assert_in_window(self, record: EvidenceRecord) -> None:
61
+ if self.window is Window.PRE_T and record.timestamp > self.cutoff:
62
+ raise ContaminationError(
63
+ f"{record.evidence_id} is dated {record.timestamp.isoformat()}, "
64
+ f"after the cutoff {self.cutoff.isoformat()}; the agent must not see it"
65
+ )
66
+ if self.window is Window.POST_T and record.timestamp <= self.cutoff:
67
+ raise ContaminationError(
68
+ f"{record.evidence_id} is dated {record.timestamp.isoformat()}, "
69
+ f"at or before the cutoff {self.cutoff.isoformat()}; "
70
+ "labels must not be computed from it"
71
+ )
72
+
73
+ @abstractmethod
74
+ def _fetch_raw(self, request: str, /, **params: object) -> Iterable[EvidenceRecord]: ...
75
+
76
+ @abstractmethod
77
+ def _resolve_raw(self, evidence_id: str) -> EvidenceRecord | None: ...
@@ -0,0 +1,79 @@
1
+ """Strip third-party credentials out of captured evidence before it is committed.
2
+
3
+ Public GitHub issues contain leaked API keys, in bug reports and in secret-scanner
4
+ test cases alike. Crawling them is fine; **redistributing them in a submitted
5
+ artifact is not**, whether or not they are still live. So the scrub runs at
6
+ capture time and every fixture in this repository has been through it.
7
+
8
+ Two tiers, because one is not enough:
9
+
10
+ 1. Any string in a recognised credential format is replaced wherever it appears.
11
+ 2. In a record where tier 1 fired, long opaque runs are replaced too. One captured
12
+ issue printed a token backwards next to the real one — a format matcher will
13
+ never catch that, and a record already known to be discussing a live secret is
14
+ the right place to be blunt about it.
15
+
16
+ Tier 2 deliberately does not run repository-wide: commit hashes and base64 blobs
17
+ are legitimate evidence, and destroying them everywhere to catch one obfuscated
18
+ token would cost more than it buys.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import re
24
+ from typing import Any
25
+
26
+ MARKER = "[REDACTED-CREDENTIAL]"
27
+
28
+ # Prefixed formats only. A pattern loose enough to catch unprefixed secrets is
29
+ # loose enough to shred ordinary evidence.
30
+ CREDENTIAL_PATTERNS = [
31
+ re.compile(r"gh[pousr]_[A-Za-z0-9]{36,255}"),
32
+ re.compile(r"github_pat_[A-Za-z0-9_]{22,255}"),
33
+ re.compile(r"sk-proj-[A-Za-z0-9_-]{20,}"),
34
+ re.compile(r"sk-ant-[A-Za-z0-9_-]{20,}"),
35
+ re.compile(r"sk-[A-Za-z0-9]{32,}"),
36
+ re.compile(r"xox[baprs]-[A-Za-z0-9-]{10,}"),
37
+ re.compile(r"AKIA[0-9A-Z]{16}"),
38
+ re.compile(r"AIza[0-9A-Za-z_-]{35}"),
39
+ ]
40
+
41
+ # Tier 2, inside an already-flagged record only.
42
+ OPAQUE_RUN = re.compile(r"\b[A-Za-z0-9]{30,}\b")
43
+
44
+
45
+ def _scrub(text: str, patterns) -> tuple[str, int]:
46
+ hits = 0
47
+ for pattern in patterns:
48
+ text, n = pattern.subn(MARKER, text)
49
+ hits += n
50
+ return text, hits
51
+
52
+
53
+ def _walk(value: Any, patterns) -> tuple[Any, int]:
54
+ if isinstance(value, str):
55
+ return _scrub(value, patterns)
56
+ if isinstance(value, dict):
57
+ out, total = {}, 0
58
+ for k, v in value.items():
59
+ out[k], n = _walk(v, patterns)
60
+ total += n
61
+ return out, total
62
+ if isinstance(value, list):
63
+ out, total = [], 0
64
+ for v in value:
65
+ scrubbed, n = _walk(v, patterns)
66
+ out.append(scrubbed)
67
+ total += n
68
+ return out, total
69
+ return value, 0
70
+
71
+
72
+ def redact_payload(payload: dict[str, Any]) -> tuple[dict[str, Any], int]:
73
+ """Return a scrubbed copy and the number of secrets removed."""
74
+ scrubbed, hits = _walk(payload, CREDENTIAL_PATTERNS)
75
+ if hits:
76
+ scrubbed, extra = _walk(scrubbed, [OPAQUE_RUN])
77
+ hits += extra
78
+ scrubbed["redacted"] = True
79
+ return scrubbed, hits