holt-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- holt/__init__.py +0 -0
- holt/agent/__init__.py +0 -0
- holt/agent/entry.py +86 -0
- holt/agent/findings.py +49 -0
- holt/agent/landing.py +154 -0
- holt/agent/pipeline.py +244 -0
- holt/agent/progression.py +408 -0
- holt/agent/signals.py +220 -0
- holt/agent/stages.py +533 -0
- holt/agent/verdict.py +226 -0
- holt/agent/verify.py +140 -0
- holt/baseline.py +89 -0
- holt/baseline_matched.py +116 -0
- holt/cli.py +616 -0
- holt/discover.py +497 -0
- holt/evidence/__init__.py +3 -0
- holt/evidence/fixtures.py +154 -0
- holt/evidence/github_graphql.py +538 -0
- holt/evidence/provider.py +77 -0
- holt/evidence/redact.py +79 -0
- holt/issues.py +41 -0
- holt/model.py +516 -0
- holt/profile.py +126 -0
- holt/report.py +157 -0
- holt/tui/__init__.py +0 -0
- holt/tui/animation.py +84 -0
- holt/tui/app.py +294 -0
- holt/tui/clipboard.py +89 -0
- holt/tui/commands.py +134 -0
- holt/tui/discovery.py +305 -0
- holt/tui/env.py +49 -0
- holt/tui/events.py +245 -0
- holt/tui/mascot.py +121 -0
- holt/tui/models.py +590 -0
- holt/tui/observe.py +297 -0
- holt/tui/screens/__init__.py +35 -0
- holt/tui/screens/assessment.py +337 -0
- holt/tui/screens/confirm.py +62 -0
- holt/tui/screens/discover.py +444 -0
- holt/tui/screens/home.py +519 -0
- holt/tui/screens/inspector.py +106 -0
- holt/tui/screens/live.py +335 -0
- holt/tui/screens/models.py +393 -0
- holt/tui/screens/next_steps.py +425 -0
- holt/tui/screens/profile.py +129 -0
- holt/tui/session.py +711 -0
- holt/tui/store.py +458 -0
- holt/tui/theme.py +479 -0
- holt/tui/visual.py +33 -0
- holt/tui/widgets/__init__.py +0 -0
- holt/tui/widgets/candidates.py +78 -0
- holt/tui/widgets/claims.py +59 -0
- holt/tui/widgets/disclosure.py +121 -0
- holt/tui/widgets/evidence.py +121 -0
- holt/tui/widgets/masthead.py +122 -0
- holt/tui/widgets/recent.py +167 -0
- holt/tui/widgets/scrolling.py +38 -0
- holt/tui/widgets/stages.py +232 -0
- holt/types.py +48 -0
- holt_cli-0.1.0.dist-info/METADATA +198 -0
- holt_cli-0.1.0.dist-info/RECORD +65 -0
- holt_cli-0.1.0.dist-info/WHEEL +4 -0
- holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
- holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
- holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
|
@@ -0,0 +1,538 @@
|
|
|
1
|
+
"""Live GitHub evidence, via GraphQL.
|
|
2
|
+
|
|
3
|
+
REST needs roughly four calls per pull request (the PR, its reviews, its
|
|
4
|
+
comments, its files). At 5,000 requests/hour that exhausts the budget well
|
|
5
|
+
before the pool is crawled. One GraphQL query returns a page of PRs with all
|
|
6
|
+
four, so the same crawl costs a couple of hundred points instead.
|
|
7
|
+
|
|
8
|
+
A pull request is decomposed into *events*, not stored whole. A PR opened in
|
|
9
|
+
April and merged in July is two facts with two timestamps: the agent may see
|
|
10
|
+
the first and must not see the second. Storing the PR as a single record with
|
|
11
|
+
a single timestamp would force a choice between leaking the merge and hiding
|
|
12
|
+
the thread. Event decomposition removes the choice.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import os
|
|
18
|
+
from collections.abc import Iterable, Iterator
|
|
19
|
+
from datetime import UTC, datetime
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
import httpx
|
|
23
|
+
|
|
24
|
+
from holt.evidence.provider import EvidenceProvider
|
|
25
|
+
from holt.types import EvidenceRecord, Window
|
|
26
|
+
|
|
27
|
+
API = "https://api.github.com/graphql"
|
|
28
|
+
|
|
29
|
+
# Comment and review bodies carry the signal Holt actually reads: tone, intent,
|
|
30
|
+
# whether a maintainer engaged. Four thousand characters is far more than any of
|
|
31
|
+
# that needs. What blows past it is log dumps and stack traces -- one observed
|
|
32
|
+
# comment ran to 74,000 characters -- which cost a judge download size and cost
|
|
33
|
+
# the model context without changing a single judgement. Truncation is recorded
|
|
34
|
+
# on the record so a reader is never silently shown a partial quote.
|
|
35
|
+
MAX_BODY_CHARS = 4000
|
|
36
|
+
|
|
37
|
+
REPO_META = """
|
|
38
|
+
query($owner:String!, $name:String!, $until:GitTimestamp!) {
|
|
39
|
+
rateLimit { remaining resetAt }
|
|
40
|
+
repository(owner:$owner, name:$name) {
|
|
41
|
+
createdAt pushedAt isArchived isMirror isFork stargazerCount
|
|
42
|
+
description homepageUrl primaryLanguage { name }
|
|
43
|
+
defaultBranchRef {
|
|
44
|
+
name
|
|
45
|
+
target {
|
|
46
|
+
... on Commit { history(until:$until, first:1) { nodes { oid committedDate } } }
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
# The README as it stood at the cutoff, not as it stands today. Reading HEAD
|
|
54
|
+
# would hand the agent a document rewritten months after the window it is
|
|
55
|
+
# supposed to be reasoning about -- a leak that would never announce itself.
|
|
56
|
+
REPO_DOCS = """
|
|
57
|
+
query($owner:String!, $name:String!, $readme:String!, $contributing:String!) {
|
|
58
|
+
rateLimit { remaining resetAt }
|
|
59
|
+
repository(owner:$owner, name:$name) {
|
|
60
|
+
readme: object(expression:$readme) { ... on Blob { text } }
|
|
61
|
+
contributing: object(expression:$contributing) { ... on Blob { text } }
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
# Date filtering happens server-side. Ordering by newest and paging until the
|
|
67
|
+
# timestamps fall past the cutoff would burn most of the rate-limit budget on
|
|
68
|
+
# records the window filter then discards.
|
|
69
|
+
PR_SEARCH = """
|
|
70
|
+
query($q:String!, $cursor:String) {
|
|
71
|
+
rateLimit { remaining resetAt }
|
|
72
|
+
search(query:$q, type:ISSUE, first:25, after:$cursor) {
|
|
73
|
+
issueCount
|
|
74
|
+
pageInfo { hasNextPage endCursor }
|
|
75
|
+
nodes {
|
|
76
|
+
... on PullRequest {
|
|
77
|
+
number title createdAt mergedAt closedAt merged
|
|
78
|
+
additions deletions changedFiles
|
|
79
|
+
author { login __typename }
|
|
80
|
+
files(first:20) { nodes { path additions deletions } }
|
|
81
|
+
reviews(first:20) { nodes { createdAt state body author { login __typename } } }
|
|
82
|
+
comments(first:30) { nodes { createdAt body author { login __typename } } }
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# Issues, for Path Finder. Decomposed the same way pull requests are: an issue
|
|
91
|
+
# opening is a pre-cutoff fact, and an issue being closed by somebody's merged
|
|
92
|
+
# pull request is a post-cutoff one. The two must not travel together.
|
|
93
|
+
ISSUE_SEARCH = """
|
|
94
|
+
query($q:String!, $cursor:String) {
|
|
95
|
+
rateLimit { remaining resetAt }
|
|
96
|
+
search(query:$q, type:ISSUE, first:50, after:$cursor) {
|
|
97
|
+
issueCount
|
|
98
|
+
pageInfo { hasNextPage endCursor }
|
|
99
|
+
nodes {
|
|
100
|
+
... on Issue {
|
|
101
|
+
number title body createdAt closedAt lastEditedAt
|
|
102
|
+
author { login __typename }
|
|
103
|
+
labels(first:12) { nodes { name } }
|
|
104
|
+
comments { totalCount }
|
|
105
|
+
closedByPullRequestsReferences(first:5, includeClosedPrs:true) {
|
|
106
|
+
nodes { number mergedAt author { login __typename } }
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
"""
|
|
113
|
+
|
|
114
|
+
MAX_ISSUE_BODY = 4000
|
|
115
|
+
|
|
116
|
+
# Repository search, for `holt discover`. Sourcing only: these results are where
|
|
117
|
+
# candidates come from, and the output says so. Nothing downstream treats search
|
|
118
|
+
# rank as a signal — the screening pass re-derives everything it uses from the
|
|
119
|
+
# contribution history.
|
|
120
|
+
REPO_SEARCH = """
|
|
121
|
+
query($q:String!, $cursor:String) {
|
|
122
|
+
rateLimit { remaining resetAt }
|
|
123
|
+
search(query:$q, type:REPOSITORY, first:25, after:$cursor) {
|
|
124
|
+
repositoryCount
|
|
125
|
+
pageInfo { hasNextPage endCursor }
|
|
126
|
+
nodes {
|
|
127
|
+
... on Repository {
|
|
128
|
+
nameWithOwner description stargazerCount pushedAt isArchived isFork
|
|
129
|
+
primaryLanguage { name }
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
"""
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _ts(value: str | None) -> datetime | None:
|
|
138
|
+
return datetime.fromisoformat(value.replace("Z", "+00:00")) if value else None
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _body(text: str | None) -> tuple[str | None, bool, int]:
|
|
142
|
+
"""Return (possibly truncated body, was_truncated, original_length)."""
|
|
143
|
+
if not text:
|
|
144
|
+
return text, False, 0
|
|
145
|
+
if len(text) <= MAX_BODY_CHARS:
|
|
146
|
+
return text, False, len(text)
|
|
147
|
+
return text[:MAX_BODY_CHARS], True, len(text)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _login(actor: dict[str, Any] | None) -> str:
|
|
151
|
+
"""Deleted accounts come back as null; bots carry a distinct __typename."""
|
|
152
|
+
if not actor:
|
|
153
|
+
return "(ghost)"
|
|
154
|
+
return actor.get("login") or "(ghost)"
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _is_bot(actor: dict[str, Any] | None) -> bool:
|
|
158
|
+
if not actor:
|
|
159
|
+
return False
|
|
160
|
+
if actor.get("__typename") == "Bot":
|
|
161
|
+
return True
|
|
162
|
+
login = (actor.get("login") or "").lower()
|
|
163
|
+
return login.endswith("[bot]") or login in {"dependabot", "renovate", "greenkeeper"}
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
class GitHubGraphQL:
|
|
167
|
+
"""Thin transport. Knows about auth, pagination and the rate-limit budget."""
|
|
168
|
+
|
|
169
|
+
def __init__(self, token: str | None = None, client: httpx.Client | None = None) -> None:
|
|
170
|
+
self.token = token or os.environ.get("GITHUB_TOKEN")
|
|
171
|
+
if not self.token:
|
|
172
|
+
raise RuntimeError(
|
|
173
|
+
"GITHUB_TOKEN is not set. Live mode needs a token; "
|
|
174
|
+
"use fixture or replay mode to run without one."
|
|
175
|
+
)
|
|
176
|
+
self._client = client or httpx.Client(timeout=30.0)
|
|
177
|
+
self.remaining: int | None = None
|
|
178
|
+
|
|
179
|
+
def query(self, document: str, **variables: object) -> dict[str, Any]:
|
|
180
|
+
response = self._client.post(
|
|
181
|
+
API,
|
|
182
|
+
headers={"Authorization": f"bearer {self.token}"},
|
|
183
|
+
json={"query": document, "variables": variables},
|
|
184
|
+
)
|
|
185
|
+
response.raise_for_status()
|
|
186
|
+
body = response.json()
|
|
187
|
+
if "errors" in body:
|
|
188
|
+
raise RuntimeError(f"GraphQL error: {body['errors']}")
|
|
189
|
+
data = body["data"]
|
|
190
|
+
if limit := data.get("rateLimit"):
|
|
191
|
+
self.remaining = limit["remaining"]
|
|
192
|
+
return data
|
|
193
|
+
|
|
194
|
+
def repo_meta(self, owner: str, name: str, until: datetime) -> dict[str, Any]:
|
|
195
|
+
repo = self.query(
|
|
196
|
+
REPO_META, owner=owner, name=name, until=until.isoformat()
|
|
197
|
+
)["repository"]
|
|
198
|
+
if repo is None:
|
|
199
|
+
raise RuntimeError(f"{owner}/{name} not found or not public")
|
|
200
|
+
return repo
|
|
201
|
+
|
|
202
|
+
def docs_at(self, owner: str, name: str, oid: str) -> dict[str, Any]:
|
|
203
|
+
"""README and CONTRIBUTING at a specific commit."""
|
|
204
|
+
return self.query(
|
|
205
|
+
REPO_DOCS,
|
|
206
|
+
owner=owner,
|
|
207
|
+
name=name,
|
|
208
|
+
readme=f"{oid}:README.md",
|
|
209
|
+
contributing=f"{oid}:CONTRIBUTING.md",
|
|
210
|
+
)["repository"]
|
|
211
|
+
|
|
212
|
+
def search_issues(self, q: str, max_pages: int = 6) -> Iterator[dict[str, Any]]:
|
|
213
|
+
cursor: str | None = None
|
|
214
|
+
for _ in range(max_pages):
|
|
215
|
+
search = self.query(ISSUE_SEARCH, q=q, cursor=cursor)["search"]
|
|
216
|
+
yield from (n for n in search["nodes"] if n)
|
|
217
|
+
page = search["pageInfo"]
|
|
218
|
+
if not page["hasNextPage"]:
|
|
219
|
+
return
|
|
220
|
+
cursor = page["endCursor"]
|
|
221
|
+
|
|
222
|
+
def search_repositories(self, q: str, max_pages: int = 2) -> Iterator[dict[str, Any]]:
|
|
223
|
+
cursor: str | None = None
|
|
224
|
+
for _ in range(max_pages):
|
|
225
|
+
search = self.query(REPO_SEARCH, q=q, cursor=cursor)["search"]
|
|
226
|
+
yield from (n for n in search["nodes"] if n)
|
|
227
|
+
page = search["pageInfo"]
|
|
228
|
+
if not page["hasNextPage"]:
|
|
229
|
+
return
|
|
230
|
+
cursor = page["endCursor"]
|
|
231
|
+
|
|
232
|
+
def search_pull_requests(self, q: str, max_pages: int = 8) -> Iterator[dict[str, Any]]:
|
|
233
|
+
cursor: str | None = None
|
|
234
|
+
for _ in range(max_pages):
|
|
235
|
+
search = self.query(PR_SEARCH, q=q, cursor=cursor)["search"]
|
|
236
|
+
yield from (n for n in search["nodes"] if n)
|
|
237
|
+
page = search["pageInfo"]
|
|
238
|
+
if not page["hasNextPage"]:
|
|
239
|
+
return
|
|
240
|
+
cursor = page["endCursor"]
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def search_query(repo_slug: str, window: Window, cutoff: datetime) -> str:
|
|
244
|
+
"""Bound the crawl by date server-side, on the side of the holdout we are on."""
|
|
245
|
+
day = cutoff.date().isoformat()
|
|
246
|
+
bound = f"created:<{day}" if window is Window.PRE_T else f"created:>={day}"
|
|
247
|
+
return f"repo:{repo_slug} is:pr {bound} sort:created-desc"
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _nodes(connection: dict[str, Any] | None) -> list[dict[str, Any]]:
|
|
251
|
+
"""A connection with nothing in it can come back as null, not as an empty list.
|
|
252
|
+
|
|
253
|
+
Observed on a pull request that changed no files: `files` was null while
|
|
254
|
+
`changedFiles` was 0. Treating null and empty as the same thing here keeps a
|
|
255
|
+
single odd pull request from aborting a repository's whole capture.
|
|
256
|
+
"""
|
|
257
|
+
if not connection:
|
|
258
|
+
return []
|
|
259
|
+
return [n for n in (connection.get("nodes") or []) if n]
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _with_body(payload: dict[str, Any], raw: str | None) -> dict[str, Any]:
|
|
263
|
+
body, truncated, original = _body(raw)
|
|
264
|
+
payload["body"] = body
|
|
265
|
+
if truncated:
|
|
266
|
+
payload["body_truncated"] = True
|
|
267
|
+
payload["body_original_chars"] = original
|
|
268
|
+
return payload
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def project(repo_slug: str, nodes: Iterable[dict[str, Any]]) -> Iterator[EvidenceRecord]:
|
|
272
|
+
"""Turn pull requests into timestamped, individually-addressable evidence."""
|
|
273
|
+
for pr in nodes:
|
|
274
|
+
number = pr["number"]
|
|
275
|
+
base = f"pr:{repo_slug}#{number}"
|
|
276
|
+
url = f"https://github.com/{repo_slug}/pull/{number}"
|
|
277
|
+
shared = {"author": _login(pr["author"]), "author_is_bot": _is_bot(pr["author"])}
|
|
278
|
+
|
|
279
|
+
yield EvidenceRecord(
|
|
280
|
+
evidence_id=f"{base}:opened",
|
|
281
|
+
source="github",
|
|
282
|
+
url=url,
|
|
283
|
+
timestamp=_ts(pr["createdAt"]),
|
|
284
|
+
payload={
|
|
285
|
+
**shared,
|
|
286
|
+
"title": pr["title"],
|
|
287
|
+
"additions": pr["additions"],
|
|
288
|
+
"deletions": pr["deletions"],
|
|
289
|
+
"changed_files": pr["changedFiles"],
|
|
290
|
+
"files": [f["path"] for f in _nodes(pr["files"])],
|
|
291
|
+
},
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
if merged_at := _ts(pr["mergedAt"]):
|
|
295
|
+
yield EvidenceRecord(
|
|
296
|
+
evidence_id=f"{base}:merged",
|
|
297
|
+
source="github",
|
|
298
|
+
url=url,
|
|
299
|
+
timestamp=merged_at,
|
|
300
|
+
payload={**shared, "merged": True},
|
|
301
|
+
)
|
|
302
|
+
elif (closed_at := _ts(pr["closedAt"])) and not pr["merged"]:
|
|
303
|
+
yield EvidenceRecord(
|
|
304
|
+
evidence_id=f"{base}:closed",
|
|
305
|
+
source="github",
|
|
306
|
+
url=url,
|
|
307
|
+
timestamp=closed_at,
|
|
308
|
+
payload={**shared, "merged": False},
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
for i, review in enumerate(_nodes(pr["reviews"])):
|
|
312
|
+
yield EvidenceRecord(
|
|
313
|
+
evidence_id=f"{base}:review:{i}",
|
|
314
|
+
source="github",
|
|
315
|
+
url=url,
|
|
316
|
+
timestamp=_ts(review["createdAt"]),
|
|
317
|
+
payload=_with_body(
|
|
318
|
+
{
|
|
319
|
+
"author": _login(review["author"]),
|
|
320
|
+
"author_is_bot": _is_bot(review["author"]),
|
|
321
|
+
"state": review["state"],
|
|
322
|
+
},
|
|
323
|
+
review["body"],
|
|
324
|
+
),
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
for i, comment in enumerate(_nodes(pr["comments"])):
|
|
328
|
+
yield EvidenceRecord(
|
|
329
|
+
evidence_id=f"{base}:comment:{i}",
|
|
330
|
+
source="github",
|
|
331
|
+
url=url,
|
|
332
|
+
timestamp=_ts(comment["createdAt"]),
|
|
333
|
+
payload=_with_body(
|
|
334
|
+
{
|
|
335
|
+
"author": _login(comment["author"]),
|
|
336
|
+
"author_is_bot": _is_bot(comment["author"]),
|
|
337
|
+
},
|
|
338
|
+
comment["body"],
|
|
339
|
+
),
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def project_repo_meta(repo_slug: str, repo: dict[str, Any]) -> EvidenceRecord:
|
|
344
|
+
"""Repository-level facts.
|
|
345
|
+
|
|
346
|
+
Mutable counters (stars) are as-of-fetch, not as-of-T: GitHub does not expose
|
|
347
|
+
a historical star count, so they cannot be reconstructed at the cutoff. The
|
|
348
|
+
payload says so. Holt's own reasoning must not lean on them; the popularity
|
|
349
|
+
diagnostic does, and that limitation is published rather than hidden.
|
|
350
|
+
"""
|
|
351
|
+
return EvidenceRecord(
|
|
352
|
+
evidence_id=f"repo:{repo_slug}:meta",
|
|
353
|
+
source="github",
|
|
354
|
+
url=f"https://github.com/{repo_slug}",
|
|
355
|
+
timestamp=_ts(repo["createdAt"]),
|
|
356
|
+
payload={
|
|
357
|
+
"pushed_at": repo["pushedAt"],
|
|
358
|
+
"is_archived": repo["isArchived"],
|
|
359
|
+
"is_mirror": repo["isMirror"],
|
|
360
|
+
"is_fork": repo["isFork"],
|
|
361
|
+
"description": repo["description"],
|
|
362
|
+
"homepage_url": repo["homepageUrl"],
|
|
363
|
+
"primary_language": (repo["primaryLanguage"] or {}).get("name"),
|
|
364
|
+
"stargazer_count": repo["stargazerCount"],
|
|
365
|
+
"_counters_are_as_of_fetch_not_cutoff": True,
|
|
366
|
+
},
|
|
367
|
+
)
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
MAX_DOC_CHARS = 12000
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def project_docs(repo_slug: str, docs: dict[str, Any], commit: dict[str, Any]) -> Iterator[EvidenceRecord]:
|
|
374
|
+
"""README and CONTRIBUTING as they stood at the cutoff commit."""
|
|
375
|
+
when = _ts(commit["committedDate"])
|
|
376
|
+
for kind in ("readme", "contributing"):
|
|
377
|
+
blob = docs.get(kind)
|
|
378
|
+
text = (blob or {}).get("text")
|
|
379
|
+
if not text:
|
|
380
|
+
continue
|
|
381
|
+
truncated = len(text) > MAX_DOC_CHARS
|
|
382
|
+
payload: dict[str, Any] = {
|
|
383
|
+
"kind": kind,
|
|
384
|
+
"text": text[:MAX_DOC_CHARS],
|
|
385
|
+
"commit_oid": commit["oid"],
|
|
386
|
+
}
|
|
387
|
+
if truncated:
|
|
388
|
+
payload["text_truncated"] = True
|
|
389
|
+
payload["text_original_chars"] = len(text)
|
|
390
|
+
yield EvidenceRecord(
|
|
391
|
+
evidence_id=f"repo:{repo_slug}:{kind}",
|
|
392
|
+
source="github",
|
|
393
|
+
url=f"https://github.com/{repo_slug}/blob/{commit['oid']}/{kind.upper()}.md",
|
|
394
|
+
timestamp=when,
|
|
395
|
+
payload=payload,
|
|
396
|
+
)
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def project_issues(repo_slug: str, nodes: Iterable[dict[str, Any]]) -> Iterator[EvidenceRecord]:
|
|
400
|
+
"""Issue events. Opening is pre-cutoff evidence; being resolved is the label."""
|
|
401
|
+
for issue in nodes:
|
|
402
|
+
number = issue["number"]
|
|
403
|
+
base = f"issue:{repo_slug}#{number}"
|
|
404
|
+
url = f"https://github.com/{repo_slug}/issues/{number}"
|
|
405
|
+
body = issue.get("body") or ""
|
|
406
|
+
|
|
407
|
+
yield EvidenceRecord(
|
|
408
|
+
evidence_id=f"{base}:opened",
|
|
409
|
+
source="github",
|
|
410
|
+
url=url,
|
|
411
|
+
timestamp=_ts(issue["createdAt"]),
|
|
412
|
+
payload={
|
|
413
|
+
"title": issue.get("title"),
|
|
414
|
+
"body": body[:MAX_ISSUE_BODY],
|
|
415
|
+
"body_truncated": len(body) > MAX_ISSUE_BODY,
|
|
416
|
+
"labels": [n["name"] for n in (issue.get("labels") or {}).get("nodes", [])],
|
|
417
|
+
"comments": (issue.get("comments") or {}).get("totalCount", 0),
|
|
418
|
+
"author": _login(issue.get("author")),
|
|
419
|
+
# The body GitHub returns is the current one, not the one that
|
|
420
|
+
# existed at the cutoff. `lastEditedAt` is null unless the body
|
|
421
|
+
# itself was edited; `updatedAt` bumps on any comment or label
|
|
422
|
+
# change, and using it measured "had activity" rather than "was
|
|
423
|
+
# edited" -- reporting a 100% leak that was not real.
|
|
424
|
+
"last_edited_at": issue.get("lastEditedAt"),
|
|
425
|
+
},
|
|
426
|
+
)
|
|
427
|
+
|
|
428
|
+
closed_at = _ts(issue.get("closedAt"))
|
|
429
|
+
if not closed_at:
|
|
430
|
+
continue
|
|
431
|
+
merged = [
|
|
432
|
+
p for p in (issue.get("closedByPullRequestsReferences") or {}).get("nodes", [])
|
|
433
|
+
if p and p.get("mergedAt")
|
|
434
|
+
]
|
|
435
|
+
yield EvidenceRecord(
|
|
436
|
+
evidence_id=f"{base}:closed",
|
|
437
|
+
source="github",
|
|
438
|
+
url=url,
|
|
439
|
+
timestamp=closed_at,
|
|
440
|
+
payload={
|
|
441
|
+
"resolved_by_merged_pr": bool(merged),
|
|
442
|
+
"closing_prs": [
|
|
443
|
+
{"number": p["number"], "author": _login(p.get("author")),
|
|
444
|
+
"author_is_bot": _is_bot(p.get("author"))}
|
|
445
|
+
for p in merged
|
|
446
|
+
],
|
|
447
|
+
},
|
|
448
|
+
)
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
class LiveGitHubProvider(EvidenceProvider):
|
|
452
|
+
"""Crawls GitHub, then hands every record to the base-class window check.
|
|
453
|
+
|
|
454
|
+
The default cutoff is **now**, not the benchmark's T. T = 2026-06-01 is an
|
|
455
|
+
evaluation device; a live reader wants everything up to today, and a caller
|
|
456
|
+
that inherited T by default reported an active repository created in July as
|
|
457
|
+
having no history at all. The evaluation and the fixture capture pass their
|
|
458
|
+
cutoff explicitly, which is the correct place for that decision to be
|
|
459
|
+
visible.
|
|
460
|
+
"""
|
|
461
|
+
|
|
462
|
+
def __init__(
|
|
463
|
+
self,
|
|
464
|
+
window: Window,
|
|
465
|
+
cutoff: datetime | None = None,
|
|
466
|
+
transport: GitHubGraphQL | None = None,
|
|
467
|
+
max_pages: int = 8,
|
|
468
|
+
) -> None:
|
|
469
|
+
super().__init__(window, cutoff or datetime.now(UTC))
|
|
470
|
+
self.transport = transport or GitHubGraphQL()
|
|
471
|
+
self.max_pages = max_pages
|
|
472
|
+
self._seen: dict[str, EvidenceRecord] = {}
|
|
473
|
+
|
|
474
|
+
def _fetch_raw(self, request: str, /, **params: object) -> Iterable[EvidenceRecord]:
|
|
475
|
+
owner, _, name = request.partition("/")
|
|
476
|
+
meta = self.transport.repo_meta(owner, name, self.cutoff)
|
|
477
|
+
records: list[EvidenceRecord] = [project_repo_meta(request, meta)]
|
|
478
|
+
|
|
479
|
+
branch = meta.get("defaultBranchRef") or {}
|
|
480
|
+
history = ((branch.get("target") or {}).get("history") or {}).get("nodes") or []
|
|
481
|
+
if history:
|
|
482
|
+
docs = self.transport.docs_at(owner, name, history[0]["oid"])
|
|
483
|
+
records.extend(project_docs(request, docs, history[0]))
|
|
484
|
+
nodes = self.transport.search_pull_requests(
|
|
485
|
+
search_query(request, self.window, self.cutoff), self.max_pages
|
|
486
|
+
)
|
|
487
|
+
records.extend(project(request, nodes))
|
|
488
|
+
|
|
489
|
+
# Slice at the source; the base-class assertion is the safety net, not
|
|
490
|
+
# the filter. A PR created before T can still carry a merge after it.
|
|
491
|
+
kept = [r for r in records if self._in_window(r)]
|
|
492
|
+
self._seen.update({r.evidence_id: r for r in kept})
|
|
493
|
+
return kept
|
|
494
|
+
|
|
495
|
+
def _in_window(self, record: EvidenceRecord) -> bool:
|
|
496
|
+
if self.window is Window.PRE_T:
|
|
497
|
+
return record.timestamp <= self.cutoff
|
|
498
|
+
return record.timestamp > self.cutoff
|
|
499
|
+
|
|
500
|
+
def _resolve_raw(self, evidence_id: str) -> EvidenceRecord | None:
|
|
501
|
+
return self._seen.get(evidence_id)
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
class LiveGitHubIssueProvider(EvidenceProvider):
|
|
505
|
+
"""Issues, through the same chokepoint as everything else.
|
|
506
|
+
|
|
507
|
+
Separate from `LiveGitHubProvider` rather than a flag on it because the two
|
|
508
|
+
answer different questions and are captured into different fixture roots. A
|
|
509
|
+
provider that returned issues or pull requests depending on a constructor
|
|
510
|
+
argument would make every window assertion harder to read for no gain.
|
|
511
|
+
"""
|
|
512
|
+
|
|
513
|
+
def __init__(
|
|
514
|
+
self,
|
|
515
|
+
window: Window,
|
|
516
|
+
cutoff: datetime | None = None,
|
|
517
|
+
transport: GitHubGraphQL | None = None,
|
|
518
|
+
max_pages: int = 6,
|
|
519
|
+
) -> None:
|
|
520
|
+
# Same default as LiveGitHubProvider, for the same reason: live means now.
|
|
521
|
+
super().__init__(window, cutoff or datetime.now(UTC))
|
|
522
|
+
self.transport = transport or GitHubGraphQL()
|
|
523
|
+
self.max_pages = max_pages
|
|
524
|
+
self._seen: dict[str, EvidenceRecord] = {}
|
|
525
|
+
|
|
526
|
+
def _fetch_raw(self, request: str, /, **params: object) -> Iterable[EvidenceRecord]:
|
|
527
|
+
# Slice at the source. `created:<T` is a server-side qualifier, so the
|
|
528
|
+
# newest-first ordering cannot fill the page with issues we must not see.
|
|
529
|
+
day = self.cutoff.date().isoformat()
|
|
530
|
+
nodes = self.transport.search_issues(
|
|
531
|
+
f"repo:{request} is:issue created:<{day}", self.max_pages
|
|
532
|
+
)
|
|
533
|
+
kept = [r for r in project_issues(request, nodes) if r.timestamp <= self.cutoff]
|
|
534
|
+
self._seen.update({r.evidence_id: r for r in kept})
|
|
535
|
+
return kept
|
|
536
|
+
|
|
537
|
+
def _resolve_raw(self, evidence_id: str) -> EvidenceRecord | None:
|
|
538
|
+
return self._seen.get(evidence_id)
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""The single chokepoint every fact passes through.
|
|
2
|
+
|
|
3
|
+
Both GitHub and web results resolve through this interface, in one of two
|
|
4
|
+
implementations: live (real network) or fixture (committed JSON). That buys
|
|
5
|
+
three things at once:
|
|
6
|
+
|
|
7
|
+
* fixture mode makes the eval reproducible with no token and no rate limits
|
|
8
|
+
* live mode is the real product
|
|
9
|
+
* one place asserts the holdout boundary, so contamination is structurally
|
|
10
|
+
impossible rather than a matter of discipline
|
|
11
|
+
|
|
12
|
+
Subclasses implement ``_fetch_raw`` / ``_resolve_raw``. They cannot skip the
|
|
13
|
+
boundary check: the public methods own it.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from abc import ABC, abstractmethod
|
|
19
|
+
from collections.abc import Iterable
|
|
20
|
+
from datetime import datetime
|
|
21
|
+
|
|
22
|
+
from holt.types import T_CUTOFF, EvidenceRecord, Window
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class ContaminationError(AssertionError):
|
|
26
|
+
"""A provider returned a record from the wrong side of the holdout.
|
|
27
|
+
|
|
28
|
+
This is a bug, not a condition to handle. It means the agent was about to
|
|
29
|
+
see post-cutoff data, or a label was about to be computed from pre-cutoff
|
|
30
|
+
data. Either invalidates the measured claim.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class EvidenceProvider(ABC):
|
|
35
|
+
def __init__(self, window: Window, cutoff: datetime = T_CUTOFF) -> None:
|
|
36
|
+
self.window = window
|
|
37
|
+
self.cutoff = cutoff
|
|
38
|
+
# Every access, in order. The trajectory deliverable asks what the agent
|
|
39
|
+
# did and how its tools responded; the model calls are only half of that.
|
|
40
|
+
self.call_log: list[tuple[str, str, int]] = []
|
|
41
|
+
|
|
42
|
+
def fetch(self, request: str, /, **params: object) -> list[EvidenceRecord]:
|
|
43
|
+
records = list(self._fetch_raw(request, **params))
|
|
44
|
+
for record in records:
|
|
45
|
+
self._assert_in_window(record)
|
|
46
|
+
self.call_log.append(("fetch", request, len(records)))
|
|
47
|
+
return records
|
|
48
|
+
|
|
49
|
+
def resolve(self, evidence_id: str) -> EvidenceRecord | None:
|
|
50
|
+
"""Return the record behind an id, or None if it does not resolve.
|
|
51
|
+
|
|
52
|
+
Stage D uses the None case to drop findings rather than soften them.
|
|
53
|
+
"""
|
|
54
|
+
record = self._resolve_raw(evidence_id)
|
|
55
|
+
if record is not None:
|
|
56
|
+
self._assert_in_window(record)
|
|
57
|
+
self.call_log.append(("resolve", evidence_id, 1 if record else 0))
|
|
58
|
+
return record
|
|
59
|
+
|
|
60
|
+
def _assert_in_window(self, record: EvidenceRecord) -> None:
|
|
61
|
+
if self.window is Window.PRE_T and record.timestamp > self.cutoff:
|
|
62
|
+
raise ContaminationError(
|
|
63
|
+
f"{record.evidence_id} is dated {record.timestamp.isoformat()}, "
|
|
64
|
+
f"after the cutoff {self.cutoff.isoformat()}; the agent must not see it"
|
|
65
|
+
)
|
|
66
|
+
if self.window is Window.POST_T and record.timestamp <= self.cutoff:
|
|
67
|
+
raise ContaminationError(
|
|
68
|
+
f"{record.evidence_id} is dated {record.timestamp.isoformat()}, "
|
|
69
|
+
f"at or before the cutoff {self.cutoff.isoformat()}; "
|
|
70
|
+
"labels must not be computed from it"
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
@abstractmethod
|
|
74
|
+
def _fetch_raw(self, request: str, /, **params: object) -> Iterable[EvidenceRecord]: ...
|
|
75
|
+
|
|
76
|
+
@abstractmethod
|
|
77
|
+
def _resolve_raw(self, evidence_id: str) -> EvidenceRecord | None: ...
|
holt/evidence/redact.py
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Strip third-party credentials out of captured evidence before it is committed.
|
|
2
|
+
|
|
3
|
+
Public GitHub issues contain leaked API keys, in bug reports and in secret-scanner
|
|
4
|
+
test cases alike. Crawling them is fine; **redistributing them in a submitted
|
|
5
|
+
artifact is not**, whether or not they are still live. So the scrub runs at
|
|
6
|
+
capture time and every fixture in this repository has been through it.
|
|
7
|
+
|
|
8
|
+
Two tiers, because one is not enough:
|
|
9
|
+
|
|
10
|
+
1. Any string in a recognised credential format is replaced wherever it appears.
|
|
11
|
+
2. In a record where tier 1 fired, long opaque runs are replaced too. One captured
|
|
12
|
+
issue printed a token backwards next to the real one — a format matcher will
|
|
13
|
+
never catch that, and a record already known to be discussing a live secret is
|
|
14
|
+
the right place to be blunt about it.
|
|
15
|
+
|
|
16
|
+
Tier 2 deliberately does not run repository-wide: commit hashes and base64 blobs
|
|
17
|
+
are legitimate evidence, and destroying them everywhere to catch one obfuscated
|
|
18
|
+
token would cost more than it buys.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import re
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
MARKER = "[REDACTED-CREDENTIAL]"
|
|
27
|
+
|
|
28
|
+
# Prefixed formats only. A pattern loose enough to catch unprefixed secrets is
|
|
29
|
+
# loose enough to shred ordinary evidence.
|
|
30
|
+
CREDENTIAL_PATTERNS = [
|
|
31
|
+
re.compile(r"gh[pousr]_[A-Za-z0-9]{36,255}"),
|
|
32
|
+
re.compile(r"github_pat_[A-Za-z0-9_]{22,255}"),
|
|
33
|
+
re.compile(r"sk-proj-[A-Za-z0-9_-]{20,}"),
|
|
34
|
+
re.compile(r"sk-ant-[A-Za-z0-9_-]{20,}"),
|
|
35
|
+
re.compile(r"sk-[A-Za-z0-9]{32,}"),
|
|
36
|
+
re.compile(r"xox[baprs]-[A-Za-z0-9-]{10,}"),
|
|
37
|
+
re.compile(r"AKIA[0-9A-Z]{16}"),
|
|
38
|
+
re.compile(r"AIza[0-9A-Za-z_-]{35}"),
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
# Tier 2, inside an already-flagged record only.
|
|
42
|
+
OPAQUE_RUN = re.compile(r"\b[A-Za-z0-9]{30,}\b")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _scrub(text: str, patterns) -> tuple[str, int]:
|
|
46
|
+
hits = 0
|
|
47
|
+
for pattern in patterns:
|
|
48
|
+
text, n = pattern.subn(MARKER, text)
|
|
49
|
+
hits += n
|
|
50
|
+
return text, hits
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _walk(value: Any, patterns) -> tuple[Any, int]:
|
|
54
|
+
if isinstance(value, str):
|
|
55
|
+
return _scrub(value, patterns)
|
|
56
|
+
if isinstance(value, dict):
|
|
57
|
+
out, total = {}, 0
|
|
58
|
+
for k, v in value.items():
|
|
59
|
+
out[k], n = _walk(v, patterns)
|
|
60
|
+
total += n
|
|
61
|
+
return out, total
|
|
62
|
+
if isinstance(value, list):
|
|
63
|
+
out, total = [], 0
|
|
64
|
+
for v in value:
|
|
65
|
+
scrubbed, n = _walk(v, patterns)
|
|
66
|
+
out.append(scrubbed)
|
|
67
|
+
total += n
|
|
68
|
+
return out, total
|
|
69
|
+
return value, 0
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def redact_payload(payload: dict[str, Any]) -> tuple[dict[str, Any], int]:
|
|
73
|
+
"""Return a scrubbed copy and the number of secrets removed."""
|
|
74
|
+
scrubbed, hits = _walk(payload, CREDENTIAL_PATTERNS)
|
|
75
|
+
if hits:
|
|
76
|
+
scrubbed, extra = _walk(scrubbed, [OPAQUE_RUN])
|
|
77
|
+
hits += extra
|
|
78
|
+
scrubbed["redacted"] = True
|
|
79
|
+
return scrubbed, hits
|