holt-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. holt/__init__.py +0 -0
  2. holt/agent/__init__.py +0 -0
  3. holt/agent/entry.py +86 -0
  4. holt/agent/findings.py +49 -0
  5. holt/agent/landing.py +154 -0
  6. holt/agent/pipeline.py +244 -0
  7. holt/agent/progression.py +408 -0
  8. holt/agent/signals.py +220 -0
  9. holt/agent/stages.py +533 -0
  10. holt/agent/verdict.py +226 -0
  11. holt/agent/verify.py +140 -0
  12. holt/baseline.py +89 -0
  13. holt/baseline_matched.py +116 -0
  14. holt/cli.py +616 -0
  15. holt/discover.py +497 -0
  16. holt/evidence/__init__.py +3 -0
  17. holt/evidence/fixtures.py +154 -0
  18. holt/evidence/github_graphql.py +538 -0
  19. holt/evidence/provider.py +77 -0
  20. holt/evidence/redact.py +79 -0
  21. holt/issues.py +41 -0
  22. holt/model.py +516 -0
  23. holt/profile.py +126 -0
  24. holt/report.py +157 -0
  25. holt/tui/__init__.py +0 -0
  26. holt/tui/animation.py +84 -0
  27. holt/tui/app.py +294 -0
  28. holt/tui/clipboard.py +89 -0
  29. holt/tui/commands.py +134 -0
  30. holt/tui/discovery.py +305 -0
  31. holt/tui/env.py +49 -0
  32. holt/tui/events.py +245 -0
  33. holt/tui/mascot.py +121 -0
  34. holt/tui/models.py +590 -0
  35. holt/tui/observe.py +297 -0
  36. holt/tui/screens/__init__.py +35 -0
  37. holt/tui/screens/assessment.py +337 -0
  38. holt/tui/screens/confirm.py +62 -0
  39. holt/tui/screens/discover.py +444 -0
  40. holt/tui/screens/home.py +519 -0
  41. holt/tui/screens/inspector.py +106 -0
  42. holt/tui/screens/live.py +335 -0
  43. holt/tui/screens/models.py +393 -0
  44. holt/tui/screens/next_steps.py +425 -0
  45. holt/tui/screens/profile.py +129 -0
  46. holt/tui/session.py +711 -0
  47. holt/tui/store.py +458 -0
  48. holt/tui/theme.py +479 -0
  49. holt/tui/visual.py +33 -0
  50. holt/tui/widgets/__init__.py +0 -0
  51. holt/tui/widgets/candidates.py +78 -0
  52. holt/tui/widgets/claims.py +59 -0
  53. holt/tui/widgets/disclosure.py +121 -0
  54. holt/tui/widgets/evidence.py +121 -0
  55. holt/tui/widgets/masthead.py +122 -0
  56. holt/tui/widgets/recent.py +167 -0
  57. holt/tui/widgets/scrolling.py +38 -0
  58. holt/tui/widgets/stages.py +232 -0
  59. holt/types.py +48 -0
  60. holt_cli-0.1.0.dist-info/METADATA +198 -0
  61. holt_cli-0.1.0.dist-info/RECORD +65 -0
  62. holt_cli-0.1.0.dist-info/WHEEL +4 -0
  63. holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
  64. holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
  65. holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
@@ -0,0 +1,408 @@
1
+ """Personalised contribution discovery: what should *this* person do next here?
2
+
3
+ The ranking is arithmetic. The model does not order anything.
4
+
5
+ That separation is the point rather than an implementation detail. `find_paths`,
6
+ the prototype this replaces, put the whole ranking inside one prompt, so when it
7
+ failed to beat GitHub's `good first issue` label there was no way to see *why* --
8
+ whether it was ignoring contributor history, weighting the wrong thing, or simply
9
+ guessing. Here every rank is a weighted sum of named features, `explain()` prints
10
+ the vector that produced it, and the model appears in exactly two places that
11
+ cannot move a result:
12
+
13
+ * `profile()` -- one call per contributor, turning their merged pull requests
14
+ and the review feedback on them into a competence profile. It feeds exactly
15
+ one feature term out of eight.
16
+ * `describe()` -- prose for the top few, after the order is fixed.
17
+
18
+ `holt_arith` (no model at all) against `holt_full` (with the profile term) is
19
+ therefore a clean measurement of whether the model adds anything over arithmetic.
20
+
21
+ Weights are fixed in `eval/PREREGISTRATION-3.md`, written before this file
22
+ existed, and are never fitted to the outcome.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import re
28
+ from dataclasses import dataclass, field
29
+
30
+ from holt.model import ModelClient
31
+ from holt.types import EvidenceRecord
32
+
33
+ # Fixed in the pre-registration. Changing one is an experiment, not a tweak.
34
+ WEIGHTS = {
35
+ "file_hit": 3.0,
36
+ "profile_hit": 2.5,
37
+ "dir_hit": 2.0,
38
+ "thread_reviewer": 1.5,
39
+ "lang_hit": 1.0,
40
+ "scope_step": 1.0,
41
+ "actionable": 0.5,
42
+ "discussion": 0.5,
43
+ }
44
+
45
+ # A path-ish token: something with an extension, or something with a slash.
46
+ PATH_TOKEN = re.compile(r"[\w][\w.-]*\.[A-Za-z][A-Za-z0-9]{0,4}\b|[\w][\w.-]*(?:/[\w.-]+)+")
47
+ MAX_ISSUE_TEXT = 4000
48
+
49
+ # Size bands. Declared here rather than tuned: a contributor who has been landing
50
+ # ~20-line changes is not obviously ready for a 2,000-line one, and "one band up"
51
+ # is the step this feature is meant to reward.
52
+ PR_BANDS = (50, 500)
53
+ ISSUE_BANDS = (300, 1200)
54
+
55
+
56
+ def _band(value: int, edges: tuple[int, int]) -> int:
57
+ return sum(1 for edge in edges if value >= edge)
58
+
59
+
60
+ def paths_in(record: EvidenceRecord) -> set[str]:
61
+ text = f"{record.payload.get('title') or ''} {record.payload.get('body') or ''}"
62
+ return set(PATH_TOKEN.findall(text[:MAX_ISSUE_TEXT]))
63
+
64
+
65
+ def _dirs(files: set[str]) -> set[str]:
66
+ return {f.rsplit("/", 1)[0] for f in files if "/" in f}
67
+
68
+
69
+ def _exts(names: set[str]) -> set[str]:
70
+ return {n.rsplit(".", 1)[-1].lower() for n in names if "." in n}
71
+
72
+
73
+ @dataclass
74
+ class Profile:
75
+ """What the contributor has demonstrably worked on. Optional by design."""
76
+
77
+ areas: list[str] = field(default_factory=list)
78
+ skills: list[str] = field(default_factory=list)
79
+ ready_for: str = ""
80
+
81
+ def terms(self) -> set[str]:
82
+ # Short tokens match everything; they would make the feature fire on all
83
+ # issues and quietly become a constant.
84
+ return {t.lower() for t in (*self.areas, *self.skills) if len(t) >= 4}
85
+
86
+
87
+ @dataclass
88
+ class Contributor:
89
+ """Everything the ranker is allowed to know, all of it from before the cutoff."""
90
+
91
+ login: str
92
+ files: set[str]
93
+ median_pr_size: int
94
+ engaged_with: set[str]
95
+ merged_count: int
96
+ profile: Profile | None = None
97
+
98
+
99
+ def features(contributor: Contributor, issue: EvidenceRecord) -> dict[str, float]:
100
+ """The whole ranking signal, as named numbers a reader can check."""
101
+ named = paths_in(issue)
102
+ payload = issue.payload
103
+ text = f"{payload.get('title') or ''} {payload.get('body') or ''}".lower()
104
+
105
+ theirs = contributor.files
106
+ their_dirs = _dirs(theirs)
107
+ their_base = {f.rsplit("/", 1)[-1] for f in theirs}
108
+
109
+ file_hit = any(
110
+ p in theirs or p.rsplit("/", 1)[-1] in their_base or any(p.endswith(f) or f.endswith(p) for f in theirs)
111
+ for p in named
112
+ )
113
+ dir_hit = any(
114
+ any(d and (p.startswith(d + "/") or d in p) for d in their_dirs) for p in named
115
+ )
116
+ lang_hit = bool(_exts(named) & _exts(theirs))
117
+
118
+ issue_band = _band(len(payload.get("body") or ""), ISSUE_BANDS)
119
+ their_band = _band(contributor.median_pr_size, PR_BANDS)
120
+
121
+ profile_hit = 0.0
122
+ if contributor.profile:
123
+ profile_hit = float(any(term in text for term in contributor.profile.terms()))
124
+
125
+ return {
126
+ "file_hit": float(file_hit),
127
+ "profile_hit": profile_hit,
128
+ "dir_hit": float(dir_hit),
129
+ # The pre-registration wrote this as "someone who reviewed their work is
130
+ # active on this issue". Captured issues carry the author but not the
131
+ # commenters, so it is narrowed to the author. Recorded rather than
132
+ # silently redefined.
133
+ "thread_reviewer": float(payload.get("author") in contributor.engaged_with),
134
+ "lang_hit": float(lang_hit),
135
+ "scope_step": float(issue_band in (their_band, their_band + 1)),
136
+ "actionable": float(bool(named)),
137
+ "discussion": min(payload.get("comments", 0), 5) / 5.0,
138
+ }
139
+
140
+
141
+ # Amendment 1. Three of the eight registered features fire on 66-83% of all
142
+ # candidates with lift at or below 1.11 -- they are constants, not features, and
143
+ # together they outweigh `dir_hit` entirely. The drop rule ("fires on >50% of
144
+ # candidates AND lift < 1.15") is a property of a feature's own distribution, was
145
+ # fitted on pool 1 alone, and changes no weight. See `eval/PREREGISTRATION-3.md`.
146
+ NEAR_CONSTANT = ("scope_step", "actionable", "discussion")
147
+ REPAIRED_WEIGHTS = {k: v for k, v in WEIGHTS.items() if k not in NEAR_CONSTANT}
148
+
149
+
150
+ def score(vector: dict[str, float], weights: dict[str, float] | None = None) -> float:
151
+ w = weights if weights is not None else WEIGHTS
152
+ return sum(w[k] * v for k, v in vector.items() if k in w)
153
+
154
+
155
+ def rank(
156
+ contributor: Contributor,
157
+ issues: dict[str, EvidenceRecord],
158
+ weights: dict[str, float] | None = None,
159
+ ) -> list[tuple[str, float, dict[str, float]]]:
160
+ """Best first. Deterministic, no model, ties broken by recency."""
161
+ scored = []
162
+ for key, issue in issues.items():
163
+ vector = features(contributor, issue)
164
+ scored.append((key, score(vector, weights), vector, issue.timestamp))
165
+ scored.sort(key=lambda row: (-row[1], -row[3].timestamp()))
166
+ return [(key, value, vector) for key, value, vector, _ in scored]
167
+
168
+
169
+ def explain(vector: dict[str, float], weights: dict[str, float] | None = None) -> str:
170
+ """Why this ranked where it did — the numbers, not a paragraph about them."""
171
+ w = weights if weights is not None else WEIGHTS
172
+ live = [(k, v) for k, v in vector.items() if v and k in w]
173
+ if not live:
174
+ return "no features fired"
175
+ return " ".join(f"{k}={v:.2g}×{w[k]}" for k, v in live) + f" = {score(vector, w):.2f}"
176
+
177
+
178
+ # What `holt next` may claim about this ranking, verbatim in the output. The
179
+ # elaborate weighted scorer above was cut for failing to beat this rule
180
+ # (hit@10 0.211 against 0.234 across 128 pairs); the rule ships because it is
181
+ # the best of five methods tried, and the interval is printed because it spans
182
+ # zero. Measured by eval/progression_harness.py; per-pair rows in
183
+ # eval/progression_results.json.
184
+ NEXT_MEASUREMENT = (
185
+ "Ranked by one deterministic rule: open issues naming a file or directory "
186
+ "you have already worked on here, newest first, then the rest by recency. "
187
+ "Measured across 128 (repository, contributor) pairs it is the best of "
188
+ "five methods we tried — hit@10 0.234 vs 0.211 for a weighted scorer, "
189
+ "0.188 for recency alone, 0.172 for chance. That is +0.06 over chance, "
190
+ "95% interval [-0.003, +0.132] — an interval that spans zero — and we "
191
+ "found nothing that beats it."
192
+ )
193
+
194
+
195
+ def overlap_tokens(files: set[str], issue: EvidenceRecord) -> set[str]:
196
+ """The path-ish tokens in an issue that name something this person touched.
197
+
198
+ Kept identical to the `overlaps` predicate the harness measured
199
+ (eval/progression_harness.py); shipping a different rule under the measured
200
+ rule's numbers would be the exact overclaim this project exists to avoid.
201
+ """
202
+ named = paths_in(issue)
203
+ dirs = _dirs(files)
204
+ return {
205
+ p for p in named
206
+ if p in files
207
+ or any(p.endswith(f) or f.endswith(p) for f in files)
208
+ or any(d and d in p for d in dirs)
209
+ }
210
+
211
+
212
+ def path_overlap_rank(
213
+ files: set[str], issues: dict[str, EvidenceRecord]
214
+ ) -> list[tuple[str, set[str]]]:
215
+ """Best first: overlapping issues in recency order, then the rest.
216
+
217
+ Returns each key with the tokens that matched, so the renderer can show
218
+ *why* a row is where it is instead of asserting that it belongs there.
219
+ """
220
+ recency = sorted(issues.items(), key=lambda kv: kv[1].timestamp, reverse=True)
221
+ matched = [(k, toks) for k, r in recency if (toks := overlap_tokens(files, r))]
222
+ rest = [(k, set()) for k, r in recency if not overlap_tokens(files, r)]
223
+ return matched + rest
224
+
225
+
226
+ def history_for(login: str, threads) -> Contributor:
227
+ """What this person has demonstrably done here, from pre-cutoff threads."""
228
+ import statistics as _stats
229
+
230
+ merged = [t for t in threads.values() if t.merged and t.author == login]
231
+ files = {f for t in merged for f in t.files}
232
+ sizes = [t.additions + t.deletions for t in merged]
233
+ engaged = {who for t in threads.values() if t.author == login
234
+ for _, who, _ in t.responses if who != login}
235
+ return Contributor(
236
+ login=login,
237
+ files=files,
238
+ median_pr_size=int(_stats.median(sizes)) if sizes else 0,
239
+ engaged_with=engaged,
240
+ merged_count=len(merged),
241
+ )
242
+
243
+
244
+ def render_next(
245
+ repo: str,
246
+ contributor: Contributor,
247
+ ranked: list[tuple[str, set[str]]],
248
+ issues: dict[str, EvidenceRecord],
249
+ top: int = 10,
250
+ ) -> str:
251
+ """The measurement is emitted here, in the only path that prints the
252
+ ranking, so no caller can show the order without the number that says how
253
+ well it works."""
254
+ lines = [f"# What to look at next in {repo} — for `{contributor.login}`", ""]
255
+ lines += [
256
+ f"You have {contributor.merged_count} merged pull request"
257
+ f"{'s' if contributor.merged_count != 1 else ''} here, touching "
258
+ f"{len(contributor.files)} file{'s' if len(contributor.files) != 1 else ''}.",
259
+ "",
260
+ NEXT_MEASUREMENT,
261
+ "",
262
+ ]
263
+ for key, tokens in ranked[:top]:
264
+ issue = issues[key]
265
+ title = issue.payload.get("title") or "(untitled)"
266
+ lines.append(f"- **{title}** — `{issue.evidence_id}`")
267
+ if tokens:
268
+ shown = ", ".join(f"`{t}`" for t in sorted(tokens)[:4])
269
+ lines.append(f" names {shown} — work you have already touched")
270
+ else:
271
+ lines.append(" no overlap with your history; ranked by recency only")
272
+ return "\n".join(lines).rstrip() + "\n"
273
+
274
+
275
+ PROFILE_SYSTEM = """You are reading one contributor's merged pull requests in a
276
+ single repository, together with what reviewers said to them.
277
+
278
+ Describe what this person has **demonstrably** worked on. Not what they might be
279
+ good at, and not a compliment: only what the merged work and the review feedback
280
+ actually show.
281
+
282
+ * areas: parts of the project they have touched. Use the vocabulary of the
283
+ repository itself -- directory names, subsystem names, feature names.
284
+ * skills: what kind of work they did. "packaging", "test fixtures",
285
+ "documentation", "API endpoints", "build configuration".
286
+ * ready_for: one sentence on the next step up in scope that their history
287
+ supports. Be specific and be conservative; if the history is thin, say so.
288
+
289
+ If two merged pull requests are all you have, say what those two show and nothing
290
+ more."""
291
+
292
+ PROFILE_SCHEMA = {
293
+ "type": "object",
294
+ "properties": {
295
+ "areas": {"type": "array", "items": {"type": "string"}},
296
+ "skills": {"type": "array", "items": {"type": "string"}},
297
+ "ready_for": {"type": "string"},
298
+ },
299
+ "required": ["areas", "skills", "ready_for"],
300
+ "additionalProperties": False,
301
+ }
302
+
303
+ MAX_PRS_SHOWN = 12
304
+
305
+
306
+ def profile(repo: str, login: str, merged_prs: list[dict], model: ModelClient) -> Profile:
307
+ """One call. Feeds one feature term out of eight; cannot reorder anything."""
308
+ if not merged_prs:
309
+ return Profile()
310
+ # Title breaks size ties: without it the order depends on how the caller
311
+ # happened to iterate, and the same contributor yields a different prompt on
312
+ # a different run.
313
+ shown = sorted(
314
+ merged_prs,
315
+ key=lambda p: (-((p.get("additions") or 0) + (p.get("deletions") or 0)),
316
+ p.get("title") or ""),
317
+ )[:MAX_PRS_SHOWN]
318
+ lines = []
319
+ for pr in shown:
320
+ files = pr.get("files") or []
321
+ lines.append(
322
+ f"--- {pr.get('title')}\n"
323
+ f" {pr.get('changed_files', len(files))} files, "
324
+ f"+{pr.get('additions', 0)}/-{pr.get('deletions', 0)}\n"
325
+ f" touched: {', '.join(files[:12]) or '(file list unavailable)'}\n"
326
+ f" reviewers said: {'; '.join(pr.get('_responders') or []) or '(nobody replied)'}"
327
+ )
328
+ result = model.complete(
329
+ label="profile",
330
+ system=PROFILE_SYSTEM,
331
+ prompt=f"Repository: {repo}\nContributor: {login}\n\n"
332
+ f"Their merged pull requests ({len(shown)} of {len(merged_prs)}):\n\n"
333
+ + "\n".join(lines),
334
+ schema=PROFILE_SCHEMA,
335
+ )
336
+ return Profile(result["areas"], result["skills"], result["ready_for"])
337
+
338
+
339
+ DESCRIBE_SYSTEM = """You are told which open issues a contributor should look at
340
+ next, and why each one scored where it did. **The order is already decided and you
341
+ must not change it.**
342
+
343
+ For each, write one sentence: what they would concretely do, and how it builds on
344
+ what they have already merged here. Refer to their actual past work.
345
+
346
+ Do not say an issue is easy. Do not promise it will be merged. If the connection
347
+ to their history is weak, say that instead of inventing one."""
348
+
349
+ DESCRIBE_SCHEMA = {
350
+ "type": "object",
351
+ "properties": {
352
+ "steps": {
353
+ "type": "array",
354
+ "items": {
355
+ "type": "object",
356
+ "properties": {
357
+ "evidence_id": {"type": "string"},
358
+ "next_step": {"type": "string"},
359
+ },
360
+ "required": ["evidence_id", "next_step"],
361
+ "additionalProperties": False,
362
+ },
363
+ }
364
+ },
365
+ "required": ["steps"],
366
+ "additionalProperties": False,
367
+ }
368
+
369
+
370
+ def describe(
371
+ repo: str,
372
+ contributor: Contributor,
373
+ ranked: list[tuple[str, float, dict[str, float]]],
374
+ issues: dict[str, EvidenceRecord],
375
+ model: ModelClient,
376
+ limit: int = 5,
377
+ ) -> list[dict]:
378
+ """Prose for an order that is already fixed."""
379
+ top = ranked[:limit]
380
+ if not top:
381
+ return []
382
+ blocks = []
383
+ for key, _, vector in top:
384
+ issue = issues[key]
385
+ body = " ".join((issue.payload.get("body") or "").split())[:500]
386
+ blocks.append(
387
+ f"--- evidence id: {issue.evidence_id}\n"
388
+ f" title: {issue.payload.get('title')}\n"
389
+ f" why it ranked here: {explain(vector)}\n"
390
+ f" {body or '(no description)'}"
391
+ )
392
+ past = sorted(contributor.files)[:20]
393
+ prof = contributor.profile
394
+ return model.complete(
395
+ label="describe",
396
+ system=DESCRIBE_SYSTEM,
397
+ prompt=(
398
+ f"Repository: {repo}\nContributor: {contributor.login}\n"
399
+ f"They have merged {contributor.merged_count} pull request(s) here, "
400
+ f"median size {contributor.median_pr_size} lines.\n"
401
+ f"Files they have touched: {', '.join(past) or '(unknown)'}\n"
402
+ + (f"Demonstrated areas: {', '.join(prof.areas)}\n"
403
+ f"Demonstrated skills: {', '.join(prof.skills)}\n" if prof else "")
404
+ + "\nThe issues, in the order they must stay in:\n\n"
405
+ + "\n".join(blocks)
406
+ ),
407
+ schema=DESCRIBE_SCHEMA,
408
+ )["steps"]
holt/agent/signals.py ADDED
@@ -0,0 +1,220 @@
1
+ """Pre-cutoff signals computed from evidence, with no model involved.
2
+
3
+ The split matters. Counting merges and measuring how long a maintainer took to
4
+ reply are arithmetic; asking a language model to do them adds cost, variance and
5
+ a chance of being wrong about a number that was sitting right there. What the
6
+ model is for is judgement -- what kind of project is this, what does the tone of
7
+ that thread mean -- which is the part arithmetic cannot reach.
8
+
9
+ Note what is deliberately absent: this module has no diff-shape rules. L1 has
10
+ those, and if the agent shared them it would agree with the label by
11
+ construction on the dimension that matters most. The agent judges what a
12
+ contribution *was* by reading, not by re-running the grader's arithmetic.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import statistics
18
+ from collections.abc import Iterable
19
+ from dataclasses import dataclass, field
20
+
21
+ from holt.types import EvidenceRecord
22
+
23
+
24
+ def pr_key(evidence_id: str) -> str:
25
+ return ":".join(evidence_id.split(":")[:2])
26
+
27
+
28
+ # GitHub only marks an account as a Bot when it is a real GitHub App. Plenty of
29
+ # automation runs on ordinary user accounts -- wingetbot on microsoft/winget-pkgs
30
+ # posts every validation log as a normal user -- and counting those as human
31
+ # engagement turns an auto-merge pipeline into a conversational project. Applied
32
+ # at read time so fixtures stay as captured.
33
+ _BOT_HINTS = ("dependabot", "renovate", "greenkeeper", "imgbot", "allcontributors",
34
+ "codecov", "sonarcloud", "netlify", "vercel", "mergify", "stale")
35
+
36
+
37
+ def looks_like_bot(login: str, flagged: bool = False) -> bool:
38
+ if flagged:
39
+ return True
40
+ low = (login or "").lower()
41
+ if low.endswith("[bot]") or low.endswith("bot") or "-bot" in low:
42
+ return True
43
+ return any(hint in low for hint in _BOT_HINTS)
44
+
45
+
46
+ @dataclass(slots=True)
47
+ class Thread:
48
+ """One pull request and everything that happened on it before the cutoff."""
49
+
50
+ key: str
51
+ number: int
52
+ author: str
53
+ author_is_bot: bool
54
+ opened_at: object
55
+ files: list[str] = field(default_factory=list)
56
+ changed_files: int = 0
57
+ additions: int = 0
58
+ deletions: int = 0
59
+ merged: bool = False
60
+ closed_unmerged: bool = False
61
+ responses: list[tuple[object, str, str]] = field(default_factory=list)
62
+
63
+ @property
64
+ def first_response_hours(self) -> float | None:
65
+ """Hours until someone other than the author first said anything."""
66
+ others = [t for t, who, _ in self.responses if who != self.author]
67
+ if not others:
68
+ return None
69
+ return (min(others) - self.opened_at).total_seconds() / 3600
70
+
71
+ @property
72
+ def engaged(self) -> bool:
73
+ return any(who != self.author for _, who, _ in self.responses)
74
+
75
+
76
+ def build_threads(records: Iterable[EvidenceRecord]) -> dict[str, Thread]:
77
+ threads: dict[str, Thread] = {}
78
+ records = list(records)
79
+
80
+ for r in records:
81
+ if not r.evidence_id.endswith(":opened"):
82
+ continue
83
+ p = r.payload
84
+ key = pr_key(r.evidence_id)
85
+ threads[key] = Thread(
86
+ key=key,
87
+ number=int(key.split("#")[-1]),
88
+ author=p.get("author", ""),
89
+ author_is_bot=looks_like_bot(p.get("author", ""), bool(p.get("author_is_bot"))),
90
+ opened_at=r.timestamp,
91
+ files=list(p.get("files") or []),
92
+ changed_files=p.get("changed_files") or 0,
93
+ additions=p.get("additions") or 0,
94
+ deletions=p.get("deletions") or 0,
95
+ )
96
+
97
+ for r in records:
98
+ key = pr_key(r.evidence_id)
99
+ thread = threads.get(key)
100
+ if thread is None:
101
+ continue
102
+ if r.evidence_id.endswith(":merged"):
103
+ thread.merged = True
104
+ elif r.evidence_id.endswith(":closed"):
105
+ thread.closed_unmerged = True
106
+ elif ":review:" in r.evidence_id or ":comment:" in r.evidence_id:
107
+ if not looks_like_bot(
108
+ r.payload.get("author", ""), bool(r.payload.get("author_is_bot"))
109
+ ):
110
+ thread.responses.append(
111
+ (r.timestamp, r.payload.get("author", ""), r.payload.get("body") or "")
112
+ )
113
+ return threads
114
+
115
+
116
+ def newcomer_threads(threads: dict[str, Thread]) -> list[Thread]:
117
+ """Threads opened by someone who had not yet landed anything here.
118
+
119
+ The obvious definition -- an outsider is anyone without a merged pull
120
+ request -- is circular inside a single window: merging is what stops you
121
+ being an outsider, so "outsider merges" is always zero. That bug produced a
122
+ column of zeros across the whole pool before it was caught.
123
+
124
+ Asked properly, the question is per-thread and time-ordered: at the moment
125
+ this pull request was opened, had its author ever landed anything here
126
+ before? That is also the question a newcomer actually has, which is the
127
+ point of the project.
128
+ """
129
+ merged_opens: dict[str, list] = {}
130
+ for t in threads.values():
131
+ if t.merged and not t.author_is_bot:
132
+ merged_opens.setdefault(t.author, []).append(t.opened_at)
133
+
134
+ return [
135
+ t
136
+ for t in threads.values()
137
+ if not t.author_is_bot
138
+ and not any(earlier < t.opened_at for earlier in merged_opens.get(t.author, []))
139
+ ]
140
+
141
+
142
+ @dataclass(slots=True)
143
+ class Signals:
144
+ total_threads: int
145
+ outsider_threads: int
146
+ outsider_merged: int
147
+ outsider_ignored: int
148
+ median_first_response_hours: float | None
149
+ bot_share: float
150
+ distinct_outsider_authors: int
151
+ distinct_merged_authors: int
152
+ # Share of merged threads where somebody other than the author, and not a
153
+ # bot, said anything at all. Mechanical, over every merge -- not the
154
+ # twelve-thread model-judged sample that the first rejection rule used and
155
+ # which is the likeliest reason that attempt failed.
156
+ reviewed_share: float | None
157
+ merge_rate: float | None
158
+ # The shape of a merged contribution: how many files it touched, and how
159
+ # many top-level directories those spanned, at the median. A catalogue entry
160
+ # is one file in one place almost by definition; a change to running
161
+ # software is not. These exist so a claim *about* that shape -- `registry`,
162
+ # `awesome_list` -- can be checked against it rather than trusted.
163
+ merged_files_median: float | None = None
164
+ merged_dirs_median: float | None = None
165
+ merged_with_files: int = 0
166
+
167
+ def as_dict(self) -> dict:
168
+ return {
169
+ "total_threads": self.total_threads,
170
+ "outsider_threads": self.outsider_threads,
171
+ "outsider_merged": self.outsider_merged,
172
+ "outsider_ignored": self.outsider_ignored,
173
+ "median_first_response_hours": self.median_first_response_hours,
174
+ "bot_share": round(self.bot_share, 3),
175
+ "distinct_outsider_authors": self.distinct_outsider_authors,
176
+ "distinct_merged_authors": self.distinct_merged_authors,
177
+ "reviewed_share": self.reviewed_share,
178
+ "merge_rate": self.merge_rate,
179
+ "merged_files_median": self.merged_files_median,
180
+ "merged_dirs_median": self.merged_dirs_median,
181
+ "merged_with_files": self.merged_with_files,
182
+ }
183
+
184
+
185
+ def compute(threads: dict[str, Thread]) -> Signals:
186
+ outsiders = newcomer_threads(threads)
187
+ merged_threads = [t for t in threads.values() if t.merged]
188
+ # Every merge with a file list, not only the outsiders': what a merged
189
+ # contribution *is* here is a property of the repository, and narrowing it
190
+ # to newcomers would measure it on a handful of threads in a repository that
191
+ # merges hundreds.
192
+ shaped = [t for t in merged_threads if t.files]
193
+ file_counts = [len(t.files) for t in shaped]
194
+ dir_counts = [len({f.split("/")[0] for f in t.files}) for t in shaped]
195
+ latencies = [h for t in outsiders if (h := t.first_response_hours) is not None]
196
+ bots = sum(1 for t in threads.values() if t.author_is_bot)
197
+
198
+ return Signals(
199
+ total_threads=len(threads),
200
+ outsider_threads=len(outsiders),
201
+ outsider_merged=sum(1 for t in outsiders if t.merged),
202
+ outsider_ignored=sum(1 for t in outsiders if not t.engaged and not t.merged),
203
+ median_first_response_hours=round(statistics.median(latencies), 1) if latencies else None,
204
+ bot_share=(bots / len(threads)) if threads else 0.0,
205
+ distinct_outsider_authors=len({t.author for t in outsiders}),
206
+ # People who actually landed something, as distinct from people who
207
+ # tried. Conflating the two produced a user-visible falsehood: a repo
208
+ # with 15 merges and 72 first-time attempters was reported as "15
209
+ # merges from 72 people".
210
+ distinct_merged_authors=len({t.author for t in outsiders if t.merged}),
211
+ reviewed_share=(
212
+ sum(1 for t in merged_threads if t.engaged) / len(merged_threads)
213
+ if merged_threads else None
214
+ ),
215
+ merge_rate=(len(outsiders) and sum(1 for t in outsiders if t.merged) / len(outsiders))
216
+ or (None if not outsiders else 0.0),
217
+ merged_files_median=statistics.median(file_counts) if file_counts else None,
218
+ merged_dirs_median=statistics.median(dir_counts) if dir_counts else None,
219
+ merged_with_files=len(shaped),
220
+ )