holt-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. holt/__init__.py +0 -0
  2. holt/agent/__init__.py +0 -0
  3. holt/agent/entry.py +86 -0
  4. holt/agent/findings.py +49 -0
  5. holt/agent/landing.py +154 -0
  6. holt/agent/pipeline.py +244 -0
  7. holt/agent/progression.py +408 -0
  8. holt/agent/signals.py +220 -0
  9. holt/agent/stages.py +533 -0
  10. holt/agent/verdict.py +226 -0
  11. holt/agent/verify.py +140 -0
  12. holt/baseline.py +89 -0
  13. holt/baseline_matched.py +116 -0
  14. holt/cli.py +616 -0
  15. holt/discover.py +497 -0
  16. holt/evidence/__init__.py +3 -0
  17. holt/evidence/fixtures.py +154 -0
  18. holt/evidence/github_graphql.py +538 -0
  19. holt/evidence/provider.py +77 -0
  20. holt/evidence/redact.py +79 -0
  21. holt/issues.py +41 -0
  22. holt/model.py +516 -0
  23. holt/profile.py +126 -0
  24. holt/report.py +157 -0
  25. holt/tui/__init__.py +0 -0
  26. holt/tui/animation.py +84 -0
  27. holt/tui/app.py +294 -0
  28. holt/tui/clipboard.py +89 -0
  29. holt/tui/commands.py +134 -0
  30. holt/tui/discovery.py +305 -0
  31. holt/tui/env.py +49 -0
  32. holt/tui/events.py +245 -0
  33. holt/tui/mascot.py +121 -0
  34. holt/tui/models.py +590 -0
  35. holt/tui/observe.py +297 -0
  36. holt/tui/screens/__init__.py +35 -0
  37. holt/tui/screens/assessment.py +337 -0
  38. holt/tui/screens/confirm.py +62 -0
  39. holt/tui/screens/discover.py +444 -0
  40. holt/tui/screens/home.py +519 -0
  41. holt/tui/screens/inspector.py +106 -0
  42. holt/tui/screens/live.py +335 -0
  43. holt/tui/screens/models.py +393 -0
  44. holt/tui/screens/next_steps.py +425 -0
  45. holt/tui/screens/profile.py +129 -0
  46. holt/tui/session.py +711 -0
  47. holt/tui/store.py +458 -0
  48. holt/tui/theme.py +479 -0
  49. holt/tui/visual.py +33 -0
  50. holt/tui/widgets/__init__.py +0 -0
  51. holt/tui/widgets/candidates.py +78 -0
  52. holt/tui/widgets/claims.py +59 -0
  53. holt/tui/widgets/disclosure.py +121 -0
  54. holt/tui/widgets/evidence.py +121 -0
  55. holt/tui/widgets/masthead.py +122 -0
  56. holt/tui/widgets/recent.py +167 -0
  57. holt/tui/widgets/scrolling.py +38 -0
  58. holt/tui/widgets/stages.py +232 -0
  59. holt/types.py +48 -0
  60. holt_cli-0.1.0.dist-info/METADATA +198 -0
  61. holt_cli-0.1.0.dist-info/RECORD +65 -0
  62. holt_cli-0.1.0.dist-info/WHEEL +4 -0
  63. holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
  64. holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
  65. holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/__init__.py ADDED
File without changes
holt/agent/__init__.py ADDED
File without changes
holt/agent/entry.py ADDED
@@ -0,0 +1,86 @@
1
+ """Entry points: which open issue an outsider should attempt first.
2
+
3
+ **Not shipped. Kept as the prototype that showed what not to build.**
4
+
5
+ It was designed, pre-registered and then measured against the comparators that
6
+ could make it unnecessary (`eval/PATHFINDER-DESIGN.md`, written before any of it
7
+ existed). It did not beat them. The measurement is a module constant here, not a
8
+ line in a document the reader will never open, because a ranking whose evaluation
9
+ lives only in the README is an unsupported ranking with the caveat filed where
10
+ nobody looks.
11
+
12
+ **Why it failed is more useful than the fact that it did.** This ranker never
13
+ sees the contributor. It produces one ranking for everybody, which means it is
14
+ answering "which issues here are generally approachable" -- exactly what a
15
+ `good first issue` label already encodes. It did not lose because the model is
16
+ weak. It lost because it was solving the label's problem, and a tie with the
17
+ label is the expected outcome of that.
18
+
19
+ Available behind `holt analyze --entry-points`, off by default, and it still
20
+ prints its own measurement when asked for. The successor question -- which issue
21
+ is a sensible next step *for this person, given what they have already merged
22
+ here* -- is a different question, and one no label answers.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ from holt.agent.signals import build_threads, compute
28
+ from holt.agent.stages import find_paths
29
+ from holt.issues import open_at_cutoff
30
+ from holt.model import ModelClient
31
+ from holt.types import EvidenceRecord
32
+
33
+ # The signals the ranker is shown. Kept as a constant because the evaluation and
34
+ # the shipped path must build byte-identical prompts, or the recorded trajectories
35
+ # stop replaying and the published number stops describing what users run.
36
+ RANKER_SIGNALS = ("outsider_merged", "outsider_threads", "median_first_response_hours")
37
+
38
+ #: Measured on both pools. Regenerate with:
39
+ #: uv run python eval/pathfinder_harness.py --replay
40
+ #: uv run python eval/pathfinder_harness.py --replay --pool eval/pool2.json \
41
+ #: --labels eval/results_labels_pool2.json
42
+ MEASURED = {
43
+ "repositories": 25,
44
+ "issues_ranked": 3613,
45
+ "precision_at_3": {"holt": 0.173, "good_first_issue": 0.187, "recency": 0.160, "random": 0.151},
46
+ "paired_ci_vs_label": (-0.133, 0.120),
47
+ "sign_test_p": 0.51,
48
+ "repos_with_no_labelled_issue": 13,
49
+ }
50
+
51
+ DISCLAIMER = (
52
+ "> **This ranking is not measurably better than picking at random.** Measured\n"
53
+ "> over 25 repositories and 3,613 issues held out before the cutoff:\n"
54
+ "> precision@3 was **0.173** for this ranking, **0.187** for GitHub's\n"
55
+ "> `good first issue` label and **0.151** for a random pick — differences well\n"
56
+ "> inside noise (paired 95% CI [−0.13, +0.12], sign test p = 0.51).\n"
57
+ ">\n"
58
+ "> It is printed anyway because **13 of those 25 repositories had no\n"
59
+ "> beginner-labelled issue at all**, so on half of them there is no free signal\n"
60
+ "> to lose to. Read it as a reading order, not a recommendation.\n"
61
+ ">\n"
62
+ "> Check it yourself: `uv run python eval/pathfinder_harness.py --replay`"
63
+ )
64
+
65
+
66
+ def rank(
67
+ repo: str,
68
+ issue_records: list[EvidenceRecord],
69
+ pull_records: list[EvidenceRecord],
70
+ model: ModelClient,
71
+ ) -> list[dict]:
72
+ """Rank the issues open at the cutoff. `[]` when there are none.
73
+
74
+ Called by both `eval/pathfinder_harness.py` and the CLI, so the ranking a
75
+ user sees is produced by the same code path as the ranking that was scored.
76
+ """
77
+ candidates = open_at_cutoff(issue_records)
78
+ if not candidates:
79
+ return []
80
+ signals = compute(build_threads(pull_records)).as_dict()
81
+ return find_paths(
82
+ repo,
83
+ list(candidates.values()),
84
+ {k: signals[k] for k in RANKER_SIGNALS},
85
+ model,
86
+ )
holt/agent/findings.py ADDED
@@ -0,0 +1,49 @@
1
+ """Typed findings: what the stages produce, before anything becomes a verdict."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from typing import Any
7
+
8
+
9
+ @dataclass(frozen=True, slots=True)
10
+ class Finding:
11
+ """One field, its value, and the evidence that supports it.
12
+
13
+ A finding with no resolvable evidence is dropped by Stage D rather than
14
+ softened, so `evidence_ids` is what keeps a claim alive.
15
+ """
16
+
17
+ field: str
18
+ value: Any
19
+ evidence_ids: tuple[str, ...] = ()
20
+ note: str = ""
21
+
22
+
23
+ @dataclass(slots=True)
24
+ class Findings:
25
+ items: list[Finding] = field(default_factory=list)
26
+
27
+ def add(self, field_name: str, value: Any, evidence_ids=(), note: str = "") -> None:
28
+ self.items.append(Finding(field_name, value, tuple(evidence_ids), note))
29
+
30
+ def drop(self, field_name: str) -> None:
31
+ """Remove a field entirely.
32
+
33
+ Used where two sources of evidence disagree: the standing rule is to
34
+ drop the contested field rather than pick a side, and a field that is
35
+ dropped must not survive in the claim list either.
36
+ """
37
+ self.items = [i for i in self.items if i.field != field_name]
38
+
39
+ def get(self, field_name: str, default: Any = None) -> Any:
40
+ for item in self.items:
41
+ if item.field == field_name:
42
+ return item.value
43
+ return default
44
+
45
+ def __iter__(self):
46
+ return iter(self.items)
47
+
48
+ def __len__(self) -> int:
49
+ return len(self.items)
holt/agent/landing.py ADDED
@@ -0,0 +1,154 @@
1
+ """Where outsider work actually landed, and where it never did.
2
+
3
+ Every pull request Holt reads carries its file list. Until now that list was used
4
+ for exactly one thing -- deciding whether a diff was substantial enough to count
5
+ -- and then discarded. This counts it instead.
6
+
7
+ The output is the sentence a newcomer most needs and cannot get anywhere:
8
+ *in a tree of two hundred thousand files, these three directories are where
9
+ strangers' work has actually been merged, and these are where strangers tried and
10
+ never succeeded.* GitHub does not show it, CONTRIBUTING does not say it, and it is
11
+ not inferable from stars, issue labels or commit frequency.
12
+
13
+ **It ranks nothing and predicts nothing.** After five capabilities cut for losing
14
+ to a cheap comparator, that is deliberate: there is no ordering to be beaten here,
15
+ only a count of what happened. Arithmetic over evidence already in hand, no model
16
+ call, and nothing that could disagree with the verdict.
17
+
18
+ Read `attempted but never landed` carefully -- it is a description of this
19
+ sample, not a prohibition. A directory can appear there because outsiders are
20
+ turned away from it, or because only two people ever tried.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from collections import Counter
26
+ from dataclasses import dataclass
27
+
28
+ from holt.agent.signals import Thread, newcomer_threads
29
+
30
+ # Two path segments. One is too coarse to act on in a monorepo (`pkgs`, `src`);
31
+ # three splits the same area into a dozen near-identical rows.
32
+ DEPTH = 2
33
+ MIN_ATTEMPTS = 2
34
+ TOP_N = 4
35
+
36
+ # When two segments barely group anything -- a plugin registry where every entry
37
+ # is its own directory produces ninety areas from a hundred pull requests -- the
38
+ # rows read as insight while carrying none. Above this ratio of areas to pull
39
+ # requests, fall back to one segment, which for that registry correctly collapses
40
+ # to a single honest row: everything lands in `plugins`.
41
+ REGROUP_ABOVE = 0.5
42
+
43
+
44
+ @dataclass(frozen=True, slots=True)
45
+ class Area:
46
+ path: str
47
+ landed: int
48
+ attempted: int
49
+
50
+ @property
51
+ def rate(self) -> float:
52
+ return self.landed / self.attempted if self.attempted else 0.0
53
+
54
+
55
+ @dataclass(frozen=True, slots=True)
56
+ class Landing:
57
+ """Where outsiders got in, and where they did not."""
58
+
59
+ landed: list[Area]
60
+ never: list[Area]
61
+ outsider_threads: int
62
+ outsider_merges: int
63
+ depth: int = DEPTH
64
+
65
+ def __bool__(self) -> bool:
66
+ return bool(self.landed or self.never)
67
+
68
+
69
+ def area_of(path: str, depth: int = DEPTH) -> str:
70
+ """`pkgs/by-name/fo/foo/package.nix` -> `pkgs/by-name`. A root file -> `(root)`."""
71
+ parts = path.split("/")
72
+ if len(parts) == 1:
73
+ return "(root)"
74
+ return "/".join(parts[:depth])
75
+
76
+
77
+ def _tally(outsiders: list[Thread], depth: int) -> tuple[Counter, Counter]:
78
+ landed: Counter = Counter()
79
+ attempted: Counter = Counter()
80
+ for thread in outsiders:
81
+ # Count each area once per pull request. A change touching forty files in
82
+ # one directory is one attempt at that directory, not forty.
83
+ for area in sorted({area_of(f, depth) for f in (thread.files or [])}):
84
+ attempted[area] += 1
85
+ if thread.merged:
86
+ landed[area] += 1
87
+ return landed, attempted
88
+
89
+
90
+ def compute(threads: dict[str, Thread]) -> Landing:
91
+ outsiders = newcomer_threads(threads)
92
+ depth = DEPTH
93
+ landed, attempted = _tally(outsiders, depth)
94
+ if outsiders and len(attempted) > REGROUP_ABOVE * len(outsiders):
95
+ depth = 1
96
+ landed, attempted = _tally(outsiders, depth)
97
+
98
+ # `Counter.most_common` breaks ties by insertion order, and insertion order
99
+ # here was set-iteration order, which varies with the process hash seed. Two
100
+ # replays of the same recording printed different fourth rows -- a report
101
+ # this project claims is a function of the evidence quietly was not. Ordered
102
+ # explicitly instead: most merges first, then the better odds, then the
103
+ # larger sample, then alphabetically, so the order is total.
104
+ got_in = [
105
+ Area(a, landed[a], attempted[a])
106
+ for a in sorted(
107
+ landed,
108
+ key=lambda a: (-landed[a], -landed[a] / attempted[a], -attempted[a], a),
109
+ )[:TOP_N]
110
+ ]
111
+ never = [
112
+ Area(a, 0, attempted[a])
113
+ for a in sorted(attempted, key=lambda a: (-attempted[a], a))
114
+ if landed.get(a, 0) == 0 and attempted[a] >= MIN_ATTEMPTS
115
+ ][:TOP_N]
116
+
117
+ return Landing(
118
+ landed=got_in,
119
+ never=never,
120
+ outsider_threads=len(outsiders),
121
+ outsider_merges=sum(1 for t in outsiders if t.merged),
122
+ depth=depth,
123
+ )
124
+
125
+
126
+ def render(landing: Landing) -> list[str]:
127
+ """Markdown lines, or nothing at all when there is nothing to say."""
128
+ if not landing:
129
+ return []
130
+ lines = ["## Where outsider work landed", ""]
131
+ lines.append(
132
+ f"Counted over the {landing.outsider_threads} pull requests opened here by "
133
+ f"people with no prior merge, of which {landing.outsider_merges} were merged. "
134
+ + ("Paths are cut to their first two segments, so a short path may name a "
135
+ "file rather than a directory." if landing.depth == 2 else
136
+ "Paths are cut to their first segment: at two, almost every pull request "
137
+ "here landed in a directory of its own, which groups nothing.")
138
+ )
139
+ lines.append("")
140
+ for area in landing.landed:
141
+ lines.append(
142
+ f"- **`{area.path}`** — {area.landed} merged "
143
+ f"of {area.attempted} attempted ({area.rate:.0%})"
144
+ )
145
+ if landing.never:
146
+ lines.append("")
147
+ joined = ", ".join(f"`{a.path}` ({a.attempted})" for a in landing.never)
148
+ lines.append(
149
+ f"Outsiders attempted these and none were merged: {joined}. "
150
+ f"That is what this sample shows, not a rule — a directory can appear "
151
+ f"here because newcomers are turned away from it, or because only a "
152
+ f"couple of people ever tried."
153
+ )
154
+ return lines
holt/agent/pipeline.py ADDED
@@ -0,0 +1,244 @@
1
+ """A -> B -> C -> D -> verdict -> E.
2
+
3
+ The ordering that matters: the verdict is computed *before* narration and handed
4
+ to Stage E as an input it cannot alter. If the report and verdict.py could
5
+ disagree, the determinism claim would be worth nothing.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass, field
11
+ from datetime import datetime
12
+
13
+ from holt.agent import landing, stages
14
+ from holt.agent.findings import Finding, Findings
15
+ from holt.agent.signals import Signals, build_threads, compute
16
+ from holt.agent.verdict import classify as decide
17
+ from holt.agent.verdict import contested_kind
18
+ from holt.agent.verify import check_quotes, verify
19
+ from holt.evidence.provider import EvidenceProvider
20
+ from holt.model import ModelClient
21
+ from holt.report import VERDICT_HEADLINES, Assessment, Claim, Verdict
22
+
23
+ MAX_CLAIM_CHARS = 240
24
+ MAX_QUOTE_CHARS = 180
25
+
26
+
27
+ def clip(text: str, limit: int) -> str:
28
+ """Cut on a word boundary. Cutting mid-word reads as a bug, because it is one."""
29
+ text = " ".join(text.split())
30
+ if len(text) <= limit:
31
+ return text
32
+ head = text[:limit]
33
+ cut = max(head.rfind(" "), head.rfind(". "))
34
+ return (head[:cut] if cut > limit // 2 else head).rstrip(" ,;:.") + "…"
35
+
36
+
37
+ @dataclass(slots=True)
38
+ class Trace:
39
+ """What happened, for the demo and the trajectory record."""
40
+
41
+ signals: Signals
42
+ before_verification: int = 0
43
+ after_verification: int = 0
44
+ dropped: list[Finding] = field(default_factory=list)
45
+ # Findings whose id resolved but whose quotation is not in the record.
46
+ invented: list[Finding] = field(default_factory=list)
47
+ rules: list[str] = field(default_factory=list)
48
+
49
+
50
+ def analyze(
51
+ repo: str,
52
+ provider: EvidenceProvider,
53
+ model: ModelClient | None,
54
+ contributor_days: int = 7,
55
+ as_of: datetime | None = None,
56
+ ) -> tuple[Assessment, Trace]:
57
+ if model is None:
58
+ return analyze_without_model(repo, provider, contributor_days, as_of)
59
+ records = provider.fetch(repo)
60
+ threads = build_threads(records)
61
+ signals = compute(threads)
62
+
63
+ findings = Findings()
64
+ stages.classify(repo, records, threads, model, findings)
65
+ stages.assess_opportunity(repo, records, model, findings)
66
+ stages.read_outcomes(repo, threads, model, findings)
67
+
68
+ before = len(findings)
69
+ findings, dropped = verify(findings, provider)
70
+
71
+ # `repo_kind` is the only model-derived field that can decide the answer by
72
+ # itself, and Stage D cannot check it -- an id resolving says nothing about
73
+ # whether a classification is true. Where the evidence contradicts the
74
+ # reason the kind rule would give, the field is dropped before it decides
75
+ # anything and the disagreement is printed. See eval/PREREGISTRATION-4.md.
76
+ meta = next((r for r in records if r.evidence_id.endswith(":meta")), None)
77
+ contested = contested_kind(findings, signals, meta.payload if meta else None)
78
+ if contested:
79
+ findings.drop("repo_kind")
80
+
81
+ verdict, rules = decide(findings, signals, contributor_days)
82
+ if contested:
83
+ rules.insert(0, contested)
84
+ # The narration prompt is deliberately held to the signal fields that existed
85
+ # when the trajectories were recorded. New signals reach the *verdict*
86
+ # immediately but only reach the prose on the next re-record, so adding one
87
+ # does not invalidate every committed trajectory and break replay for a judge.
88
+ # When a new signal changes the outcome it still reaches the narrator, via
89
+ # the rule trace.
90
+ narrated_signals = {
91
+ k: v for k, v in signals.as_dict().items()
92
+ if k not in ("reviewed_share", "merge_rate", "merged_files_median",
93
+ "merged_dirs_median", "merged_with_files")
94
+ }
95
+ narrated = stages.narrate(
96
+ repo, verdict.value, rules, findings, narrated_signals, model
97
+ )
98
+
99
+ # The evidence list is built from verified findings, not written by the
100
+ # model. Stage E supplies prose; it cannot introduce a citation.
101
+ #
102
+ # The quote check runs here rather than inside Stage D on purpose. A claim
103
+ # whose id does not resolve is worthless to everyone, narrator included, so
104
+ # `verify` removes it before anything else runs. A claim whose id resolves
105
+ # but whose words are not in the record is a different failure: the thread
106
+ # is real and the outcome may well be right, and what must not reach the
107
+ # reader is the quotation. Filtering the claim list is exactly that, and it
108
+ # leaves the narration prompt byte-identical, so every committed trajectory
109
+ # still replays -- a guarantee that would otherwise cost a re-record of the
110
+ # frozen benchmark to buy.
111
+ quoting, invented = check_quotes(findings, records)
112
+ claims: list[Claim] = []
113
+ for item in quoting:
114
+ if item.field == "thread_outcome":
115
+ outcome = item.value["outcome"].replace("_", " ")
116
+ quote = (item.value.get("quote") or "").strip()
117
+ # An empty quote used to render as a pair of quotation marks with
118
+ # nothing between them, which reads as a bug because it is one.
119
+ text = (f"{outcome} — “{clip(quote, MAX_QUOTE_CHARS)}”" if quote
120
+ else f"{outcome}, nothing said")
121
+ else:
122
+ text = f"{item.field.replace('_', ' ')}: {item.value}" + (
123
+ f" — {clip(item.note, MAX_CLAIM_CHARS)}" if item.note else "")
124
+ claims.append(Claim(text=text, evidence_id=item.evidence_ids[0]))
125
+
126
+ assessment = Assessment(
127
+ repo=repo,
128
+ verdict=verdict,
129
+ summary=narrated["what_the_evidence_shows"],
130
+ bottom_line=narrated["bottom_line"],
131
+ limits=narrated["what_could_not_be_determined"],
132
+ rules=list(rules),
133
+ contributor_days=contributor_days,
134
+ as_of=as_of,
135
+ landing=landing.render(landing.compute(threads)),
136
+ claims=claims,
137
+ method="holt (A classify, B opportunity, C outcomes, D verify, deterministic verdict, E narrate)",
138
+ replayed=model.replayed,
139
+ models=list(model.usage.models),
140
+ dropped_claims=len(dropped) + len(invented),
141
+ )
142
+ return assessment, Trace(
143
+ signals=signals,
144
+ before_verification=before,
145
+ after_verification=len(findings),
146
+ dropped=dropped,
147
+ invented=invented,
148
+ rules=rules,
149
+ )
150
+
151
+
152
+ # --- degraded mode -----------------------------------------------------------
153
+ #
154
+ # The project measured its own model stages at +0.01 MCC over the arithmetic
155
+ # (Iteration 22). A finding that large about your own architecture should change
156
+ # the architecture, not just the write-up: if the rules decide the verdict, the
157
+ # verdict must be obtainable without a model, and the reader must be told what
158
+ # they lost. That is this function.
159
+ #
160
+ # It is not a second implementation of the verdict. It calls the same `decide`
161
+ # on the same `Signals`, so the two modes cannot disagree about a repository
162
+ # they both have findings for -- a test asserts exactly that over the pool.
163
+ # What it does not do is write: Stages A, B, C and E never run, so there are no
164
+ # thread quotes, no narration, and no `repo_kind`. That absence is the point of
165
+ # `eval/evidence_integrity.py`'s yield column, and it is stated on the report
166
+ # rather than left for the reader to notice.
167
+
168
+ NO_MODEL_METHOD = (
169
+ "holt --no-model (deterministic verdict from arithmetic; "
170
+ "stages A, B, C and E did not run)"
171
+ )
172
+
173
+
174
+ def analyze_without_model(
175
+ repo: str,
176
+ provider: EvidenceProvider,
177
+ contributor_days: int = 7,
178
+ as_of: datetime | None = None,
179
+ ) -> tuple[Assessment, Trace]:
180
+ """The verdict, with no model call anywhere and the cost of that printed."""
181
+ records = provider.fetch(repo)
182
+ threads = build_threads(records)
183
+ signals = compute(threads)
184
+
185
+ findings = Findings()
186
+ # `is_archived` is a structured GitHub field. Stage A was asking a model to
187
+ # read a boolean the provider already had, which is the clearest single
188
+ # illustration of why the model stages measured +0.01: some of what they
189
+ # were doing did not need a model at all. Here it is taken from the record
190
+ # and cited to it.
191
+ meta = next((r for r in records if r.evidence_id.endswith(":meta")), None)
192
+ if meta is not None and meta.payload.get("is_archived"):
193
+ findings.add("is_archived", True, (meta.evidence_id,),
194
+ "GitHub reports this repository as archived")
195
+
196
+ verdict, rules = decide(findings, signals, contributor_days)
197
+
198
+ s = signals.as_dict()
199
+ if signals.outsider_threads:
200
+ summary = (
201
+ f"{s['outsider_merged']} of {s['outsider_threads']} outsider pull "
202
+ f"requests merged, by {s['distinct_merged_authors']} distinct people "
203
+ f"out of {s['distinct_outsider_authors']} who tried"
204
+ )
205
+ if s["median_first_response_hours"] is not None:
206
+ summary += f"; median first response {s['median_first_response_hours']}h"
207
+ summary += (
208
+ f"; {s['outsider_ignored']} drew no response at all. "
209
+ "Counted from the pull request record, not judged."
210
+ )
211
+ else:
212
+ summary = (
213
+ "No outsider pull requests in the period read, so the arithmetic has "
214
+ "nothing to count."
215
+ )
216
+
217
+ return Assessment(
218
+ repo=repo,
219
+ verdict=verdict,
220
+ summary=summary,
221
+ bottom_line=f"{VERDICT_HEADLINES[verdict]}. " + (rules[0] if rules else ""),
222
+ limits=(
223
+ "No model ran. This report is the verdict and the rule that produced "
224
+ "it; the parts a model writes — what specific threads said, who was "
225
+ "welcoming, what kind of project this is, and the prose explaining "
226
+ "any of it — are absent, not merely brief. Measured: this mode scores "
227
+ "MCC +0.60 against the full pipeline's +0.61 in sample, but +0.55 "
228
+ "against +0.63 out of sample, and writes 0 citable statements against "
229
+ "its 11.8 (eval/evidence_integrity.py). Run without --no-model for a "
230
+ "report you can check against the record."
231
+ ),
232
+ rules=list(rules),
233
+ contributor_days=contributor_days,
234
+ as_of=as_of,
235
+ landing=landing.render(landing.compute(threads)),
236
+ claims=[
237
+ Claim(text=f"{i.field.replace('_', ' ')}: {i.value}", evidence_id=i.evidence_ids[0])
238
+ for i in findings
239
+ ],
240
+ method=NO_MODEL_METHOD,
241
+ replayed=False,
242
+ models=[],
243
+ dropped_claims=0,
244
+ ), Trace(signals=signals, rules=rules)