holt-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. holt/__init__.py +0 -0
  2. holt/agent/__init__.py +0 -0
  3. holt/agent/entry.py +86 -0
  4. holt/agent/findings.py +49 -0
  5. holt/agent/landing.py +154 -0
  6. holt/agent/pipeline.py +244 -0
  7. holt/agent/progression.py +408 -0
  8. holt/agent/signals.py +220 -0
  9. holt/agent/stages.py +533 -0
  10. holt/agent/verdict.py +226 -0
  11. holt/agent/verify.py +140 -0
  12. holt/baseline.py +89 -0
  13. holt/baseline_matched.py +116 -0
  14. holt/cli.py +616 -0
  15. holt/discover.py +497 -0
  16. holt/evidence/__init__.py +3 -0
  17. holt/evidence/fixtures.py +154 -0
  18. holt/evidence/github_graphql.py +538 -0
  19. holt/evidence/provider.py +77 -0
  20. holt/evidence/redact.py +79 -0
  21. holt/issues.py +41 -0
  22. holt/model.py +516 -0
  23. holt/profile.py +126 -0
  24. holt/report.py +157 -0
  25. holt/tui/__init__.py +0 -0
  26. holt/tui/animation.py +84 -0
  27. holt/tui/app.py +294 -0
  28. holt/tui/clipboard.py +89 -0
  29. holt/tui/commands.py +134 -0
  30. holt/tui/discovery.py +305 -0
  31. holt/tui/env.py +49 -0
  32. holt/tui/events.py +245 -0
  33. holt/tui/mascot.py +121 -0
  34. holt/tui/models.py +590 -0
  35. holt/tui/observe.py +297 -0
  36. holt/tui/screens/__init__.py +35 -0
  37. holt/tui/screens/assessment.py +337 -0
  38. holt/tui/screens/confirm.py +62 -0
  39. holt/tui/screens/discover.py +444 -0
  40. holt/tui/screens/home.py +519 -0
  41. holt/tui/screens/inspector.py +106 -0
  42. holt/tui/screens/live.py +335 -0
  43. holt/tui/screens/models.py +393 -0
  44. holt/tui/screens/next_steps.py +425 -0
  45. holt/tui/screens/profile.py +129 -0
  46. holt/tui/session.py +711 -0
  47. holt/tui/store.py +458 -0
  48. holt/tui/theme.py +479 -0
  49. holt/tui/visual.py +33 -0
  50. holt/tui/widgets/__init__.py +0 -0
  51. holt/tui/widgets/candidates.py +78 -0
  52. holt/tui/widgets/claims.py +59 -0
  53. holt/tui/widgets/disclosure.py +121 -0
  54. holt/tui/widgets/evidence.py +121 -0
  55. holt/tui/widgets/masthead.py +122 -0
  56. holt/tui/widgets/recent.py +167 -0
  57. holt/tui/widgets/scrolling.py +38 -0
  58. holt/tui/widgets/stages.py +232 -0
  59. holt/types.py +48 -0
  60. holt_cli-0.1.0.dist-info/METADATA +198 -0
  61. holt_cli-0.1.0.dist-info/RECORD +65 -0
  62. holt_cli-0.1.0.dist-info/WHEEL +4 -0
  63. holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
  64. holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
  65. holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/agent/verdict.py ADDED
@@ -0,0 +1,226 @@
1
+ """The only path from findings to a verdict, and it is a plain function.
2
+
3
+ No model runs here. Three reasons, all of which matter:
4
+
5
+ * a judge who reruns Holt gets our numbers, not a resample of them
6
+ * scoring across a pool is not polluted by model variance
7
+ * "is this just a wrapper around a prompt?" is answered by a file rather than
8
+ an argument
9
+
10
+ Stage E writes prose *around* this decision and is handed the result as an input
11
+ it cannot change. A test asserts the rendered report and this function agree.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from holt.agent.findings import Findings
17
+ from holt.agent.signals import Signals
18
+ from holt.report import Verdict
19
+
20
+ # Kinds where a merged pull request is not a software contribution. Landing work
21
+ # in these is easy and means nothing for the question being asked.
22
+ NON_SOFTWARE_KINDS = {"registry", "awesome_list", "portfolio", "course_material"}
23
+
24
+ # Kinds where outside contribution is not accepted regardless of activity.
25
+ CLOSED_KINDS = {"mirror"}
26
+
27
+ # --- contesting the one field that can decide alone -------------------------
28
+ #
29
+ # `repo_kind` is the only model-derived input either of the rules above reads,
30
+ # and Stage D cannot check it: it verifies that a cited id resolves, and a
31
+ # classification is not a quotation. A model that answers `mirror` while citing
32
+ # a real README has made a claim that verifies perfectly and is false -- which
33
+ # is exactly what happened to `pytorch/pytorch` under a 3B local model, and to
34
+ # `aden-hive/hive`, which was called a registry.
35
+ #
36
+ # So the two kinds that can flip a verdict are checked against the evidence they
37
+ # implicitly claim something about. What is contested is the rule's stated
38
+ # *reason*, never what the repository "really is": a catalogue entry is one file
39
+ # in one place, and a mirror does not merge outsiders' pull requests. Where the
40
+ # evidence disagrees the field is dropped rather than overridden -- no verdict
41
+ # is asserted in its place, `classify` falls through to the arithmetic, and the
42
+ # disagreement is printed for the reader.
43
+ #
44
+ # Pre-registered with its predictions in eval/PREREGISTRATION-4.md; thresholds
45
+ # chosen on pool 1 and that fitting disclosed there.
46
+ CATALOGUE_KINDS = {"registry", "awesome_list"}
47
+
48
+ # `portfolio` and `course_material` are deliberately not contested this way.
49
+ # Their reason is about whose project it is, not the shape of a diff, and a
50
+ # portfolio being real code is not a contradiction.
51
+ MIN_MERGES_FOR_SHAPE = 5
52
+
53
+ # One directory. Not "few enough files": that criterion was pre-registered,
54
+ # failed out-of-sample, and is gone. `microsoft/winget-pkgs` is a real registry
55
+ # whose every entry is three YAML manifests -- installer, locale, version -- in
56
+ # one package directory, so a median-files test called a correct classification
57
+ # a hallucination on all four of its recordings. What survived is the criterion
58
+ # that did not misfire: a catalogue entry lands in one place, whatever it
59
+ # weighs. The narrowing was chosen after seeing that failure and is disclosed as
60
+ # such in eval/PREREGISTRATION-4.md; it has no untouched holdout behind it.
61
+ CATALOGUE_DIRS_MAX = 2
62
+
63
+ # How long the contributor has. Everything time-shaped scales from this, because
64
+ # "is this repository worth my time" has no answer independent of how much time
65
+ # you have: a maintainer who replies in five days is fine if you have three
66
+ # months and useless if you have three days.
67
+ DEFAULT_CONTRIBUTOR_DAYS = 7
68
+
69
+ # Rubber-stamp rejection. Validated out-of-sample on pool 2 -- specificity 0.58
70
+ # to 0.83 with all three pre-registered predictions holding. Thresholds were
71
+ # chosen on pool 1 and that fitting is disclosed in eval/PREREGISTRATION-2.md.
72
+ #
73
+ # Both halves are required. Landing easily alone describes a welcoming project;
74
+ # going unreviewed alone describes a project whose review happens elsewhere,
75
+ # which is what killed the first rejection rule when nixpkgs was withheld. It is
76
+ # the conjunction that describes work being waved through unread.
77
+ RUBBER_STAMP_REVIEWED_MAX = 0.20
78
+ RUBBER_STAMP_MERGE_RATE_MIN = 0.60
79
+
80
+ # One merge from one person is an anecdote; two people is a pattern.
81
+ MIN_MERGES = 2
82
+ MIN_DISTINCT_AUTHORS = 2
83
+
84
+ # Below this share of ignored attempts, silence is noise rather than a policy.
85
+ IGNORED_SHARE = 0.7
86
+
87
+ # ...and below this many attempts there is no share worth speaking of. Four
88
+ # ignored pull requests out of four is not evidence of hostility, it is four
89
+ # data points. tensorflow/tensorflow reaches exactly that shape: 97% of its pull
90
+ # request traffic is automation, leaving a handful of outsider threads. Without
91
+ # this guard the rule turns a thin sample into a confident accusation.
92
+ MIN_ATTEMPTS_FOR_HOSTILE = 8
93
+
94
+
95
+ def contested_kind(
96
+ findings: Findings, signals: Signals, meta: dict | None = None
97
+ ) -> str | None:
98
+ """Why the claimed `repo_kind` disagrees with the evidence, or None.
99
+
100
+ Returns the sentence a reader should see, not a boolean: a field being
101
+ dropped is a thing that happened to their report and it is printed.
102
+ """
103
+ kind = findings.get("repo_kind")
104
+
105
+ if kind in CATALOGUE_KINDS:
106
+ if (
107
+ signals.merged_with_files >= MIN_MERGES_FOR_SHAPE
108
+ and signals.merged_dirs_median is not None
109
+ and signals.merged_dirs_median >= CATALOGUE_DIRS_MAX
110
+ ):
111
+ return (
112
+ f"repo_kind={kind} was claimed, but merged work here spans a "
113
+ f"median of {signals.merged_dirs_median:g} top-level directories "
114
+ "rather than landing in one place, which is not a catalogue "
115
+ "entry; the field is dropped and decided nothing"
116
+ )
117
+
118
+ if kind in CLOSED_KINDS:
119
+ # `is_mirror` alone would not be enough -- GitHub sets it only for
120
+ # repositories created as mirrors, so a genuine mirror can report
121
+ # false. The merges are what disprove the claim being made.
122
+ if (
123
+ (meta or {}).get("is_mirror") is False
124
+ and signals.outsider_merged >= MIN_MERGES
125
+ and signals.distinct_merged_authors >= MIN_DISTINCT_AUTHORS
126
+ ):
127
+ return (
128
+ f"repo_kind={kind} was claimed, but GitHub does not report this "
129
+ f"repository as a mirror and {signals.outsider_merged} outside "
130
+ f"pull requests by {signals.distinct_merged_authors} people were "
131
+ "merged in the period read; the field is dropped and decided nothing"
132
+ )
133
+
134
+ return None
135
+
136
+
137
+ def classify(
138
+ findings: Findings,
139
+ signals: Signals,
140
+ contributor_days: int = DEFAULT_CONTRIBUTOR_DAYS,
141
+ ) -> tuple[Verdict, list[str]]:
142
+ """Return a verdict and the rule trace that produced it.
143
+
144
+ `contributor_days` is the time the person actually has. Re-running this
145
+ function with a different budget costs nothing and calls no model, because
146
+ the findings are already computed -- which is a thing a single prompt cannot
147
+ do without paying for the whole assessment again.
148
+ """
149
+ trace: list[str] = []
150
+ slow_response_hours = contributor_days * 24.0
151
+ kind = findings.get("repo_kind")
152
+
153
+ if findings.get("is_archived"):
154
+ trace.append("archived: no longer accepting work")
155
+ return Verdict.NOT_VIABLE, trace
156
+
157
+ if kind in CLOSED_KINDS:
158
+ trace.append(f"repo_kind={kind}: outside pull requests are not the contribution path")
159
+ return Verdict.NOT_VIABLE, trace
160
+
161
+ if kind in NON_SOFTWARE_KINDS:
162
+ trace.append(f"repo_kind={kind}: merged work here is not a software contribution")
163
+ return Verdict.NOT_VIABLE, trace
164
+
165
+ if signals.outsider_threads == 0:
166
+ # "In the period read", not "before the cutoff": the cutoff is an
167
+ # evaluation device, and this line is printed verbatim to users.
168
+ trace.append("no outsider attempts in the period read: nothing to judge from")
169
+ return Verdict.INSUFFICIENT_EVIDENCE, trace
170
+
171
+ ignored_share = signals.outsider_ignored / signals.outsider_threads
172
+ if (
173
+ signals.outsider_merged == 0
174
+ and ignored_share > IGNORED_SHARE
175
+ and signals.outsider_threads >= MIN_ATTEMPTS_FOR_HOSTILE
176
+ ):
177
+ trace.append(
178
+ f"{signals.outsider_ignored}/{signals.outsider_threads} outsider attempts "
179
+ "drew no response and none merged"
180
+ )
181
+ return Verdict.NOT_VIABLE, trace
182
+
183
+ slow = (
184
+ signals.median_first_response_hours is not None
185
+ and signals.median_first_response_hours > slow_response_hours
186
+ )
187
+ if (
188
+ signals.outsider_merged >= MIN_MERGES
189
+ and signals.distinct_outsider_authors >= MIN_DISTINCT_AUTHORS
190
+ and not slow
191
+ ):
192
+ trace.append(
193
+ f"{signals.outsider_merged} first-time merges by "
194
+ f"{signals.distinct_merged_authors} distinct people, out of "
195
+ f"{signals.outsider_threads} attempts by "
196
+ f"{signals.distinct_outsider_authors}; median first response "
197
+ f"{signals.median_first_response_hours}h"
198
+ )
199
+ if (
200
+ signals.reviewed_share is not None
201
+ and signals.merge_rate is not None
202
+ and signals.reviewed_share < RUBBER_STAMP_REVIEWED_MAX
203
+ and signals.merge_rate > RUBBER_STAMP_MERGE_RATE_MIN
204
+ ):
205
+ trace.append(
206
+ f"but only {signals.reviewed_share:.0%} of merges drew any human "
207
+ f"reply while {signals.merge_rate:.0%} of attempts landed: work is "
208
+ "being waved through unread, so a contribution here buys no review"
209
+ )
210
+ return Verdict.NOT_VIABLE, trace
211
+ return Verdict.VIABLE, trace
212
+
213
+ if slow:
214
+ trace.append(
215
+ f"median first response {signals.median_first_response_hours}h "
216
+ f"exceeds the {slow_response_hours:.0f}h a {contributor_days}-day "
217
+ "budget allows"
218
+ )
219
+ if signals.outsider_merged == 0 and ignored_share > IGNORED_SHARE:
220
+ trace.append(
221
+ f"{signals.outsider_ignored}/{signals.outsider_threads} attempts ignored, "
222
+ f"but fewer than {MIN_ATTEMPTS_FOR_HOSTILE} attempts is too thin to call hostile"
223
+ )
224
+ elif signals.outsider_merged < MIN_MERGES:
225
+ trace.append(f"only {signals.outsider_merged} outsider merges in the period read")
226
+ return Verdict.INSUFFICIENT_EVIDENCE, trace
holt/agent/verify.py ADDED
@@ -0,0 +1,140 @@
1
+ """Stage D — the only stage that removes things.
2
+
3
+ Two mechanical checks, both of which delete rather than soften.
4
+
5
+ **Resolution.** Every finding carries the evidence ids that support it. This
6
+ resolves each one against the provider. A finding whose evidence does not
7
+ resolve is dropped: the alternative is prose that hedges around a claim nobody
8
+ can check, which is how unsupported statements survive into reports.
9
+
10
+ **Quotation.** A claim can cite a pull request that exists and still put words
11
+ in its mouth. `eval/evidence_integrity.py` has measured that gap since Iteration
12
+ 11 -- it is the difference between an id that resolves and evidence that says
13
+ what the claim says -- and the check here is the same function the metric uses,
14
+ so the number the eval reports and the guarantee the reader gets cannot drift
15
+ apart. A quote that is not in the record takes its claim with it, for the same
16
+ reason: keeping the outcome while deleting the fabricated quote is softening.
17
+
18
+ No model runs here. Resolution is a lookup and quotation is string matching.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import re
24
+ from collections.abc import Iterable
25
+
26
+ from holt.agent.findings import Finding, Findings
27
+ from holt.evidence.provider import EvidenceProvider
28
+ from holt.types import EvidenceRecord
29
+
30
+ # A quote counts as present if a long-enough run of its words appears verbatim
31
+ # in the record. Models normalise whitespace and clip mid-sentence; penalising
32
+ # that would measure formatting rather than fidelity.
33
+ SHINGLE = 6
34
+
35
+ # `stages._render_thread` prefixes every reply with the speaker -- `[AUTHOR]`,
36
+ # `[octocat]` -- and a model quoting that reply routinely copies the prefix.
37
+ # Nobody said those words: they are ours. Stripping the tag before matching
38
+ # stops the guard blaming the model for our own formatting, which is the same
39
+ # lesson NO_REPLIES taught. What is left still has to be in the record, and a
40
+ # "quote" that is nothing *but* our scaffolding quotes nothing at all.
41
+ SPEAKER_TAG = re.compile(r"^\s*\[[^\]]{1,40}\]\s*")
42
+
43
+
44
+ def verify(findings: Findings, provider: EvidenceProvider) -> tuple[Findings, list[Finding]]:
45
+ """Return (surviving findings, dropped findings)."""
46
+ kept, dropped = Findings(), []
47
+ for item in findings:
48
+ if not item.evidence_ids:
49
+ dropped.append(item)
50
+ continue
51
+ resolved = tuple(e for e in item.evidence_ids if provider.resolve(e) is not None)
52
+ if not resolved:
53
+ dropped.append(item)
54
+ continue
55
+ kept.add(item.field, item.value, resolved, item.note)
56
+ return kept, dropped
57
+
58
+
59
+ # Punctuation is folded away with the whitespace, for the same reason. A model
60
+ # that quotes ``[`nixpkgs-review`]``(...) as ``[`nixpkgs-review`].`` -- closing a
61
+ # markdown link it truncated -- has quoted the thread; rejecting it would be
62
+ # measuring punctuation. What survives folding is the words, and those still
63
+ # have to be the record's words, in the record's order.
64
+ _NON_WORD = re.compile(r"[^\w]+", re.UNICODE)
65
+
66
+
67
+ def normalise(text: str) -> str:
68
+ return " ".join(_NON_WORD.sub(" ", (text or "").lower()).split())
69
+
70
+
71
+ def spoken_part(quote: str) -> str:
72
+ """The words attributed to a person, with our speaker tag removed."""
73
+ return SPEAKER_TAG.sub("", quote or "", count=1).strip()
74
+
75
+
76
+ def quote_supported(quote: str, haystack: str) -> bool:
77
+ q, h = normalise(quote), normalise(haystack)
78
+ if not q:
79
+ return False
80
+ words = q.split()
81
+ if len(words) <= SHINGLE:
82
+ return q in h
83
+ return any(
84
+ " ".join(words[i : i + SHINGLE]) in h for i in range(len(words) - SHINGLE + 1)
85
+ )
86
+
87
+
88
+ def spoken_words(records: Iterable[EvidenceRecord]) -> dict[str, str]:
89
+ """Everything said anywhere on each pull request, keyed by its number.
90
+
91
+ Quotes come from reviews and comments, which live in their own records
92
+ (`#12:review:0`, `#12:comment:1`) rather than in the `:opened` record a
93
+ Stage C finding cites. Checking a quote against the cited record alone would
94
+ reject every real quotation, so the haystack is the whole thread.
95
+ """
96
+ said: dict[str, list[str]] = {}
97
+ for record in records:
98
+ number = _pr_number(record.evidence_id)
99
+ if number is None:
100
+ continue
101
+ payload = record.payload
102
+ text = f"{payload.get('title') or ''} {payload.get('body') or ''}"
103
+ if text.strip():
104
+ said.setdefault(number, []).append(text)
105
+ return {k: " ".join(v) for k, v in said.items()}
106
+
107
+
108
+ def _pr_number(evidence_id: str) -> str | None:
109
+ if "#" not in evidence_id:
110
+ return None
111
+ return evidence_id.split("#")[-1].split(":")[0]
112
+
113
+
114
+ def check_quotes(
115
+ findings: Findings, records: Iterable[EvidenceRecord]
116
+ ) -> tuple[list[Finding], list[Finding]]:
117
+ """Split findings into (quoting the record, putting words in its mouth).
118
+
119
+ Findings that quote nothing pass through untouched -- there is nothing to
120
+ check and nothing to accuse them of.
121
+ """
122
+ said = spoken_words(records)
123
+ supported, invented = [], []
124
+ for item in findings:
125
+ raw = item.value.get("quote", "") if isinstance(item.value, dict) else ""
126
+ if not raw or not str(raw).strip():
127
+ supported.append(item)
128
+ continue
129
+ quote = spoken_part(str(raw))
130
+ if not quote:
131
+ # Entirely our own scaffolding. There is no quotation here to check,
132
+ # and printing `“[octocat]”` to a reader is not evidence of anything.
133
+ invented.append(item)
134
+ continue
135
+ number = _pr_number(item.evidence_ids[0]) if item.evidence_ids else None
136
+ if number is not None and quote_supported(quote, said.get(number, "")):
137
+ supported.append(item)
138
+ else:
139
+ invented.append(item)
140
+ return supported, invented
holt/baseline.py ADDED
@@ -0,0 +1,89 @@
1
+ """The baseline solution: one prompt, the README, and the repository metadata.
2
+
3
+ This is a *solution*, not a comparator -- it has its own entry point
4
+ (`holt analyze --baseline`) and produces the same Assessment the full pipeline
5
+ does. It is what a competent engineer would build in an afternoon, and it is the
6
+ thing the rest of the system has to beat.
7
+
8
+ It reads through the same EvidenceProvider, so it is bound by the same holdout
9
+ and runs in the same fixture and replay modes. Same task, same cases, same
10
+ evidence, different method.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from holt.evidence.provider import EvidenceProvider
16
+ from holt.model import ModelClient
17
+ from holt.report import Assessment, Claim, Verdict
18
+
19
+ SYSTEM = """You assess whether a GitHub repository is a worthwhile place for an \
20
+ outside developer -- someone with no prior connection to the project -- to spend \
21
+ a week contributing.
22
+
23
+ Answer with one of three verdicts:
24
+ viable an outsider could realistically land a meaningful change
25
+ not_viable an outsider could not, or the work would not be software
26
+ insufficient_evidence the material does not support a call either way
27
+
28
+ Prefer insufficient_evidence to a confident guess."""
29
+
30
+ SCHEMA = {
31
+ "type": "object",
32
+ "properties": {
33
+ "verdict": {"type": "string", "enum": [v.value for v in Verdict]},
34
+ "summary": {"type": "string"},
35
+ "reasons": {"type": "array", "items": {"type": "string"}},
36
+ },
37
+ "required": ["verdict", "summary", "reasons"],
38
+ "additionalProperties": False,
39
+ }
40
+
41
+
42
+ def _prompt(repo: str, meta: dict, readme: str | None) -> str:
43
+ parts = [f"Repository: {repo}", "", "Metadata:"]
44
+ for key in (
45
+ "description",
46
+ "primary_language",
47
+ "stargazer_count",
48
+ "pushed_at",
49
+ "is_archived",
50
+ "is_fork",
51
+ "is_mirror",
52
+ "homepage_url",
53
+ ):
54
+ parts.append(f" {key}: {meta.get(key)!r}")
55
+ parts += ["", "README:", readme or "(no README found at the cutoff)"]
56
+ return "\n".join(parts)
57
+
58
+
59
+ def assess(repo: str, provider: EvidenceProvider, model: ModelClient) -> Assessment:
60
+ records = provider.fetch(repo)
61
+ meta = next(
62
+ (r.payload for r in records if r.evidence_id.endswith(":meta")),
63
+ {},
64
+ )
65
+ readme = next(
66
+ (r.payload.get("text") for r in records if r.evidence_id.endswith(":readme")),
67
+ None,
68
+ )
69
+
70
+ result = model.complete(
71
+ label="baseline",
72
+ system=SYSTEM,
73
+ prompt=_prompt(repo, meta, readme),
74
+ schema=SCHEMA,
75
+ )
76
+
77
+ # The baseline cites nothing beyond what it was shown. That is the honest
78
+ # rendering of a method that never looked at a pull request: its reasons are
79
+ # impressions of a README, and the report should not dress them as evidence.
80
+ claims = [Claim(text=r, evidence_id=None) for r in result.get("reasons", [])]
81
+ return Assessment(
82
+ repo=repo,
83
+ verdict=Verdict(result["verdict"]),
84
+ summary=result["summary"],
85
+ claims=claims,
86
+ method="baseline (single prompt over README and metadata)",
87
+ replayed=model.replayed,
88
+ models=list(model.usage.models),
89
+ )
@@ -0,0 +1,116 @@
1
+ """An evidence-matched baseline: one prompt, everything Holt sees.
2
+
3
+ The original baseline reads a README and repository metadata. Holt reads two
4
+ hundred pull request threads. Comparing them measures mostly "pull request
5
+ history beats a landing page", which is true but is not the claim the
6
+ architecture is making.
7
+
8
+ This baseline closes that gap. It receives the *same* arithmetic signals Holt
9
+ computes and the *same* twelve-thread digest Stage C reads, and is asked for the
10
+ same three-valued verdict in a single call. What differs is only the
11
+ architecture: one model call decides, instead of typed findings passing through
12
+ verification into a model-free verdict function.
13
+
14
+ If Holt does not beat this, the pipeline is not earning its complexity, and that
15
+ is worth knowing and publishing.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from holt.agent.signals import build_threads, compute
21
+ from holt.agent.stages import _render_thread
22
+ from holt.evidence.provider import EvidenceProvider
23
+ from holt.model import ModelClient
24
+ from holt.report import Assessment, Claim, Verdict
25
+
26
+ SYSTEM = """You assess whether a GitHub repository is a worthwhile place for an \
27
+ outside developer -- someone with no prior connection to the project -- to spend \
28
+ a week contributing.
29
+
30
+ You are given the repository's README, its metadata, arithmetic measured from its
31
+ pull request history before the cutoff, and a sample of its pull request threads.
32
+
33
+ Judge what a merged contribution here actually *is*. A repository where merged
34
+ work means appending an entry to a catalogue -- package manifests, domain
35
+ records, plugin listings -- is easy to contribute to and is not a place to spend
36
+ a week writing software.
37
+
38
+ Answer with one of three verdicts:
39
+ viable an outsider could realistically land a meaningful change
40
+ not_viable an outsider could not, or the work would not be software
41
+ insufficient_evidence the material does not support a call either way
42
+
43
+ Prefer insufficient_evidence to a confident guess. Cite evidence ids you were
44
+ given for the reasons you list."""
45
+
46
+ SCHEMA = {
47
+ "type": "object",
48
+ "properties": {
49
+ "verdict": {"type": "string", "enum": [v.value for v in Verdict]},
50
+ "summary": {"type": "string"},
51
+ "reasons": {"type": "array", "items": {"type": "string"}},
52
+ },
53
+ "required": ["verdict", "summary", "reasons"],
54
+ "additionalProperties": False,
55
+ }
56
+
57
+ THREAD_SAMPLE = 12
58
+
59
+
60
+ RECORDED_SIGNAL_FIELDS = (
61
+ "total_threads",
62
+ "outsider_threads",
63
+ "outsider_merged",
64
+ "outsider_ignored",
65
+ "median_first_response_hours",
66
+ "bot_share",
67
+ "distinct_outsider_authors",
68
+ "distinct_merged_authors",
69
+ "reviewed_share",
70
+ "merge_rate",
71
+ )
72
+
73
+
74
+ def assess(repo: str, provider: EvidenceProvider, model: ModelClient) -> Assessment:
75
+ records = provider.fetch(repo)
76
+ meta = next((r.payload for r in records if r.evidence_id.endswith(":meta")), {})
77
+ readme = next(
78
+ (r.payload.get("text") for r in records if r.evidence_id.endswith(":readme")), None
79
+ )
80
+ threads = build_threads(records)
81
+ signals = compute(threads)
82
+
83
+ talkative = sorted(
84
+ (t for t in threads.values() if not t.author_is_bot),
85
+ key=lambda t: (len(t.responses), t.additions + t.deletions),
86
+ reverse=True,
87
+ )[:THREAD_SAMPLE]
88
+
89
+ parts = [f"Repository: {repo}", "", "Metadata:"]
90
+ for key in ("description", "primary_language", "stargazer_count", "pushed_at",
91
+ "is_archived", "is_fork", "is_mirror", "homepage_url"):
92
+ parts.append(f" {key}: {meta.get(key)!r}")
93
+ parts += ["", "Measured from pull request history before the cutoff:"]
94
+ # The exact ten fields this prompt was recorded with, listed rather than
95
+ # taken from `as_dict()`. This is the *comparison* method: a signal added to
96
+ # the pipeline must not silently change what the baseline is shown, or the
97
+ # ablation stops being the same experiment -- and every recorded
98
+ # `baseline_matched` call becomes a replay miss the moment anyone adds one.
99
+ measured = signals.as_dict()
100
+ parts += [f" {k}: {measured[k]}" for k in RECORDED_SIGNAL_FIELDS]
101
+ parts += ["", "README:", (readme or "(none at the cutoff)")[:6000], ""]
102
+ parts += ["Pull request threads:", ""]
103
+ parts += [_render_thread(t) for t in talkative]
104
+
105
+ result = model.complete(
106
+ label="baseline_matched", system=SYSTEM, prompt="\n".join(parts), schema=SCHEMA
107
+ )
108
+ return Assessment(
109
+ repo=repo,
110
+ verdict=Verdict(result["verdict"]),
111
+ summary=result["summary"],
112
+ claims=[Claim(text=r, evidence_id=None) for r in result.get("reasons", [])],
113
+ method="baseline, evidence-matched (one prompt, the same signals and threads Holt reads)",
114
+ replayed=model.replayed,
115
+ models=list(model.usage.models),
116
+ )