holt-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- holt/__init__.py +0 -0
- holt/agent/__init__.py +0 -0
- holt/agent/entry.py +86 -0
- holt/agent/findings.py +49 -0
- holt/agent/landing.py +154 -0
- holt/agent/pipeline.py +244 -0
- holt/agent/progression.py +408 -0
- holt/agent/signals.py +220 -0
- holt/agent/stages.py +533 -0
- holt/agent/verdict.py +226 -0
- holt/agent/verify.py +140 -0
- holt/baseline.py +89 -0
- holt/baseline_matched.py +116 -0
- holt/cli.py +616 -0
- holt/discover.py +497 -0
- holt/evidence/__init__.py +3 -0
- holt/evidence/fixtures.py +154 -0
- holt/evidence/github_graphql.py +538 -0
- holt/evidence/provider.py +77 -0
- holt/evidence/redact.py +79 -0
- holt/issues.py +41 -0
- holt/model.py +516 -0
- holt/profile.py +126 -0
- holt/report.py +157 -0
- holt/tui/__init__.py +0 -0
- holt/tui/animation.py +84 -0
- holt/tui/app.py +294 -0
- holt/tui/clipboard.py +89 -0
- holt/tui/commands.py +134 -0
- holt/tui/discovery.py +305 -0
- holt/tui/env.py +49 -0
- holt/tui/events.py +245 -0
- holt/tui/mascot.py +121 -0
- holt/tui/models.py +590 -0
- holt/tui/observe.py +297 -0
- holt/tui/screens/__init__.py +35 -0
- holt/tui/screens/assessment.py +337 -0
- holt/tui/screens/confirm.py +62 -0
- holt/tui/screens/discover.py +444 -0
- holt/tui/screens/home.py +519 -0
- holt/tui/screens/inspector.py +106 -0
- holt/tui/screens/live.py +335 -0
- holt/tui/screens/models.py +393 -0
- holt/tui/screens/next_steps.py +425 -0
- holt/tui/screens/profile.py +129 -0
- holt/tui/session.py +711 -0
- holt/tui/store.py +458 -0
- holt/tui/theme.py +479 -0
- holt/tui/visual.py +33 -0
- holt/tui/widgets/__init__.py +0 -0
- holt/tui/widgets/candidates.py +78 -0
- holt/tui/widgets/claims.py +59 -0
- holt/tui/widgets/disclosure.py +121 -0
- holt/tui/widgets/evidence.py +121 -0
- holt/tui/widgets/masthead.py +122 -0
- holt/tui/widgets/recent.py +167 -0
- holt/tui/widgets/scrolling.py +38 -0
- holt/tui/widgets/stages.py +232 -0
- holt/types.py +48 -0
- holt_cli-0.1.0.dist-info/METADATA +198 -0
- holt_cli-0.1.0.dist-info/RECORD +65 -0
- holt_cli-0.1.0.dist-info/WHEEL +4 -0
- holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
- holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
- holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/agent/verdict.py
ADDED
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
"""The only path from findings to a verdict, and it is a plain function.
|
|
2
|
+
|
|
3
|
+
No model runs here. Three reasons, all of which matter:
|
|
4
|
+
|
|
5
|
+
* a judge who reruns Holt gets our numbers, not a resample of them
|
|
6
|
+
* scoring across a pool is not polluted by model variance
|
|
7
|
+
* "is this just a wrapper around a prompt?" is answered by a file rather than
|
|
8
|
+
an argument
|
|
9
|
+
|
|
10
|
+
Stage E writes prose *around* this decision and is handed the result as an input
|
|
11
|
+
it cannot change. A test asserts the rendered report and this function agree.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from holt.agent.findings import Findings
|
|
17
|
+
from holt.agent.signals import Signals
|
|
18
|
+
from holt.report import Verdict
|
|
19
|
+
|
|
20
|
+
# Kinds where a merged pull request is not a software contribution. Landing work
|
|
21
|
+
# in these is easy and means nothing for the question being asked.
|
|
22
|
+
NON_SOFTWARE_KINDS = {"registry", "awesome_list", "portfolio", "course_material"}
|
|
23
|
+
|
|
24
|
+
# Kinds where outside contribution is not accepted regardless of activity.
|
|
25
|
+
CLOSED_KINDS = {"mirror"}
|
|
26
|
+
|
|
27
|
+
# --- contesting the one field that can decide alone -------------------------
|
|
28
|
+
#
|
|
29
|
+
# `repo_kind` is the only model-derived input either of the rules above reads,
|
|
30
|
+
# and Stage D cannot check it: it verifies that a cited id resolves, and a
|
|
31
|
+
# classification is not a quotation. A model that answers `mirror` while citing
|
|
32
|
+
# a real README has made a claim that verifies perfectly and is false -- which
|
|
33
|
+
# is exactly what happened to `pytorch/pytorch` under a 3B local model, and to
|
|
34
|
+
# `aden-hive/hive`, which was called a registry.
|
|
35
|
+
#
|
|
36
|
+
# So the two kinds that can flip a verdict are checked against the evidence they
|
|
37
|
+
# implicitly claim something about. What is contested is the rule's stated
|
|
38
|
+
# *reason*, never what the repository "really is": a catalogue entry is one file
|
|
39
|
+
# in one place, and a mirror does not merge outsiders' pull requests. Where the
|
|
40
|
+
# evidence disagrees the field is dropped rather than overridden -- no verdict
|
|
41
|
+
# is asserted in its place, `classify` falls through to the arithmetic, and the
|
|
42
|
+
# disagreement is printed for the reader.
|
|
43
|
+
#
|
|
44
|
+
# Pre-registered with its predictions in eval/PREREGISTRATION-4.md; thresholds
|
|
45
|
+
# chosen on pool 1 and that fitting disclosed there.
|
|
46
|
+
CATALOGUE_KINDS = {"registry", "awesome_list"}
|
|
47
|
+
|
|
48
|
+
# `portfolio` and `course_material` are deliberately not contested this way.
|
|
49
|
+
# Their reason is about whose project it is, not the shape of a diff, and a
|
|
50
|
+
# portfolio being real code is not a contradiction.
|
|
51
|
+
MIN_MERGES_FOR_SHAPE = 5
|
|
52
|
+
|
|
53
|
+
# One directory. Not "few enough files": that criterion was pre-registered,
|
|
54
|
+
# failed out-of-sample, and is gone. `microsoft/winget-pkgs` is a real registry
|
|
55
|
+
# whose every entry is three YAML manifests -- installer, locale, version -- in
|
|
56
|
+
# one package directory, so a median-files test called a correct classification
|
|
57
|
+
# a hallucination on all four of its recordings. What survived is the criterion
|
|
58
|
+
# that did not misfire: a catalogue entry lands in one place, whatever it
|
|
59
|
+
# weighs. The narrowing was chosen after seeing that failure and is disclosed as
|
|
60
|
+
# such in eval/PREREGISTRATION-4.md; it has no untouched holdout behind it.
|
|
61
|
+
CATALOGUE_DIRS_MAX = 2
|
|
62
|
+
|
|
63
|
+
# How long the contributor has. Everything time-shaped scales from this, because
|
|
64
|
+
# "is this repository worth my time" has no answer independent of how much time
|
|
65
|
+
# you have: a maintainer who replies in five days is fine if you have three
|
|
66
|
+
# months and useless if you have three days.
|
|
67
|
+
DEFAULT_CONTRIBUTOR_DAYS = 7
|
|
68
|
+
|
|
69
|
+
# Rubber-stamp rejection. Validated out-of-sample on pool 2 -- specificity 0.58
|
|
70
|
+
# to 0.83 with all three pre-registered predictions holding. Thresholds were
|
|
71
|
+
# chosen on pool 1 and that fitting is disclosed in eval/PREREGISTRATION-2.md.
|
|
72
|
+
#
|
|
73
|
+
# Both halves are required. Landing easily alone describes a welcoming project;
|
|
74
|
+
# going unreviewed alone describes a project whose review happens elsewhere,
|
|
75
|
+
# which is what killed the first rejection rule when nixpkgs was withheld. It is
|
|
76
|
+
# the conjunction that describes work being waved through unread.
|
|
77
|
+
RUBBER_STAMP_REVIEWED_MAX = 0.20
|
|
78
|
+
RUBBER_STAMP_MERGE_RATE_MIN = 0.60
|
|
79
|
+
|
|
80
|
+
# One merge from one person is an anecdote; two people is a pattern.
|
|
81
|
+
MIN_MERGES = 2
|
|
82
|
+
MIN_DISTINCT_AUTHORS = 2
|
|
83
|
+
|
|
84
|
+
# Below this share of ignored attempts, silence is noise rather than a policy.
|
|
85
|
+
IGNORED_SHARE = 0.7
|
|
86
|
+
|
|
87
|
+
# ...and below this many attempts there is no share worth speaking of. Four
|
|
88
|
+
# ignored pull requests out of four is not evidence of hostility, it is four
|
|
89
|
+
# data points. tensorflow/tensorflow reaches exactly that shape: 97% of its pull
|
|
90
|
+
# request traffic is automation, leaving a handful of outsider threads. Without
|
|
91
|
+
# this guard the rule turns a thin sample into a confident accusation.
|
|
92
|
+
MIN_ATTEMPTS_FOR_HOSTILE = 8
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def contested_kind(
|
|
96
|
+
findings: Findings, signals: Signals, meta: dict | None = None
|
|
97
|
+
) -> str | None:
|
|
98
|
+
"""Why the claimed `repo_kind` disagrees with the evidence, or None.
|
|
99
|
+
|
|
100
|
+
Returns the sentence a reader should see, not a boolean: a field being
|
|
101
|
+
dropped is a thing that happened to their report and it is printed.
|
|
102
|
+
"""
|
|
103
|
+
kind = findings.get("repo_kind")
|
|
104
|
+
|
|
105
|
+
if kind in CATALOGUE_KINDS:
|
|
106
|
+
if (
|
|
107
|
+
signals.merged_with_files >= MIN_MERGES_FOR_SHAPE
|
|
108
|
+
and signals.merged_dirs_median is not None
|
|
109
|
+
and signals.merged_dirs_median >= CATALOGUE_DIRS_MAX
|
|
110
|
+
):
|
|
111
|
+
return (
|
|
112
|
+
f"repo_kind={kind} was claimed, but merged work here spans a "
|
|
113
|
+
f"median of {signals.merged_dirs_median:g} top-level directories "
|
|
114
|
+
"rather than landing in one place, which is not a catalogue "
|
|
115
|
+
"entry; the field is dropped and decided nothing"
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
if kind in CLOSED_KINDS:
|
|
119
|
+
# `is_mirror` alone would not be enough -- GitHub sets it only for
|
|
120
|
+
# repositories created as mirrors, so a genuine mirror can report
|
|
121
|
+
# false. The merges are what disprove the claim being made.
|
|
122
|
+
if (
|
|
123
|
+
(meta or {}).get("is_mirror") is False
|
|
124
|
+
and signals.outsider_merged >= MIN_MERGES
|
|
125
|
+
and signals.distinct_merged_authors >= MIN_DISTINCT_AUTHORS
|
|
126
|
+
):
|
|
127
|
+
return (
|
|
128
|
+
f"repo_kind={kind} was claimed, but GitHub does not report this "
|
|
129
|
+
f"repository as a mirror and {signals.outsider_merged} outside "
|
|
130
|
+
f"pull requests by {signals.distinct_merged_authors} people were "
|
|
131
|
+
"merged in the period read; the field is dropped and decided nothing"
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
return None
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def classify(
|
|
138
|
+
findings: Findings,
|
|
139
|
+
signals: Signals,
|
|
140
|
+
contributor_days: int = DEFAULT_CONTRIBUTOR_DAYS,
|
|
141
|
+
) -> tuple[Verdict, list[str]]:
|
|
142
|
+
"""Return a verdict and the rule trace that produced it.
|
|
143
|
+
|
|
144
|
+
`contributor_days` is the time the person actually has. Re-running this
|
|
145
|
+
function with a different budget costs nothing and calls no model, because
|
|
146
|
+
the findings are already computed -- which is a thing a single prompt cannot
|
|
147
|
+
do without paying for the whole assessment again.
|
|
148
|
+
"""
|
|
149
|
+
trace: list[str] = []
|
|
150
|
+
slow_response_hours = contributor_days * 24.0
|
|
151
|
+
kind = findings.get("repo_kind")
|
|
152
|
+
|
|
153
|
+
if findings.get("is_archived"):
|
|
154
|
+
trace.append("archived: no longer accepting work")
|
|
155
|
+
return Verdict.NOT_VIABLE, trace
|
|
156
|
+
|
|
157
|
+
if kind in CLOSED_KINDS:
|
|
158
|
+
trace.append(f"repo_kind={kind}: outside pull requests are not the contribution path")
|
|
159
|
+
return Verdict.NOT_VIABLE, trace
|
|
160
|
+
|
|
161
|
+
if kind in NON_SOFTWARE_KINDS:
|
|
162
|
+
trace.append(f"repo_kind={kind}: merged work here is not a software contribution")
|
|
163
|
+
return Verdict.NOT_VIABLE, trace
|
|
164
|
+
|
|
165
|
+
if signals.outsider_threads == 0:
|
|
166
|
+
# "In the period read", not "before the cutoff": the cutoff is an
|
|
167
|
+
# evaluation device, and this line is printed verbatim to users.
|
|
168
|
+
trace.append("no outsider attempts in the period read: nothing to judge from")
|
|
169
|
+
return Verdict.INSUFFICIENT_EVIDENCE, trace
|
|
170
|
+
|
|
171
|
+
ignored_share = signals.outsider_ignored / signals.outsider_threads
|
|
172
|
+
if (
|
|
173
|
+
signals.outsider_merged == 0
|
|
174
|
+
and ignored_share > IGNORED_SHARE
|
|
175
|
+
and signals.outsider_threads >= MIN_ATTEMPTS_FOR_HOSTILE
|
|
176
|
+
):
|
|
177
|
+
trace.append(
|
|
178
|
+
f"{signals.outsider_ignored}/{signals.outsider_threads} outsider attempts "
|
|
179
|
+
"drew no response and none merged"
|
|
180
|
+
)
|
|
181
|
+
return Verdict.NOT_VIABLE, trace
|
|
182
|
+
|
|
183
|
+
slow = (
|
|
184
|
+
signals.median_first_response_hours is not None
|
|
185
|
+
and signals.median_first_response_hours > slow_response_hours
|
|
186
|
+
)
|
|
187
|
+
if (
|
|
188
|
+
signals.outsider_merged >= MIN_MERGES
|
|
189
|
+
and signals.distinct_outsider_authors >= MIN_DISTINCT_AUTHORS
|
|
190
|
+
and not slow
|
|
191
|
+
):
|
|
192
|
+
trace.append(
|
|
193
|
+
f"{signals.outsider_merged} first-time merges by "
|
|
194
|
+
f"{signals.distinct_merged_authors} distinct people, out of "
|
|
195
|
+
f"{signals.outsider_threads} attempts by "
|
|
196
|
+
f"{signals.distinct_outsider_authors}; median first response "
|
|
197
|
+
f"{signals.median_first_response_hours}h"
|
|
198
|
+
)
|
|
199
|
+
if (
|
|
200
|
+
signals.reviewed_share is not None
|
|
201
|
+
and signals.merge_rate is not None
|
|
202
|
+
and signals.reviewed_share < RUBBER_STAMP_REVIEWED_MAX
|
|
203
|
+
and signals.merge_rate > RUBBER_STAMP_MERGE_RATE_MIN
|
|
204
|
+
):
|
|
205
|
+
trace.append(
|
|
206
|
+
f"but only {signals.reviewed_share:.0%} of merges drew any human "
|
|
207
|
+
f"reply while {signals.merge_rate:.0%} of attempts landed: work is "
|
|
208
|
+
"being waved through unread, so a contribution here buys no review"
|
|
209
|
+
)
|
|
210
|
+
return Verdict.NOT_VIABLE, trace
|
|
211
|
+
return Verdict.VIABLE, trace
|
|
212
|
+
|
|
213
|
+
if slow:
|
|
214
|
+
trace.append(
|
|
215
|
+
f"median first response {signals.median_first_response_hours}h "
|
|
216
|
+
f"exceeds the {slow_response_hours:.0f}h a {contributor_days}-day "
|
|
217
|
+
"budget allows"
|
|
218
|
+
)
|
|
219
|
+
if signals.outsider_merged == 0 and ignored_share > IGNORED_SHARE:
|
|
220
|
+
trace.append(
|
|
221
|
+
f"{signals.outsider_ignored}/{signals.outsider_threads} attempts ignored, "
|
|
222
|
+
f"but fewer than {MIN_ATTEMPTS_FOR_HOSTILE} attempts is too thin to call hostile"
|
|
223
|
+
)
|
|
224
|
+
elif signals.outsider_merged < MIN_MERGES:
|
|
225
|
+
trace.append(f"only {signals.outsider_merged} outsider merges in the period read")
|
|
226
|
+
return Verdict.INSUFFICIENT_EVIDENCE, trace
|
holt/agent/verify.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""Stage D — the only stage that removes things.
|
|
2
|
+
|
|
3
|
+
Two mechanical checks, both of which delete rather than soften.
|
|
4
|
+
|
|
5
|
+
**Resolution.** Every finding carries the evidence ids that support it. This
|
|
6
|
+
resolves each one against the provider. A finding whose evidence does not
|
|
7
|
+
resolve is dropped: the alternative is prose that hedges around a claim nobody
|
|
8
|
+
can check, which is how unsupported statements survive into reports.
|
|
9
|
+
|
|
10
|
+
**Quotation.** A claim can cite a pull request that exists and still put words
|
|
11
|
+
in its mouth. `eval/evidence_integrity.py` has measured that gap since Iteration
|
|
12
|
+
11 -- it is the difference between an id that resolves and evidence that says
|
|
13
|
+
what the claim says -- and the check here is the same function the metric uses,
|
|
14
|
+
so the number the eval reports and the guarantee the reader gets cannot drift
|
|
15
|
+
apart. A quote that is not in the record takes its claim with it, for the same
|
|
16
|
+
reason: keeping the outcome while deleting the fabricated quote is softening.
|
|
17
|
+
|
|
18
|
+
No model runs here. Resolution is a lookup and quotation is string matching.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import re
|
|
24
|
+
from collections.abc import Iterable
|
|
25
|
+
|
|
26
|
+
from holt.agent.findings import Finding, Findings
|
|
27
|
+
from holt.evidence.provider import EvidenceProvider
|
|
28
|
+
from holt.types import EvidenceRecord
|
|
29
|
+
|
|
30
|
+
# A quote counts as present if a long-enough run of its words appears verbatim
|
|
31
|
+
# in the record. Models normalise whitespace and clip mid-sentence; penalising
|
|
32
|
+
# that would measure formatting rather than fidelity.
|
|
33
|
+
SHINGLE = 6
|
|
34
|
+
|
|
35
|
+
# `stages._render_thread` prefixes every reply with the speaker -- `[AUTHOR]`,
|
|
36
|
+
# `[octocat]` -- and a model quoting that reply routinely copies the prefix.
|
|
37
|
+
# Nobody said those words: they are ours. Stripping the tag before matching
|
|
38
|
+
# stops the guard blaming the model for our own formatting, which is the same
|
|
39
|
+
# lesson NO_REPLIES taught. What is left still has to be in the record, and a
|
|
40
|
+
# "quote" that is nothing *but* our scaffolding quotes nothing at all.
|
|
41
|
+
SPEAKER_TAG = re.compile(r"^\s*\[[^\]]{1,40}\]\s*")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def verify(findings: Findings, provider: EvidenceProvider) -> tuple[Findings, list[Finding]]:
|
|
45
|
+
"""Return (surviving findings, dropped findings)."""
|
|
46
|
+
kept, dropped = Findings(), []
|
|
47
|
+
for item in findings:
|
|
48
|
+
if not item.evidence_ids:
|
|
49
|
+
dropped.append(item)
|
|
50
|
+
continue
|
|
51
|
+
resolved = tuple(e for e in item.evidence_ids if provider.resolve(e) is not None)
|
|
52
|
+
if not resolved:
|
|
53
|
+
dropped.append(item)
|
|
54
|
+
continue
|
|
55
|
+
kept.add(item.field, item.value, resolved, item.note)
|
|
56
|
+
return kept, dropped
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# Punctuation is folded away with the whitespace, for the same reason. A model
|
|
60
|
+
# that quotes ``[`nixpkgs-review`]``(...) as ``[`nixpkgs-review`].`` -- closing a
|
|
61
|
+
# markdown link it truncated -- has quoted the thread; rejecting it would be
|
|
62
|
+
# measuring punctuation. What survives folding is the words, and those still
|
|
63
|
+
# have to be the record's words, in the record's order.
|
|
64
|
+
_NON_WORD = re.compile(r"[^\w]+", re.UNICODE)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def normalise(text: str) -> str:
|
|
68
|
+
return " ".join(_NON_WORD.sub(" ", (text or "").lower()).split())
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def spoken_part(quote: str) -> str:
|
|
72
|
+
"""The words attributed to a person, with our speaker tag removed."""
|
|
73
|
+
return SPEAKER_TAG.sub("", quote or "", count=1).strip()
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def quote_supported(quote: str, haystack: str) -> bool:
|
|
77
|
+
q, h = normalise(quote), normalise(haystack)
|
|
78
|
+
if not q:
|
|
79
|
+
return False
|
|
80
|
+
words = q.split()
|
|
81
|
+
if len(words) <= SHINGLE:
|
|
82
|
+
return q in h
|
|
83
|
+
return any(
|
|
84
|
+
" ".join(words[i : i + SHINGLE]) in h for i in range(len(words) - SHINGLE + 1)
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def spoken_words(records: Iterable[EvidenceRecord]) -> dict[str, str]:
|
|
89
|
+
"""Everything said anywhere on each pull request, keyed by its number.
|
|
90
|
+
|
|
91
|
+
Quotes come from reviews and comments, which live in their own records
|
|
92
|
+
(`#12:review:0`, `#12:comment:1`) rather than in the `:opened` record a
|
|
93
|
+
Stage C finding cites. Checking a quote against the cited record alone would
|
|
94
|
+
reject every real quotation, so the haystack is the whole thread.
|
|
95
|
+
"""
|
|
96
|
+
said: dict[str, list[str]] = {}
|
|
97
|
+
for record in records:
|
|
98
|
+
number = _pr_number(record.evidence_id)
|
|
99
|
+
if number is None:
|
|
100
|
+
continue
|
|
101
|
+
payload = record.payload
|
|
102
|
+
text = f"{payload.get('title') or ''} {payload.get('body') or ''}"
|
|
103
|
+
if text.strip():
|
|
104
|
+
said.setdefault(number, []).append(text)
|
|
105
|
+
return {k: " ".join(v) for k, v in said.items()}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _pr_number(evidence_id: str) -> str | None:
|
|
109
|
+
if "#" not in evidence_id:
|
|
110
|
+
return None
|
|
111
|
+
return evidence_id.split("#")[-1].split(":")[0]
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def check_quotes(
|
|
115
|
+
findings: Findings, records: Iterable[EvidenceRecord]
|
|
116
|
+
) -> tuple[list[Finding], list[Finding]]:
|
|
117
|
+
"""Split findings into (quoting the record, putting words in its mouth).
|
|
118
|
+
|
|
119
|
+
Findings that quote nothing pass through untouched -- there is nothing to
|
|
120
|
+
check and nothing to accuse them of.
|
|
121
|
+
"""
|
|
122
|
+
said = spoken_words(records)
|
|
123
|
+
supported, invented = [], []
|
|
124
|
+
for item in findings:
|
|
125
|
+
raw = item.value.get("quote", "") if isinstance(item.value, dict) else ""
|
|
126
|
+
if not raw or not str(raw).strip():
|
|
127
|
+
supported.append(item)
|
|
128
|
+
continue
|
|
129
|
+
quote = spoken_part(str(raw))
|
|
130
|
+
if not quote:
|
|
131
|
+
# Entirely our own scaffolding. There is no quotation here to check,
|
|
132
|
+
# and printing `“[octocat]”` to a reader is not evidence of anything.
|
|
133
|
+
invented.append(item)
|
|
134
|
+
continue
|
|
135
|
+
number = _pr_number(item.evidence_ids[0]) if item.evidence_ids else None
|
|
136
|
+
if number is not None and quote_supported(quote, said.get(number, "")):
|
|
137
|
+
supported.append(item)
|
|
138
|
+
else:
|
|
139
|
+
invented.append(item)
|
|
140
|
+
return supported, invented
|
holt/baseline.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""The baseline solution: one prompt, the README, and the repository metadata.
|
|
2
|
+
|
|
3
|
+
This is a *solution*, not a comparator -- it has its own entry point
|
|
4
|
+
(`holt analyze --baseline`) and produces the same Assessment the full pipeline
|
|
5
|
+
does. It is what a competent engineer would build in an afternoon, and it is the
|
|
6
|
+
thing the rest of the system has to beat.
|
|
7
|
+
|
|
8
|
+
It reads through the same EvidenceProvider, so it is bound by the same holdout
|
|
9
|
+
and runs in the same fixture and replay modes. Same task, same cases, same
|
|
10
|
+
evidence, different method.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from holt.evidence.provider import EvidenceProvider
|
|
16
|
+
from holt.model import ModelClient
|
|
17
|
+
from holt.report import Assessment, Claim, Verdict
|
|
18
|
+
|
|
19
|
+
SYSTEM = """You assess whether a GitHub repository is a worthwhile place for an \
|
|
20
|
+
outside developer -- someone with no prior connection to the project -- to spend \
|
|
21
|
+
a week contributing.
|
|
22
|
+
|
|
23
|
+
Answer with one of three verdicts:
|
|
24
|
+
viable an outsider could realistically land a meaningful change
|
|
25
|
+
not_viable an outsider could not, or the work would not be software
|
|
26
|
+
insufficient_evidence the material does not support a call either way
|
|
27
|
+
|
|
28
|
+
Prefer insufficient_evidence to a confident guess."""
|
|
29
|
+
|
|
30
|
+
SCHEMA = {
|
|
31
|
+
"type": "object",
|
|
32
|
+
"properties": {
|
|
33
|
+
"verdict": {"type": "string", "enum": [v.value for v in Verdict]},
|
|
34
|
+
"summary": {"type": "string"},
|
|
35
|
+
"reasons": {"type": "array", "items": {"type": "string"}},
|
|
36
|
+
},
|
|
37
|
+
"required": ["verdict", "summary", "reasons"],
|
|
38
|
+
"additionalProperties": False,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _prompt(repo: str, meta: dict, readme: str | None) -> str:
|
|
43
|
+
parts = [f"Repository: {repo}", "", "Metadata:"]
|
|
44
|
+
for key in (
|
|
45
|
+
"description",
|
|
46
|
+
"primary_language",
|
|
47
|
+
"stargazer_count",
|
|
48
|
+
"pushed_at",
|
|
49
|
+
"is_archived",
|
|
50
|
+
"is_fork",
|
|
51
|
+
"is_mirror",
|
|
52
|
+
"homepage_url",
|
|
53
|
+
):
|
|
54
|
+
parts.append(f" {key}: {meta.get(key)!r}")
|
|
55
|
+
parts += ["", "README:", readme or "(no README found at the cutoff)"]
|
|
56
|
+
return "\n".join(parts)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def assess(repo: str, provider: EvidenceProvider, model: ModelClient) -> Assessment:
|
|
60
|
+
records = provider.fetch(repo)
|
|
61
|
+
meta = next(
|
|
62
|
+
(r.payload for r in records if r.evidence_id.endswith(":meta")),
|
|
63
|
+
{},
|
|
64
|
+
)
|
|
65
|
+
readme = next(
|
|
66
|
+
(r.payload.get("text") for r in records if r.evidence_id.endswith(":readme")),
|
|
67
|
+
None,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
result = model.complete(
|
|
71
|
+
label="baseline",
|
|
72
|
+
system=SYSTEM,
|
|
73
|
+
prompt=_prompt(repo, meta, readme),
|
|
74
|
+
schema=SCHEMA,
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
# The baseline cites nothing beyond what it was shown. That is the honest
|
|
78
|
+
# rendering of a method that never looked at a pull request: its reasons are
|
|
79
|
+
# impressions of a README, and the report should not dress them as evidence.
|
|
80
|
+
claims = [Claim(text=r, evidence_id=None) for r in result.get("reasons", [])]
|
|
81
|
+
return Assessment(
|
|
82
|
+
repo=repo,
|
|
83
|
+
verdict=Verdict(result["verdict"]),
|
|
84
|
+
summary=result["summary"],
|
|
85
|
+
claims=claims,
|
|
86
|
+
method="baseline (single prompt over README and metadata)",
|
|
87
|
+
replayed=model.replayed,
|
|
88
|
+
models=list(model.usage.models),
|
|
89
|
+
)
|
holt/baseline_matched.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""An evidence-matched baseline: one prompt, everything Holt sees.
|
|
2
|
+
|
|
3
|
+
The original baseline reads a README and repository metadata. Holt reads two
|
|
4
|
+
hundred pull request threads. Comparing them measures mostly "pull request
|
|
5
|
+
history beats a landing page", which is true but is not the claim the
|
|
6
|
+
architecture is making.
|
|
7
|
+
|
|
8
|
+
This baseline closes that gap. It receives the *same* arithmetic signals Holt
|
|
9
|
+
computes and the *same* twelve-thread digest Stage C reads, and is asked for the
|
|
10
|
+
same three-valued verdict in a single call. What differs is only the
|
|
11
|
+
architecture: one model call decides, instead of typed findings passing through
|
|
12
|
+
verification into a model-free verdict function.
|
|
13
|
+
|
|
14
|
+
If Holt does not beat this, the pipeline is not earning its complexity, and that
|
|
15
|
+
is worth knowing and publishing.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from holt.agent.signals import build_threads, compute
|
|
21
|
+
from holt.agent.stages import _render_thread
|
|
22
|
+
from holt.evidence.provider import EvidenceProvider
|
|
23
|
+
from holt.model import ModelClient
|
|
24
|
+
from holt.report import Assessment, Claim, Verdict
|
|
25
|
+
|
|
26
|
+
SYSTEM = """You assess whether a GitHub repository is a worthwhile place for an \
|
|
27
|
+
outside developer -- someone with no prior connection to the project -- to spend \
|
|
28
|
+
a week contributing.
|
|
29
|
+
|
|
30
|
+
You are given the repository's README, its metadata, arithmetic measured from its
|
|
31
|
+
pull request history before the cutoff, and a sample of its pull request threads.
|
|
32
|
+
|
|
33
|
+
Judge what a merged contribution here actually *is*. A repository where merged
|
|
34
|
+
work means appending an entry to a catalogue -- package manifests, domain
|
|
35
|
+
records, plugin listings -- is easy to contribute to and is not a place to spend
|
|
36
|
+
a week writing software.
|
|
37
|
+
|
|
38
|
+
Answer with one of three verdicts:
|
|
39
|
+
viable an outsider could realistically land a meaningful change
|
|
40
|
+
not_viable an outsider could not, or the work would not be software
|
|
41
|
+
insufficient_evidence the material does not support a call either way
|
|
42
|
+
|
|
43
|
+
Prefer insufficient_evidence to a confident guess. Cite evidence ids you were
|
|
44
|
+
given for the reasons you list."""
|
|
45
|
+
|
|
46
|
+
SCHEMA = {
|
|
47
|
+
"type": "object",
|
|
48
|
+
"properties": {
|
|
49
|
+
"verdict": {"type": "string", "enum": [v.value for v in Verdict]},
|
|
50
|
+
"summary": {"type": "string"},
|
|
51
|
+
"reasons": {"type": "array", "items": {"type": "string"}},
|
|
52
|
+
},
|
|
53
|
+
"required": ["verdict", "summary", "reasons"],
|
|
54
|
+
"additionalProperties": False,
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
THREAD_SAMPLE = 12
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
RECORDED_SIGNAL_FIELDS = (
|
|
61
|
+
"total_threads",
|
|
62
|
+
"outsider_threads",
|
|
63
|
+
"outsider_merged",
|
|
64
|
+
"outsider_ignored",
|
|
65
|
+
"median_first_response_hours",
|
|
66
|
+
"bot_share",
|
|
67
|
+
"distinct_outsider_authors",
|
|
68
|
+
"distinct_merged_authors",
|
|
69
|
+
"reviewed_share",
|
|
70
|
+
"merge_rate",
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def assess(repo: str, provider: EvidenceProvider, model: ModelClient) -> Assessment:
|
|
75
|
+
records = provider.fetch(repo)
|
|
76
|
+
meta = next((r.payload for r in records if r.evidence_id.endswith(":meta")), {})
|
|
77
|
+
readme = next(
|
|
78
|
+
(r.payload.get("text") for r in records if r.evidence_id.endswith(":readme")), None
|
|
79
|
+
)
|
|
80
|
+
threads = build_threads(records)
|
|
81
|
+
signals = compute(threads)
|
|
82
|
+
|
|
83
|
+
talkative = sorted(
|
|
84
|
+
(t for t in threads.values() if not t.author_is_bot),
|
|
85
|
+
key=lambda t: (len(t.responses), t.additions + t.deletions),
|
|
86
|
+
reverse=True,
|
|
87
|
+
)[:THREAD_SAMPLE]
|
|
88
|
+
|
|
89
|
+
parts = [f"Repository: {repo}", "", "Metadata:"]
|
|
90
|
+
for key in ("description", "primary_language", "stargazer_count", "pushed_at",
|
|
91
|
+
"is_archived", "is_fork", "is_mirror", "homepage_url"):
|
|
92
|
+
parts.append(f" {key}: {meta.get(key)!r}")
|
|
93
|
+
parts += ["", "Measured from pull request history before the cutoff:"]
|
|
94
|
+
# The exact ten fields this prompt was recorded with, listed rather than
|
|
95
|
+
# taken from `as_dict()`. This is the *comparison* method: a signal added to
|
|
96
|
+
# the pipeline must not silently change what the baseline is shown, or the
|
|
97
|
+
# ablation stops being the same experiment -- and every recorded
|
|
98
|
+
# `baseline_matched` call becomes a replay miss the moment anyone adds one.
|
|
99
|
+
measured = signals.as_dict()
|
|
100
|
+
parts += [f" {k}: {measured[k]}" for k in RECORDED_SIGNAL_FIELDS]
|
|
101
|
+
parts += ["", "README:", (readme or "(none at the cutoff)")[:6000], ""]
|
|
102
|
+
parts += ["Pull request threads:", ""]
|
|
103
|
+
parts += [_render_thread(t) for t in talkative]
|
|
104
|
+
|
|
105
|
+
result = model.complete(
|
|
106
|
+
label="baseline_matched", system=SYSTEM, prompt="\n".join(parts), schema=SCHEMA
|
|
107
|
+
)
|
|
108
|
+
return Assessment(
|
|
109
|
+
repo=repo,
|
|
110
|
+
verdict=Verdict(result["verdict"]),
|
|
111
|
+
summary=result["summary"],
|
|
112
|
+
claims=[Claim(text=r, evidence_id=None) for r in result.get("reasons", [])],
|
|
113
|
+
method="baseline, evidence-matched (one prompt, the same signals and threads Holt reads)",
|
|
114
|
+
replayed=model.replayed,
|
|
115
|
+
models=list(model.usage.models),
|
|
116
|
+
)
|