holt-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- holt/__init__.py +0 -0
- holt/agent/__init__.py +0 -0
- holt/agent/entry.py +86 -0
- holt/agent/findings.py +49 -0
- holt/agent/landing.py +154 -0
- holt/agent/pipeline.py +244 -0
- holt/agent/progression.py +408 -0
- holt/agent/signals.py +220 -0
- holt/agent/stages.py +533 -0
- holt/agent/verdict.py +226 -0
- holt/agent/verify.py +140 -0
- holt/baseline.py +89 -0
- holt/baseline_matched.py +116 -0
- holt/cli.py +616 -0
- holt/discover.py +497 -0
- holt/evidence/__init__.py +3 -0
- holt/evidence/fixtures.py +154 -0
- holt/evidence/github_graphql.py +538 -0
- holt/evidence/provider.py +77 -0
- holt/evidence/redact.py +79 -0
- holt/issues.py +41 -0
- holt/model.py +516 -0
- holt/profile.py +126 -0
- holt/report.py +157 -0
- holt/tui/__init__.py +0 -0
- holt/tui/animation.py +84 -0
- holt/tui/app.py +294 -0
- holt/tui/clipboard.py +89 -0
- holt/tui/commands.py +134 -0
- holt/tui/discovery.py +305 -0
- holt/tui/env.py +49 -0
- holt/tui/events.py +245 -0
- holt/tui/mascot.py +121 -0
- holt/tui/models.py +590 -0
- holt/tui/observe.py +297 -0
- holt/tui/screens/__init__.py +35 -0
- holt/tui/screens/assessment.py +337 -0
- holt/tui/screens/confirm.py +62 -0
- holt/tui/screens/discover.py +444 -0
- holt/tui/screens/home.py +519 -0
- holt/tui/screens/inspector.py +106 -0
- holt/tui/screens/live.py +335 -0
- holt/tui/screens/models.py +393 -0
- holt/tui/screens/next_steps.py +425 -0
- holt/tui/screens/profile.py +129 -0
- holt/tui/session.py +711 -0
- holt/tui/store.py +458 -0
- holt/tui/theme.py +479 -0
- holt/tui/visual.py +33 -0
- holt/tui/widgets/__init__.py +0 -0
- holt/tui/widgets/candidates.py +78 -0
- holt/tui/widgets/claims.py +59 -0
- holt/tui/widgets/disclosure.py +121 -0
- holt/tui/widgets/evidence.py +121 -0
- holt/tui/widgets/masthead.py +122 -0
- holt/tui/widgets/recent.py +167 -0
- holt/tui/widgets/scrolling.py +38 -0
- holt/tui/widgets/stages.py +232 -0
- holt/types.py +48 -0
- holt_cli-0.1.0.dist-info/METADATA +198 -0
- holt_cli-0.1.0.dist-info/RECORD +65 -0
- holt_cli-0.1.0.dist-info/WHEEL +4 -0
- holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
- holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
- holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/agent/stages.py
ADDED
|
@@ -0,0 +1,533 @@
|
|
|
1
|
+
"""The stages that need a model.
|
|
2
|
+
|
|
3
|
+
Each returns typed findings carrying evidence ids. Nothing here decides a
|
|
4
|
+
verdict: that is verdict.py, and it runs no model. What these stages do is turn
|
|
5
|
+
evidence into fields a rule can act on.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import random
|
|
11
|
+
from collections.abc import Iterable
|
|
12
|
+
|
|
13
|
+
from holt.agent.findings import Findings
|
|
14
|
+
from holt.agent.signals import Thread
|
|
15
|
+
from holt.model import ModelClient
|
|
16
|
+
from holt.types import EvidenceRecord
|
|
17
|
+
|
|
18
|
+
REPO_KINDS = [
|
|
19
|
+
"real_software",
|
|
20
|
+
"registry",
|
|
21
|
+
"awesome_list",
|
|
22
|
+
"portfolio",
|
|
23
|
+
"course_material",
|
|
24
|
+
"docs",
|
|
25
|
+
"mirror",
|
|
26
|
+
"unclear",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
CLASSIFY_SYSTEM = """You identify what kind of GitHub repository you are looking at.
|
|
30
|
+
|
|
31
|
+
The distinction that matters is what a merged pull request *is* here:
|
|
32
|
+
|
|
33
|
+
real_software changes to code that runs: features, fixes, refactors
|
|
34
|
+
registry entries in a catalogue -- package manifests, domain records,
|
|
35
|
+
plugin listings, adapter stubs. Merges are easy and frequent
|
|
36
|
+
and change no software.
|
|
37
|
+
awesome_list a curated list of links
|
|
38
|
+
portfolio someone's personal work, coursework, or a collection of demos
|
|
39
|
+
course_material exercises or teaching material
|
|
40
|
+
docs a documentation site
|
|
41
|
+
mirror a read-only copy of a project developed elsewhere
|
|
42
|
+
unclear the evidence does not settle it
|
|
43
|
+
|
|
44
|
+
Registries are the common trap: they look extremely healthy on every activity
|
|
45
|
+
metric precisely because contributing to them is trivial. Judge by what the
|
|
46
|
+
merged diffs touch, not by how many there are.
|
|
47
|
+
|
|
48
|
+
Cite evidence ids for what you claim. Only cite ids you were given. If the
|
|
49
|
+
evidence does not settle the question, answer unclear rather than guessing."""
|
|
50
|
+
|
|
51
|
+
CLASSIFY_SCHEMA = {
|
|
52
|
+
"type": "object",
|
|
53
|
+
"properties": {
|
|
54
|
+
"repo_kind": {"type": "string", "enum": REPO_KINDS},
|
|
55
|
+
"confidence": {"type": "string", "enum": ["high", "medium", "low"]},
|
|
56
|
+
"rationale": {"type": "string"},
|
|
57
|
+
"evidence_ids": {"type": "array", "items": {"type": "string"}},
|
|
58
|
+
"governance_flags": {
|
|
59
|
+
"type": "array",
|
|
60
|
+
"items": {
|
|
61
|
+
"type": "string",
|
|
62
|
+
"enum": ["cla_required", "corporate_controlled", "read_only", "none"],
|
|
63
|
+
},
|
|
64
|
+
},
|
|
65
|
+
},
|
|
66
|
+
"required": ["repo_kind", "confidence", "rationale", "evidence_ids", "governance_flags"],
|
|
67
|
+
"additionalProperties": False,
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _doc(records: Iterable[EvidenceRecord], suffix: str) -> tuple[str, str] | None:
|
|
72
|
+
for r in records:
|
|
73
|
+
if r.evidence_id.endswith(suffix):
|
|
74
|
+
return r.payload.get("text", ""), r.evidence_id
|
|
75
|
+
return None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _merged_path_sample(threads: dict[str, Thread], n: int = 25) -> list[tuple[str, list[str]]]:
|
|
79
|
+
"""What merged contributions actually touched -- the registry tell."""
|
|
80
|
+
merged = [t for t in threads.values() if t.merged and t.files]
|
|
81
|
+
rng = random.Random(0)
|
|
82
|
+
sample = merged if len(merged) <= n else rng.sample(merged, n)
|
|
83
|
+
return [(t.key, t.files[:4]) for t in sorted(sample, key=lambda t: t.key)]
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def classify(
|
|
87
|
+
repo: str,
|
|
88
|
+
records: list[EvidenceRecord],
|
|
89
|
+
threads: dict[str, Thread],
|
|
90
|
+
model: ModelClient,
|
|
91
|
+
findings: Findings,
|
|
92
|
+
) -> None:
|
|
93
|
+
meta = next((r for r in records if r.evidence_id.endswith(":meta")), None)
|
|
94
|
+
readme = _doc(records, ":readme")
|
|
95
|
+
contributing = _doc(records, ":contributing")
|
|
96
|
+
paths = _merged_path_sample(threads)
|
|
97
|
+
|
|
98
|
+
parts = [f"Repository: {repo}", ""]
|
|
99
|
+
if meta:
|
|
100
|
+
p = meta.payload
|
|
101
|
+
parts += [
|
|
102
|
+
f"Metadata (evidence id: {meta.evidence_id})",
|
|
103
|
+
f" description: {p.get('description')!r}",
|
|
104
|
+
f" primary language: {p.get('primary_language')!r}",
|
|
105
|
+
f" homepage: {p.get('homepage_url')!r}",
|
|
106
|
+
f" archived: {p.get('is_archived')} fork: {p.get('is_fork')} mirror: {p.get('is_mirror')}",
|
|
107
|
+
"",
|
|
108
|
+
]
|
|
109
|
+
if readme:
|
|
110
|
+
parts += [f"README (evidence id: {readme[1]})", readme[0][:6000], ""]
|
|
111
|
+
if contributing:
|
|
112
|
+
parts += [f"CONTRIBUTING (evidence id: {contributing[1]})", contributing[0][:3000], ""]
|
|
113
|
+
if paths:
|
|
114
|
+
parts += ["Files touched by merged pull requests (evidence ids shown):"]
|
|
115
|
+
parts += [f" {key} {files}" for key, files in paths]
|
|
116
|
+
else:
|
|
117
|
+
parts += ["No merged pull requests with file information were available."]
|
|
118
|
+
|
|
119
|
+
result = model.complete(
|
|
120
|
+
label="classify",
|
|
121
|
+
system=CLASSIFY_SYSTEM,
|
|
122
|
+
prompt="\n".join(parts),
|
|
123
|
+
schema=CLASSIFY_SCHEMA,
|
|
124
|
+
)
|
|
125
|
+
# Stage A cites pull requests too, and shortens them the same way Stage C
|
|
126
|
+
# does. Normalising here as well means a real citation is not thrown away
|
|
127
|
+
# for being written in the wrong shape.
|
|
128
|
+
cited = tuple(normalise_citation(repo, e) for e in result.get("evidence_ids", ()))
|
|
129
|
+
findings.add(
|
|
130
|
+
"repo_kind",
|
|
131
|
+
result["repo_kind"],
|
|
132
|
+
evidence_ids=cited,
|
|
133
|
+
note=result.get("rationale", ""),
|
|
134
|
+
)
|
|
135
|
+
flags = [f for f in result.get("governance_flags", []) if f != "none"]
|
|
136
|
+
if flags:
|
|
137
|
+
findings.add("governance_flags", flags, evidence_ids=cited)
|
|
138
|
+
if meta is not None and meta.payload.get("is_archived"):
|
|
139
|
+
findings.add("is_archived", True, evidence_ids=(meta.evidence_id,))
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
OUTCOMES_SYSTEM = """You read pull request threads and judge what each one reveals
|
|
143
|
+
about an outsider's chances of landing meaningful work in this repository.
|
|
144
|
+
|
|
145
|
+
This is not sentiment. A polite refusal and an impatient acceptance point in
|
|
146
|
+
opposite directions from how they sound. Judge the path the contributor was left
|
|
147
|
+
on, not the tone of the words.
|
|
148
|
+
|
|
149
|
+
Two cases that are easy to get backwards:
|
|
150
|
+
|
|
151
|
+
"Thanks for taking the time. We're moving this into the new architecture, so
|
|
152
|
+
closing." -- warm words, but the contributor is told this class of work is not
|
|
153
|
+
wanted. That is discouraging.
|
|
154
|
+
|
|
155
|
+
"This isn't right yet. Change X and Y and I'll merge it." -- a rejection at
|
|
156
|
+
this moment, and strong evidence of a working contribution process. That is
|
|
157
|
+
welcoming.
|
|
158
|
+
|
|
159
|
+
Outcomes:
|
|
160
|
+
Cite the exact evidence id shown for each thread, in full, including the
|
|
161
|
+
":opened" suffix. Do not abbreviate it to a number.
|
|
162
|
+
|
|
163
|
+
merged_after_review merged, with substantive human feedback on the way
|
|
164
|
+
merged_without_engagement merged, nobody said anything of substance
|
|
165
|
+
changes_requested not merged yet, but a maintainer gave a route in
|
|
166
|
+
closed_with_guidance closed, and the contributor was told where to go instead
|
|
167
|
+
closed_dismissive closed with no route forward
|
|
168
|
+
ignored nobody replied at all
|
|
169
|
+
|
|
170
|
+
Signal is what the thread tells a prospective contributor: welcoming, neutral,
|
|
171
|
+
or discouraging.
|
|
172
|
+
|
|
173
|
+
Quote the words you judged from, verbatim and short, copied exactly from the
|
|
174
|
+
thread. If a thread shows NO_REPLIES there is nothing to quote: return an empty
|
|
175
|
+
quote rather than describing the silence. Never quote the scaffolding around the
|
|
176
|
+
thread -- only what a person actually wrote. Cite only pull request ids you were
|
|
177
|
+
given."""
|
|
178
|
+
|
|
179
|
+
OUTCOMES_SCHEMA = {
|
|
180
|
+
"type": "object",
|
|
181
|
+
"properties": {
|
|
182
|
+
"threads": {
|
|
183
|
+
"type": "array",
|
|
184
|
+
"items": {
|
|
185
|
+
"type": "object",
|
|
186
|
+
"properties": {
|
|
187
|
+
"pr_id": {"type": "string"},
|
|
188
|
+
"outcome": {
|
|
189
|
+
"type": "string",
|
|
190
|
+
"enum": [
|
|
191
|
+
"merged_after_review",
|
|
192
|
+
"merged_without_engagement",
|
|
193
|
+
"changes_requested",
|
|
194
|
+
"closed_with_guidance",
|
|
195
|
+
"closed_dismissive",
|
|
196
|
+
"ignored",
|
|
197
|
+
],
|
|
198
|
+
},
|
|
199
|
+
"signal": {
|
|
200
|
+
"type": "string",
|
|
201
|
+
"enum": ["welcoming", "neutral", "discouraging"],
|
|
202
|
+
},
|
|
203
|
+
"quote": {"type": "string"},
|
|
204
|
+
},
|
|
205
|
+
"required": ["pr_id", "outcome", "signal", "quote"],
|
|
206
|
+
"additionalProperties": False,
|
|
207
|
+
},
|
|
208
|
+
},
|
|
209
|
+
"posture": {"type": "string", "enum": ["welcoming", "mixed", "discouraging", "absent"]},
|
|
210
|
+
"posture_rationale": {"type": "string"},
|
|
211
|
+
},
|
|
212
|
+
"required": ["threads", "posture", "posture_rationale"],
|
|
213
|
+
"additionalProperties": False,
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def cite_id(thread_key: str) -> str:
|
|
218
|
+
"""A thread key is not itself an evidence id -- only its events are.
|
|
219
|
+
|
|
220
|
+
Stage C reasons about whole threads, but the provider holds `#12:opened`,
|
|
221
|
+
`#12:merged` and so on. Citing the bare key would make every Stage C finding
|
|
222
|
+
unresolvable and Stage D would correctly delete the entire stage's output.
|
|
223
|
+
"""
|
|
224
|
+
return f"{thread_key}:opened"
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def normalise_citation(repo: str, cited: str) -> str:
|
|
228
|
+
"""Repair the shapes a model reaches for when asked to quote an id.
|
|
229
|
+
|
|
230
|
+
Models shorten. Given `pr:owner/name#381843:opened` they will often answer
|
|
231
|
+
`381843`. Repairing the format is not the same as excusing the claim: the
|
|
232
|
+
repaired id is still resolved against real evidence, and still dropped if
|
|
233
|
+
nothing is there.
|
|
234
|
+
"""
|
|
235
|
+
cited = (cited or "").strip()
|
|
236
|
+
if cited.isdigit():
|
|
237
|
+
return f"pr:{repo}#{cited}:opened"
|
|
238
|
+
if cited.startswith("pr:") and cited.count(":") == 1:
|
|
239
|
+
return f"{cited}:opened"
|
|
240
|
+
if cited.startswith("#") and cited[1:].isdigit():
|
|
241
|
+
return f"pr:{repo}#{cited[1:]}:opened"
|
|
242
|
+
return cited
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _render_thread(t: Thread) -> str:
|
|
246
|
+
state = "merged" if t.merged else "closed unmerged" if t.closed_unmerged else "open"
|
|
247
|
+
lines = [
|
|
248
|
+
f"--- evidence id: {cite_id(t.key)} ({state})",
|
|
249
|
+
f" opened by {t.author}; {t.changed_files} files, +{t.additions}/-{t.deletions}",
|
|
250
|
+
f" files: {t.files[:4]}",
|
|
251
|
+
]
|
|
252
|
+
if not t.responses:
|
|
253
|
+
# Deliberately not a quotable sentence. The previous wording read like
|
|
254
|
+
# thread content and the model quoted it back as evidence, which the
|
|
255
|
+
# evidence-integrity check caught: 80 of 528 quotes were this scaffold.
|
|
256
|
+
lines.append(" NO_REPLIES")
|
|
257
|
+
for when, who, body in sorted(t.responses)[:6]:
|
|
258
|
+
speaker = "AUTHOR" if who == t.author else who
|
|
259
|
+
lines.append(f" [{speaker}] {' '.join((body or '').split())[:600]}")
|
|
260
|
+
return "\n".join(lines)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def read_outcomes(
|
|
264
|
+
repo: str,
|
|
265
|
+
threads: dict[str, Thread],
|
|
266
|
+
model: ModelClient,
|
|
267
|
+
findings: Findings,
|
|
268
|
+
sample: int = 12,
|
|
269
|
+
) -> None:
|
|
270
|
+
"""Read the threads with the most conversation -- silence is already counted."""
|
|
271
|
+
talkative = sorted(
|
|
272
|
+
(t for t in threads.values() if not t.author_is_bot),
|
|
273
|
+
key=lambda t: (len(t.responses), t.additions + t.deletions),
|
|
274
|
+
reverse=True,
|
|
275
|
+
)[:sample]
|
|
276
|
+
if not talkative:
|
|
277
|
+
findings.add("outsider_posture", "absent", note="no threads available to read")
|
|
278
|
+
return
|
|
279
|
+
|
|
280
|
+
prompt = "\n".join(
|
|
281
|
+
[f"Repository: {repo}", "", "Pull request threads:", ""]
|
|
282
|
+
+ [_render_thread(t) for t in talkative]
|
|
283
|
+
)
|
|
284
|
+
result = model.complete(
|
|
285
|
+
label="outcomes",
|
|
286
|
+
system=OUTCOMES_SYSTEM,
|
|
287
|
+
prompt=prompt,
|
|
288
|
+
schema=OUTCOMES_SCHEMA,
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
per_thread = result.get("threads", [])
|
|
292
|
+
findings.add(
|
|
293
|
+
"outsider_posture",
|
|
294
|
+
result["posture"],
|
|
295
|
+
evidence_ids=tuple(normalise_citation(repo, t["pr_id"]) for t in per_thread),
|
|
296
|
+
note=result.get("posture_rationale", ""),
|
|
297
|
+
)
|
|
298
|
+
for entry in per_thread:
|
|
299
|
+
findings.add(
|
|
300
|
+
"thread_outcome",
|
|
301
|
+
{"outcome": entry["outcome"], "signal": entry["signal"], "quote": entry["quote"]},
|
|
302
|
+
evidence_ids=(normalise_citation(repo, entry["pr_id"]),),
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
OPPORTUNITY_SYSTEM = """You judge whether a repository offers an outsider a real
|
|
307
|
+
route in, using its own onboarding material.
|
|
308
|
+
|
|
309
|
+
What counts as a real route: a documented setup that someone could follow, a
|
|
310
|
+
described process for proposing work, named places where help is wanted, some
|
|
311
|
+
indication of who to ask. What does not: a CONTRIBUTING file that only restates
|
|
312
|
+
a code of conduct, a README that is purely marketing, or instructions that
|
|
313
|
+
assume commit access.
|
|
314
|
+
|
|
315
|
+
Answer from the material given. If there is no onboarding material at all, say
|
|
316
|
+
so rather than inferring from the project's fame. Cite only evidence ids you
|
|
317
|
+
were given."""
|
|
318
|
+
|
|
319
|
+
OPPORTUNITY_SCHEMA = {
|
|
320
|
+
"type": "object",
|
|
321
|
+
"properties": {
|
|
322
|
+
"onboarding": {
|
|
323
|
+
"type": "string",
|
|
324
|
+
"enum": ["substantive", "boilerplate", "absent", "assumes_insider"],
|
|
325
|
+
},
|
|
326
|
+
"rationale": {"type": "string"},
|
|
327
|
+
"evidence_ids": {"type": "array", "items": {"type": "string"}},
|
|
328
|
+
},
|
|
329
|
+
"required": ["onboarding", "rationale", "evidence_ids"],
|
|
330
|
+
"additionalProperties": False,
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def assess_opportunity(
|
|
335
|
+
repo: str, records: list[EvidenceRecord], model: ModelClient, findings: Findings
|
|
336
|
+
) -> None:
|
|
337
|
+
readme = _doc(records, ":readme")
|
|
338
|
+
contributing = _doc(records, ":contributing")
|
|
339
|
+
parts = [f"Repository: {repo}", ""]
|
|
340
|
+
if contributing:
|
|
341
|
+
parts += [f"CONTRIBUTING (evidence id: {contributing[1]})", contributing[0][:6000], ""]
|
|
342
|
+
else:
|
|
343
|
+
parts += ["No CONTRIBUTING file was present at the cutoff.", ""]
|
|
344
|
+
if readme:
|
|
345
|
+
parts += [f"README (evidence id: {readme[1]})", readme[0][:4000]]
|
|
346
|
+
|
|
347
|
+
result = model.complete(
|
|
348
|
+
label="opportunity",
|
|
349
|
+
system=OPPORTUNITY_SYSTEM,
|
|
350
|
+
prompt="\n".join(parts),
|
|
351
|
+
schema=OPPORTUNITY_SCHEMA,
|
|
352
|
+
)
|
|
353
|
+
findings.add(
|
|
354
|
+
"onboarding",
|
|
355
|
+
result["onboarding"],
|
|
356
|
+
evidence_ids=tuple(
|
|
357
|
+
normalise_citation(repo, e) for e in result.get("evidence_ids", ())
|
|
358
|
+
),
|
|
359
|
+
note=result.get("rationale", ""),
|
|
360
|
+
)
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
NARRATE_SYSTEM = """You write the assessment a careful contributor would leave
|
|
364
|
+
after an afternoon reading a repository's pull requests, for someone deciding
|
|
365
|
+
where to spend a limited number of days. You are not told how many days they
|
|
366
|
+
have; say what the repository is like, not how long it would take them.
|
|
367
|
+
|
|
368
|
+
You are given a verdict that has already been decided. You do not revisit it,
|
|
369
|
+
soften it, or argue with it -- you explain what it rests on.
|
|
370
|
+
|
|
371
|
+
Three parts, each doing a different job:
|
|
372
|
+
|
|
373
|
+
**bottom_line** -- two sentences at most, addressed to the reader as "you", and
|
|
374
|
+
concrete about what they would be walking into. Not a restatement of the verdict
|
|
375
|
+
word. "Your pull request will get a reply within a day, but half of newcomers
|
|
376
|
+
here never got one" beats "this repository appears welcoming".
|
|
377
|
+
|
|
378
|
+
**what_the_evidence_shows** -- plain prose, no headings and no bullet lists, the
|
|
379
|
+
way you would write to a colleague. **Two or three short paragraphs, separated by
|
|
380
|
+
a blank line, and under 200 words in total.** One unbroken block is something a
|
|
381
|
+
reader has to mine rather than read. Lead with the fact that mattered most rather
|
|
382
|
+
than with a summary of everything, and leave out anything that does not change
|
|
383
|
+
the decision. Quote a maintainer where a quote does the work
|
|
384
|
+
better than a paraphrase. Numbers are only useful next to what they mean: "0.8
|
|
385
|
+
hours to first reply" is worth writing as fast, "63 threads ignored" is worth
|
|
386
|
+
writing as most.
|
|
387
|
+
|
|
388
|
+
**what_could_not_be_determined** -- one sentence, or empty if there is nothing
|
|
389
|
+
honest to put there. Do not invent a limitation to look rigorous, and do not
|
|
390
|
+
list something the evidence in front of you actually settles.
|
|
391
|
+
|
|
392
|
+
Never say a contribution will be accepted. Never describe work as easy. Where the
|
|
393
|
+
evidence is thin, say it is thin -- "I could not determine this" is a better
|
|
394
|
+
sentence than a confident one that outruns what was read.
|
|
395
|
+
|
|
396
|
+
Never use the word "cutoff" -- it is internal jargon and means nothing to the
|
|
397
|
+
reader. The measurements you are given describe the window of history that was
|
|
398
|
+
read; say "in the period read", "in this sample", or name no window at all."""
|
|
399
|
+
|
|
400
|
+
# Three fields rather than one blob. The old single `summary` produced a
|
|
401
|
+
# 250-word paragraph that a reader had to mine for the decision, which is the
|
|
402
|
+
# shape of an answer nobody proof-read.
|
|
403
|
+
NARRATE_SCHEMA = {
|
|
404
|
+
"type": "object",
|
|
405
|
+
"properties": {
|
|
406
|
+
"bottom_line": {"type": "string"},
|
|
407
|
+
"what_the_evidence_shows": {"type": "string"},
|
|
408
|
+
"what_could_not_be_determined": {"type": "string"},
|
|
409
|
+
},
|
|
410
|
+
"required": ["bottom_line", "what_the_evidence_shows", "what_could_not_be_determined"],
|
|
411
|
+
"additionalProperties": False,
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def narrate(
|
|
416
|
+
repo: str, verdict: str, trace: list[str], findings: Findings, signals_dict: dict,
|
|
417
|
+
model: ModelClient,
|
|
418
|
+
) -> dict:
|
|
419
|
+
# The contributor's day budget is deliberately absent from this prompt. It
|
|
420
|
+
# reaches the reader through the renderer's headline and through the rule
|
|
421
|
+
# trace below, both of which are computed without a model. Putting it here
|
|
422
|
+
# made the prompt vary with `--days`, which turned every non-default budget
|
|
423
|
+
# into a replay miss and quietly broke the one claim that re-answering the
|
|
424
|
+
# question costs nothing.
|
|
425
|
+
lines = [f"Repository: {repo}", f"Verdict (already decided, do not change): {verdict}", ""]
|
|
426
|
+
lines += ["Why the rules landed there:"] + [f" - {t}" for t in trace]
|
|
427
|
+
lines += ["", "Measured in the sampled window:"]
|
|
428
|
+
lines += [f" {k}: {v}" for k, v in signals_dict.items()]
|
|
429
|
+
lines += ["", "Verified findings:"]
|
|
430
|
+
for item in findings:
|
|
431
|
+
note = f" -- {item.note}" if item.note else ""
|
|
432
|
+
lines.append(f" {item.field} = {item.value}{note}")
|
|
433
|
+
return model.complete(
|
|
434
|
+
label="narrate",
|
|
435
|
+
system=NARRATE_SYSTEM,
|
|
436
|
+
prompt="\n".join(lines),
|
|
437
|
+
schema=NARRATE_SCHEMA,
|
|
438
|
+
)
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
PATHFINDER_SYSTEM = """You are helping an outside developer -- someone with no
|
|
442
|
+
prior connection to a project -- choose which open issue to attempt first.
|
|
443
|
+
|
|
444
|
+
You are given issues that were open at a fixed point in time, and evidence about
|
|
445
|
+
how the project treated outside contributions before that point.
|
|
446
|
+
|
|
447
|
+
Rank the issues by one thing only: **how likely is it that an outsider, starting
|
|
448
|
+
from nothing, lands a merged pull request resolving this issue?**
|
|
449
|
+
|
|
450
|
+
That is not the same as "which issue is most important", and it is not the same
|
|
451
|
+
as "which issue is easiest". Weigh:
|
|
452
|
+
|
|
453
|
+
* whether the issue states a concrete, bounded outcome rather than a wish
|
|
454
|
+
* whether someone could act on it without private context or a design decision
|
|
455
|
+
only a maintainer can make
|
|
456
|
+
* whether the report contains enough to reproduce or locate the problem
|
|
457
|
+
* whether the project's history suggests work of this shape gets merged
|
|
458
|
+
|
|
459
|
+
An issue labelled for beginners is not automatically a good entry point; many
|
|
460
|
+
are aspirational one-liners nobody has scoped. Judge the text, not the label.
|
|
461
|
+
|
|
462
|
+
Return at most five, best first. For each, say in one sentence what the person
|
|
463
|
+
would actually do, and cite the issue's evidence id."""
|
|
464
|
+
|
|
465
|
+
PATHFINDER_SCHEMA = {
|
|
466
|
+
"type": "object",
|
|
467
|
+
"properties": {
|
|
468
|
+
"ranked": {
|
|
469
|
+
"type": "array",
|
|
470
|
+
"items": {
|
|
471
|
+
"type": "object",
|
|
472
|
+
"properties": {
|
|
473
|
+
"evidence_id": {"type": "string"},
|
|
474
|
+
"first_step": {"type": "string"},
|
|
475
|
+
"why": {"type": "string"},
|
|
476
|
+
},
|
|
477
|
+
"required": ["evidence_id", "first_step", "why"],
|
|
478
|
+
"additionalProperties": False,
|
|
479
|
+
},
|
|
480
|
+
}
|
|
481
|
+
},
|
|
482
|
+
"required": ["ranked"],
|
|
483
|
+
"additionalProperties": False,
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
MAX_ISSUES_SHOWN = 40
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def _render_issue(record: EvidenceRecord) -> str:
|
|
490
|
+
p = record.payload
|
|
491
|
+
body = " ".join((p.get("body") or "").split())[:700]
|
|
492
|
+
return "\n".join([
|
|
493
|
+
f"--- evidence id: {record.evidence_id}",
|
|
494
|
+
f" opened {record.timestamp.date()} by {p.get('author')}; "
|
|
495
|
+
f"{p.get('comments', 0)} comments; labels: {p.get('labels') or 'none'}",
|
|
496
|
+
f" title: {p.get('title')}",
|
|
497
|
+
f" {body or '(no description)'}",
|
|
498
|
+
])
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def find_paths(
|
|
502
|
+
repo: str,
|
|
503
|
+
issues: list[EvidenceRecord],
|
|
504
|
+
signals_summary: dict,
|
|
505
|
+
model: ModelClient,
|
|
506
|
+
) -> list[dict]:
|
|
507
|
+
"""Rank candidate issues. Returns [] when there is nothing worth ranking.
|
|
508
|
+
|
|
509
|
+
Shipped, and shipped losing. It does not beat the comparators it was
|
|
510
|
+
pre-registered against, and `holt.agent.entry` prints that result in the same
|
|
511
|
+
output as the ranking rather than filing it in a document. Call through
|
|
512
|
+
`entry.rank` rather than directly: it is what both the CLI and
|
|
513
|
+
`eval/pathfinder_harness.py` use, which is the only reason the published
|
|
514
|
+
precision describes something a user can actually run.
|
|
515
|
+
"""
|
|
516
|
+
if not issues:
|
|
517
|
+
return []
|
|
518
|
+
# Most-discussed first: an issue nobody has said anything about is usually
|
|
519
|
+
# unscoped, and the sample has to fit in one call.
|
|
520
|
+
shown = sorted(
|
|
521
|
+
issues, key=lambda r: r.payload.get("comments", 0), reverse=True
|
|
522
|
+
)[:MAX_ISSUES_SHOWN]
|
|
523
|
+
|
|
524
|
+
prompt = "\n".join(
|
|
525
|
+
[f"Repository: {repo}", "", "How this project treated outsiders before the cutoff:"]
|
|
526
|
+
+ [f" {k}: {v}" for k, v in signals_summary.items()]
|
|
527
|
+
+ ["", f"Issues open at the cutoff ({len(shown)} of {len(issues)} shown):", ""]
|
|
528
|
+
+ [_render_issue(r) for r in shown]
|
|
529
|
+
)
|
|
530
|
+
return model.complete(
|
|
531
|
+
label="pathfinder", system=PATHFINDER_SYSTEM, prompt=prompt,
|
|
532
|
+
schema=PATHFINDER_SCHEMA,
|
|
533
|
+
)["ranked"]
|