holt-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- holt/__init__.py +0 -0
- holt/agent/__init__.py +0 -0
- holt/agent/entry.py +86 -0
- holt/agent/findings.py +49 -0
- holt/agent/landing.py +154 -0
- holt/agent/pipeline.py +244 -0
- holt/agent/progression.py +408 -0
- holt/agent/signals.py +220 -0
- holt/agent/stages.py +533 -0
- holt/agent/verdict.py +226 -0
- holt/agent/verify.py +140 -0
- holt/baseline.py +89 -0
- holt/baseline_matched.py +116 -0
- holt/cli.py +616 -0
- holt/discover.py +497 -0
- holt/evidence/__init__.py +3 -0
- holt/evidence/fixtures.py +154 -0
- holt/evidence/github_graphql.py +538 -0
- holt/evidence/provider.py +77 -0
- holt/evidence/redact.py +79 -0
- holt/issues.py +41 -0
- holt/model.py +516 -0
- holt/profile.py +126 -0
- holt/report.py +157 -0
- holt/tui/__init__.py +0 -0
- holt/tui/animation.py +84 -0
- holt/tui/app.py +294 -0
- holt/tui/clipboard.py +89 -0
- holt/tui/commands.py +134 -0
- holt/tui/discovery.py +305 -0
- holt/tui/env.py +49 -0
- holt/tui/events.py +245 -0
- holt/tui/mascot.py +121 -0
- holt/tui/models.py +590 -0
- holt/tui/observe.py +297 -0
- holt/tui/screens/__init__.py +35 -0
- holt/tui/screens/assessment.py +337 -0
- holt/tui/screens/confirm.py +62 -0
- holt/tui/screens/discover.py +444 -0
- holt/tui/screens/home.py +519 -0
- holt/tui/screens/inspector.py +106 -0
- holt/tui/screens/live.py +335 -0
- holt/tui/screens/models.py +393 -0
- holt/tui/screens/next_steps.py +425 -0
- holt/tui/screens/profile.py +129 -0
- holt/tui/session.py +711 -0
- holt/tui/store.py +458 -0
- holt/tui/theme.py +479 -0
- holt/tui/visual.py +33 -0
- holt/tui/widgets/__init__.py +0 -0
- holt/tui/widgets/candidates.py +78 -0
- holt/tui/widgets/claims.py +59 -0
- holt/tui/widgets/disclosure.py +121 -0
- holt/tui/widgets/evidence.py +121 -0
- holt/tui/widgets/masthead.py +122 -0
- holt/tui/widgets/recent.py +167 -0
- holt/tui/widgets/scrolling.py +38 -0
- holt/tui/widgets/stages.py +232 -0
- holt/types.py +48 -0
- holt_cli-0.1.0.dist-info/METADATA +198 -0
- holt_cli-0.1.0.dist-info/RECORD +65 -0
- holt_cli-0.1.0.dist-info/WHEEL +4 -0
- holt_cli-0.1.0.dist-info/entry_points.txt +2 -0
- holt_cli-0.1.0.dist-info/licenses/LICENSE +201 -0
- holt_cli-0.1.0.dist-info/licenses/NOTICE +4 -0
holt/__init__.py
ADDED
|
File without changes
|
holt/agent/__init__.py
ADDED
|
File without changes
|
holt/agent/entry.py
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Entry points: which open issue an outsider should attempt first.
|
|
2
|
+
|
|
3
|
+
**Not shipped. Kept as the prototype that showed what not to build.**
|
|
4
|
+
|
|
5
|
+
It was designed, pre-registered and then measured against the comparators that
|
|
6
|
+
could make it unnecessary (`eval/PATHFINDER-DESIGN.md`, written before any of it
|
|
7
|
+
existed). It did not beat them. The measurement is a module constant here, not a
|
|
8
|
+
line in a document the reader will never open, because a ranking whose evaluation
|
|
9
|
+
lives only in the README is an unsupported ranking with the caveat filed where
|
|
10
|
+
nobody looks.
|
|
11
|
+
|
|
12
|
+
**Why it failed is more useful than the fact that it did.** This ranker never
|
|
13
|
+
sees the contributor. It produces one ranking for everybody, which means it is
|
|
14
|
+
answering "which issues here are generally approachable" -- exactly what a
|
|
15
|
+
`good first issue` label already encodes. It did not lose because the model is
|
|
16
|
+
weak. It lost because it was solving the label's problem, and a tie with the
|
|
17
|
+
label is the expected outcome of that.
|
|
18
|
+
|
|
19
|
+
Available behind `holt analyze --entry-points`, off by default, and it still
|
|
20
|
+
prints its own measurement when asked for. The successor question -- which issue
|
|
21
|
+
is a sensible next step *for this person, given what they have already merged
|
|
22
|
+
here* -- is a different question, and one no label answers.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from holt.agent.signals import build_threads, compute
|
|
28
|
+
from holt.agent.stages import find_paths
|
|
29
|
+
from holt.issues import open_at_cutoff
|
|
30
|
+
from holt.model import ModelClient
|
|
31
|
+
from holt.types import EvidenceRecord
|
|
32
|
+
|
|
33
|
+
# The signals the ranker is shown. Kept as a constant because the evaluation and
|
|
34
|
+
# the shipped path must build byte-identical prompts, or the recorded trajectories
|
|
35
|
+
# stop replaying and the published number stops describing what users run.
|
|
36
|
+
RANKER_SIGNALS = ("outsider_merged", "outsider_threads", "median_first_response_hours")
|
|
37
|
+
|
|
38
|
+
#: Measured on both pools. Regenerate with:
|
|
39
|
+
#: uv run python eval/pathfinder_harness.py --replay
|
|
40
|
+
#: uv run python eval/pathfinder_harness.py --replay --pool eval/pool2.json \
|
|
41
|
+
#: --labels eval/results_labels_pool2.json
|
|
42
|
+
MEASURED = {
|
|
43
|
+
"repositories": 25,
|
|
44
|
+
"issues_ranked": 3613,
|
|
45
|
+
"precision_at_3": {"holt": 0.173, "good_first_issue": 0.187, "recency": 0.160, "random": 0.151},
|
|
46
|
+
"paired_ci_vs_label": (-0.133, 0.120),
|
|
47
|
+
"sign_test_p": 0.51,
|
|
48
|
+
"repos_with_no_labelled_issue": 13,
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
DISCLAIMER = (
|
|
52
|
+
"> **This ranking is not measurably better than picking at random.** Measured\n"
|
|
53
|
+
"> over 25 repositories and 3,613 issues held out before the cutoff:\n"
|
|
54
|
+
"> precision@3 was **0.173** for this ranking, **0.187** for GitHub's\n"
|
|
55
|
+
"> `good first issue` label and **0.151** for a random pick — differences well\n"
|
|
56
|
+
"> inside noise (paired 95% CI [−0.13, +0.12], sign test p = 0.51).\n"
|
|
57
|
+
">\n"
|
|
58
|
+
"> It is printed anyway because **13 of those 25 repositories had no\n"
|
|
59
|
+
"> beginner-labelled issue at all**, so on half of them there is no free signal\n"
|
|
60
|
+
"> to lose to. Read it as a reading order, not a recommendation.\n"
|
|
61
|
+
">\n"
|
|
62
|
+
"> Check it yourself: `uv run python eval/pathfinder_harness.py --replay`"
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def rank(
|
|
67
|
+
repo: str,
|
|
68
|
+
issue_records: list[EvidenceRecord],
|
|
69
|
+
pull_records: list[EvidenceRecord],
|
|
70
|
+
model: ModelClient,
|
|
71
|
+
) -> list[dict]:
|
|
72
|
+
"""Rank the issues open at the cutoff. `[]` when there are none.
|
|
73
|
+
|
|
74
|
+
Called by both `eval/pathfinder_harness.py` and the CLI, so the ranking a
|
|
75
|
+
user sees is produced by the same code path as the ranking that was scored.
|
|
76
|
+
"""
|
|
77
|
+
candidates = open_at_cutoff(issue_records)
|
|
78
|
+
if not candidates:
|
|
79
|
+
return []
|
|
80
|
+
signals = compute(build_threads(pull_records)).as_dict()
|
|
81
|
+
return find_paths(
|
|
82
|
+
repo,
|
|
83
|
+
list(candidates.values()),
|
|
84
|
+
{k: signals[k] for k in RANKER_SIGNALS},
|
|
85
|
+
model,
|
|
86
|
+
)
|
holt/agent/findings.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Typed findings: what the stages produce, before anything becomes a verdict."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True, slots=True)
|
|
10
|
+
class Finding:
|
|
11
|
+
"""One field, its value, and the evidence that supports it.
|
|
12
|
+
|
|
13
|
+
A finding with no resolvable evidence is dropped by Stage D rather than
|
|
14
|
+
softened, so `evidence_ids` is what keeps a claim alive.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
field: str
|
|
18
|
+
value: Any
|
|
19
|
+
evidence_ids: tuple[str, ...] = ()
|
|
20
|
+
note: str = ""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(slots=True)
|
|
24
|
+
class Findings:
|
|
25
|
+
items: list[Finding] = field(default_factory=list)
|
|
26
|
+
|
|
27
|
+
def add(self, field_name: str, value: Any, evidence_ids=(), note: str = "") -> None:
|
|
28
|
+
self.items.append(Finding(field_name, value, tuple(evidence_ids), note))
|
|
29
|
+
|
|
30
|
+
def drop(self, field_name: str) -> None:
|
|
31
|
+
"""Remove a field entirely.
|
|
32
|
+
|
|
33
|
+
Used where two sources of evidence disagree: the standing rule is to
|
|
34
|
+
drop the contested field rather than pick a side, and a field that is
|
|
35
|
+
dropped must not survive in the claim list either.
|
|
36
|
+
"""
|
|
37
|
+
self.items = [i for i in self.items if i.field != field_name]
|
|
38
|
+
|
|
39
|
+
def get(self, field_name: str, default: Any = None) -> Any:
|
|
40
|
+
for item in self.items:
|
|
41
|
+
if item.field == field_name:
|
|
42
|
+
return item.value
|
|
43
|
+
return default
|
|
44
|
+
|
|
45
|
+
def __iter__(self):
|
|
46
|
+
return iter(self.items)
|
|
47
|
+
|
|
48
|
+
def __len__(self) -> int:
|
|
49
|
+
return len(self.items)
|
holt/agent/landing.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""Where outsider work actually landed, and where it never did.
|
|
2
|
+
|
|
3
|
+
Every pull request Holt reads carries its file list. Until now that list was used
|
|
4
|
+
for exactly one thing -- deciding whether a diff was substantial enough to count
|
|
5
|
+
-- and then discarded. This counts it instead.
|
|
6
|
+
|
|
7
|
+
The output is the sentence a newcomer most needs and cannot get anywhere:
|
|
8
|
+
*in a tree of two hundred thousand files, these three directories are where
|
|
9
|
+
strangers' work has actually been merged, and these are where strangers tried and
|
|
10
|
+
never succeeded.* GitHub does not show it, CONTRIBUTING does not say it, and it is
|
|
11
|
+
not inferable from stars, issue labels or commit frequency.
|
|
12
|
+
|
|
13
|
+
**It ranks nothing and predicts nothing.** After five capabilities cut for losing
|
|
14
|
+
to a cheap comparator, that is deliberate: there is no ordering to be beaten here,
|
|
15
|
+
only a count of what happened. Arithmetic over evidence already in hand, no model
|
|
16
|
+
call, and nothing that could disagree with the verdict.
|
|
17
|
+
|
|
18
|
+
Read `attempted but never landed` carefully -- it is a description of this
|
|
19
|
+
sample, not a prohibition. A directory can appear there because outsiders are
|
|
20
|
+
turned away from it, or because only two people ever tried.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from collections import Counter
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
|
|
28
|
+
from holt.agent.signals import Thread, newcomer_threads
|
|
29
|
+
|
|
30
|
+
# Two path segments. One is too coarse to act on in a monorepo (`pkgs`, `src`);
|
|
31
|
+
# three splits the same area into a dozen near-identical rows.
|
|
32
|
+
DEPTH = 2
|
|
33
|
+
MIN_ATTEMPTS = 2
|
|
34
|
+
TOP_N = 4
|
|
35
|
+
|
|
36
|
+
# When two segments barely group anything -- a plugin registry where every entry
|
|
37
|
+
# is its own directory produces ninety areas from a hundred pull requests -- the
|
|
38
|
+
# rows read as insight while carrying none. Above this ratio of areas to pull
|
|
39
|
+
# requests, fall back to one segment, which for that registry correctly collapses
|
|
40
|
+
# to a single honest row: everything lands in `plugins`.
|
|
41
|
+
REGROUP_ABOVE = 0.5
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True, slots=True)
|
|
45
|
+
class Area:
|
|
46
|
+
path: str
|
|
47
|
+
landed: int
|
|
48
|
+
attempted: int
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def rate(self) -> float:
|
|
52
|
+
return self.landed / self.attempted if self.attempted else 0.0
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass(frozen=True, slots=True)
|
|
56
|
+
class Landing:
|
|
57
|
+
"""Where outsiders got in, and where they did not."""
|
|
58
|
+
|
|
59
|
+
landed: list[Area]
|
|
60
|
+
never: list[Area]
|
|
61
|
+
outsider_threads: int
|
|
62
|
+
outsider_merges: int
|
|
63
|
+
depth: int = DEPTH
|
|
64
|
+
|
|
65
|
+
def __bool__(self) -> bool:
|
|
66
|
+
return bool(self.landed or self.never)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def area_of(path: str, depth: int = DEPTH) -> str:
|
|
70
|
+
"""`pkgs/by-name/fo/foo/package.nix` -> `pkgs/by-name`. A root file -> `(root)`."""
|
|
71
|
+
parts = path.split("/")
|
|
72
|
+
if len(parts) == 1:
|
|
73
|
+
return "(root)"
|
|
74
|
+
return "/".join(parts[:depth])
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _tally(outsiders: list[Thread], depth: int) -> tuple[Counter, Counter]:
|
|
78
|
+
landed: Counter = Counter()
|
|
79
|
+
attempted: Counter = Counter()
|
|
80
|
+
for thread in outsiders:
|
|
81
|
+
# Count each area once per pull request. A change touching forty files in
|
|
82
|
+
# one directory is one attempt at that directory, not forty.
|
|
83
|
+
for area in sorted({area_of(f, depth) for f in (thread.files or [])}):
|
|
84
|
+
attempted[area] += 1
|
|
85
|
+
if thread.merged:
|
|
86
|
+
landed[area] += 1
|
|
87
|
+
return landed, attempted
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def compute(threads: dict[str, Thread]) -> Landing:
|
|
91
|
+
outsiders = newcomer_threads(threads)
|
|
92
|
+
depth = DEPTH
|
|
93
|
+
landed, attempted = _tally(outsiders, depth)
|
|
94
|
+
if outsiders and len(attempted) > REGROUP_ABOVE * len(outsiders):
|
|
95
|
+
depth = 1
|
|
96
|
+
landed, attempted = _tally(outsiders, depth)
|
|
97
|
+
|
|
98
|
+
# `Counter.most_common` breaks ties by insertion order, and insertion order
|
|
99
|
+
# here was set-iteration order, which varies with the process hash seed. Two
|
|
100
|
+
# replays of the same recording printed different fourth rows -- a report
|
|
101
|
+
# this project claims is a function of the evidence quietly was not. Ordered
|
|
102
|
+
# explicitly instead: most merges first, then the better odds, then the
|
|
103
|
+
# larger sample, then alphabetically, so the order is total.
|
|
104
|
+
got_in = [
|
|
105
|
+
Area(a, landed[a], attempted[a])
|
|
106
|
+
for a in sorted(
|
|
107
|
+
landed,
|
|
108
|
+
key=lambda a: (-landed[a], -landed[a] / attempted[a], -attempted[a], a),
|
|
109
|
+
)[:TOP_N]
|
|
110
|
+
]
|
|
111
|
+
never = [
|
|
112
|
+
Area(a, 0, attempted[a])
|
|
113
|
+
for a in sorted(attempted, key=lambda a: (-attempted[a], a))
|
|
114
|
+
if landed.get(a, 0) == 0 and attempted[a] >= MIN_ATTEMPTS
|
|
115
|
+
][:TOP_N]
|
|
116
|
+
|
|
117
|
+
return Landing(
|
|
118
|
+
landed=got_in,
|
|
119
|
+
never=never,
|
|
120
|
+
outsider_threads=len(outsiders),
|
|
121
|
+
outsider_merges=sum(1 for t in outsiders if t.merged),
|
|
122
|
+
depth=depth,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def render(landing: Landing) -> list[str]:
|
|
127
|
+
"""Markdown lines, or nothing at all when there is nothing to say."""
|
|
128
|
+
if not landing:
|
|
129
|
+
return []
|
|
130
|
+
lines = ["## Where outsider work landed", ""]
|
|
131
|
+
lines.append(
|
|
132
|
+
f"Counted over the {landing.outsider_threads} pull requests opened here by "
|
|
133
|
+
f"people with no prior merge, of which {landing.outsider_merges} were merged. "
|
|
134
|
+
+ ("Paths are cut to their first two segments, so a short path may name a "
|
|
135
|
+
"file rather than a directory." if landing.depth == 2 else
|
|
136
|
+
"Paths are cut to their first segment: at two, almost every pull request "
|
|
137
|
+
"here landed in a directory of its own, which groups nothing.")
|
|
138
|
+
)
|
|
139
|
+
lines.append("")
|
|
140
|
+
for area in landing.landed:
|
|
141
|
+
lines.append(
|
|
142
|
+
f"- **`{area.path}`** — {area.landed} merged "
|
|
143
|
+
f"of {area.attempted} attempted ({area.rate:.0%})"
|
|
144
|
+
)
|
|
145
|
+
if landing.never:
|
|
146
|
+
lines.append("")
|
|
147
|
+
joined = ", ".join(f"`{a.path}` ({a.attempted})" for a in landing.never)
|
|
148
|
+
lines.append(
|
|
149
|
+
f"Outsiders attempted these and none were merged: {joined}. "
|
|
150
|
+
f"That is what this sample shows, not a rule — a directory can appear "
|
|
151
|
+
f"here because newcomers are turned away from it, or because only a "
|
|
152
|
+
f"couple of people ever tried."
|
|
153
|
+
)
|
|
154
|
+
return lines
|
holt/agent/pipeline.py
ADDED
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
"""A -> B -> C -> D -> verdict -> E.
|
|
2
|
+
|
|
3
|
+
The ordering that matters: the verdict is computed *before* narration and handed
|
|
4
|
+
to Stage E as an input it cannot alter. If the report and verdict.py could
|
|
5
|
+
disagree, the determinism claim would be worth nothing.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from datetime import datetime
|
|
12
|
+
|
|
13
|
+
from holt.agent import landing, stages
|
|
14
|
+
from holt.agent.findings import Finding, Findings
|
|
15
|
+
from holt.agent.signals import Signals, build_threads, compute
|
|
16
|
+
from holt.agent.verdict import classify as decide
|
|
17
|
+
from holt.agent.verdict import contested_kind
|
|
18
|
+
from holt.agent.verify import check_quotes, verify
|
|
19
|
+
from holt.evidence.provider import EvidenceProvider
|
|
20
|
+
from holt.model import ModelClient
|
|
21
|
+
from holt.report import VERDICT_HEADLINES, Assessment, Claim, Verdict
|
|
22
|
+
|
|
23
|
+
MAX_CLAIM_CHARS = 240
|
|
24
|
+
MAX_QUOTE_CHARS = 180
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def clip(text: str, limit: int) -> str:
|
|
28
|
+
"""Cut on a word boundary. Cutting mid-word reads as a bug, because it is one."""
|
|
29
|
+
text = " ".join(text.split())
|
|
30
|
+
if len(text) <= limit:
|
|
31
|
+
return text
|
|
32
|
+
head = text[:limit]
|
|
33
|
+
cut = max(head.rfind(" "), head.rfind(". "))
|
|
34
|
+
return (head[:cut] if cut > limit // 2 else head).rstrip(" ,;:.") + "…"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(slots=True)
|
|
38
|
+
class Trace:
|
|
39
|
+
"""What happened, for the demo and the trajectory record."""
|
|
40
|
+
|
|
41
|
+
signals: Signals
|
|
42
|
+
before_verification: int = 0
|
|
43
|
+
after_verification: int = 0
|
|
44
|
+
dropped: list[Finding] = field(default_factory=list)
|
|
45
|
+
# Findings whose id resolved but whose quotation is not in the record.
|
|
46
|
+
invented: list[Finding] = field(default_factory=list)
|
|
47
|
+
rules: list[str] = field(default_factory=list)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def analyze(
|
|
51
|
+
repo: str,
|
|
52
|
+
provider: EvidenceProvider,
|
|
53
|
+
model: ModelClient | None,
|
|
54
|
+
contributor_days: int = 7,
|
|
55
|
+
as_of: datetime | None = None,
|
|
56
|
+
) -> tuple[Assessment, Trace]:
|
|
57
|
+
if model is None:
|
|
58
|
+
return analyze_without_model(repo, provider, contributor_days, as_of)
|
|
59
|
+
records = provider.fetch(repo)
|
|
60
|
+
threads = build_threads(records)
|
|
61
|
+
signals = compute(threads)
|
|
62
|
+
|
|
63
|
+
findings = Findings()
|
|
64
|
+
stages.classify(repo, records, threads, model, findings)
|
|
65
|
+
stages.assess_opportunity(repo, records, model, findings)
|
|
66
|
+
stages.read_outcomes(repo, threads, model, findings)
|
|
67
|
+
|
|
68
|
+
before = len(findings)
|
|
69
|
+
findings, dropped = verify(findings, provider)
|
|
70
|
+
|
|
71
|
+
# `repo_kind` is the only model-derived field that can decide the answer by
|
|
72
|
+
# itself, and Stage D cannot check it -- an id resolving says nothing about
|
|
73
|
+
# whether a classification is true. Where the evidence contradicts the
|
|
74
|
+
# reason the kind rule would give, the field is dropped before it decides
|
|
75
|
+
# anything and the disagreement is printed. See eval/PREREGISTRATION-4.md.
|
|
76
|
+
meta = next((r for r in records if r.evidence_id.endswith(":meta")), None)
|
|
77
|
+
contested = contested_kind(findings, signals, meta.payload if meta else None)
|
|
78
|
+
if contested:
|
|
79
|
+
findings.drop("repo_kind")
|
|
80
|
+
|
|
81
|
+
verdict, rules = decide(findings, signals, contributor_days)
|
|
82
|
+
if contested:
|
|
83
|
+
rules.insert(0, contested)
|
|
84
|
+
# The narration prompt is deliberately held to the signal fields that existed
|
|
85
|
+
# when the trajectories were recorded. New signals reach the *verdict*
|
|
86
|
+
# immediately but only reach the prose on the next re-record, so adding one
|
|
87
|
+
# does not invalidate every committed trajectory and break replay for a judge.
|
|
88
|
+
# When a new signal changes the outcome it still reaches the narrator, via
|
|
89
|
+
# the rule trace.
|
|
90
|
+
narrated_signals = {
|
|
91
|
+
k: v for k, v in signals.as_dict().items()
|
|
92
|
+
if k not in ("reviewed_share", "merge_rate", "merged_files_median",
|
|
93
|
+
"merged_dirs_median", "merged_with_files")
|
|
94
|
+
}
|
|
95
|
+
narrated = stages.narrate(
|
|
96
|
+
repo, verdict.value, rules, findings, narrated_signals, model
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
# The evidence list is built from verified findings, not written by the
|
|
100
|
+
# model. Stage E supplies prose; it cannot introduce a citation.
|
|
101
|
+
#
|
|
102
|
+
# The quote check runs here rather than inside Stage D on purpose. A claim
|
|
103
|
+
# whose id does not resolve is worthless to everyone, narrator included, so
|
|
104
|
+
# `verify` removes it before anything else runs. A claim whose id resolves
|
|
105
|
+
# but whose words are not in the record is a different failure: the thread
|
|
106
|
+
# is real and the outcome may well be right, and what must not reach the
|
|
107
|
+
# reader is the quotation. Filtering the claim list is exactly that, and it
|
|
108
|
+
# leaves the narration prompt byte-identical, so every committed trajectory
|
|
109
|
+
# still replays -- a guarantee that would otherwise cost a re-record of the
|
|
110
|
+
# frozen benchmark to buy.
|
|
111
|
+
quoting, invented = check_quotes(findings, records)
|
|
112
|
+
claims: list[Claim] = []
|
|
113
|
+
for item in quoting:
|
|
114
|
+
if item.field == "thread_outcome":
|
|
115
|
+
outcome = item.value["outcome"].replace("_", " ")
|
|
116
|
+
quote = (item.value.get("quote") or "").strip()
|
|
117
|
+
# An empty quote used to render as a pair of quotation marks with
|
|
118
|
+
# nothing between them, which reads as a bug because it is one.
|
|
119
|
+
text = (f"{outcome} — “{clip(quote, MAX_QUOTE_CHARS)}”" if quote
|
|
120
|
+
else f"{outcome}, nothing said")
|
|
121
|
+
else:
|
|
122
|
+
text = f"{item.field.replace('_', ' ')}: {item.value}" + (
|
|
123
|
+
f" — {clip(item.note, MAX_CLAIM_CHARS)}" if item.note else "")
|
|
124
|
+
claims.append(Claim(text=text, evidence_id=item.evidence_ids[0]))
|
|
125
|
+
|
|
126
|
+
assessment = Assessment(
|
|
127
|
+
repo=repo,
|
|
128
|
+
verdict=verdict,
|
|
129
|
+
summary=narrated["what_the_evidence_shows"],
|
|
130
|
+
bottom_line=narrated["bottom_line"],
|
|
131
|
+
limits=narrated["what_could_not_be_determined"],
|
|
132
|
+
rules=list(rules),
|
|
133
|
+
contributor_days=contributor_days,
|
|
134
|
+
as_of=as_of,
|
|
135
|
+
landing=landing.render(landing.compute(threads)),
|
|
136
|
+
claims=claims,
|
|
137
|
+
method="holt (A classify, B opportunity, C outcomes, D verify, deterministic verdict, E narrate)",
|
|
138
|
+
replayed=model.replayed,
|
|
139
|
+
models=list(model.usage.models),
|
|
140
|
+
dropped_claims=len(dropped) + len(invented),
|
|
141
|
+
)
|
|
142
|
+
return assessment, Trace(
|
|
143
|
+
signals=signals,
|
|
144
|
+
before_verification=before,
|
|
145
|
+
after_verification=len(findings),
|
|
146
|
+
dropped=dropped,
|
|
147
|
+
invented=invented,
|
|
148
|
+
rules=rules,
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
# --- degraded mode -----------------------------------------------------------
|
|
153
|
+
#
|
|
154
|
+
# The project measured its own model stages at +0.01 MCC over the arithmetic
|
|
155
|
+
# (Iteration 22). A finding that large about your own architecture should change
|
|
156
|
+
# the architecture, not just the write-up: if the rules decide the verdict, the
|
|
157
|
+
# verdict must be obtainable without a model, and the reader must be told what
|
|
158
|
+
# they lost. That is this function.
|
|
159
|
+
#
|
|
160
|
+
# It is not a second implementation of the verdict. It calls the same `decide`
|
|
161
|
+
# on the same `Signals`, so the two modes cannot disagree about a repository
|
|
162
|
+
# they both have findings for -- a test asserts exactly that over the pool.
|
|
163
|
+
# What it does not do is write: Stages A, B, C and E never run, so there are no
|
|
164
|
+
# thread quotes, no narration, and no `repo_kind`. That absence is the point of
|
|
165
|
+
# `eval/evidence_integrity.py`'s yield column, and it is stated on the report
|
|
166
|
+
# rather than left for the reader to notice.
|
|
167
|
+
|
|
168
|
+
NO_MODEL_METHOD = (
|
|
169
|
+
"holt --no-model (deterministic verdict from arithmetic; "
|
|
170
|
+
"stages A, B, C and E did not run)"
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def analyze_without_model(
|
|
175
|
+
repo: str,
|
|
176
|
+
provider: EvidenceProvider,
|
|
177
|
+
contributor_days: int = 7,
|
|
178
|
+
as_of: datetime | None = None,
|
|
179
|
+
) -> tuple[Assessment, Trace]:
|
|
180
|
+
"""The verdict, with no model call anywhere and the cost of that printed."""
|
|
181
|
+
records = provider.fetch(repo)
|
|
182
|
+
threads = build_threads(records)
|
|
183
|
+
signals = compute(threads)
|
|
184
|
+
|
|
185
|
+
findings = Findings()
|
|
186
|
+
# `is_archived` is a structured GitHub field. Stage A was asking a model to
|
|
187
|
+
# read a boolean the provider already had, which is the clearest single
|
|
188
|
+
# illustration of why the model stages measured +0.01: some of what they
|
|
189
|
+
# were doing did not need a model at all. Here it is taken from the record
|
|
190
|
+
# and cited to it.
|
|
191
|
+
meta = next((r for r in records if r.evidence_id.endswith(":meta")), None)
|
|
192
|
+
if meta is not None and meta.payload.get("is_archived"):
|
|
193
|
+
findings.add("is_archived", True, (meta.evidence_id,),
|
|
194
|
+
"GitHub reports this repository as archived")
|
|
195
|
+
|
|
196
|
+
verdict, rules = decide(findings, signals, contributor_days)
|
|
197
|
+
|
|
198
|
+
s = signals.as_dict()
|
|
199
|
+
if signals.outsider_threads:
|
|
200
|
+
summary = (
|
|
201
|
+
f"{s['outsider_merged']} of {s['outsider_threads']} outsider pull "
|
|
202
|
+
f"requests merged, by {s['distinct_merged_authors']} distinct people "
|
|
203
|
+
f"out of {s['distinct_outsider_authors']} who tried"
|
|
204
|
+
)
|
|
205
|
+
if s["median_first_response_hours"] is not None:
|
|
206
|
+
summary += f"; median first response {s['median_first_response_hours']}h"
|
|
207
|
+
summary += (
|
|
208
|
+
f"; {s['outsider_ignored']} drew no response at all. "
|
|
209
|
+
"Counted from the pull request record, not judged."
|
|
210
|
+
)
|
|
211
|
+
else:
|
|
212
|
+
summary = (
|
|
213
|
+
"No outsider pull requests in the period read, so the arithmetic has "
|
|
214
|
+
"nothing to count."
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
return Assessment(
|
|
218
|
+
repo=repo,
|
|
219
|
+
verdict=verdict,
|
|
220
|
+
summary=summary,
|
|
221
|
+
bottom_line=f"{VERDICT_HEADLINES[verdict]}. " + (rules[0] if rules else ""),
|
|
222
|
+
limits=(
|
|
223
|
+
"No model ran. This report is the verdict and the rule that produced "
|
|
224
|
+
"it; the parts a model writes — what specific threads said, who was "
|
|
225
|
+
"welcoming, what kind of project this is, and the prose explaining "
|
|
226
|
+
"any of it — are absent, not merely brief. Measured: this mode scores "
|
|
227
|
+
"MCC +0.60 against the full pipeline's +0.61 in sample, but +0.55 "
|
|
228
|
+
"against +0.63 out of sample, and writes 0 citable statements against "
|
|
229
|
+
"its 11.8 (eval/evidence_integrity.py). Run without --no-model for a "
|
|
230
|
+
"report you can check against the record."
|
|
231
|
+
),
|
|
232
|
+
rules=list(rules),
|
|
233
|
+
contributor_days=contributor_days,
|
|
234
|
+
as_of=as_of,
|
|
235
|
+
landing=landing.render(landing.compute(threads)),
|
|
236
|
+
claims=[
|
|
237
|
+
Claim(text=f"{i.field.replace('_', ' ')}: {i.value}", evidence_id=i.evidence_ids[0])
|
|
238
|
+
for i in findings
|
|
239
|
+
],
|
|
240
|
+
method=NO_MODEL_METHOD,
|
|
241
|
+
replayed=False,
|
|
242
|
+
models=[],
|
|
243
|
+
dropped_claims=0,
|
|
244
|
+
), Trace(signals=signals, rules=rules)
|