detectorproof 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- detectorproof/__init__.py +101 -0
- detectorproof/cli.py +153 -0
- detectorproof/domain.py +314 -0
- detectorproof/panel.py +434 -0
- detectorproof/report.py +102 -0
- detectorproof/stats.py +92 -0
- detectorproof-0.1.0.dist-info/METADATA +192 -0
- detectorproof-0.1.0.dist-info/RECORD +12 -0
- detectorproof-0.1.0.dist-info/WHEEL +5 -0
- detectorproof-0.1.0.dist-info/entry_points.txt +2 -0
- detectorproof-0.1.0.dist-info/licenses/LICENSE +21 -0
- detectorproof-0.1.0.dist-info/top_level.txt +1 -0
detectorproof/panel.py
ADDED
|
@@ -0,0 +1,434 @@
|
|
|
1
|
+
"""`panel` - measure the detectors before trusting any of them.
|
|
2
|
+
|
|
3
|
+
This is the gate everything else in the instrument depends on, and it exists
|
|
4
|
+
because of a specific measured failure: on one corpus, three widely benchmarked
|
|
5
|
+
detectors sat at chance while their model cards advertised sub-1% equal error
|
|
6
|
+
rates. A panel that quietly averaged them in would have reported a confident
|
|
7
|
+
number built partly on models that could not do the task at all.
|
|
8
|
+
|
|
9
|
+
So `panel` scores a **declared** set of frozen detectors over a labelled control
|
|
10
|
+
set, reports the area under the ROC curve for each, and refuses to let an
|
|
11
|
+
incompetent one contribute anything downstream.
|
|
12
|
+
|
|
13
|
+
Five refusals are enforced here rather than described in a README. Each one was
|
|
14
|
+
bought by a real error:
|
|
15
|
+
|
|
16
|
+
1. **A detector at or near chance is NOT COMPETENT ON THIS MATERIAL.** It is
|
|
17
|
+
reported in full - never dropped, never footnoted - and excluded from every
|
|
18
|
+
aggregate. `competent_detectors()` is the only accessor downstream code
|
|
19
|
+
should use.
|
|
20
|
+
|
|
21
|
+
2. **The panel is declared before the run.** A score for a detector nobody
|
|
22
|
+
declared is a fault. A declared detector with no scores is NOT MEASURED and
|
|
23
|
+
stays in the report as that. This is the rule that makes the panel a
|
|
24
|
+
pre-registration rather than a selection.
|
|
25
|
+
|
|
26
|
+
3. **Overlapping crops of one file are not independent samples.** Two scores
|
|
27
|
+
for the same detector, item and condition are a fault, not something to
|
|
28
|
+
average. Collapsing windows to one value per item is the caller's job and
|
|
29
|
+
must happen before the statistic, because doing it afterwards is exactly the
|
|
30
|
+
error that inflated one published proportion from 75% to 86%.
|
|
31
|
+
|
|
32
|
+
4. **NOT MEASURED is a third state.** It is never zero, never an absence of
|
|
33
|
+
effect, and it propagates to the report as itself.
|
|
34
|
+
|
|
35
|
+
5. **A near-miss at the band edge is recorded, not re-banded.** When a detector
|
|
36
|
+
lands just outside the chance band, the report says so. In the study behind
|
|
37
|
+
this package one detector landed at 0.447 against a band of [0.45, 0.55] -
|
|
38
|
+
marginally outside, on the inverted side - and the honest response was to
|
|
39
|
+
record the edge rather than widen the band after seeing the number.
|
|
40
|
+
|
|
41
|
+
There is no function here that recommends a transformation, and there will not
|
|
42
|
+
be one. Optimising audio against a detector score is measured, in the work this
|
|
43
|
+
came from, to have made the audio *more* separable, not less - one intervention
|
|
44
|
+
tuned to a chosen statistic moved a detector's AUC from 0.576 to 0.763. The
|
|
45
|
+
instrument reports; it does not advise.
|
|
46
|
+
"""
|
|
47
|
+
from __future__ import annotations
|
|
48
|
+
|
|
49
|
+
from dataclasses import dataclass, field
|
|
50
|
+
from typing import Iterable, Mapping, Sequence
|
|
51
|
+
|
|
52
|
+
from .domain import (
|
|
53
|
+
BONAFIDE,
|
|
54
|
+
DETECTOR,
|
|
55
|
+
HIGHER_SYNTHETIC,
|
|
56
|
+
ITEM,
|
|
57
|
+
LABEL,
|
|
58
|
+
ORIENTATION,
|
|
59
|
+
ORIENTATION_UNRESOLVED,
|
|
60
|
+
score_is_measured,
|
|
61
|
+
validate_rows,
|
|
62
|
+
)
|
|
63
|
+
from .stats import auc
|
|
64
|
+
|
|
65
|
+
# The pre-declared chance band. A detector whose AUC falls inside it is not
|
|
66
|
+
# competent on the material it was scored against. The default is the band the
|
|
67
|
+
# originating protocol declared before any score existed; a caller may declare a
|
|
68
|
+
# different one, but it is an argument to the run and not something to adjust
|
|
69
|
+
# afterwards.
|
|
70
|
+
CHANCE_BAND = (0.45, 0.55)
|
|
71
|
+
|
|
72
|
+
# How far outside the band still counts as "at the edge" for reporting purposes.
|
|
73
|
+
# This changes no verdict. It exists so that a marginal result is visible as
|
|
74
|
+
# marginal instead of reading like a clean pass.
|
|
75
|
+
BAND_EDGE_MARGIN = 0.01
|
|
76
|
+
|
|
77
|
+
COMPETENT = "COMPETENT"
|
|
78
|
+
NOT_COMPETENT = "NOT COMPETENT ON THIS MATERIAL"
|
|
79
|
+
NOT_MEASURED_VERDICT = "NOT MEASURED"
|
|
80
|
+
UNRESOLVED_ORIENTATION = "ORIENTATION UNRESOLVED"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class PanelFault(ValueError):
|
|
84
|
+
"""A fault in the panel declaration or the score table.
|
|
85
|
+
|
|
86
|
+
Distinct from `DomainError`, which is a single field outside its domain.
|
|
87
|
+
A `PanelFault` is a table that parses cleanly and still cannot support the
|
|
88
|
+
measurement asked of it - a duplicate key, an undeclared detector, a control
|
|
89
|
+
set with one class.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass(frozen=True)
|
|
94
|
+
class Detector:
|
|
95
|
+
"""One declared panel member.
|
|
96
|
+
|
|
97
|
+
`identity` is whatever pins the weights: a commit, a revision, a file hash.
|
|
98
|
+
It is free text because the sources differ, but it is required, because a
|
|
99
|
+
panel that cannot say which checkpoint produced a number is not frozen.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
name: str
|
|
103
|
+
orientation: str
|
|
104
|
+
identity: str
|
|
105
|
+
source: str = ""
|
|
106
|
+
|
|
107
|
+
def __post_init__(self) -> None:
|
|
108
|
+
object.__setattr__(self, "name", DETECTOR.parse(self.name))
|
|
109
|
+
object.__setattr__(self, "orientation", ORIENTATION.parse(self.orientation))
|
|
110
|
+
identity = (self.identity or "").strip()
|
|
111
|
+
if not identity:
|
|
112
|
+
raise PanelFault(
|
|
113
|
+
f"detector {self.name!r} has no identity. A panel member must name the "
|
|
114
|
+
"checkpoint that produced its scores - a revision, a commit or a file "
|
|
115
|
+
"hash. A panel that cannot say which weights it ran is not frozen."
|
|
116
|
+
)
|
|
117
|
+
object.__setattr__(self, "identity", identity)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@dataclass(frozen=True)
|
|
121
|
+
class DetectorResult:
|
|
122
|
+
"""What the panel found for one declared detector."""
|
|
123
|
+
|
|
124
|
+
detector: str
|
|
125
|
+
verdict: str
|
|
126
|
+
auc: float | None
|
|
127
|
+
n_bonafide: int
|
|
128
|
+
n_synthetic: int
|
|
129
|
+
orientation: str
|
|
130
|
+
identity: str
|
|
131
|
+
band_edge: bool = False
|
|
132
|
+
note: str = ""
|
|
133
|
+
|
|
134
|
+
@property
|
|
135
|
+
def competent(self) -> bool:
|
|
136
|
+
return self.verdict == COMPETENT
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
@dataclass(frozen=True)
|
|
140
|
+
class PanelReport:
|
|
141
|
+
"""The full panel result. Every declared detector appears exactly once."""
|
|
142
|
+
|
|
143
|
+
results: tuple[DetectorResult, ...]
|
|
144
|
+
chance_band: tuple[float, float]
|
|
145
|
+
control_items: int
|
|
146
|
+
faults: tuple[str, ...] = field(default=())
|
|
147
|
+
|
|
148
|
+
def competent_detectors(self) -> tuple[DetectorResult, ...]:
|
|
149
|
+
"""The only accessor downstream code may aggregate over.
|
|
150
|
+
|
|
151
|
+
Incompetent and unmeasured detectors are still in `results` and still in
|
|
152
|
+
every printed report. They are absent from here, which is what "excluded
|
|
153
|
+
from every downstream answer" means in code.
|
|
154
|
+
"""
|
|
155
|
+
return tuple(r for r in self.results if r.competent)
|
|
156
|
+
|
|
157
|
+
def excluded(self) -> tuple[DetectorResult, ...]:
|
|
158
|
+
return tuple(r for r in self.results if not r.competent)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def canonical_labels(labels: Mapping[str, str] | Iterable[tuple[str, str]]) -> dict[str, str]:
|
|
162
|
+
"""Canonicalise the control set, refusing any item labelled two ways.
|
|
163
|
+
|
|
164
|
+
**Refusal 6, and it is the reason this function exists rather than a dict
|
|
165
|
+
comprehension.** The control set is what defines the two classes, so an item
|
|
166
|
+
carrying two labels does not make the answer slightly less certain - it makes
|
|
167
|
+
the answer depend on which row happened to be read last.
|
|
168
|
+
|
|
169
|
+
Measured, before this refusal existed: three items scored 3, 2, 1, with item
|
|
170
|
+
`a` labelled both ways and `b` bona fide and `c` synthetic. Reversing only
|
|
171
|
+
the two conflicting rows for `a` moved the result from **AUC 0.5, exit 1** to
|
|
172
|
+
**AUC 1.0, exit 0**. Neither input was refused. A verdict that depends on row
|
|
173
|
+
order is not a measurement.
|
|
174
|
+
|
|
175
|
+
The collision is detected **after** canonicalisation, because items are
|
|
176
|
+
compared case-insensitively after stripping: `A` and `a` are one item here,
|
|
177
|
+
so labelling them differently is the same fault wearing different case.
|
|
178
|
+
|
|
179
|
+
Repeating an item with the *same* label is accepted. It is redundant rather
|
|
180
|
+
than ambiguous, and no ordering of those rows can change any verdict.
|
|
181
|
+
"""
|
|
182
|
+
pairs = labels.items() if isinstance(labels, Mapping) else labels
|
|
183
|
+
|
|
184
|
+
canonical: dict[str, str] = {}
|
|
185
|
+
original: dict[str, str] = {}
|
|
186
|
+
for raw_item, raw_label in pairs:
|
|
187
|
+
item = ITEM.parse(raw_item)
|
|
188
|
+
key = item.lower()
|
|
189
|
+
label = LABEL.parse(raw_label)
|
|
190
|
+
if key in canonical and canonical[key] != label:
|
|
191
|
+
raise PanelFault(
|
|
192
|
+
f"control item {item!r} is labelled both {canonical[key]!r} and "
|
|
193
|
+
f"{label!r}"
|
|
194
|
+
+ (
|
|
195
|
+
f" (as {original[key]!r} and {item!r}, which are one item after "
|
|
196
|
+
"canonicalisation)"
|
|
197
|
+
if original[key] != item
|
|
198
|
+
else ""
|
|
199
|
+
)
|
|
200
|
+
+ ". A conflicting label is refused rather than resolved: taking "
|
|
201
|
+
"either one would make the verdict depend on the order the rows "
|
|
202
|
+
"were read in. Correct the control set."
|
|
203
|
+
)
|
|
204
|
+
canonical[key] = label
|
|
205
|
+
original.setdefault(key, item)
|
|
206
|
+
return canonical
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _label_of(item: str, labels: Mapping[str, str]) -> str:
|
|
210
|
+
try:
|
|
211
|
+
raw = labels[item]
|
|
212
|
+
except KeyError:
|
|
213
|
+
raise PanelFault(
|
|
214
|
+
f"control item {item!r} has no label. Every item in the control set must be "
|
|
215
|
+
f"declared {BONAFIDE!r} or {'synthetic'!r}; an unlabelled item cannot be "
|
|
216
|
+
"scored and must not be silently dropped."
|
|
217
|
+
) from None
|
|
218
|
+
return LABEL.parse(raw)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def run_panel(
|
|
222
|
+
detectors: Sequence[Detector],
|
|
223
|
+
scores: Iterable[Mapping[str, object]],
|
|
224
|
+
labels: Mapping[str, str] | Iterable[tuple[str, str]],
|
|
225
|
+
*,
|
|
226
|
+
condition: str,
|
|
227
|
+
chance_band: tuple[float, float] = CHANCE_BAND,
|
|
228
|
+
) -> PanelReport:
|
|
229
|
+
"""Score a declared panel over a labelled control set.
|
|
230
|
+
|
|
231
|
+
`detectors` is the declaration, fixed before the run. `scores` is a table of
|
|
232
|
+
rows carrying `detector`, `item`, `condition`, `score`. `labels` maps each
|
|
233
|
+
control item to `bonafide` or `synthetic`. `condition` selects which rows of
|
|
234
|
+
the table are the control set - the same table usually carries the
|
|
235
|
+
transformed conditions too, and they are not what this gate measures.
|
|
236
|
+
|
|
237
|
+
Every declared detector appears in the result exactly once, whatever
|
|
238
|
+
happened to it.
|
|
239
|
+
"""
|
|
240
|
+
low, high = chance_band
|
|
241
|
+
if not (0.0 <= low < high <= 1.0):
|
|
242
|
+
raise PanelFault(
|
|
243
|
+
f"chance band {chance_band} is not an interval inside [0, 1]. "
|
|
244
|
+
"The band is declared before the run; it is not a result."
|
|
245
|
+
)
|
|
246
|
+
if not detectors:
|
|
247
|
+
raise PanelFault(
|
|
248
|
+
"the panel is empty. A panel is declared before it is run, so an empty "
|
|
249
|
+
"declaration is a fault rather than a result with no rows."
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
declared: dict[str, Detector] = {}
|
|
253
|
+
for det in detectors:
|
|
254
|
+
if det.name.lower() in declared:
|
|
255
|
+
raise PanelFault(
|
|
256
|
+
f"detector {det.name!r} is declared twice. A panel member appears once; "
|
|
257
|
+
"two declarations of one name make its result ambiguous."
|
|
258
|
+
)
|
|
259
|
+
declared[det.name.lower()] = det
|
|
260
|
+
|
|
261
|
+
rows = validate_rows(scores)
|
|
262
|
+
wanted = condition.strip().lower()
|
|
263
|
+
|
|
264
|
+
# Refusal 3: one score per (detector, item). A second score for the same
|
|
265
|
+
# pair is a fault. This is the guard against overlapping windows arriving as
|
|
266
|
+
# independent observations.
|
|
267
|
+
by_detector: dict[str, dict[str, float | object]] = {k: {} for k in declared}
|
|
268
|
+
faults: list[str] = []
|
|
269
|
+
for row in rows:
|
|
270
|
+
if row.condition.lower() != wanted:
|
|
271
|
+
continue
|
|
272
|
+
key = row.detector.lower()
|
|
273
|
+
if key not in declared:
|
|
274
|
+
raise PanelFault(
|
|
275
|
+
f"scores contain detector {row.detector!r}, which the panel did not "
|
|
276
|
+
"declare. The panel is fixed before the run: a detector that appears "
|
|
277
|
+
"only in the results is a selection, not a measurement."
|
|
278
|
+
)
|
|
279
|
+
seen = by_detector[key]
|
|
280
|
+
item_key = row.item.lower()
|
|
281
|
+
if item_key in seen:
|
|
282
|
+
raise PanelFault(
|
|
283
|
+
f"detector {row.detector!r} has more than one score for item "
|
|
284
|
+
f"{row.item!r} under condition {condition!r}. Overlapping analysis "
|
|
285
|
+
"windows are not independent samples and must be collapsed to one "
|
|
286
|
+
"value per item before this gate, not averaged inside it."
|
|
287
|
+
)
|
|
288
|
+
seen[item_key] = row.score
|
|
289
|
+
|
|
290
|
+
# Canonicalise through the collision-refusing helper. A dict comprehension
|
|
291
|
+
# here was the blocking defect: it silently kept whichever label was read
|
|
292
|
+
# last, so reversing two rows could flip a verdict.
|
|
293
|
+
control = canonical_labels(labels)
|
|
294
|
+
if not control:
|
|
295
|
+
raise PanelFault("the control set is empty; there is nothing to score against.")
|
|
296
|
+
|
|
297
|
+
results: list[DetectorResult] = []
|
|
298
|
+
for key, det in declared.items():
|
|
299
|
+
seen = by_detector[key]
|
|
300
|
+
|
|
301
|
+
# Refusal 4: separate the two ways a number can be absent. A detector
|
|
302
|
+
# never run is NOT MEASURED. A detector run and recorded as the sentinel
|
|
303
|
+
# is also NOT MEASURED. Neither is zero.
|
|
304
|
+
bonafide_scores: list[float] = []
|
|
305
|
+
synthetic_scores: list[float] = []
|
|
306
|
+
for item, label in control.items():
|
|
307
|
+
if item not in seen:
|
|
308
|
+
continue
|
|
309
|
+
value = seen[item]
|
|
310
|
+
if not score_is_measured(value):
|
|
311
|
+
continue
|
|
312
|
+
if label == BONAFIDE:
|
|
313
|
+
bonafide_scores.append(float(value))
|
|
314
|
+
else:
|
|
315
|
+
synthetic_scores.append(float(value))
|
|
316
|
+
measured = len(bonafide_scores) + len(synthetic_scores)
|
|
317
|
+
|
|
318
|
+
# Orientation decides which class the detector is supposed to score
|
|
319
|
+
# HIGHER, and therefore which is the positive class for the AUC. Both
|
|
320
|
+
# branches produce an AUC where above 0.5 means "separates correctly",
|
|
321
|
+
# so a single chance band applies to the whole panel regardless of which
|
|
322
|
+
# way each model's native scale runs.
|
|
323
|
+
if det.orientation == HIGHER_SYNTHETIC:
|
|
324
|
+
positive, negative = synthetic_scores, bonafide_scores
|
|
325
|
+
else:
|
|
326
|
+
positive, negative = bonafide_scores, synthetic_scores
|
|
327
|
+
|
|
328
|
+
if measured == 0:
|
|
329
|
+
results.append(
|
|
330
|
+
DetectorResult(
|
|
331
|
+
detector=det.name,
|
|
332
|
+
verdict=NOT_MEASURED_VERDICT,
|
|
333
|
+
auc=None,
|
|
334
|
+
n_bonafide=0,
|
|
335
|
+
n_synthetic=0,
|
|
336
|
+
orientation=det.orientation,
|
|
337
|
+
identity=det.identity,
|
|
338
|
+
note=(
|
|
339
|
+
"declared but produced no measurement on this control set. "
|
|
340
|
+
"NOT MEASURED is not zero and not an absence of effect."
|
|
341
|
+
),
|
|
342
|
+
)
|
|
343
|
+
)
|
|
344
|
+
continue
|
|
345
|
+
|
|
346
|
+
if not bonafide_scores or not synthetic_scores:
|
|
347
|
+
results.append(
|
|
348
|
+
DetectorResult(
|
|
349
|
+
detector=det.name,
|
|
350
|
+
verdict=NOT_MEASURED_VERDICT,
|
|
351
|
+
auc=None,
|
|
352
|
+
n_bonafide=len(bonafide_scores),
|
|
353
|
+
n_synthetic=len(synthetic_scores),
|
|
354
|
+
orientation=det.orientation,
|
|
355
|
+
identity=det.identity,
|
|
356
|
+
note=(
|
|
357
|
+
"only one class of the control set was measured, so AUC is "
|
|
358
|
+
"undefined. An undefined AUC is not 0.5 and is not at chance."
|
|
359
|
+
),
|
|
360
|
+
)
|
|
361
|
+
)
|
|
362
|
+
continue
|
|
363
|
+
|
|
364
|
+
value = auc(positive, negative)
|
|
365
|
+
n_bona = len(bonafide_scores)
|
|
366
|
+
n_synth = len(synthetic_scores)
|
|
367
|
+
|
|
368
|
+
# Refusal 5: the band is applied by the letter, and a near miss is
|
|
369
|
+
# recorded as a near miss rather than resolved by moving the band.
|
|
370
|
+
at_chance = low <= value <= high
|
|
371
|
+
edge = (not at_chance) and (
|
|
372
|
+
low - BAND_EDGE_MARGIN <= value < low or high < value <= high + BAND_EDGE_MARGIN
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
if det.orientation == ORIENTATION_UNRESOLVED:
|
|
376
|
+
verdict = UNRESOLVED_ORIENTATION
|
|
377
|
+
note = (
|
|
378
|
+
"orientation is not resolved from documentation, so the DIRECTION of "
|
|
379
|
+
"this detector's score is not interpreted. The value is reported; its "
|
|
380
|
+
"sign is not. Excluded from aggregates."
|
|
381
|
+
)
|
|
382
|
+
elif at_chance:
|
|
383
|
+
verdict = NOT_COMPETENT
|
|
384
|
+
note = (
|
|
385
|
+
f"AUC {value:.4f} is inside the declared chance band "
|
|
386
|
+
f"[{low}, {high}]. Reported in full and excluded from every "
|
|
387
|
+
"downstream answer; never averaged in, never used to break a tie."
|
|
388
|
+
)
|
|
389
|
+
else:
|
|
390
|
+
verdict = COMPETENT
|
|
391
|
+
note = ""
|
|
392
|
+
if edge:
|
|
393
|
+
note = (
|
|
394
|
+
f"AUC {value:.4f} is within {BAND_EDGE_MARGIN} of the declared band "
|
|
395
|
+
f"[{low}, {high}]. Recorded as a band-edge result. The band was "
|
|
396
|
+
"declared before the run and is not widened to absorb it."
|
|
397
|
+
)
|
|
398
|
+
|
|
399
|
+
results.append(
|
|
400
|
+
DetectorResult(
|
|
401
|
+
detector=det.name,
|
|
402
|
+
verdict=verdict,
|
|
403
|
+
auc=value,
|
|
404
|
+
n_bonafide=n_bona,
|
|
405
|
+
n_synthetic=n_synth,
|
|
406
|
+
orientation=det.orientation,
|
|
407
|
+
identity=det.identity,
|
|
408
|
+
band_edge=edge,
|
|
409
|
+
note=note,
|
|
410
|
+
)
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
return PanelReport(
|
|
414
|
+
results=tuple(results),
|
|
415
|
+
chance_band=(low, high),
|
|
416
|
+
control_items=len(control),
|
|
417
|
+
faults=tuple(faults),
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
__all__ = [
|
|
422
|
+
"BAND_EDGE_MARGIN",
|
|
423
|
+
"canonical_labels",
|
|
424
|
+
"CHANCE_BAND",
|
|
425
|
+
"COMPETENT",
|
|
426
|
+
"Detector",
|
|
427
|
+
"DetectorResult",
|
|
428
|
+
"NOT_COMPETENT",
|
|
429
|
+
"NOT_MEASURED_VERDICT",
|
|
430
|
+
"PanelFault",
|
|
431
|
+
"PanelReport",
|
|
432
|
+
"UNRESOLVED_ORIENTATION",
|
|
433
|
+
"run_panel",
|
|
434
|
+
]
|
detectorproof/report.py
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Rendering a panel result as text.
|
|
2
|
+
|
|
3
|
+
The wording here is part of the instrument, not decoration. Three rules govern
|
|
4
|
+
every line this module prints, and each was bought by a real reporting error in
|
|
5
|
+
the study behind the package:
|
|
6
|
+
|
|
7
|
+
**A null is "no evidence of a difference", never "no difference".** A
|
|
8
|
+
non-significant test at a small n is absence of evidence. The one time this was
|
|
9
|
+
written the other way round it had to be corrected in four documents.
|
|
10
|
+
|
|
11
|
+
**NOT MEASURED is printed as itself.** Never as 0, never as a blank cell, never
|
|
12
|
+
omitted from the table. A reader scanning a column of numbers must be able to
|
|
13
|
+
see that a cell is missing rather than small.
|
|
14
|
+
|
|
15
|
+
**An excluded detector is printed in full.** The refusal removes it from the
|
|
16
|
+
aggregates, not from the page. A panel that hid its incompetent members would be
|
|
17
|
+
indistinguishable from a panel chosen to agree.
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from .panel import PanelReport
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _fmt_auc(value: float | None) -> str:
|
|
25
|
+
return "NOT MEASURED" if value is None else f"{value:.4f}"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def render(report: PanelReport) -> str:
|
|
29
|
+
"""A plain-text panel report. Every declared detector appears exactly once."""
|
|
30
|
+
low, high = report.chance_band
|
|
31
|
+
lines: list[str] = []
|
|
32
|
+
lines.append("PANEL")
|
|
33
|
+
lines.append("=====")
|
|
34
|
+
lines.append("")
|
|
35
|
+
lines.append(f"control items declared : {report.control_items}")
|
|
36
|
+
lines.append(f"chance band (declared) : [{low}, {high}]")
|
|
37
|
+
lines.append(f"detectors declared : {len(report.results)}")
|
|
38
|
+
lines.append("")
|
|
39
|
+
|
|
40
|
+
width = max((len(r.detector) for r in report.results), default=8)
|
|
41
|
+
width = max(width, 8)
|
|
42
|
+
header = f"{'detector'.ljust(width)} {'AUC':>12} {'bona':>5} {'synth':>5} verdict"
|
|
43
|
+
lines.append(header)
|
|
44
|
+
lines.append("-" * len(header))
|
|
45
|
+
for r in sorted(report.results, key=lambda x: (x.auc is None, -(x.auc or 0.0))):
|
|
46
|
+
lines.append(
|
|
47
|
+
f"{r.detector.ljust(width)} {_fmt_auc(r.auc):>12} "
|
|
48
|
+
f"{r.n_bonafide:>5} {r.n_synthetic:>5} {r.verdict}"
|
|
49
|
+
)
|
|
50
|
+
lines.append("")
|
|
51
|
+
|
|
52
|
+
# The declared identity of every panel member, printed rather than merely
|
|
53
|
+
# promised. A verdict is a statement about a specific checkpoint, so a
|
|
54
|
+
# report that names the detector but not the weights that produced its
|
|
55
|
+
# scores cannot be checked by the person reading it. An earlier version of
|
|
56
|
+
# this renderer said identities were "recorded above" and then printed none
|
|
57
|
+
# of them.
|
|
58
|
+
lines.append("DECLARED IDENTITIES")
|
|
59
|
+
lines.append("-------------------")
|
|
60
|
+
for r in sorted(report.results, key=lambda x: x.detector.lower()):
|
|
61
|
+
lines.append(f" {r.detector.ljust(width)} {r.orientation:<17} {r.identity}")
|
|
62
|
+
lines.append("")
|
|
63
|
+
|
|
64
|
+
notes = [r for r in report.results if r.note]
|
|
65
|
+
if notes:
|
|
66
|
+
lines.append("NOTES")
|
|
67
|
+
lines.append("-----")
|
|
68
|
+
for r in notes:
|
|
69
|
+
lines.append(f" {r.detector}: {r.note}")
|
|
70
|
+
lines.append("")
|
|
71
|
+
|
|
72
|
+
competent = report.competent_detectors()
|
|
73
|
+
excluded = report.excluded()
|
|
74
|
+
lines.append("GATE")
|
|
75
|
+
lines.append("----")
|
|
76
|
+
lines.append(f" competent on this material : {len(competent)} of {len(report.results)}")
|
|
77
|
+
if competent:
|
|
78
|
+
lines.append(" " + ", ".join(r.detector for r in competent))
|
|
79
|
+
if excluded:
|
|
80
|
+
lines.append(f" excluded from every downstream answer : {len(excluded)}")
|
|
81
|
+
for r in excluded:
|
|
82
|
+
lines.append(f" {r.detector} - {r.verdict}")
|
|
83
|
+
lines.append("")
|
|
84
|
+
|
|
85
|
+
if not competent:
|
|
86
|
+
lines.append(
|
|
87
|
+
" No detector in this declared panel is competent on this control set.\n"
|
|
88
|
+
" That is a reportable result about the panel and this material, and it is\n"
|
|
89
|
+
" NOT evidence that the material is genuine or that detection is impossible.\n"
|
|
90
|
+
" It is absence of evidence, not evidence of absence. Nothing downstream may\n"
|
|
91
|
+
" run on an empty competent set."
|
|
92
|
+
)
|
|
93
|
+
else:
|
|
94
|
+
lines.append(
|
|
95
|
+
" Each verdict above is a statement about the checkpoint named in DECLARED\n"
|
|
96
|
+
" IDENTITIES, measured on this control set - 'not competent on this\n"
|
|
97
|
+
" material' - and never about a detector's general quality."
|
|
98
|
+
)
|
|
99
|
+
return "\n".join(lines)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
__all__ = ["render"]
|
detectorproof/stats.py
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Exact, dependency-free statistics.
|
|
2
|
+
|
|
3
|
+
Everything here is computed from the numbers a study already has. There is no
|
|
4
|
+
numpy and no scipy, so this package installs and runs on a machine that will
|
|
5
|
+
never hold the audio - the same reason spkproof's core has no dependencies.
|
|
6
|
+
|
|
7
|
+
Two properties are deliberate and both are load-bearing:
|
|
8
|
+
|
|
9
|
+
**Ties are handled by midranks, everywhere.** A detector whose scores are
|
|
10
|
+
compressed near zero produces long runs of equal values, and a rank statistic
|
|
11
|
+
that breaks ties arbitrarily reports a different answer depending on the order
|
|
12
|
+
its input happened to arrive in. One detector in the study this package came
|
|
13
|
+
out of had real scores of 0.003 against 0.002; its rank statistic is the only
|
|
14
|
+
interpretable thing about it, so the ranks have to be right.
|
|
15
|
+
|
|
16
|
+
**AUC is computed from the rank sum, not by counting pairs.** They agree, but
|
|
17
|
+
the rank-sum form is O(n log n) rather than O(n^2) and, more importantly, it
|
|
18
|
+
makes the tie correction a property of the ranking rather than a special case
|
|
19
|
+
bolted onto a comparison.
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from typing import Sequence
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def midranks(values: Sequence[float]) -> list[float]:
|
|
27
|
+
"""Ranks, 1-based, with tied values sharing the mean of their positions."""
|
|
28
|
+
order = sorted(range(len(values)), key=lambda i: values[i])
|
|
29
|
+
ranks = [0.0] * len(values)
|
|
30
|
+
i = 0
|
|
31
|
+
while i < len(order):
|
|
32
|
+
j = i
|
|
33
|
+
while j + 1 < len(order) and values[order[j + 1]] == values[order[i]]:
|
|
34
|
+
j += 1
|
|
35
|
+
# Positions i..j (0-based) are tied; their 1-based ranks average to this.
|
|
36
|
+
shared = (i + j + 2) / 2.0
|
|
37
|
+
for k in range(i, j + 1):
|
|
38
|
+
ranks[order[k]] = shared
|
|
39
|
+
i = j + 1
|
|
40
|
+
return ranks
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def auc(positive: Sequence[float], negative: Sequence[float]) -> float:
|
|
44
|
+
"""Area under the ROC curve, as P(positive scores above negative).
|
|
45
|
+
|
|
46
|
+
`positive` are the scores of the class the detector is supposed to give
|
|
47
|
+
HIGHER values to; `negative` the other class. Ties contribute 0.5, which is
|
|
48
|
+
what the midrank form produces without a special case.
|
|
49
|
+
|
|
50
|
+
Returns a value in [0, 1]. Both classes must be non-empty: an AUC computed
|
|
51
|
+
against an empty class is not a weak measurement, it is undefined, and
|
|
52
|
+
returning 0.5 for it would put a fabricated "at chance" into a report.
|
|
53
|
+
"""
|
|
54
|
+
n_pos, n_neg = len(positive), len(negative)
|
|
55
|
+
if n_pos == 0 or n_neg == 0:
|
|
56
|
+
raise ValueError(
|
|
57
|
+
"AUC needs at least one item of each class; "
|
|
58
|
+
f"got {n_pos} positive and {n_neg} negative. "
|
|
59
|
+
"An undefined AUC is not 0.5."
|
|
60
|
+
)
|
|
61
|
+
ranks = midranks(list(positive) + list(negative))
|
|
62
|
+
rank_sum_pos = sum(ranks[:n_pos])
|
|
63
|
+
return (rank_sum_pos - n_pos * (n_pos + 1) / 2.0) / (n_pos * n_neg)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def holm_adjusted(pvalues: Sequence[float]) -> list[float]:
|
|
67
|
+
"""Holm-Bonferroni adjusted p-values, in the caller's original order.
|
|
68
|
+
|
|
69
|
+
Sorted ascending, multiplied by the descending family size, then made
|
|
70
|
+
monotone by a running maximum and capped at 1. The running maximum is the
|
|
71
|
+
step people leave out; without it an adjusted p-value can come out below
|
|
72
|
+
the one before it, which is not a Holm result.
|
|
73
|
+
"""
|
|
74
|
+
n = len(pvalues)
|
|
75
|
+
if n == 0:
|
|
76
|
+
return []
|
|
77
|
+
order = sorted(range(n), key=lambda i: pvalues[i])
|
|
78
|
+
adjusted = [0.0] * n
|
|
79
|
+
running = 0.0
|
|
80
|
+
for position, index in enumerate(order):
|
|
81
|
+
scaled = pvalues[index] * (n - position)
|
|
82
|
+
running = max(running, scaled)
|
|
83
|
+
adjusted[index] = min(1.0, running)
|
|
84
|
+
return adjusted
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def holm(pvalues: Sequence[float], alpha: float = 0.05) -> list[bool]:
|
|
88
|
+
"""Which tests survive Holm-Bonferroni at family-wise `alpha`."""
|
|
89
|
+
return [p <= alpha for p in holm_adjusted(pvalues)]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
__all__ = ["auc", "holm", "holm_adjusted", "midranks"]
|