gpu-seal 0.1.0a0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gpu_seal/__init__.py +8 -0
- gpu_seal/__main__.py +4 -0
- gpu_seal/analysis/__init__.py +40 -0
- gpu_seal/analysis/separability.py +310 -0
- gpu_seal/analysis/statistics.py +170 -0
- gpu_seal/cli.py +213 -0
- gpu_seal/controller/__init__.py +100 -0
- gpu_seal/controller/budget.py +263 -0
- gpu_seal/controller/disclosure.py +218 -0
- gpu_seal/controller/evidence_store.py +279 -0
- gpu_seal/controller/fake_runtime.py +103 -0
- gpu_seal/controller/native_runner.py +362 -0
- gpu_seal/controller/orchestrator.py +217 -0
- gpu_seal/controller/policy_matrix.py +270 -0
- gpu_seal/controller/scheduler.py +273 -0
- gpu_seal/cuda/__init__.py +34 -0
- gpu_seal/cuda/backend.py +660 -0
- gpu_seal/cuda/nvml.py +190 -0
- gpu_seal/evidence/__init__.py +32 -0
- gpu_seal/evidence/observation.py +181 -0
- gpu_seal/evidence/result.py +386 -0
- gpu_seal/evidence/signing.py +275 -0
- gpu_seal/probes/__init__.py +127 -0
- gpu_seal/probes/allocation_model.py +554 -0
- gpu_seal/probes/attestation.py +407 -0
- gpu_seal/probes/device_exposure.py +596 -0
- gpu_seal/probes/environment.py +414 -0
- gpu_seal/probes/framework_allocator.py +218 -0
- gpu_seal/probes/location.py +275 -0
- gpu_seal/probes/memory_global.py +423 -0
- gpu_seal/probes/memory_local.py +294 -0
- gpu_seal/probes/mig_temporal.py +395 -0
- gpu_seal/probes/self_canary.py +296 -0
- gpu_seal/probes/topology.py +656 -0
- gpu_seal/reporting/__init__.py +33 -0
- gpu_seal/reporting/report_card.py +622 -0
- gpu_seal/resources.py +35 -0
- gpu_seal/safety/__init__.py +60 -0
- gpu_seal/safety/aggregation.py +573 -0
- gpu_seal/safety/buffer.py +332 -0
- gpu_seal/safety/campaign.py +88 -0
- gpu_seal/safety/canary.py +475 -0
- gpu_seal/safety/errors.py +127 -0
- gpu_seal/safety/metadata.py +98 -0
- gpu_seal/safety/policy.py +303 -0
- gpu_seal/schemas/experiment.schema.json +181 -0
- gpu_seal/schemas/provider-policy.schema.json +106 -0
- gpu_seal/schemas/report-card.schema.json +84 -0
- gpu_seal/schemas/result.schema.json +360 -0
- gpu_seal-0.1.0a0.dist-info/METADATA +400 -0
- gpu_seal-0.1.0a0.dist-info/RECORD +55 -0
- gpu_seal-0.1.0a0.dist-info/WHEEL +5 -0
- gpu_seal-0.1.0a0.dist-info/entry_points.txt +2 -0
- gpu_seal-0.1.0a0.dist-info/licenses/LICENSE +202 -0
- gpu_seal-0.1.0a0.dist-info/top_level.txt +1 -0
gpu_seal/__init__.py
ADDED
gpu_seal/__main__.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Statistics and analysis — CHARTER.md §12, §18.
|
|
2
|
+
|
|
3
|
+
"Don't overinterpret one allocation."
|
|
4
|
+
|
|
5
|
+
The math lives in the installable package rather than in notebooks, so that
|
|
6
|
+
every figure in the paper is produced by code that the test suite exercises.
|
|
7
|
+
A number that only exists inside a notebook cell is a number nobody can
|
|
8
|
+
reproduce.
|
|
9
|
+
|
|
10
|
+
Two modules:
|
|
11
|
+
|
|
12
|
+
``statistics``
|
|
13
|
+
Bootstrap confidence intervals and the counts §12 requires with every
|
|
14
|
+
conclusion (sample / success / failure / error / exclusion).
|
|
15
|
+
|
|
16
|
+
``separability``
|
|
17
|
+
Contribution **D5** — same-model die separation (§9.8b). Pairwise
|
|
18
|
+
separability, leave-one-out classification, and ROC/AUC over topology
|
|
19
|
+
certificates, plus the explicit statement of when the claim fails.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from .separability import (
|
|
23
|
+
SeparabilityReport,
|
|
24
|
+
certificate_distance,
|
|
25
|
+
evaluate_separability,
|
|
26
|
+
)
|
|
27
|
+
from .statistics import (
|
|
28
|
+
ObservationCounts,
|
|
29
|
+
bootstrap_ci,
|
|
30
|
+
proportion_ci,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"ObservationCounts",
|
|
35
|
+
"bootstrap_ci",
|
|
36
|
+
"proportion_ci",
|
|
37
|
+
"SeparabilityReport",
|
|
38
|
+
"certificate_distance",
|
|
39
|
+
"evaluate_separability",
|
|
40
|
+
]
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
"""Same-model die separation — CHARTER.md §9.8b, contribution **D5**.
|
|
2
|
+
|
|
3
|
+
This is the open problem Alpay & Alpay handed the project. Their §12 says it
|
|
4
|
+
plainly:
|
|
5
|
+
|
|
6
|
+
"The cross-die identity experiment separates two different Blackwell
|
|
7
|
+
*products*. Same-model die separation is left to prior GPU
|
|
8
|
+
fingerprinting."
|
|
9
|
+
|
|
10
|
+
They can tell a B200 from a 5090. They have not shown they can tell *your*
|
|
11
|
+
H100 from *another* H100 — and §9.5's self-vs-self canary does not work
|
|
12
|
+
without that, because a clean result across two allocations of the same
|
|
13
|
+
advertised model is equally consistent with "the provider sanitised" and "the
|
|
14
|
+
provider gave us a different chip." D1's strongest result is inconclusive
|
|
15
|
+
until this is resolved either way.
|
|
16
|
+
|
|
17
|
+
**What this module is.** The evaluation half of D5: given a set of topology
|
|
18
|
+
certificates with known device labels, it measures whether they separate.
|
|
19
|
+
Pairwise distances, leave-one-out nearest-neighbour classification, and an
|
|
20
|
+
ROC/AUC over same-device versus different-device pairs.
|
|
21
|
+
|
|
22
|
+
**What this module is not.** It is not the study. §9.8b requires N rented
|
|
23
|
+
instances of a single advertised model, and that is Phase 2b work needing a
|
|
24
|
+
budget this project does not yet have. What exists here is the instrument that
|
|
25
|
+
will grade the data when it arrives, tested against synthetic certificates
|
|
26
|
+
with known ground truth so that it is known to work before it is pointed at
|
|
27
|
+
anything expensive.
|
|
28
|
+
|
|
29
|
+
**A negative result is a result.** §9.8b: *"Report honestly if it fails.
|
|
30
|
+
'Same-model die separation is not achievable at tenant privilege under
|
|
31
|
+
conditions X, Y, Z' is itself a publishable, useful negative that bounds the
|
|
32
|
+
whole research area."* :meth:`SeparabilityReport.verdict` is written to make
|
|
33
|
+
that outcome as easy to state as the positive one.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
from collections.abc import Sequence
|
|
39
|
+
from dataclasses import dataclass
|
|
40
|
+
from typing import Any
|
|
41
|
+
|
|
42
|
+
from ..probes.topology import TopologyCertificate
|
|
43
|
+
|
|
44
|
+
__all__ = [
|
|
45
|
+
"SeparabilityReport",
|
|
46
|
+
"certificate_distance",
|
|
47
|
+
"evaluate_separability",
|
|
48
|
+
]
|
|
49
|
+
|
|
50
|
+
#: Leave-one-out accuracy at or above which the classifier is treated as
|
|
51
|
+
#: usable for gating §9.5 conclusions. Set at the level the source paper
|
|
52
|
+
#: reached for cross-*product* separation (100%), discounted for the harder
|
|
53
|
+
#: same-model case: below this, §13.1 grade A stays unreachable.
|
|
54
|
+
#:
|
|
55
|
+
#: This is a pre-registered threshold (§12 "pre-register scoring"). It is set
|
|
56
|
+
#: here, before any same-model data exists, precisely so it cannot be chosen
|
|
57
|
+
#: after seeing the result.
|
|
58
|
+
VALIDATION_ACCURACY_THRESHOLD = 0.95
|
|
59
|
+
|
|
60
|
+
#: And the AUC it must clear alongside accuracy. Accuracy alone can be carried
|
|
61
|
+
#: by an unbalanced pair set.
|
|
62
|
+
VALIDATION_AUC_THRESHOLD = 0.98
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def certificate_distance(a: TopologyCertificate, b: TopologyCertificate) -> float:
|
|
66
|
+
"""Mean absolute distance between two row-normalised fingerprints.
|
|
67
|
+
|
|
68
|
+
Shape-only, matching the source paper's shape-only classification: absolute
|
|
69
|
+
latency tracks clock and thermal state, and comparing raw cycles would
|
|
70
|
+
separate a warm chip from a cold one rather than one die from another.
|
|
71
|
+
|
|
72
|
+
Returns ``inf`` for certificates that are not comparable at all — different
|
|
73
|
+
kernels, or different matrix geometry. Infinity rather than a large finite
|
|
74
|
+
number so an incomparable pair can never be silently averaged into a
|
|
75
|
+
distance distribution.
|
|
76
|
+
"""
|
|
77
|
+
if a.kernel_hash != b.kernel_hash:
|
|
78
|
+
return float("inf")
|
|
79
|
+
fa, fb = a.shape_features, b.shape_features
|
|
80
|
+
if not fa or len(fa) != len(fb):
|
|
81
|
+
return float("inf")
|
|
82
|
+
return sum(abs(x - y) for x, y in zip(fa, fb, strict=True)) / len(fa)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass(frozen=True)
|
|
86
|
+
class SeparabilityReport:
|
|
87
|
+
"""Whether same-model dies separated, and under what conditions."""
|
|
88
|
+
|
|
89
|
+
#: Certificates evaluated, and how many distinct physical devices they
|
|
90
|
+
#: came from. Both matter: 20 certificates from 2 dies is a much weaker
|
|
91
|
+
#: study than 20 from 10, and the numbers must travel together.
|
|
92
|
+
certificates: int
|
|
93
|
+
devices: int
|
|
94
|
+
|
|
95
|
+
same_device_pairs: int
|
|
96
|
+
different_device_pairs: int
|
|
97
|
+
|
|
98
|
+
#: Distance distributions, which are the actual finding. Overlap between
|
|
99
|
+
#: them is what "does not separate" means.
|
|
100
|
+
same_device_distances: list[float]
|
|
101
|
+
different_device_distances: list[float]
|
|
102
|
+
|
|
103
|
+
#: Leave-one-out nearest-neighbour accuracy over device labels.
|
|
104
|
+
leave_one_out_accuracy: float
|
|
105
|
+
#: Area under the ROC for the same-vs-different decision.
|
|
106
|
+
auc: float
|
|
107
|
+
#: Distance threshold maximising Youden's J, and its error rates.
|
|
108
|
+
best_threshold: float
|
|
109
|
+
false_positive_rate: float
|
|
110
|
+
false_negative_rate: float
|
|
111
|
+
|
|
112
|
+
#: Conditions the study ran under. Named explicitly so that the negative
|
|
113
|
+
#: result, if that is what this is, is bounded rather than universal.
|
|
114
|
+
conditions: list[str] = None # type: ignore[assignment]
|
|
115
|
+
|
|
116
|
+
def __post_init__(self) -> None:
|
|
117
|
+
if self.conditions is None:
|
|
118
|
+
object.__setattr__(self, "conditions", [])
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def validated(self) -> bool:
|
|
122
|
+
"""Whether D5 has landed well enough to gate §9.5 on.
|
|
123
|
+
|
|
124
|
+
Read by :class:`gpu_seal.reporting.MemoryHygieneEvidence` as
|
|
125
|
+
``same_model_classifier_validated``. Both thresholds are
|
|
126
|
+
pre-registered above; neither is adjustable from a result.
|
|
127
|
+
"""
|
|
128
|
+
return (
|
|
129
|
+
self.leave_one_out_accuracy >= VALIDATION_ACCURACY_THRESHOLD
|
|
130
|
+
and self.auc >= VALIDATION_AUC_THRESHOLD
|
|
131
|
+
and self.devices >= 2
|
|
132
|
+
and self.same_device_pairs > 0
|
|
133
|
+
and self.different_device_pairs > 0
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
def verdict(self) -> str:
|
|
137
|
+
"""One sentence, publishable either way."""
|
|
138
|
+
if self.devices < 2:
|
|
139
|
+
return (
|
|
140
|
+
f"Not evaluable: {self.certificates} certificate(s) from "
|
|
141
|
+
f"{self.devices} device(s). Same-model separation requires at "
|
|
142
|
+
f"least two physical dies of one advertised model (§9.8b)."
|
|
143
|
+
)
|
|
144
|
+
if self.validated:
|
|
145
|
+
return (
|
|
146
|
+
f"Same-model die separation achieved at tenant privilege: "
|
|
147
|
+
f"leave-one-out accuracy {self.leave_one_out_accuracy:.1%}, "
|
|
148
|
+
f"AUC {self.auc:.3f}, over {self.devices} dies and "
|
|
149
|
+
f"{self.certificates} certificates. §9.5 conclusions are no "
|
|
150
|
+
f"longer gated (CHARTER.md §13.1)."
|
|
151
|
+
)
|
|
152
|
+
return (
|
|
153
|
+
f"Same-model die separation NOT achieved under these conditions: "
|
|
154
|
+
f"leave-one-out accuracy {self.leave_one_out_accuracy:.1%} "
|
|
155
|
+
f"(pre-registered threshold {VALIDATION_ACCURACY_THRESHOLD:.0%}), "
|
|
156
|
+
f"AUC {self.auc:.3f} (threshold {VALIDATION_AUC_THRESHOLD}). "
|
|
157
|
+
f"Conditions: {'; '.join(self.conditions) or 'unstated'}. This is a "
|
|
158
|
+
f"bounded negative result, not a universal one — it constrains "
|
|
159
|
+
f"§9.5 and §13.1 grade A, which remain capped at U."
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
def to_dict(self) -> dict[str, Any]:
|
|
163
|
+
return {
|
|
164
|
+
"certificates": self.certificates,
|
|
165
|
+
"devices": self.devices,
|
|
166
|
+
"same_device_pairs": self.same_device_pairs,
|
|
167
|
+
"different_device_pairs": self.different_device_pairs,
|
|
168
|
+
"leave_one_out_accuracy": round(self.leave_one_out_accuracy, 4),
|
|
169
|
+
"auc": round(self.auc, 4),
|
|
170
|
+
"best_threshold": round(self.best_threshold, 6),
|
|
171
|
+
"false_positive_rate": round(self.false_positive_rate, 4),
|
|
172
|
+
"false_negative_rate": round(self.false_negative_rate, 4),
|
|
173
|
+
"validated": self.validated,
|
|
174
|
+
"pre_registered_thresholds": {
|
|
175
|
+
"leave_one_out_accuracy": VALIDATION_ACCURACY_THRESHOLD,
|
|
176
|
+
"auc": VALIDATION_AUC_THRESHOLD,
|
|
177
|
+
},
|
|
178
|
+
"conditions": self.conditions,
|
|
179
|
+
"verdict": self.verdict(),
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def evaluate_separability(
|
|
184
|
+
labelled: Sequence[tuple[str, TopologyCertificate]],
|
|
185
|
+
*,
|
|
186
|
+
conditions: Sequence[str] = (),
|
|
187
|
+
) -> SeparabilityReport:
|
|
188
|
+
"""Evaluate whether certificates separate by physical device.
|
|
189
|
+
|
|
190
|
+
Args:
|
|
191
|
+
labelled: ``(device_label, certificate)`` pairs. The label is ground
|
|
192
|
+
truth — which physical die the certificate came from — and in a
|
|
193
|
+
real §9.8b study it comes from the researcher's own record of
|
|
194
|
+
which instance was rented when, not from the instrument.
|
|
195
|
+
conditions: what the study held fixed (model, driver version, region,
|
|
196
|
+
load state). Carried into the verdict so a negative result is
|
|
197
|
+
bounded by the conditions that produced it.
|
|
198
|
+
"""
|
|
199
|
+
if len(labelled) < 2:
|
|
200
|
+
raise ValueError("separability needs at least two certificates")
|
|
201
|
+
|
|
202
|
+
labels = [label for label, _ in labelled]
|
|
203
|
+
certs = [cert for _, cert in labelled]
|
|
204
|
+
devices = len(set(labels))
|
|
205
|
+
|
|
206
|
+
same: list[float] = []
|
|
207
|
+
different: list[float] = []
|
|
208
|
+
for i in range(len(certs)):
|
|
209
|
+
for j in range(i + 1, len(certs)):
|
|
210
|
+
distance = certificate_distance(certs[i], certs[j])
|
|
211
|
+
if distance == float("inf"):
|
|
212
|
+
# Incomparable pairs are dropped, not scored. Including them
|
|
213
|
+
# would inflate separation with certificates that were never
|
|
214
|
+
# measuring the same thing.
|
|
215
|
+
continue
|
|
216
|
+
(same if labels[i] == labels[j] else different).append(distance)
|
|
217
|
+
|
|
218
|
+
accuracy = _leave_one_out_accuracy(labels, certs)
|
|
219
|
+
auc = _auc(same, different)
|
|
220
|
+
threshold, fpr, fnr = _best_threshold(same, different)
|
|
221
|
+
|
|
222
|
+
return SeparabilityReport(
|
|
223
|
+
certificates=len(certs),
|
|
224
|
+
devices=devices,
|
|
225
|
+
same_device_pairs=len(same),
|
|
226
|
+
different_device_pairs=len(different),
|
|
227
|
+
same_device_distances=same,
|
|
228
|
+
different_device_distances=different,
|
|
229
|
+
leave_one_out_accuracy=accuracy,
|
|
230
|
+
auc=auc,
|
|
231
|
+
best_threshold=threshold,
|
|
232
|
+
false_positive_rate=fpr,
|
|
233
|
+
false_negative_rate=fnr,
|
|
234
|
+
conditions=list(conditions),
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
# ---------------------------------------------------------------------------
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _leave_one_out_accuracy(
|
|
242
|
+
labels: Sequence[str], certs: Sequence[TopologyCertificate]
|
|
243
|
+
) -> float:
|
|
244
|
+
"""Nearest-neighbour accuracy, holding each certificate out in turn.
|
|
245
|
+
|
|
246
|
+
The metric the source paper reports (100% across Blackwell dies), so that
|
|
247
|
+
a same-model number is directly comparable to their cross-product one.
|
|
248
|
+
"""
|
|
249
|
+
correct = 0
|
|
250
|
+
scored = 0
|
|
251
|
+
for i in range(len(certs)):
|
|
252
|
+
best_distance = float("inf")
|
|
253
|
+
best_label: str | None = None
|
|
254
|
+
for j in range(len(certs)):
|
|
255
|
+
if i == j:
|
|
256
|
+
continue
|
|
257
|
+
distance = certificate_distance(certs[i], certs[j])
|
|
258
|
+
if distance < best_distance:
|
|
259
|
+
best_distance, best_label = distance, labels[j]
|
|
260
|
+
if best_label is None:
|
|
261
|
+
continue
|
|
262
|
+
scored += 1
|
|
263
|
+
if best_label == labels[i]:
|
|
264
|
+
correct += 1
|
|
265
|
+
return (correct / scored) if scored else 0.0
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _auc(same: Sequence[float], different: Sequence[float]) -> float:
|
|
269
|
+
"""Probability a random different-device pair is farther than a same one.
|
|
270
|
+
|
|
271
|
+
Computed directly from the Mann-Whitney U interpretation rather than by
|
|
272
|
+
integrating a sampled ROC curve — exact on small samples, which is what
|
|
273
|
+
§9.8b will have.
|
|
274
|
+
"""
|
|
275
|
+
if not same or not different:
|
|
276
|
+
return 0.0
|
|
277
|
+
wins = 0.0
|
|
278
|
+
for s in same:
|
|
279
|
+
for d in different:
|
|
280
|
+
if d > s:
|
|
281
|
+
wins += 1.0
|
|
282
|
+
elif d == s:
|
|
283
|
+
wins += 0.5
|
|
284
|
+
return wins / (len(same) * len(different))
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _best_threshold(
|
|
288
|
+
same: Sequence[float], different: Sequence[float]
|
|
289
|
+
) -> tuple[float, float, float]:
|
|
290
|
+
"""Threshold maximising Youden's J, with the error rates it produces.
|
|
291
|
+
|
|
292
|
+
Reported rather than tuned away: the false-positive rate here is the rate
|
|
293
|
+
at which two *different* dies would be called the same one, which is
|
|
294
|
+
exactly the error that would turn "the provider gave us a fresh GPU" into
|
|
295
|
+
a false claim of physical continuity.
|
|
296
|
+
"""
|
|
297
|
+
if not same or not different:
|
|
298
|
+
return 0.0, 1.0, 1.0
|
|
299
|
+
|
|
300
|
+
candidates = sorted(set(list(same) + list(different)))
|
|
301
|
+
best = (0.0, 1.0, 1.0)
|
|
302
|
+
best_j = -1.0
|
|
303
|
+
for threshold in candidates:
|
|
304
|
+
true_positive = sum(1 for s in same if s <= threshold) / len(same)
|
|
305
|
+
false_positive = sum(1 for d in different if d <= threshold) / len(different)
|
|
306
|
+
j = true_positive - false_positive
|
|
307
|
+
if j > best_j:
|
|
308
|
+
best_j = j
|
|
309
|
+
best = (threshold, false_positive, 1 - true_positive)
|
|
310
|
+
return best
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""Statistical requirements — CHARTER.md §12.
|
|
2
|
+
|
|
3
|
+
"For each conclusion report sample / success / failure / error / exclusion
|
|
4
|
+
counts, confidence interval (bootstrap CIs where appropriate), variance,
|
|
5
|
+
fingerprint stability, classifier confidence."
|
|
6
|
+
|
|
7
|
+
The important function here is the least interesting one.
|
|
8
|
+
:class:`ObservationCounts` exists so that a result cannot be reported without
|
|
9
|
+
its denominator. "No canary recovered" over 3 usable cycles out of 10 attempted
|
|
10
|
+
is a different claim from the same sentence over 10 out of 10, and the second
|
|
11
|
+
number is the one that goes missing when people write prose instead of
|
|
12
|
+
carrying a struct around.
|
|
13
|
+
|
|
14
|
+
Deliberately dependency-light: NumPy is used when present for speed, and the
|
|
15
|
+
pure-Python path gives identical answers, because a statistic that only
|
|
16
|
+
computes inside one environment is not reproducible either.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import math
|
|
22
|
+
import random
|
|
23
|
+
from collections.abc import Callable, Sequence
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"ObservationCounts",
|
|
29
|
+
"bootstrap_ci",
|
|
30
|
+
"proportion_ci",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True)
|
|
35
|
+
class ObservationCounts:
|
|
36
|
+
"""The denominator, carried alongside every conclusion."""
|
|
37
|
+
|
|
38
|
+
attempted: int
|
|
39
|
+
usable: int
|
|
40
|
+
positive: int
|
|
41
|
+
errored: int = 0
|
|
42
|
+
excluded: int = 0
|
|
43
|
+
exclusion_reasons: Sequence[str] = ()
|
|
44
|
+
|
|
45
|
+
def __post_init__(self) -> None:
|
|
46
|
+
if self.usable > self.attempted:
|
|
47
|
+
raise ValueError(
|
|
48
|
+
f"usable ({self.usable}) exceeds attempted ({self.attempted})"
|
|
49
|
+
)
|
|
50
|
+
if self.positive > self.usable:
|
|
51
|
+
raise ValueError(
|
|
52
|
+
f"positive ({self.positive}) exceeds usable ({self.usable})"
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def rate(self) -> float | None:
|
|
57
|
+
"""Positive rate over *usable* cycles. ``None`` when nothing is usable.
|
|
58
|
+
|
|
59
|
+
Returning ``None`` rather than 0.0 is the whole point: a run where
|
|
60
|
+
every cycle errored has a positive rate of *unknown*, and reporting it
|
|
61
|
+
as zero would turn a failed measurement into a clean result.
|
|
62
|
+
"""
|
|
63
|
+
return (self.positive / self.usable) if self.usable else None
|
|
64
|
+
|
|
65
|
+
def confidence_interval(self, confidence: float = 0.95) -> tuple[float, float] | None:
|
|
66
|
+
if not self.usable:
|
|
67
|
+
return None
|
|
68
|
+
return proportion_ci(self.positive, self.usable, confidence=confidence)
|
|
69
|
+
|
|
70
|
+
def to_dict(self) -> dict[str, Any]:
|
|
71
|
+
interval = self.confidence_interval()
|
|
72
|
+
return {
|
|
73
|
+
"attempted": self.attempted,
|
|
74
|
+
"usable": self.usable,
|
|
75
|
+
"positive": self.positive,
|
|
76
|
+
"errored": self.errored,
|
|
77
|
+
"excluded": self.excluded,
|
|
78
|
+
"exclusion_reasons": list(self.exclusion_reasons),
|
|
79
|
+
"rate": self.rate,
|
|
80
|
+
"confidence_interval_95": list(interval) if interval else None,
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def proportion_ci(
|
|
85
|
+
successes: int, trials: int, *, confidence: float = 0.95
|
|
86
|
+
) -> tuple[float, float]:
|
|
87
|
+
"""Wilson score interval for a proportion.
|
|
88
|
+
|
|
89
|
+
Wilson rather than the normal approximation because GPU-SEAL's headline
|
|
90
|
+
results are proportions near zero over small samples — 0 canaries in 10
|
|
91
|
+
cycles — and the normal approximation returns the degenerate interval
|
|
92
|
+
[0, 0] there. Claiming a 95% CI of exactly zero from ten observations is
|
|
93
|
+
the precise overclaim §12 exists to prevent. Wilson gives [0, 0.28], which
|
|
94
|
+
is the honest answer and a considerably less exciting one.
|
|
95
|
+
"""
|
|
96
|
+
if trials <= 0:
|
|
97
|
+
raise ValueError("trials must be positive")
|
|
98
|
+
if not 0 <= successes <= trials:
|
|
99
|
+
raise ValueError("successes must be within [0, trials]")
|
|
100
|
+
|
|
101
|
+
z = _z_for(confidence)
|
|
102
|
+
phat = successes / trials
|
|
103
|
+
denominator = 1 + z * z / trials
|
|
104
|
+
centre = (phat + z * z / (2 * trials)) / denominator
|
|
105
|
+
margin = (
|
|
106
|
+
z
|
|
107
|
+
* math.sqrt(phat * (1 - phat) / trials + z * z / (4 * trials * trials))
|
|
108
|
+
/ denominator
|
|
109
|
+
)
|
|
110
|
+
return max(0.0, centre - margin), min(1.0, centre + margin)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def bootstrap_ci(
|
|
114
|
+
samples: Sequence[float],
|
|
115
|
+
*,
|
|
116
|
+
statistic: Callable[[Sequence[float]], float] | None = None,
|
|
117
|
+
resamples: int = 2000,
|
|
118
|
+
confidence: float = 0.95,
|
|
119
|
+
seed: int = 0x5EA1,
|
|
120
|
+
) -> tuple[float, float]:
|
|
121
|
+
"""Percentile bootstrap CI for an arbitrary statistic.
|
|
122
|
+
|
|
123
|
+
Used for fingerprint stability and timing distributions, where the
|
|
124
|
+
sampling distribution has no closed form worth trusting. Seeded, because
|
|
125
|
+
a confidence interval that moves between runs of the analysis is not a
|
|
126
|
+
reproducible result (§10 reproducibility fields).
|
|
127
|
+
"""
|
|
128
|
+
if not samples:
|
|
129
|
+
raise ValueError("cannot bootstrap an empty sample")
|
|
130
|
+
|
|
131
|
+
values = list(samples)
|
|
132
|
+
estimator = statistic or _mean
|
|
133
|
+
# A seeded Mersenne Twister, on purpose: resampling an already-collected
|
|
134
|
+
# sample is a numerical procedure, not a security one, and it has to be
|
|
135
|
+
# reproducible. A cryptographic generator would make the confidence
|
|
136
|
+
# interval move between runs of the same analysis.
|
|
137
|
+
rng = random.Random(seed) # noqa: S311
|
|
138
|
+
n = len(values)
|
|
139
|
+
|
|
140
|
+
estimates = [
|
|
141
|
+
estimator([values[rng.randrange(n)] for _ in range(n)])
|
|
142
|
+
for _ in range(resamples)
|
|
143
|
+
]
|
|
144
|
+
estimates.sort()
|
|
145
|
+
|
|
146
|
+
tail = (1 - confidence) / 2
|
|
147
|
+
lower = estimates[int(tail * resamples)]
|
|
148
|
+
upper = estimates[min(int((1 - tail) * resamples), resamples - 1)]
|
|
149
|
+
return lower, upper
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _mean(values: Sequence[float]) -> float:
|
|
153
|
+
return sum(values) / len(values)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
#: Two-sided z scores. A lookup rather than an inverse-normal implementation:
|
|
157
|
+
#: only these three levels are ever reported, and a table cannot be subtly
|
|
158
|
+
#: wrong in a way that survives review.
|
|
159
|
+
_Z_SCORES = {0.90: 1.6449, 0.95: 1.9600, 0.99: 2.5758}
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _z_for(confidence: float) -> float:
|
|
163
|
+
try:
|
|
164
|
+
return _Z_SCORES[round(confidence, 2)]
|
|
165
|
+
except KeyError:
|
|
166
|
+
raise ValueError(
|
|
167
|
+
f"confidence must be one of {sorted(_Z_SCORES)}; got {confidence}. "
|
|
168
|
+
f"Reporting an unusual level invites the reader to wonder which "
|
|
169
|
+
f"one was chosen after seeing the data."
|
|
170
|
+
) from None
|