proofbundle 1.8.0__tar.gz → 1.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. proofbundle-1.9.0/PKG-INFO +190 -0
  2. proofbundle-1.9.0/README.md +142 -0
  3. {proofbundle-1.8.0 → proofbundle-1.9.0}/pyproject.toml +1 -1
  4. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/__init__.py +4 -1
  5. proofbundle-1.9.0/src/proofbundle/beacon.py +109 -0
  6. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/cli.py +36 -8
  7. proofbundle-1.9.0/src/proofbundle.egg-info/PKG-INFO +190 -0
  8. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/SOURCES.txt +2 -0
  9. proofbundle-1.9.0/tests/test_beacon.py +116 -0
  10. proofbundle-1.8.0/PKG-INFO +0 -604
  11. proofbundle-1.8.0/README.md +0 -556
  12. proofbundle-1.8.0/src/proofbundle.egg-info/PKG-INFO +0 -604
  13. {proofbundle-1.8.0 → proofbundle-1.9.0}/LICENSE +0 -0
  14. {proofbundle-1.8.0 → proofbundle-1.9.0}/setup.cfg +0 -0
  15. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/_inspect_registry.py +0 -0
  16. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/_integration.py +0 -0
  17. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/adapters/__init__.py +0 -0
  18. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/adapters/_provenance.py +0 -0
  19. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/adapters/eee.py +0 -0
  20. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/adapters/inspect_ai.py +0 -0
  21. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/adapters/lm_eval.py +0 -0
  22. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/adapters/promptfoo.py +0 -0
  23. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/adapters/samples.py +0 -0
  24. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/bundle.py +0 -0
  25. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/checkpoint.py +0 -0
  26. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/demo.py +0 -0
  27. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/dsse.py +0 -0
  28. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/eee_eval_schema.json +0 -0
  29. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/emit.py +0 -0
  30. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/errors.py +0 -0
  31. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/evalclaim.py +0 -0
  32. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/hf_evals.py +0 -0
  33. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/inspect_hook.py +0 -0
  34. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/intoto.py +0 -0
  35. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/kbjwt.py +0 -0
  36. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/merkle.py +0 -0
  37. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/persample.py +0 -0
  38. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/prereg.py +0 -0
  39. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/py.typed +0 -0
  40. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/pytest_plugin.py +0 -0
  41. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/sdjwt.py +0 -0
  42. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/sdjwt_issue.py +0 -0
  43. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/signature.py +0 -0
  44. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/statuslist.py +0 -0
  45. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle/tlogproof.py +0 -0
  46. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/dependency_links.txt +0 -0
  47. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/entry_points.txt +0 -0
  48. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/requires.txt +0 -0
  49. {proofbundle-1.8.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/top_level.txt +0 -0
  50. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_adapters.py +0 -0
  51. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_adversarial.py +0 -0
  52. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_bundle.py +0 -0
  53. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_bundle_robustness.py +0 -0
  54. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_checkpoint.py +0 -0
  55. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_cli.py +0 -0
  56. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_cli_eval.py +0 -0
  57. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_cosignature.py +0 -0
  58. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_cosignature_mldsa.py +0 -0
  59. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_demo.py +0 -0
  60. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_eee.py +0 -0
  61. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_emit.py +0 -0
  62. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_eval_claim_schema.py +0 -0
  63. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_evalclaim.py +0 -0
  64. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_examples.py +0 -0
  65. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_fuzz_parsers.py +0 -0
  66. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_hf_evals.py +0 -0
  67. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_inspect_hook.py +0 -0
  68. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_intoto.py +0 -0
  69. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_intoto_dsse.py +0 -0
  70. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_kbjwt.py +0 -0
  71. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_merkle.py +0 -0
  72. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_merkle_property.py +0 -0
  73. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_persample.py +0 -0
  74. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_prereg.py +0 -0
  75. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_promptfoo.py +0 -0
  76. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_provenance.py +0 -0
  77. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_pytest_plugin.py +0 -0
  78. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_rekor_interop.py +0 -0
  79. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_rfc6962_external_vectors.py +0 -0
  80. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_schema.py +0 -0
  81. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_sdjwt_issue.py +0 -0
  82. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_sdjwt_reference.py +0 -0
  83. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_signature.py +0 -0
  84. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_statuslist.py +0 -0
  85. {proofbundle-1.8.0 → proofbundle-1.9.0}/tests/test_tlogproof.py +0 -0
@@ -0,0 +1,190 @@
1
+ Metadata-Version: 2.4
2
+ Name: proofbundle
3
+ Version: 1.9.0
4
+ Summary: Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT.
5
+ Author: Konrad Gruszka
6
+ License: MIT
7
+ Project-URL: Homepage, https://b7n0de.com
8
+ Project-URL: Repository, https://github.com/b7n0de/proofbundle
9
+ Project-URL: Issues, https://github.com/b7n0de/proofbundle/issues
10
+ Project-URL: Changelog, https://github.com/b7n0de/proofbundle/blob/main/CHANGELOG.md
11
+ Project-URL: Documentation, https://github.com/b7n0de/proofbundle#readme
12
+ Keywords: cryptography,merkle,transparency-log,ed25519,sd-jwt,verifiable-credentials,attestation,provenance,rfc6962
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Programming Language :: Python :: 3.14
22
+ Classifier: Topic :: Security :: Cryptography
23
+ Requires-Python: >=3.10
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: cryptography>=42
27
+ Provides-Extra: sdjwt
28
+ Provides-Extra: eval
29
+ Requires-Dist: rfc8785>=0.1.4; extra == "eval"
30
+ Provides-Extra: adapters
31
+ Provides-Extra: pq
32
+ Requires-Dist: cryptography>=48; extra == "pq"
33
+ Provides-Extra: pytest
34
+ Requires-Dist: pytest>=7; extra == "pytest"
35
+ Provides-Extra: inspect
36
+ Requires-Dist: inspect_ai<0.4,>=0.3.112; extra == "inspect"
37
+ Provides-Extra: dev
38
+ Requires-Dist: pytest>=7; extra == "dev"
39
+ Requires-Dist: ruff>=0.5; extra == "dev"
40
+ Requires-Dist: jsonschema>=4; extra == "dev"
41
+ Requires-Dist: mypy>=1.8; extra == "dev"
42
+ Requires-Dist: build>=1; extra == "dev"
43
+ Requires-Dist: hypothesis>=6; extra == "dev"
44
+ Requires-Dist: rfc8785>=0.1.4; extra == "dev"
45
+ Requires-Dist: sd-jwt>=0.10; extra == "dev"
46
+ Requires-Dist: inspect_ai<0.4,>=0.3.112; extra == "dev"
47
+ Dynamic: license-file
48
+
49
+ <div align="center">
50
+
51
+ <picture>
52
+ <source media="(prefers-color-scheme: dark)" srcset="assets/b7n0de-logo-dark.svg">
53
+ <img alt="b7n0de, Verified AI Work" src="assets/b7n0de-logo.svg" height="60">
54
+ </picture>
55
+
56
+ <h1>proofbundle</h1>
57
+
58
+ **Turn an AI eval result into one portable, offline-verifiable receipt.**
59
+ It proves *who signed these exact bytes* and *that nothing changed since* — not that the number is
60
+ true. Ed25519 + RFC 6962 Merkle, one file, no server, no network.
61
+
62
+ [![CI](https://github.com/b7n0de/proofbundle/actions/workflows/ci.yml/badge.svg)](https://github.com/b7n0de/proofbundle/actions/workflows/ci.yml)
63
+ [![License: MIT](https://img.shields.io/badge/license-MIT-D6248A.svg)](LICENSE)
64
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
65
+ [![Mutation tested](https://img.shields.io/badge/tests-mutation_gated-D6248A.svg)](scripts/mutation_check.py)
66
+ <!-- PyPI / Downloads / SLSA / PEP 740 badges are enabled on the first PyPI release — see RELEASE.md. -->
67
+
68
+ </div>
69
+
70
+ ## The problem
71
+
72
+ Every AI eval number you read — a safety benchmark, a capability score, a leaderboard entry — is an
73
+ **unverifiable claim**. You trust the lab. There's no portable way to check, offline, that a result
74
+ was signed by a stated party, hasn't been altered, and covers the samples it claims.
75
+
76
+ proofbundle is that check. It's a small MIT-licensed Python tool (a compact, auditable trusted core,
77
+ depends only on [`cryptography`](https://cryptography.io)) that turns a result into a signed
78
+ receipt anyone can verify from a single file — and it's honest about the line it does not cross.
79
+
80
+ ## 60-second try (offline, no setup)
81
+
82
+ ```bash
83
+ pip install "proofbundle[eval]"
84
+ proofbundle demo
85
+ ```
86
+
87
+ You'll see an honest receipt verify `=> OK`, then six independent tampers each verify `FAILED`, then
88
+ a swapped sample get caught — all in memory. The command exits non-zero if any tamper slips through,
89
+ so it's also a self-test. Full walkthrough: **[docs/DEMO.md](docs/DEMO.md)**.
90
+
91
+ ```bash
92
+ # your own receipt, from a signed payload:
93
+ proofbundle emit --payload-file result.json --new-key signer.key --out receipt.json
94
+ proofbundle verify receipt.json # exit 0 = OK, 1 = failed, 2 = malformed
95
+ ```
96
+
97
+ ## What a receipt proves — and what it doesn't
98
+
99
+ | ✅ It proves | ❌ It does **not** prove |
100
+ |---|---|
101
+ | These exact bytes were signed by this key (**authorship**) | That the number is **true** |
102
+ | Nothing changed since signing (**integrity**, Ed25519 + RFC 6962) | That the **issuer is honest** |
103
+ | The result is attributable to a stated issuer | That the **eval was well-designed** |
104
+ | A threshold was met while hiding the model/dataset (salted commitments) | That there was **no cherry-picking** — unless pre-registered |
105
+ | Optionally: individual samples, offline-auditable (per-sample Merkle) | That the **computation was correct** — that needs a TEE or independent reproduction |
106
+
107
+ This boundary is the point, not a weakness. A receipt makes a claim **attributable, tamper-evident,
108
+ and — with pre-registration and per-sample auditing — bounded and spot-checkable**. Full detail:
109
+ **[THREAT_MODEL.md](THREAT_MODEL.md)**.
110
+
111
+ ## How it fits together
112
+
113
+ ```mermaid
114
+ flowchart LR
115
+ H["eval harness<br/>inspect_ai · lm-eval · promptfoo · pytest"] --> A["adapter → signed claim<br/>salted commitments · provenance · samples root"]
116
+ A --> R["receipt<br/>one portable file"]
117
+ R --> V{{"proofbundle verify — offline"}}
118
+ V --> C["signature · Merkle inclusion · SD-JWT/KB ·<br/>witness quorum · status list · sample openings"]
119
+ C --> OK(["=> OK / FAILED"])
120
+ style V fill:#D6248A,stroke:#D6248A,color:#fff
121
+ style OK fill:#D6248A,stroke:#D6248A,color:#fff
122
+ ```
123
+
124
+ ## What's in the box
125
+
126
+ - **Core** — Ed25519 signature + RFC 6962 / 9162 Merkle inclusion, verified fully offline. Checks a
127
+ real [Sigstore Rekor](https://docs.sigstore.dev/) proof, so correctness isn't self-referential.
128
+ - **Eval receipts** — a signed claim (`metric ⋈ threshold`, `n`, salted model/dataset commitments,
129
+ assurance level, provenance) from your run. See [EVAL_CLAIM.md](EVAL_CLAIM.md).
130
+ - **Selective disclosure** — SD-JWT ([RFC 9901](https://datatracker.ietf.org/doc/rfc9901/)) with Key
131
+ Binding: prove a threshold while withholding the exact score.
132
+ - **Transparency-log interop** — C2SP `tlog-checkpoint` / cosignature / `.tlog-proof`, with
133
+ post-quantum **ML-DSA-44** witness cosignatures. Optional Token-Status-List revocation snapshots.
134
+ - **Per-sample audit** — commit to every sample; an auditor challenges random indices (with a fresh
135
+ nonce or a **public randomness beacon**, v1.9) and openings must bind to the signed root. Catches
136
+ 1% sample-doctoring with 95% confidence at 300 samples, regardless of run size.
137
+ - **Pre-registration** — `proofbundle prereg <plan>` commits to the protocol before the run, so
138
+ best-of-many publishing becomes visible.
139
+ - **Integrations** — opt-in inspect_ai end-of-task hook and pytest plugin (emit only when
140
+ `PROOFBUNDLE_EMIT=1` / `--proofbundle`), plus a Hugging Face Community Evals bridge. See
141
+ [INTEGRATIONS.md](INTEGRATIONS.md).
142
+
143
+ ## Docs
144
+
145
+ | For… | Read |
146
+ |---|---|
147
+ | Skeptics (why not SHA-256 / Sigstore / trust the issuer) | [docs/FAQ.md](docs/FAQ.md) |
148
+ | Reviewers (30-minute adversarial audit path) | [docs/REVIEWERS.md](docs/REVIEWERS.md) |
149
+ | Where every trust anchor comes from | [docs/TRUST_ANCHORS.md](docs/TRUST_ANCHORS.md) |
150
+ | The demos, tier by tier | [docs/DEMO.md](docs/DEMO.md) |
151
+ | The normative format + verification order | [SPEC.md](SPEC.md) |
152
+ | Honest comparison to Rekor / in-toto / OMS / ValiChord | [INTEROP.md](INTEROP.md) |
153
+ | Regulatory mapping (and what to never claim) | [COMPLIANCE.md](COMPLIANCE.md) |
154
+ | Funders / role fit | [docs/PROJECT_BRIEF.md](docs/PROJECT_BRIEF.md) |
155
+
156
+ ## Install
157
+
158
+ ```bash
159
+ pip install proofbundle # core: offline verify + plain emit (dependency-free)
160
+ pip install "proofbundle[eval]" # + eval receipts, prereg, and the demo (adds rfc8785 JCS)
161
+ pip install "proofbundle[eval]" # emit eval receipts (adds an RFC 8785 canonicalizer)
162
+ pip install "proofbundle[inspect]" # inspect_ai adapter + hook
163
+ pip install "proofbundle[pq]" # verify ML-DSA-44 (post-quantum) witness cosignatures
164
+ ```
165
+
166
+ Requires Python 3.10+. The verify path never rolls its own crypto — Ed25519 comes from
167
+ `cryptography`; Merkle hashing is RFC 6962.
168
+
169
+ ## Status & scope
170
+
171
+ Beta, SemVer-committed, 299 tests + a CI mutation gate + property-based parser fuzzing. Correctness
172
+ is anchored to external RFC 6962 vectors and a real Rekor proof, not just its own bundles. It is
173
+ **not** a log service, a full in-toto client, a TEE, a consensus network, or a compliance product
174
+ by itself — it is the small, offline, standards-native receipt layer between them. Security policy:
175
+ [SECURITY.md](SECURITY.md).
176
+
177
+ ## Contributing
178
+
179
+ See [CONTRIBUTING.md](CONTRIBUTING.md) and the [Code of Conduct](CODE_OF_CONDUCT.md). Good first
180
+ issues are labeled [`good-first-issue`](https://github.com/b7n0de/proofbundle/labels/good-first-issue);
181
+ security findings go through [SECURITY.md](SECURITY.md). The verifier core aims to stay small,
182
+ dependency-light, and correct.
183
+
184
+ ## License
185
+
186
+ MIT — see [LICENSE](LICENSE).
187
+
188
+ ---
189
+
190
+ <p align="center"><sub>proofbundle is part of <b>b7n0de</b>, Verified AI Work · <a href="https://b7n0de.com">b7n0de.com</a></sub></p>
@@ -0,0 +1,142 @@
1
+ <div align="center">
2
+
3
+ <picture>
4
+ <source media="(prefers-color-scheme: dark)" srcset="assets/b7n0de-logo-dark.svg">
5
+ <img alt="b7n0de, Verified AI Work" src="assets/b7n0de-logo.svg" height="60">
6
+ </picture>
7
+
8
+ <h1>proofbundle</h1>
9
+
10
+ **Turn an AI eval result into one portable, offline-verifiable receipt.**
11
+ It proves *who signed these exact bytes* and *that nothing changed since* — not that the number is
12
+ true. Ed25519 + RFC 6962 Merkle, one file, no server, no network.
13
+
14
+ [![CI](https://github.com/b7n0de/proofbundle/actions/workflows/ci.yml/badge.svg)](https://github.com/b7n0de/proofbundle/actions/workflows/ci.yml)
15
+ [![License: MIT](https://img.shields.io/badge/license-MIT-D6248A.svg)](LICENSE)
16
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
17
+ [![Mutation tested](https://img.shields.io/badge/tests-mutation_gated-D6248A.svg)](scripts/mutation_check.py)
18
+ <!-- PyPI / Downloads / SLSA / PEP 740 badges are enabled on the first PyPI release — see RELEASE.md. -->
19
+
20
+ </div>
21
+
22
+ ## The problem
23
+
24
+ Every AI eval number you read — a safety benchmark, a capability score, a leaderboard entry — is an
25
+ **unverifiable claim**. You trust the lab. There's no portable way to check, offline, that a result
26
+ was signed by a stated party, hasn't been altered, and covers the samples it claims.
27
+
28
+ proofbundle is that check. It's a small MIT-licensed Python tool (a compact, auditable trusted core,
29
+ depends only on [`cryptography`](https://cryptography.io)) that turns a result into a signed
30
+ receipt anyone can verify from a single file — and it's honest about the line it does not cross.
31
+
32
+ ## 60-second try (offline, no setup)
33
+
34
+ ```bash
35
+ pip install "proofbundle[eval]"
36
+ proofbundle demo
37
+ ```
38
+
39
+ You'll see an honest receipt verify `=> OK`, then six independent tampers each verify `FAILED`, then
40
+ a swapped sample get caught — all in memory. The command exits non-zero if any tamper slips through,
41
+ so it's also a self-test. Full walkthrough: **[docs/DEMO.md](docs/DEMO.md)**.
42
+
43
+ ```bash
44
+ # your own receipt, from a signed payload:
45
+ proofbundle emit --payload-file result.json --new-key signer.key --out receipt.json
46
+ proofbundle verify receipt.json # exit 0 = OK, 1 = failed, 2 = malformed
47
+ ```
48
+
49
+ ## What a receipt proves — and what it doesn't
50
+
51
+ | ✅ It proves | ❌ It does **not** prove |
52
+ |---|---|
53
+ | These exact bytes were signed by this key (**authorship**) | That the number is **true** |
54
+ | Nothing changed since signing (**integrity**, Ed25519 + RFC 6962) | That the **issuer is honest** |
55
+ | The result is attributable to a stated issuer | That the **eval was well-designed** |
56
+ | A threshold was met while hiding the model/dataset (salted commitments) | That there was **no cherry-picking** — unless pre-registered |
57
+ | Optionally: individual samples, offline-auditable (per-sample Merkle) | That the **computation was correct** — that needs a TEE or independent reproduction |
58
+
59
+ This boundary is the point, not a weakness. A receipt makes a claim **attributable, tamper-evident,
60
+ and — with pre-registration and per-sample auditing — bounded and spot-checkable**. Full detail:
61
+ **[THREAT_MODEL.md](THREAT_MODEL.md)**.
62
+
63
+ ## How it fits together
64
+
65
+ ```mermaid
66
+ flowchart LR
67
+ H["eval harness<br/>inspect_ai · lm-eval · promptfoo · pytest"] --> A["adapter → signed claim<br/>salted commitments · provenance · samples root"]
68
+ A --> R["receipt<br/>one portable file"]
69
+ R --> V{{"proofbundle verify — offline"}}
70
+ V --> C["signature · Merkle inclusion · SD-JWT/KB ·<br/>witness quorum · status list · sample openings"]
71
+ C --> OK(["=> OK / FAILED"])
72
+ style V fill:#D6248A,stroke:#D6248A,color:#fff
73
+ style OK fill:#D6248A,stroke:#D6248A,color:#fff
74
+ ```
75
+
76
+ ## What's in the box
77
+
78
+ - **Core** — Ed25519 signature + RFC 6962 / 9162 Merkle inclusion, verified fully offline. Checks a
79
+ real [Sigstore Rekor](https://docs.sigstore.dev/) proof, so correctness isn't self-referential.
80
+ - **Eval receipts** — a signed claim (`metric ⋈ threshold`, `n`, salted model/dataset commitments,
81
+ assurance level, provenance) from your run. See [EVAL_CLAIM.md](EVAL_CLAIM.md).
82
+ - **Selective disclosure** — SD-JWT ([RFC 9901](https://datatracker.ietf.org/doc/rfc9901/)) with Key
83
+ Binding: prove a threshold while withholding the exact score.
84
+ - **Transparency-log interop** — C2SP `tlog-checkpoint` / cosignature / `.tlog-proof`, with
85
+ post-quantum **ML-DSA-44** witness cosignatures. Optional Token-Status-List revocation snapshots.
86
+ - **Per-sample audit** — commit to every sample; an auditor challenges random indices (with a fresh
87
+ nonce or a **public randomness beacon**, v1.9) and openings must bind to the signed root. Catches
88
+ 1% sample-doctoring with 95% confidence at 300 samples, regardless of run size.
89
+ - **Pre-registration** — `proofbundle prereg <plan>` commits to the protocol before the run, so
90
+ best-of-many publishing becomes visible.
91
+ - **Integrations** — opt-in inspect_ai end-of-task hook and pytest plugin (emit only when
92
+ `PROOFBUNDLE_EMIT=1` / `--proofbundle`), plus a Hugging Face Community Evals bridge. See
93
+ [INTEGRATIONS.md](INTEGRATIONS.md).
94
+
95
+ ## Docs
96
+
97
+ | For… | Read |
98
+ |---|---|
99
+ | Skeptics (why not SHA-256 / Sigstore / trust the issuer) | [docs/FAQ.md](docs/FAQ.md) |
100
+ | Reviewers (30-minute adversarial audit path) | [docs/REVIEWERS.md](docs/REVIEWERS.md) |
101
+ | Where every trust anchor comes from | [docs/TRUST_ANCHORS.md](docs/TRUST_ANCHORS.md) |
102
+ | The demos, tier by tier | [docs/DEMO.md](docs/DEMO.md) |
103
+ | The normative format + verification order | [SPEC.md](SPEC.md) |
104
+ | Honest comparison to Rekor / in-toto / OMS / ValiChord | [INTEROP.md](INTEROP.md) |
105
+ | Regulatory mapping (and what to never claim) | [COMPLIANCE.md](COMPLIANCE.md) |
106
+ | Funders / role fit | [docs/PROJECT_BRIEF.md](docs/PROJECT_BRIEF.md) |
107
+
108
+ ## Install
109
+
110
+ ```bash
111
+ pip install proofbundle # core: offline verify + plain emit (dependency-free)
112
+ pip install "proofbundle[eval]" # + eval receipts, prereg, and the demo (adds rfc8785 JCS)
113
+ pip install "proofbundle[eval]" # emit eval receipts (adds an RFC 8785 canonicalizer)
114
+ pip install "proofbundle[inspect]" # inspect_ai adapter + hook
115
+ pip install "proofbundle[pq]" # verify ML-DSA-44 (post-quantum) witness cosignatures
116
+ ```
117
+
118
+ Requires Python 3.10+. The verify path never rolls its own crypto — Ed25519 comes from
119
+ `cryptography`; Merkle hashing is RFC 6962.
120
+
121
+ ## Status & scope
122
+
123
+ Beta, SemVer-committed, 299 tests + a CI mutation gate + property-based parser fuzzing. Correctness
124
+ is anchored to external RFC 6962 vectors and a real Rekor proof, not just its own bundles. It is
125
+ **not** a log service, a full in-toto client, a TEE, a consensus network, or a compliance product
126
+ by itself — it is the small, offline, standards-native receipt layer between them. Security policy:
127
+ [SECURITY.md](SECURITY.md).
128
+
129
+ ## Contributing
130
+
131
+ See [CONTRIBUTING.md](CONTRIBUTING.md) and the [Code of Conduct](CODE_OF_CONDUCT.md). Good first
132
+ issues are labeled [`good-first-issue`](https://github.com/b7n0de/proofbundle/labels/good-first-issue);
133
+ security findings go through [SECURITY.md](SECURITY.md). The verifier core aims to stay small,
134
+ dependency-light, and correct.
135
+
136
+ ## License
137
+
138
+ MIT — see [LICENSE](LICENSE).
139
+
140
+ ---
141
+
142
+ <p align="center"><sub>proofbundle is part of <b>b7n0de</b>, Verified AI Work · <a href="https://b7n0de.com">b7n0de.com</a></sub></p>
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "proofbundle"
7
- version = "1.8.0"
7
+ version = "1.9.0"
8
8
  description = "Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -13,7 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  from typing import TYPE_CHECKING
15
15
 
16
- __version__ = "1.8.0"
16
+ __version__ = "1.9.0"
17
17
 
18
18
  __all__ = [
19
19
  "__version__",
@@ -38,6 +38,7 @@ __all__ = [
38
38
  "audit_challenge",
39
39
  "prereg_hash",
40
40
  "verify_prereg",
41
+ "beacon_audit_challenge",
41
42
  "VerificationResult",
42
43
  "Check",
43
44
  "ProofBundleError",
@@ -63,6 +64,7 @@ _LAZY = {
63
64
  "audit_challenge": ".persample",
64
65
  "prereg_hash": ".prereg",
65
66
  "verify_prereg": ".prereg",
67
+ "beacon_audit_challenge": ".beacon",
66
68
  }
67
69
 
68
70
  if TYPE_CHECKING: # static analysers + IDEs see the real names/types; runtime stays lazy
@@ -74,6 +76,7 @@ if TYPE_CHECKING: # static analysers + IDEs see the real names/types; runtime s
74
76
  from .hf_evals import receipt_token, verify_receipt_token
75
77
  from .persample import (audit_challenge, build_sample_tree, sample_opening,
76
78
  verify_sample_opening)
79
+ from .beacon import beacon_audit_challenge
77
80
  from .prereg import prereg_hash, verify_prereg
78
81
  from .statuslist import verify_status_snapshot
79
82
  from .tlogproof import verify_tlog_proof
@@ -0,0 +1,109 @@
1
+ """Public-randomness audit challenges (v1.9) — non-interactive, publicly re-derivable.
2
+
3
+ The per-sample audit (SPEC §7g) has three challenge modes. Two shipped in v1.5: an *auditor
4
+ nonce* (interactive, grinding-impossible) and *self-challenge* (a documented-grindable sanity
5
+ check). This module formalizes the third — a **public randomness beacon** — so an audit needs no
6
+ live auditor and anyone can re-derive the same challenged indices from published data.
7
+
8
+ Why a beacon RESISTS grinding without a live auditor, AND WHAT IT ASSUMES: the producer signs the receipt (with its
9
+ ``samples_root``) at time T. A beacon pulse from a round whose emission time is *after* T did not
10
+ exist when the producer committed, so the producer cannot have chosen the samples to fit the
11
+ challenge. This is the RFC 3797 pattern ("derive selections from pre-specified future public
12
+ randomness") applied to sample auditing. Established beacons: the drand League of Entropy
13
+ (``randomness`` = 32 bytes per round) and the NIST Interoperable Randomness Beacon
14
+ (``outputValue`` = 64 bytes per pulse).
15
+
16
+ Offline-first: proofbundle never fetches. The relying party obtains the pulse out of band (or it
17
+ is bundled) and passes its raw bytes here. The returned ``AuditRequest`` records the beacon id
18
+ and round so a third party can fetch the *same* pulse and re-run ``audit_challenge`` to the
19
+ identical indices — the audit is reproducible without trusting the auditor.
20
+
21
+ Soundness caveats, stated honestly (the beacon mode is grinding-resistant ONLY under these, otherwise it
22
+ is no stronger than the documented-grindable self-challenge, so a relying party MUST check them):
23
+ 1. Beacon signature: this module does NOT verify the beacon's own signature (drand pulses are BLS-signed;
24
+ NIST pulses are RSA-signed) — that is a separate trust anchor the relying party validates with the
25
+ beacon's public key out of band, like every anchor in docs/TRUST_ANCHORS.md.
26
+ 2. Ordering depends on a SELF-DECLARED, UNVERIFIED timestamp. The "the round emitted after time T" argument
27
+ uses the receipt's own ``timestamp`` field, which the producer writes and signs but which nothing here
28
+ proves is the true commit time. A dishonest producer can BACKDATE ``timestamp``, wait for a round R whose
29
+ randomness is already public, grind sample trees against R, and sign a receipt claiming a timestamp before
30
+ R's emission. Ed25519 only proves the false timestamp was signed, not that it is true. So the relying party
31
+ must corroborate the ordering from an INDEPENDENT source (a transparency-log inclusion time for the receipt,
32
+ a witnessed/notarized timestamp, or a pre-registered round id chosen before the run) — not from the receipt's
33
+ own timestamp alone. Without independent corroboration the beacon mode does NOT close producer-side grinding.
34
+ 3. Round independence: the round id must be fixed BEFORE the samples are committed (a future round), not chosen
35
+ by the producer after seeing published randomness; this module records the round but cannot enforce that it
36
+ was pre-committed.
37
+ """
38
+
39
+ from __future__ import annotations
40
+
41
+ import hashlib
42
+ from typing import List
43
+
44
+ from .errors import BundleFormatError
45
+ from .persample import audit_challenge
46
+
47
+ __all__ = ["AuditRequest", "beacon_nonce", "beacon_audit_challenge"]
48
+
49
+ # Beacon randomness lengths we accept (raw bytes). drand = 32, NIST = 64. Other lengths are
50
+ # allowed too (any >=16 bytes of published randomness), but these are the named, verified ones.
51
+ _MIN_PULSE_BYTES = 16
52
+ _BEACON_DOMAIN = b"proofbundle/v1.9/beacon-nonce\x00"
53
+
54
+
55
+ class AuditRequest:
56
+ """A reproducible audit challenge derived from a public beacon pulse.
57
+
58
+ Everything a third party needs to RE-derive the same indices: the beacon id, the round, and
59
+ the resulting indices. ``as_dict`` is JSON-serializable for publishing alongside the receipt.
60
+ """
61
+
62
+ __slots__ = ("beacon", "round", "n", "k", "indices")
63
+
64
+ def __init__(self, beacon: str, round_: int, n: int, k: int, indices: List[int]):
65
+ self.beacon = beacon
66
+ self.round = round_
67
+ self.n = n
68
+ self.k = k
69
+ self.indices = indices
70
+
71
+ def as_dict(self) -> dict:
72
+ return {"beacon": self.beacon, "round": self.round, "n": self.n, "k": self.k,
73
+ "indices": list(self.indices)}
74
+
75
+ def __repr__(self) -> str: # pragma: no cover - cosmetic
76
+ return f"AuditRequest(beacon={self.beacon!r}, round={self.round}, k={self.k}, n={self.n})"
77
+
78
+
79
+ def beacon_nonce(pulse_randomness: bytes, beacon: str, round_: int) -> bytes:
80
+ """Derive the ``audit_challenge`` nonce from a beacon pulse, binding the beacon id + round.
81
+
82
+ The nonce is ``SHA-256(domain ‖ beacon ‖ 0x00 ‖ u64(round) ‖ pulse_randomness)`` — so two
83
+ different beacons or rounds never collide to the same challenge, and the nonce is a fixed
84
+ 32 bytes regardless of the beacon's own randomness length.
85
+ """
86
+ if not isinstance(pulse_randomness, (bytes, bytearray)) or len(pulse_randomness) < _MIN_PULSE_BYTES:
87
+ raise BundleFormatError(
88
+ f"beacon pulse randomness must be at least {_MIN_PULSE_BYTES} bytes")
89
+ if not beacon or "\x00" in beacon:
90
+ raise BundleFormatError("beacon id must be a non-empty string without NUL")
91
+ if isinstance(round_, bool) or not isinstance(round_, int) or round_ < 0 or round_ >= 2**64:
92
+ raise BundleFormatError("beacon round must be a u64 (0 <= round < 2**64)")
93
+ return hashlib.sha256(_BEACON_DOMAIN + beacon.encode("utf-8") + b"\x00"
94
+ + round_.to_bytes(8, "big") + bytes(pulse_randomness)).digest()
95
+
96
+
97
+ def beacon_audit_challenge(root, n: int, k: int, *, pulse_randomness: bytes, beacon: str,
98
+ round_: int) -> AuditRequest:
99
+ """Derive a reproducible per-sample audit challenge from a public beacon pulse.
100
+
101
+ ``root``/``n``/``k`` are the receipt's signed samples root, committed count, and the number
102
+ of samples to challenge. ``pulse_randomness`` is the raw randomness of a beacon pulse whose
103
+ round emits AFTER the receipt's signed timestamp (the relying party checks that). Returns an
104
+ :class:`AuditRequest` recording the beacon id + round + indices, so the challenge is
105
+ publicly re-derivable — no live auditor, no trust in who ran it.
106
+ """
107
+ nonce = beacon_nonce(pulse_randomness, beacon, round_)
108
+ indices = audit_challenge(root, n, k, nonce)
109
+ return AuditRequest(beacon=beacon, round_=round_, n=n, k=k, indices=indices)
@@ -187,20 +187,43 @@ def _cmd_hf_token(args: argparse.Namespace) -> int:
187
187
 
188
188
  def _cmd_audit_challenge(args: argparse.Namespace) -> int:
189
189
  from .persample import audit_challenge # noqa: PLC0415
190
+ # No silent downgrade: partial beacon flags must not fall through to the weakest self-challenge mode, and the
191
+ # two strong modes (auditor nonce vs beacon) must not be silently mixed with beacon quietly winning.
192
+ _beacon_flags = (args.beacon_randomness, args.beacon, args.round)
193
+ if any(f is not None for f in _beacon_flags) and not all(f is not None for f in _beacon_flags):
194
+ print("ERROR: beacon mode needs --beacon-randomness, --beacon and --round together "
195
+ "(partial flags would silently downgrade to the grindable self-challenge mode)", file=sys.stderr)
196
+ return 2
197
+ if args.beacon_randomness is not None and args.nonce is not None:
198
+ print("ERROR: --nonce and --beacon-randomness are mutually exclusive — pick one challenge mode",
199
+ file=sys.stderr)
200
+ return 2
190
201
  try:
191
- nonce = bytes.fromhex(args.nonce) if args.nonce else b""
192
- indices = audit_challenge(args.root, args.n, args.k, nonce)
202
+ if args.beacon_randomness is not None:
203
+ from .beacon import beacon_audit_challenge # noqa: PLC0415
204
+ req = beacon_audit_challenge(
205
+ args.root, args.n, args.k,
206
+ pulse_randomness=bytes.fromhex(args.beacon_randomness),
207
+ beacon=args.beacon, round_=args.round)
208
+ indices, mode = req.indices, "beacon"
209
+ else:
210
+ nonce = bytes.fromhex(args.nonce) if args.nonce else b""
211
+ indices = audit_challenge(args.root, args.n, args.k, nonce)
212
+ mode = "auditor-nonce" if args.nonce else "self-challenge"
193
213
  except (ProofBundleError, ValueError) as exc:
194
214
  print(f"ERROR: {exc}", file=sys.stderr)
195
215
  return 2
196
216
  if args.json:
197
- print(json.dumps({"indices": indices, "n": args.n, "k": args.k,
198
- "mode": "auditor-nonce" if args.nonce else "self-challenge"}))
217
+ out = {"indices": indices, "n": args.n, "k": args.k, "mode": mode}
218
+ if mode == "beacon":
219
+ out["beacon"] = args.beacon
220
+ out["round"] = args.round
221
+ print(json.dumps(out))
199
222
  else:
200
- if not args.nonce:
201
- print("WARNING: self-challenge mode (no --nonce) is a sanity check only — "
202
- "a producer can grind by re-salting; real audits supply a fresh nonce",
203
- file=sys.stderr)
223
+ if mode == "self-challenge":
224
+ print("WARNING: self-challenge mode (no --nonce/--beacon) is a sanity check only — "
225
+ "a producer can grind by re-salting; real audits supply a fresh nonce or a "
226
+ "public beacon pulse from a round AFTER the receipt timestamp", file=sys.stderr)
204
227
  print(" ".join(str(i) for i in indices))
205
228
  return 0
206
229
 
@@ -330,6 +353,11 @@ def build_parser() -> argparse.ArgumentParser:
330
353
  challenge.add_argument("n", type=int, help="committed sample count")
331
354
  challenge.add_argument("k", type=int, help="number of samples to challenge")
332
355
  challenge.add_argument("--nonce", help="fresh auditor nonce (hex, >=32 hex chars recommended)")
356
+ challenge.add_argument("--beacon-randomness",
357
+ help="raw randomness (hex) of a public beacon pulse — non-interactive, "
358
+ "publicly re-derivable (use a round AFTER the receipt timestamp)")
359
+ challenge.add_argument("--beacon", help="beacon id (e.g. 'drand:<chain-hash>' or 'nist')")
360
+ challenge.add_argument("--round", type=int, help="the beacon round/pulse index")
333
361
  challenge.add_argument("--json", action="store_true", help="machine readable output")
334
362
  challenge.set_defaults(func=_cmd_audit_challenge)
335
363