proofbundle 1.7.0__tar.gz → 1.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. proofbundle-1.9.0/PKG-INFO +190 -0
  2. proofbundle-1.9.0/README.md +142 -0
  3. {proofbundle-1.7.0 → proofbundle-1.9.0}/pyproject.toml +1 -1
  4. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/__init__.py +9 -1
  5. proofbundle-1.9.0/src/proofbundle/adapters/_provenance.py +63 -0
  6. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/adapters/inspect_ai.py +9 -0
  7. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/adapters/lm_eval.py +10 -0
  8. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/adapters/promptfoo.py +7 -0
  9. proofbundle-1.9.0/src/proofbundle/beacon.py +109 -0
  10. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/cli.py +76 -8
  11. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/evalclaim.py +10 -1
  12. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/hf_evals.py +43 -1
  13. proofbundle-1.9.0/src/proofbundle/prereg.py +58 -0
  14. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/tlogproof.py +3 -1
  15. proofbundle-1.9.0/src/proofbundle.egg-info/PKG-INFO +190 -0
  16. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/SOURCES.txt +7 -0
  17. proofbundle-1.9.0/tests/test_beacon.py +116 -0
  18. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_evalclaim.py +13 -0
  19. proofbundle-1.9.0/tests/test_fuzz_parsers.py +88 -0
  20. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_hf_evals.py +35 -0
  21. proofbundle-1.9.0/tests/test_prereg.py +122 -0
  22. proofbundle-1.9.0/tests/test_provenance.py +77 -0
  23. proofbundle-1.7.0/PKG-INFO +0 -604
  24. proofbundle-1.7.0/README.md +0 -556
  25. proofbundle-1.7.0/src/proofbundle.egg-info/PKG-INFO +0 -604
  26. {proofbundle-1.7.0 → proofbundle-1.9.0}/LICENSE +0 -0
  27. {proofbundle-1.7.0 → proofbundle-1.9.0}/setup.cfg +0 -0
  28. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/_inspect_registry.py +0 -0
  29. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/_integration.py +0 -0
  30. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/adapters/__init__.py +0 -0
  31. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/adapters/eee.py +0 -0
  32. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/adapters/samples.py +0 -0
  33. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/bundle.py +0 -0
  34. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/checkpoint.py +0 -0
  35. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/demo.py +0 -0
  36. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/dsse.py +0 -0
  37. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/eee_eval_schema.json +0 -0
  38. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/emit.py +0 -0
  39. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/errors.py +0 -0
  40. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/inspect_hook.py +0 -0
  41. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/intoto.py +0 -0
  42. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/kbjwt.py +0 -0
  43. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/merkle.py +0 -0
  44. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/persample.py +0 -0
  45. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/py.typed +0 -0
  46. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/pytest_plugin.py +0 -0
  47. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/sdjwt.py +0 -0
  48. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/sdjwt_issue.py +0 -0
  49. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/signature.py +0 -0
  50. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle/statuslist.py +0 -0
  51. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/dependency_links.txt +0 -0
  52. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/entry_points.txt +0 -0
  53. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/requires.txt +0 -0
  54. {proofbundle-1.7.0 → proofbundle-1.9.0}/src/proofbundle.egg-info/top_level.txt +0 -0
  55. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_adapters.py +0 -0
  56. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_adversarial.py +0 -0
  57. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_bundle.py +0 -0
  58. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_bundle_robustness.py +0 -0
  59. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_checkpoint.py +0 -0
  60. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_cli.py +0 -0
  61. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_cli_eval.py +0 -0
  62. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_cosignature.py +0 -0
  63. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_cosignature_mldsa.py +0 -0
  64. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_demo.py +0 -0
  65. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_eee.py +0 -0
  66. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_emit.py +0 -0
  67. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_eval_claim_schema.py +0 -0
  68. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_examples.py +0 -0
  69. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_inspect_hook.py +0 -0
  70. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_intoto.py +0 -0
  71. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_intoto_dsse.py +0 -0
  72. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_kbjwt.py +0 -0
  73. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_merkle.py +0 -0
  74. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_merkle_property.py +0 -0
  75. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_persample.py +0 -0
  76. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_promptfoo.py +0 -0
  77. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_pytest_plugin.py +0 -0
  78. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_rekor_interop.py +0 -0
  79. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_rfc6962_external_vectors.py +0 -0
  80. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_schema.py +0 -0
  81. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_sdjwt_issue.py +0 -0
  82. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_sdjwt_reference.py +0 -0
  83. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_signature.py +0 -0
  84. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_statuslist.py +0 -0
  85. {proofbundle-1.7.0 → proofbundle-1.9.0}/tests/test_tlogproof.py +0 -0
@@ -0,0 +1,190 @@
1
+ Metadata-Version: 2.4
2
+ Name: proofbundle
3
+ Version: 1.9.0
4
+ Summary: Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT.
5
+ Author: Konrad Gruszka
6
+ License: MIT
7
+ Project-URL: Homepage, https://b7n0de.com
8
+ Project-URL: Repository, https://github.com/b7n0de/proofbundle
9
+ Project-URL: Issues, https://github.com/b7n0de/proofbundle/issues
10
+ Project-URL: Changelog, https://github.com/b7n0de/proofbundle/blob/main/CHANGELOG.md
11
+ Project-URL: Documentation, https://github.com/b7n0de/proofbundle#readme
12
+ Keywords: cryptography,merkle,transparency-log,ed25519,sd-jwt,verifiable-credentials,attestation,provenance,rfc6962
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Programming Language :: Python :: 3.14
22
+ Classifier: Topic :: Security :: Cryptography
23
+ Requires-Python: >=3.10
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: cryptography>=42
27
+ Provides-Extra: sdjwt
28
+ Provides-Extra: eval
29
+ Requires-Dist: rfc8785>=0.1.4; extra == "eval"
30
+ Provides-Extra: adapters
31
+ Provides-Extra: pq
32
+ Requires-Dist: cryptography>=48; extra == "pq"
33
+ Provides-Extra: pytest
34
+ Requires-Dist: pytest>=7; extra == "pytest"
35
+ Provides-Extra: inspect
36
+ Requires-Dist: inspect_ai<0.4,>=0.3.112; extra == "inspect"
37
+ Provides-Extra: dev
38
+ Requires-Dist: pytest>=7; extra == "dev"
39
+ Requires-Dist: ruff>=0.5; extra == "dev"
40
+ Requires-Dist: jsonschema>=4; extra == "dev"
41
+ Requires-Dist: mypy>=1.8; extra == "dev"
42
+ Requires-Dist: build>=1; extra == "dev"
43
+ Requires-Dist: hypothesis>=6; extra == "dev"
44
+ Requires-Dist: rfc8785>=0.1.4; extra == "dev"
45
+ Requires-Dist: sd-jwt>=0.10; extra == "dev"
46
+ Requires-Dist: inspect_ai<0.4,>=0.3.112; extra == "dev"
47
+ Dynamic: license-file
48
+
49
+ <div align="center">
50
+
51
+ <picture>
52
+ <source media="(prefers-color-scheme: dark)" srcset="assets/b7n0de-logo-dark.svg">
53
+ <img alt="b7n0de, Verified AI Work" src="assets/b7n0de-logo.svg" height="60">
54
+ </picture>
55
+
56
+ <h1>proofbundle</h1>
57
+
58
+ **Turn an AI eval result into one portable, offline-verifiable receipt.**
59
+ It proves *who signed these exact bytes* and *that nothing changed since* — not that the number is
60
+ true. Ed25519 + RFC 6962 Merkle, one file, no server, no network.
61
+
62
+ [![CI](https://github.com/b7n0de/proofbundle/actions/workflows/ci.yml/badge.svg)](https://github.com/b7n0de/proofbundle/actions/workflows/ci.yml)
63
+ [![License: MIT](https://img.shields.io/badge/license-MIT-D6248A.svg)](LICENSE)
64
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
65
+ [![Mutation tested](https://img.shields.io/badge/tests-mutation_gated-D6248A.svg)](scripts/mutation_check.py)
66
+ <!-- PyPI / Downloads / SLSA / PEP 740 badges are enabled on the first PyPI release — see RELEASE.md. -->
67
+
68
+ </div>
69
+
70
+ ## The problem
71
+
72
+ Every AI eval number you read — a safety benchmark, a capability score, a leaderboard entry — is an
73
+ **unverifiable claim**. You trust the lab. There's no portable way to check, offline, that a result
74
+ was signed by a stated party, hasn't been altered, and covers the samples it claims.
75
+
76
+ proofbundle is that check. It's a small MIT-licensed Python tool (a compact, auditable trusted core,
77
+ depends only on [`cryptography`](https://cryptography.io)) that turns a result into a signed
78
+ receipt anyone can verify from a single file — and it's honest about the line it does not cross.
79
+
80
+ ## 60-second try (offline, no setup)
81
+
82
+ ```bash
83
+ pip install "proofbundle[eval]"
84
+ proofbundle demo
85
+ ```
86
+
87
+ You'll see an honest receipt verify `=> OK`, then six independent tampers each verify `FAILED`, then
88
+ a swapped sample get caught — all in memory. The command exits non-zero if any tamper slips through,
89
+ so it's also a self-test. Full walkthrough: **[docs/DEMO.md](docs/DEMO.md)**.
90
+
91
+ ```bash
92
+ # your own receipt, from a signed payload:
93
+ proofbundle emit --payload-file result.json --new-key signer.key --out receipt.json
94
+ proofbundle verify receipt.json # exit 0 = OK, 1 = failed, 2 = malformed
95
+ ```
96
+
97
+ ## What a receipt proves — and what it doesn't
98
+
99
+ | ✅ It proves | ❌ It does **not** prove |
100
+ |---|---|
101
+ | These exact bytes were signed by this key (**authorship**) | That the number is **true** |
102
+ | Nothing changed since signing (**integrity**, Ed25519 + RFC 6962) | That the **issuer is honest** |
103
+ | The result is attributable to a stated issuer | That the **eval was well-designed** |
104
+ | A threshold was met while hiding the model/dataset (salted commitments) | That there was **no cherry-picking** — unless pre-registered |
105
+ | Optionally: individual samples, offline-auditable (per-sample Merkle) | That the **computation was correct** — that needs a TEE or independent reproduction |
106
+
107
+ This boundary is the point, not a weakness. A receipt makes a claim **attributable, tamper-evident,
108
+ and — with pre-registration and per-sample auditing — bounded and spot-checkable**. Full detail:
109
+ **[THREAT_MODEL.md](THREAT_MODEL.md)**.
110
+
111
+ ## How it fits together
112
+
113
+ ```mermaid
114
+ flowchart LR
115
+ H["eval harness<br/>inspect_ai · lm-eval · promptfoo · pytest"] --> A["adapter → signed claim<br/>salted commitments · provenance · samples root"]
116
+ A --> R["receipt<br/>one portable file"]
117
+ R --> V{{"proofbundle verify — offline"}}
118
+ V --> C["signature · Merkle inclusion · SD-JWT/KB ·<br/>witness quorum · status list · sample openings"]
119
+ C --> OK(["=> OK / FAILED"])
120
+ style V fill:#D6248A,stroke:#D6248A,color:#fff
121
+ style OK fill:#D6248A,stroke:#D6248A,color:#fff
122
+ ```
123
+
124
+ ## What's in the box
125
+
126
+ - **Core** — Ed25519 signature + RFC 6962 / 9162 Merkle inclusion, verified fully offline. Checks a
127
+ real [Sigstore Rekor](https://docs.sigstore.dev/) proof, so correctness isn't self-referential.
128
+ - **Eval receipts** — a signed claim (`metric ⋈ threshold`, `n`, salted model/dataset commitments,
129
+ assurance level, provenance) from your run. See [EVAL_CLAIM.md](EVAL_CLAIM.md).
130
+ - **Selective disclosure** — SD-JWT ([RFC 9901](https://datatracker.ietf.org/doc/rfc9901/)) with Key
131
+ Binding: prove a threshold while withholding the exact score.
132
+ - **Transparency-log interop** — C2SP `tlog-checkpoint` / cosignature / `.tlog-proof`, with
133
+ post-quantum **ML-DSA-44** witness cosignatures. Optional Token-Status-List revocation snapshots.
134
+ - **Per-sample audit** — commit to every sample; an auditor challenges random indices (with a fresh
135
+ nonce or a **public randomness beacon**, v1.9) and openings must bind to the signed root. Catches
136
+ 1% sample-doctoring with 95% confidence at 300 samples, regardless of run size.
137
+ - **Pre-registration** — `proofbundle prereg <plan>` commits to the protocol before the run, so
138
+ best-of-many publishing becomes visible.
139
+ - **Integrations** — opt-in inspect_ai end-of-task hook and pytest plugin (emit only when
140
+ `PROOFBUNDLE_EMIT=1` / `--proofbundle`), plus a Hugging Face Community Evals bridge. See
141
+ [INTEGRATIONS.md](INTEGRATIONS.md).
142
+
143
+ ## Docs
144
+
145
+ | For… | Read |
146
+ |---|---|
147
+ | Skeptics (why not SHA-256 / Sigstore / trust the issuer) | [docs/FAQ.md](docs/FAQ.md) |
148
+ | Reviewers (30-minute adversarial audit path) | [docs/REVIEWERS.md](docs/REVIEWERS.md) |
149
+ | Where every trust anchor comes from | [docs/TRUST_ANCHORS.md](docs/TRUST_ANCHORS.md) |
150
+ | The demos, tier by tier | [docs/DEMO.md](docs/DEMO.md) |
151
+ | The normative format + verification order | [SPEC.md](SPEC.md) |
152
+ | Honest comparison to Rekor / in-toto / OMS / ValiChord | [INTEROP.md](INTEROP.md) |
153
+ | Regulatory mapping (and what to never claim) | [COMPLIANCE.md](COMPLIANCE.md) |
154
+ | Funders / role fit | [docs/PROJECT_BRIEF.md](docs/PROJECT_BRIEF.md) |
155
+
156
+ ## Install
157
+
158
+ ```bash
159
+ pip install proofbundle # core: offline verify + plain emit (dependency-free)
160
+ pip install "proofbundle[eval]" # + eval receipts, prereg, and the demo (adds rfc8785 JCS)
161
+ pip install "proofbundle[eval]" # emit eval receipts (adds an RFC 8785 canonicalizer)
162
+ pip install "proofbundle[inspect]" # inspect_ai adapter + hook
163
+ pip install "proofbundle[pq]" # verify ML-DSA-44 (post-quantum) witness cosignatures
164
+ ```
165
+
166
+ Requires Python 3.10+. The verify path never rolls its own crypto — Ed25519 comes from
167
+ `cryptography`; Merkle hashing is RFC 6962.
168
+
169
+ ## Status & scope
170
+
171
+ Beta, SemVer-committed, 299 tests + a CI mutation gate + property-based parser fuzzing. Correctness
172
+ is anchored to external RFC 6962 vectors and a real Rekor proof, not just its own bundles. It is
173
+ **not** a log service, a full in-toto client, a TEE, a consensus network, or a compliance product
174
+ by itself — it is the small, offline, standards-native receipt layer between them. Security policy:
175
+ [SECURITY.md](SECURITY.md).
176
+
177
+ ## Contributing
178
+
179
+ See [CONTRIBUTING.md](CONTRIBUTING.md) and the [Code of Conduct](CODE_OF_CONDUCT.md). Good first
180
+ issues are labeled [`good-first-issue`](https://github.com/b7n0de/proofbundle/labels/good-first-issue);
181
+ security findings go through [SECURITY.md](SECURITY.md). The verifier core aims to stay small,
182
+ dependency-light, and correct.
183
+
184
+ ## License
185
+
186
+ MIT — see [LICENSE](LICENSE).
187
+
188
+ ---
189
+
190
+ <p align="center"><sub>proofbundle is part of <b>b7n0de</b>, Verified AI Work · <a href="https://b7n0de.com">b7n0de.com</a></sub></p>
@@ -0,0 +1,142 @@
1
+ <div align="center">
2
+
3
+ <picture>
4
+ <source media="(prefers-color-scheme: dark)" srcset="assets/b7n0de-logo-dark.svg">
5
+ <img alt="b7n0de, Verified AI Work" src="assets/b7n0de-logo.svg" height="60">
6
+ </picture>
7
+
8
+ <h1>proofbundle</h1>
9
+
10
+ **Turn an AI eval result into one portable, offline-verifiable receipt.**
11
+ It proves *who signed these exact bytes* and *that nothing changed since* — not that the number is
12
+ true. Ed25519 + RFC 6962 Merkle, one file, no server, no network.
13
+
14
+ [![CI](https://github.com/b7n0de/proofbundle/actions/workflows/ci.yml/badge.svg)](https://github.com/b7n0de/proofbundle/actions/workflows/ci.yml)
15
+ [![License: MIT](https://img.shields.io/badge/license-MIT-D6248A.svg)](LICENSE)
16
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
17
+ [![Mutation tested](https://img.shields.io/badge/tests-mutation_gated-D6248A.svg)](scripts/mutation_check.py)
18
+ <!-- PyPI / Downloads / SLSA / PEP 740 badges are enabled on the first PyPI release — see RELEASE.md. -->
19
+
20
+ </div>
21
+
22
+ ## The problem
23
+
24
+ Every AI eval number you read — a safety benchmark, a capability score, a leaderboard entry — is an
25
+ **unverifiable claim**. You trust the lab. There's no portable way to check, offline, that a result
26
+ was signed by a stated party, hasn't been altered, and covers the samples it claims.
27
+
28
+ proofbundle is that check. It's a small MIT-licensed Python tool (a compact, auditable trusted core,
29
+ depends only on [`cryptography`](https://cryptography.io)) that turns a result into a signed
30
+ receipt anyone can verify from a single file — and it's honest about the line it does not cross.
31
+
32
+ ## 60-second try (offline, no setup)
33
+
34
+ ```bash
35
+ pip install "proofbundle[eval]"
36
+ proofbundle demo
37
+ ```
38
+
39
+ You'll see an honest receipt verify `=> OK`, then six independent tampers each verify `FAILED`, then
40
+ a swapped sample get caught — all in memory. The command exits non-zero if any tamper slips through,
41
+ so it's also a self-test. Full walkthrough: **[docs/DEMO.md](docs/DEMO.md)**.
42
+
43
+ ```bash
44
+ # your own receipt, from a signed payload:
45
+ proofbundle emit --payload-file result.json --new-key signer.key --out receipt.json
46
+ proofbundle verify receipt.json # exit 0 = OK, 1 = failed, 2 = malformed
47
+ ```
48
+
49
+ ## What a receipt proves — and what it doesn't
50
+
51
+ | ✅ It proves | ❌ It does **not** prove |
52
+ |---|---|
53
+ | These exact bytes were signed by this key (**authorship**) | That the number is **true** |
54
+ | Nothing changed since signing (**integrity**, Ed25519 + RFC 6962) | That the **issuer is honest** |
55
+ | The result is attributable to a stated issuer | That the **eval was well-designed** |
56
+ | A threshold was met while hiding the model/dataset (salted commitments) | That there was **no cherry-picking** — unless pre-registered |
57
+ | Optionally: individual samples, offline-auditable (per-sample Merkle) | That the **computation was correct** — that needs a TEE or independent reproduction |
58
+
59
+ This boundary is the point, not a weakness. A receipt makes a claim **attributable, tamper-evident,
60
+ and — with pre-registration and per-sample auditing — bounded and spot-checkable**. Full detail:
61
+ **[THREAT_MODEL.md](THREAT_MODEL.md)**.
62
+
63
+ ## How it fits together
64
+
65
+ ```mermaid
66
+ flowchart LR
67
+ H["eval harness<br/>inspect_ai · lm-eval · promptfoo · pytest"] --> A["adapter → signed claim<br/>salted commitments · provenance · samples root"]
68
+ A --> R["receipt<br/>one portable file"]
69
+ R --> V{{"proofbundle verify — offline"}}
70
+ V --> C["signature · Merkle inclusion · SD-JWT/KB ·<br/>witness quorum · status list · sample openings"]
71
+ C --> OK(["=> OK / FAILED"])
72
+ style V fill:#D6248A,stroke:#D6248A,color:#fff
73
+ style OK fill:#D6248A,stroke:#D6248A,color:#fff
74
+ ```
75
+
76
+ ## What's in the box
77
+
78
+ - **Core** — Ed25519 signature + RFC 6962 / 9162 Merkle inclusion, verified fully offline. Checks a
79
+ real [Sigstore Rekor](https://docs.sigstore.dev/) proof, so correctness isn't self-referential.
80
+ - **Eval receipts** — a signed claim (`metric ⋈ threshold`, `n`, salted model/dataset commitments,
81
+ assurance level, provenance) from your run. See [EVAL_CLAIM.md](EVAL_CLAIM.md).
82
+ - **Selective disclosure** — SD-JWT ([RFC 9901](https://datatracker.ietf.org/doc/rfc9901/)) with Key
83
+ Binding: prove a threshold while withholding the exact score.
84
+ - **Transparency-log interop** — C2SP `tlog-checkpoint` / cosignature / `.tlog-proof`, with
85
+ post-quantum **ML-DSA-44** witness cosignatures. Optional Token-Status-List revocation snapshots.
86
+ - **Per-sample audit** — commit to every sample; an auditor challenges random indices (with a fresh
87
+ nonce or a **public randomness beacon**, v1.9) and openings must bind to the signed root. Catches
88
+ 1% sample-doctoring with 95% confidence at 300 samples, regardless of run size.
89
+ - **Pre-registration** — `proofbundle prereg <plan>` commits to the protocol before the run, so
90
+ best-of-many publishing becomes visible.
91
+ - **Integrations** — opt-in inspect_ai end-of-task hook and pytest plugin (emit only when
92
+ `PROOFBUNDLE_EMIT=1` / `--proofbundle`), plus a Hugging Face Community Evals bridge. See
93
+ [INTEGRATIONS.md](INTEGRATIONS.md).
94
+
95
+ ## Docs
96
+
97
+ | For… | Read |
98
+ |---|---|
99
+ | Skeptics (why not SHA-256 / Sigstore / trust the issuer) | [docs/FAQ.md](docs/FAQ.md) |
100
+ | Reviewers (30-minute adversarial audit path) | [docs/REVIEWERS.md](docs/REVIEWERS.md) |
101
+ | Where every trust anchor comes from | [docs/TRUST_ANCHORS.md](docs/TRUST_ANCHORS.md) |
102
+ | The demos, tier by tier | [docs/DEMO.md](docs/DEMO.md) |
103
+ | The normative format + verification order | [SPEC.md](SPEC.md) |
104
+ | Honest comparison to Rekor / in-toto / OMS / ValiChord | [INTEROP.md](INTEROP.md) |
105
+ | Regulatory mapping (and what to never claim) | [COMPLIANCE.md](COMPLIANCE.md) |
106
+ | Funders / role fit | [docs/PROJECT_BRIEF.md](docs/PROJECT_BRIEF.md) |
107
+
108
+ ## Install
109
+
110
+ ```bash
111
+ pip install proofbundle # core: offline verify + plain emit (dependency-free)
112
+ pip install "proofbundle[eval]" # + eval receipts, prereg, and the demo (adds rfc8785 JCS)
113
+ pip install "proofbundle[eval]" # emit eval receipts (adds an RFC 8785 canonicalizer)
114
+ pip install "proofbundle[inspect]" # inspect_ai adapter + hook
115
+ pip install "proofbundle[pq]" # verify ML-DSA-44 (post-quantum) witness cosignatures
116
+ ```
117
+
118
+ Requires Python 3.10+. The verify path never rolls its own crypto — Ed25519 comes from
119
+ `cryptography`; Merkle hashing is RFC 6962.
120
+
121
+ ## Status & scope
122
+
123
+ Beta, SemVer-committed, 299 tests + a CI mutation gate + property-based parser fuzzing. Correctness
124
+ is anchored to external RFC 6962 vectors and a real Rekor proof, not just its own bundles. It is
125
+ **not** a log service, a full in-toto client, a TEE, a consensus network, or a compliance product
126
+ by itself — it is the small, offline, standards-native receipt layer between them. Security policy:
127
+ [SECURITY.md](SECURITY.md).
128
+
129
+ ## Contributing
130
+
131
+ See [CONTRIBUTING.md](CONTRIBUTING.md) and the [Code of Conduct](CODE_OF_CONDUCT.md). Good first
132
+ issues are labeled [`good-first-issue`](https://github.com/b7n0de/proofbundle/labels/good-first-issue);
133
+ security findings go through [SECURITY.md](SECURITY.md). The verifier core aims to stay small,
134
+ dependency-light, and correct.
135
+
136
+ ## License
137
+
138
+ MIT — see [LICENSE](LICENSE).
139
+
140
+ ---
141
+
142
+ <p align="center"><sub>proofbundle is part of <b>b7n0de</b>, Verified AI Work · <a href="https://b7n0de.com">b7n0de.com</a></sub></p>
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "proofbundle"
7
- version = "1.7.0"
7
+ version = "1.9.0"
8
8
  description = "Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -13,7 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  from typing import TYPE_CHECKING
15
15
 
16
- __version__ = "1.7.0"
16
+ __version__ = "1.9.0"
17
17
 
18
18
  __all__ = [
19
19
  "__version__",
@@ -36,6 +36,9 @@ __all__ = [
36
36
  "sample_opening",
37
37
  "verify_sample_opening",
38
38
  "audit_challenge",
39
+ "prereg_hash",
40
+ "verify_prereg",
41
+ "beacon_audit_challenge",
39
42
  "VerificationResult",
40
43
  "Check",
41
44
  "ProofBundleError",
@@ -59,6 +62,9 @@ _LAZY = {
59
62
  "sample_opening": ".persample",
60
63
  "verify_sample_opening": ".persample",
61
64
  "audit_challenge": ".persample",
65
+ "prereg_hash": ".prereg",
66
+ "verify_prereg": ".prereg",
67
+ "beacon_audit_challenge": ".beacon",
62
68
  }
63
69
 
64
70
  if TYPE_CHECKING: # static analysers + IDEs see the real names/types; runtime stays lazy
@@ -70,6 +76,8 @@ if TYPE_CHECKING: # static analysers + IDEs see the real names/types; runtime s
70
76
  from .hf_evals import receipt_token, verify_receipt_token
71
77
  from .persample import (audit_challenge, build_sample_tree, sample_opening,
72
78
  verify_sample_opening)
79
+ from .beacon import beacon_audit_challenge
80
+ from .prereg import prereg_hash, verify_prereg
73
81
  from .statuslist import verify_status_snapshot
74
82
  from .tlogproof import verify_tlog_proof
75
83
  from .merkle import verify_consistency, verify_inclusion
@@ -0,0 +1,63 @@
1
+ """Shared provenance helpers for adapters (v1.8).
2
+
3
+ The external review found that run-id and config-hash were missing from nearly every adapter and
4
+ that the two flagship adapters took the timestamp from the caller rather than the eval log. These
5
+ helpers close that: each adapter now records, where the framework exposes it, a stable RUN id, a
6
+ CONFIG hash, and the LOG-NATIVE timestamp — so a receipt is traceable back to the exact run.
7
+
8
+ Design notes (verified against framework source, 2026-07):
9
+ - No framework ships a canonical config hash, so we compute our own. Config is an in-memory
10
+ object re-serialized non-deterministically, so it MUST be canonicalized before hashing —
11
+ RFC 8785 JCS via the same `rfc8785` extra the emit path already needs. If that extra is
12
+ absent we fall back to a deterministic `json.dumps(sort_keys=True)` and LABEL the hash
13
+ algorithm accordingly, so a verifier is never misled about how the hash was formed.
14
+ - The hash is over the config's JSON, prefixed with a domain tag, hex sha256. It is provenance
15
+ metadata (traceability), NOT a security commitment — it is not salted and reveals structure;
16
+ it exists so two receipts from the same config are linkable and a changed config is visible.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import hashlib
22
+ import json
23
+ from typing import Optional
24
+
25
+ _CONFIG_DOMAIN = b"proofbundle/v1.8/config-hash\x00"
26
+
27
+
28
+ def config_hash(config) -> Optional[str]:
29
+ """Return ``"<alg>:<hex>"`` over the canonical JSON of a config object, or None if it is
30
+ empty/None. ``<alg>`` is ``sha256-jcs`` when RFC 8785 is available, else ``sha256-sortkeys``
31
+ (both deterministic; the label tells a verifier which normalization produced the hex)."""
32
+ if config is None or config == {} or config == []:
33
+ return None
34
+ try:
35
+ import rfc8785 # noqa: PLC0415 — same optional dep as the emit path
36
+ canonical = rfc8785.dumps(config)
37
+ alg = "sha256-jcs"
38
+ except (ImportError, ValueError, TypeError):
39
+ # rfc8785 rejects non-JCS-able values (e.g. floats it deems unsafe); fall back to a
40
+ # deterministic stdlib serialization and label it so the difference is never hidden.
41
+ try:
42
+ canonical = json.dumps(config, sort_keys=True, separators=(",", ":"),
43
+ ensure_ascii=False).encode("utf-8")
44
+ except (TypeError, ValueError):
45
+ return None
46
+ alg = "sha256-sortkeys"
47
+ return f"{alg}:{hashlib.sha256(_CONFIG_DOMAIN + canonical).hexdigest()}"
48
+
49
+
50
+ def add_provenance(provenance: dict, *, run_id=None, config=None, log_timestamp=None,
51
+ config_hash_value: Optional[str] = None) -> dict:
52
+ """Merge the standard traceability fields into a provenance dict, skipping absent ones.
53
+
54
+ ``config_hash_value`` lets a caller pass a precomputed hash (e.g. over already-canonical
55
+ material) instead of a config object; otherwise ``config`` is hashed here."""
56
+ if run_id:
57
+ provenance["run_id"] = str(run_id)
58
+ if log_timestamp is not None:
59
+ provenance["run_timestamp"] = str(log_timestamp)
60
+ ch = config_hash_value if config_hash_value is not None else config_hash(config)
61
+ if ch:
62
+ provenance["config_hash"] = ch
63
+ return provenance
@@ -90,6 +90,15 @@ def from_inspect_ai_log(path, metric: str, *, comparator: str, threshold: str, t
90
90
  if tv is not None:
91
91
  provenance["task_version"] = str(tv)
92
92
 
93
+ # v1.8 (external review): run-id + config-hash + LOG-NATIVE timestamp so a receipt traces back
94
+ # to the exact run. inspect_ai: eval.run_id (unique run id), eval.created (UTC datetime string),
95
+ # eval.task_args (the config material — no native config hash exists, so we compute one).
96
+ from ._provenance import add_provenance # noqa: PLC0415
97
+ task_args = getattr(ev, "task_args", None)
98
+ add_provenance(provenance, run_id=getattr(ev, "run_id", None),
99
+ config=task_args if isinstance(task_args, dict) else None,
100
+ log_timestamp=getattr(ev, "created", None))
101
+
93
102
  return build_eval_claim(
94
103
  suite=suite, suite_version=str(getattr(ev, "task_version", "1")),
95
104
  metric=metric, comparator=comparator, threshold=threshold, score=_score_str(value),
@@ -71,6 +71,16 @@ def from_lm_eval_results(path, task: str, metric: str, *, comparator: str, thres
71
71
  if stderr is not None:
72
72
  provenance["stderr"] = repr(stderr) if not isinstance(stderr, str) else stderr
73
73
 
74
+ # v1.8 (external review): config-hash + LOG-NATIVE timestamp. lm-eval has no dedicated run-id;
75
+ # its `date` is a Unix float (distinct from the ISO filename stamp). `config` is the run config
76
+ # block (model/args/seeds/gen_kwargs); no native hash exists, so we compute one.
77
+ from ._provenance import add_provenance # noqa: PLC0415
78
+ add_provenance(provenance, config=cfg if isinstance(cfg, dict) and cfg else None,
79
+ log_timestamp=data.get("date"))
80
+ task_hashes = data.get("task_hashes", {})
81
+ if isinstance(task_hashes, dict) and task_hashes.get(task):
82
+ provenance["task_hash"] = str(task_hashes[task]) # lm-eval's native per-task sample hash
83
+
74
84
  return build_eval_claim(
75
85
  suite=task, suite_version=str(data.get("versions", {}).get(task, "lm-eval")),
76
86
  metric=metric, comparator=comparator, threshold=threshold, score=str(score), n=n,
@@ -124,6 +124,13 @@ def from_promptfoo_results(path, *, comparator: str, threshold: str, timestamp:
124
124
  if summary.get("timestamp"):
125
125
  provenance["run_timestamp"] = str(summary["timestamp"])
126
126
 
127
+ # v1.8 (external review): a uniform run_id key across adapters + a config-hash over the FULL
128
+ # resolved suite config (providers/prompts/tests/…), not just the tests-derived dataset id.
129
+ from ._provenance import add_provenance # noqa: PLC0415
130
+ add_provenance(provenance, run_id=eval_id,
131
+ config=config if isinstance(config, dict) and config else None,
132
+ log_timestamp=metadata.get("evaluationCreatedAt"))
133
+
127
134
  return build_eval_claim(
128
135
  suite=suite, suite_version=f"promptfoo-summary-v{version}",
129
136
  metric="pass_rate", comparator=comparator, threshold=threshold, score=score, n=n,
@@ -0,0 +1,109 @@
1
+ """Public-randomness audit challenges (v1.9) — non-interactive, publicly re-derivable.
2
+
3
+ The per-sample audit (SPEC §7g) has three challenge modes. Two shipped in v1.5: an *auditor
4
+ nonce* (interactive, grinding-impossible) and *self-challenge* (a documented-grindable sanity
5
+ check). This module formalizes the third — a **public randomness beacon** — so an audit needs no
6
+ live auditor and anyone can re-derive the same challenged indices from published data.
7
+
8
+ Why a beacon RESISTS grinding without a live auditor, AND WHAT IT ASSUMES: the producer signs the receipt (with its
9
+ ``samples_root``) at time T. A beacon pulse from a round whose emission time is *after* T did not
10
+ exist when the producer committed, so the producer cannot have chosen the samples to fit the
11
+ challenge. This is the RFC 3797 pattern ("derive selections from pre-specified future public
12
+ randomness") applied to sample auditing. Established beacons: the drand League of Entropy
13
+ (``randomness`` = 32 bytes per round) and the NIST Interoperable Randomness Beacon
14
+ (``outputValue`` = 64 bytes per pulse).
15
+
16
+ Offline-first: proofbundle never fetches. The relying party obtains the pulse out of band (or it
17
+ is bundled) and passes its raw bytes here. The returned ``AuditRequest`` records the beacon id
18
+ and round so a third party can fetch the *same* pulse and re-run ``audit_challenge`` to the
19
+ identical indices — the audit is reproducible without trusting the auditor.
20
+
21
+ Soundness caveats, stated honestly (the beacon mode is grinding-resistant ONLY under these, otherwise it
22
+ is no stronger than the documented-grindable self-challenge, so a relying party MUST check them):
23
+ 1. Beacon signature: this module does NOT verify the beacon's own signature (drand pulses are BLS-signed;
24
+ NIST pulses are RSA-signed) — that is a separate trust anchor the relying party validates with the
25
+ beacon's public key out of band, like every anchor in docs/TRUST_ANCHORS.md.
26
+ 2. Ordering depends on a SELF-DECLARED, UNVERIFIED timestamp. The "the round emitted after time T" argument
27
+ uses the receipt's own ``timestamp`` field, which the producer writes and signs but which nothing here
28
+ proves is the true commit time. A dishonest producer can BACKDATE ``timestamp``, wait for a round R whose
29
+ randomness is already public, grind sample trees against R, and sign a receipt claiming a timestamp before
30
+ R's emission. Ed25519 only proves the false timestamp was signed, not that it is true. So the relying party
31
+ must corroborate the ordering from an INDEPENDENT source (a transparency-log inclusion time for the receipt,
32
+ a witnessed/notarized timestamp, or a pre-registered round id chosen before the run) — not from the receipt's
33
+ own timestamp alone. Without independent corroboration the beacon mode does NOT close producer-side grinding.
34
+ 3. Round independence: the round id must be fixed BEFORE the samples are committed (a future round), not chosen
35
+ by the producer after seeing published randomness; this module records the round but cannot enforce that it
36
+ was pre-committed.
37
+ """
38
+
39
+ from __future__ import annotations
40
+
41
+ import hashlib
42
+ from typing import List
43
+
44
+ from .errors import BundleFormatError
45
+ from .persample import audit_challenge
46
+
47
+ __all__ = ["AuditRequest", "beacon_nonce", "beacon_audit_challenge"]
48
+
49
+ # Beacon randomness lengths we accept (raw bytes). drand = 32, NIST = 64. Other lengths are
50
+ # allowed too (any >=16 bytes of published randomness), but these are the named, verified ones.
51
+ _MIN_PULSE_BYTES = 16
52
+ _BEACON_DOMAIN = b"proofbundle/v1.9/beacon-nonce\x00"
53
+
54
+
55
+ class AuditRequest:
56
+ """A reproducible audit challenge derived from a public beacon pulse.
57
+
58
+ Everything a third party needs to RE-derive the same indices: the beacon id, the round, and
59
+ the resulting indices. ``as_dict`` is JSON-serializable for publishing alongside the receipt.
60
+ """
61
+
62
+ __slots__ = ("beacon", "round", "n", "k", "indices")
63
+
64
+ def __init__(self, beacon: str, round_: int, n: int, k: int, indices: List[int]):
65
+ self.beacon = beacon
66
+ self.round = round_
67
+ self.n = n
68
+ self.k = k
69
+ self.indices = indices
70
+
71
+ def as_dict(self) -> dict:
72
+ return {"beacon": self.beacon, "round": self.round, "n": self.n, "k": self.k,
73
+ "indices": list(self.indices)}
74
+
75
+ def __repr__(self) -> str: # pragma: no cover - cosmetic
76
+ return f"AuditRequest(beacon={self.beacon!r}, round={self.round}, k={self.k}, n={self.n})"
77
+
78
+
79
+ def beacon_nonce(pulse_randomness: bytes, beacon: str, round_: int) -> bytes:
80
+ """Derive the ``audit_challenge`` nonce from a beacon pulse, binding the beacon id + round.
81
+
82
+ The nonce is ``SHA-256(domain ‖ beacon ‖ 0x00 ‖ u64(round) ‖ pulse_randomness)`` — so two
83
+ different beacons or rounds never collide to the same challenge, and the nonce is a fixed
84
+ 32 bytes regardless of the beacon's own randomness length.
85
+ """
86
+ if not isinstance(pulse_randomness, (bytes, bytearray)) or len(pulse_randomness) < _MIN_PULSE_BYTES:
87
+ raise BundleFormatError(
88
+ f"beacon pulse randomness must be at least {_MIN_PULSE_BYTES} bytes")
89
+ if not beacon or "\x00" in beacon:
90
+ raise BundleFormatError("beacon id must be a non-empty string without NUL")
91
+ if isinstance(round_, bool) or not isinstance(round_, int) or round_ < 0 or round_ >= 2**64:
92
+ raise BundleFormatError("beacon round must be a u64 (0 <= round < 2**64)")
93
+ return hashlib.sha256(_BEACON_DOMAIN + beacon.encode("utf-8") + b"\x00"
94
+ + round_.to_bytes(8, "big") + bytes(pulse_randomness)).digest()
95
+
96
+
97
+ def beacon_audit_challenge(root, n: int, k: int, *, pulse_randomness: bytes, beacon: str,
98
+ round_: int) -> AuditRequest:
99
+ """Derive a reproducible per-sample audit challenge from a public beacon pulse.
100
+
101
+ ``root``/``n``/``k`` are the receipt's signed samples root, committed count, and the number
102
+ of samples to challenge. ``pulse_randomness`` is the raw randomness of a beacon pulse whose
103
+ round emits AFTER the receipt's signed timestamp (the relying party checks that). Returns an
104
+ :class:`AuditRequest` recording the beacon id + round + indices, so the challenge is
105
+ publicly re-derivable — no live auditor, no trust in who ran it.
106
+ """
107
+ nonce = beacon_nonce(pulse_randomness, beacon, round_)
108
+ indices = audit_challenge(root, n, k, nonce)
109
+ return AuditRequest(beacon=beacon, round_=round_, n=n, k=k, indices=indices)