proofbundle 1.7.0__tar.gz → 1.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {proofbundle-1.7.0/src/proofbundle.egg-info → proofbundle-1.8.0}/PKG-INFO +2 -2
- {proofbundle-1.7.0 → proofbundle-1.8.0}/README.md +1 -1
- {proofbundle-1.7.0 → proofbundle-1.8.0}/pyproject.toml +1 -1
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/__init__.py +6 -1
- proofbundle-1.8.0/src/proofbundle/adapters/_provenance.py +63 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/inspect_ai.py +9 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/lm_eval.py +10 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/promptfoo.py +7 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/cli.py +40 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/evalclaim.py +10 -1
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/hf_evals.py +43 -1
- proofbundle-1.8.0/src/proofbundle/prereg.py +58 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/tlogproof.py +3 -1
- {proofbundle-1.7.0 → proofbundle-1.8.0/src/proofbundle.egg-info}/PKG-INFO +2 -2
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/SOURCES.txt +5 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_evalclaim.py +13 -0
- proofbundle-1.8.0/tests/test_fuzz_parsers.py +88 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_hf_evals.py +35 -0
- proofbundle-1.8.0/tests/test_prereg.py +122 -0
- proofbundle-1.8.0/tests/test_provenance.py +77 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/LICENSE +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/setup.cfg +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/_inspect_registry.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/_integration.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/__init__.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/eee.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/samples.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/bundle.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/checkpoint.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/demo.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/dsse.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/eee_eval_schema.json +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/emit.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/errors.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/inspect_hook.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/intoto.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/kbjwt.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/merkle.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/persample.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/py.typed +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/pytest_plugin.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/sdjwt.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/sdjwt_issue.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/signature.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/statuslist.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/dependency_links.txt +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/entry_points.txt +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/requires.txt +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/top_level.txt +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_adapters.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_adversarial.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_bundle.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_bundle_robustness.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_checkpoint.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_cli.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_cli_eval.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_cosignature.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_cosignature_mldsa.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_demo.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_eee.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_emit.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_eval_claim_schema.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_examples.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_inspect_hook.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_intoto.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_intoto_dsse.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_kbjwt.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_merkle.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_merkle_property.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_persample.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_promptfoo.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_pytest_plugin.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_rekor_interop.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_rfc6962_external_vectors.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_schema.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_sdjwt_issue.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_sdjwt_reference.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_signature.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_statuslist.py +0 -0
- {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_tlogproof.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: proofbundle
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.8.0
|
|
4
4
|
Summary: Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT.
|
|
5
5
|
Author: Konrad Gruszka
|
|
6
6
|
License: MIT
|
|
@@ -84,7 +84,7 @@ file, no server, no network.**
|
|
|
84
84
|
|
|
85
85
|
**At a glance:** `proofbundle emit` signs and anchors a payload; `proofbundle
|
|
86
86
|
verify` checks one self-contained `bundle.json` with three offline cryptographic
|
|
87
|
-
checks → `OK` or `FAILED`. No network, no daemon, no own crypto.
|
|
87
|
+
checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 289 tests + a CI mutation gate.
|
|
88
88
|
|
|
89
89
|
## Contents
|
|
90
90
|
|
|
@@ -36,7 +36,7 @@ file, no server, no network.**
|
|
|
36
36
|
|
|
37
37
|
**At a glance:** `proofbundle emit` signs and anchors a payload; `proofbundle
|
|
38
38
|
verify` checks one self-contained `bundle.json` with three offline cryptographic
|
|
39
|
-
checks → `OK` or `FAILED`. No network, no daemon, no own crypto.
|
|
39
|
+
checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 289 tests + a CI mutation gate.
|
|
40
40
|
|
|
41
41
|
## Contents
|
|
42
42
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "proofbundle"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.8.0"
|
|
8
8
|
description = "Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -13,7 +13,7 @@ from __future__ import annotations
|
|
|
13
13
|
|
|
14
14
|
from typing import TYPE_CHECKING
|
|
15
15
|
|
|
16
|
-
__version__ = "1.
|
|
16
|
+
__version__ = "1.8.0"
|
|
17
17
|
|
|
18
18
|
__all__ = [
|
|
19
19
|
"__version__",
|
|
@@ -36,6 +36,8 @@ __all__ = [
|
|
|
36
36
|
"sample_opening",
|
|
37
37
|
"verify_sample_opening",
|
|
38
38
|
"audit_challenge",
|
|
39
|
+
"prereg_hash",
|
|
40
|
+
"verify_prereg",
|
|
39
41
|
"VerificationResult",
|
|
40
42
|
"Check",
|
|
41
43
|
"ProofBundleError",
|
|
@@ -59,6 +61,8 @@ _LAZY = {
|
|
|
59
61
|
"sample_opening": ".persample",
|
|
60
62
|
"verify_sample_opening": ".persample",
|
|
61
63
|
"audit_challenge": ".persample",
|
|
64
|
+
"prereg_hash": ".prereg",
|
|
65
|
+
"verify_prereg": ".prereg",
|
|
62
66
|
}
|
|
63
67
|
|
|
64
68
|
if TYPE_CHECKING: # static analysers + IDEs see the real names/types; runtime stays lazy
|
|
@@ -70,6 +74,7 @@ if TYPE_CHECKING: # static analysers + IDEs see the real names/types; runtime s
|
|
|
70
74
|
from .hf_evals import receipt_token, verify_receipt_token
|
|
71
75
|
from .persample import (audit_challenge, build_sample_tree, sample_opening,
|
|
72
76
|
verify_sample_opening)
|
|
77
|
+
from .prereg import prereg_hash, verify_prereg
|
|
73
78
|
from .statuslist import verify_status_snapshot
|
|
74
79
|
from .tlogproof import verify_tlog_proof
|
|
75
80
|
from .merkle import verify_consistency, verify_inclusion
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Shared provenance helpers for adapters (v1.8).
|
|
2
|
+
|
|
3
|
+
The external review found that run-id and config-hash were missing from nearly every adapter and
|
|
4
|
+
that the two flagship adapters took the timestamp from the caller rather than the eval log. These
|
|
5
|
+
helpers close that: each adapter now records, where the framework exposes it, a stable RUN id, a
|
|
6
|
+
CONFIG hash, and the LOG-NATIVE timestamp — so a receipt is traceable back to the exact run.
|
|
7
|
+
|
|
8
|
+
Design notes (verified against framework source, 2026-07):
|
|
9
|
+
- No framework ships a canonical config hash, so we compute our own. Config is an in-memory
|
|
10
|
+
object re-serialized non-deterministically, so it MUST be canonicalized before hashing —
|
|
11
|
+
RFC 8785 JCS via the same `rfc8785` extra the emit path already needs. If that extra is
|
|
12
|
+
absent we fall back to a deterministic `json.dumps(sort_keys=True)` and LABEL the hash
|
|
13
|
+
algorithm accordingly, so a verifier is never misled about how the hash was formed.
|
|
14
|
+
- The hash is over the config's JSON, prefixed with a domain tag, hex sha256. It is provenance
|
|
15
|
+
metadata (traceability), NOT a security commitment — it is not salted and reveals structure;
|
|
16
|
+
it exists so two receipts from the same config are linkable and a changed config is visible.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import hashlib
|
|
22
|
+
import json
|
|
23
|
+
from typing import Optional
|
|
24
|
+
|
|
25
|
+
_CONFIG_DOMAIN = b"proofbundle/v1.8/config-hash\x00"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def config_hash(config) -> Optional[str]:
|
|
29
|
+
"""Return ``"<alg>:<hex>"`` over the canonical JSON of a config object, or None if it is
|
|
30
|
+
empty/None. ``<alg>`` is ``sha256-jcs`` when RFC 8785 is available, else ``sha256-sortkeys``
|
|
31
|
+
(both deterministic; the label tells a verifier which normalization produced the hex)."""
|
|
32
|
+
if config is None or config == {} or config == []:
|
|
33
|
+
return None
|
|
34
|
+
try:
|
|
35
|
+
import rfc8785 # noqa: PLC0415 — same optional dep as the emit path
|
|
36
|
+
canonical = rfc8785.dumps(config)
|
|
37
|
+
alg = "sha256-jcs"
|
|
38
|
+
except (ImportError, ValueError, TypeError):
|
|
39
|
+
# rfc8785 rejects non-JCS-able values (e.g. floats it deems unsafe); fall back to a
|
|
40
|
+
# deterministic stdlib serialization and label it so the difference is never hidden.
|
|
41
|
+
try:
|
|
42
|
+
canonical = json.dumps(config, sort_keys=True, separators=(",", ":"),
|
|
43
|
+
ensure_ascii=False).encode("utf-8")
|
|
44
|
+
except (TypeError, ValueError):
|
|
45
|
+
return None
|
|
46
|
+
alg = "sha256-sortkeys"
|
|
47
|
+
return f"{alg}:{hashlib.sha256(_CONFIG_DOMAIN + canonical).hexdigest()}"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def add_provenance(provenance: dict, *, run_id=None, config=None, log_timestamp=None,
|
|
51
|
+
config_hash_value: Optional[str] = None) -> dict:
|
|
52
|
+
"""Merge the standard traceability fields into a provenance dict, skipping absent ones.
|
|
53
|
+
|
|
54
|
+
``config_hash_value`` lets a caller pass a precomputed hash (e.g. over already-canonical
|
|
55
|
+
material) instead of a config object; otherwise ``config`` is hashed here."""
|
|
56
|
+
if run_id:
|
|
57
|
+
provenance["run_id"] = str(run_id)
|
|
58
|
+
if log_timestamp is not None:
|
|
59
|
+
provenance["run_timestamp"] = str(log_timestamp)
|
|
60
|
+
ch = config_hash_value if config_hash_value is not None else config_hash(config)
|
|
61
|
+
if ch:
|
|
62
|
+
provenance["config_hash"] = ch
|
|
63
|
+
return provenance
|
|
@@ -90,6 +90,15 @@ def from_inspect_ai_log(path, metric: str, *, comparator: str, threshold: str, t
|
|
|
90
90
|
if tv is not None:
|
|
91
91
|
provenance["task_version"] = str(tv)
|
|
92
92
|
|
|
93
|
+
# v1.8 (external review): run-id + config-hash + LOG-NATIVE timestamp so a receipt traces back
|
|
94
|
+
# to the exact run. inspect_ai: eval.run_id (unique run id), eval.created (UTC datetime string),
|
|
95
|
+
# eval.task_args (the config material — no native config hash exists, so we compute one).
|
|
96
|
+
from ._provenance import add_provenance # noqa: PLC0415
|
|
97
|
+
task_args = getattr(ev, "task_args", None)
|
|
98
|
+
add_provenance(provenance, run_id=getattr(ev, "run_id", None),
|
|
99
|
+
config=task_args if isinstance(task_args, dict) else None,
|
|
100
|
+
log_timestamp=getattr(ev, "created", None))
|
|
101
|
+
|
|
93
102
|
return build_eval_claim(
|
|
94
103
|
suite=suite, suite_version=str(getattr(ev, "task_version", "1")),
|
|
95
104
|
metric=metric, comparator=comparator, threshold=threshold, score=_score_str(value),
|
|
@@ -71,6 +71,16 @@ def from_lm_eval_results(path, task: str, metric: str, *, comparator: str, thres
|
|
|
71
71
|
if stderr is not None:
|
|
72
72
|
provenance["stderr"] = repr(stderr) if not isinstance(stderr, str) else stderr
|
|
73
73
|
|
|
74
|
+
# v1.8 (external review): config-hash + LOG-NATIVE timestamp. lm-eval has no dedicated run-id;
|
|
75
|
+
# its `date` is a Unix float (distinct from the ISO filename stamp). `config` is the run config
|
|
76
|
+
# block (model/args/seeds/gen_kwargs); no native hash exists, so we compute one.
|
|
77
|
+
from ._provenance import add_provenance # noqa: PLC0415
|
|
78
|
+
add_provenance(provenance, config=cfg if isinstance(cfg, dict) and cfg else None,
|
|
79
|
+
log_timestamp=data.get("date"))
|
|
80
|
+
task_hashes = data.get("task_hashes", {})
|
|
81
|
+
if isinstance(task_hashes, dict) and task_hashes.get(task):
|
|
82
|
+
provenance["task_hash"] = str(task_hashes[task]) # lm-eval's native per-task sample hash
|
|
83
|
+
|
|
74
84
|
return build_eval_claim(
|
|
75
85
|
suite=task, suite_version=str(data.get("versions", {}).get(task, "lm-eval")),
|
|
76
86
|
metric=metric, comparator=comparator, threshold=threshold, score=str(score), n=n,
|
|
@@ -124,6 +124,13 @@ def from_promptfoo_results(path, *, comparator: str, threshold: str, timestamp:
|
|
|
124
124
|
if summary.get("timestamp"):
|
|
125
125
|
provenance["run_timestamp"] = str(summary["timestamp"])
|
|
126
126
|
|
|
127
|
+
# v1.8 (external review): a uniform run_id key across adapters + a config-hash over the FULL
|
|
128
|
+
# resolved suite config (providers/prompts/tests/…), not just the tests-derived dataset id.
|
|
129
|
+
from ._provenance import add_provenance # noqa: PLC0415
|
|
130
|
+
add_provenance(provenance, run_id=eval_id,
|
|
131
|
+
config=config if isinstance(config, dict) and config else None,
|
|
132
|
+
log_timestamp=metadata.get("evaluationCreatedAt"))
|
|
133
|
+
|
|
127
134
|
return build_eval_claim(
|
|
128
135
|
suite=suite, suite_version=f"promptfoo-summary-v{version}",
|
|
129
136
|
metric="pass_rate", comparator=comparator, threshold=threshold, score=score, n=n,
|
|
@@ -229,6 +229,37 @@ def _cmd_demo(args: argparse.Namespace) -> int:
|
|
|
229
229
|
return run_demo(as_json=args.json)
|
|
230
230
|
|
|
231
231
|
|
|
232
|
+
def _cmd_prereg(args: argparse.Namespace) -> int:
|
|
233
|
+
from .prereg import prereg_hash, verify_prereg # noqa: PLC0415
|
|
234
|
+
try:
|
|
235
|
+
if args.check is not None:
|
|
236
|
+
from .evalclaim import decode_eval_claim # noqa: PLC0415
|
|
237
|
+
# Release-review CRITICAL: --check MUST verify the receipt (Ed25519 + Merkle) BEFORE trusting its
|
|
238
|
+
# prereg_sha256 — the old load_bundle+manual-decode read an UNAUTHENTICATED claim, so a forged/unsigned
|
|
239
|
+
# bundle with a doctored prereg_sha256 got a false PASS (the exact anti-cherry-picking bypass this guards).
|
|
240
|
+
claim = decode_eval_claim(args.check)
|
|
241
|
+
if claim is None:
|
|
242
|
+
print("=> FAILED: not a valid, issuer-bound eval receipt", file=sys.stderr)
|
|
243
|
+
return 1
|
|
244
|
+
res = verify_prereg(args.protocol, claim)
|
|
245
|
+
if args.json:
|
|
246
|
+
print(json.dumps(res))
|
|
247
|
+
else:
|
|
248
|
+
print(f"[{'PASS' if res['ok'] else 'FAIL'}] prereg: {res['detail']}")
|
|
249
|
+
return 0 if res["ok"] else 1
|
|
250
|
+
h = prereg_hash(args.protocol)
|
|
251
|
+
if args.json:
|
|
252
|
+
print(json.dumps({"prereg_sha256": h}))
|
|
253
|
+
else:
|
|
254
|
+
print(h)
|
|
255
|
+
print("place this in the eval claim's prereg_sha256 BEFORE running the eval",
|
|
256
|
+
file=sys.stderr)
|
|
257
|
+
return 0
|
|
258
|
+
except (ProofBundleError, OSError, ValueError, KeyError) as exc:
|
|
259
|
+
print(f"ERROR: {exc}", file=sys.stderr)
|
|
260
|
+
return 2
|
|
261
|
+
|
|
262
|
+
|
|
232
263
|
def build_parser() -> argparse.ArgumentParser:
|
|
233
264
|
parser = argparse.ArgumentParser(
|
|
234
265
|
prog="proofbundle",
|
|
@@ -317,6 +348,15 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
317
348
|
demo.add_argument("--json", action="store_true", help="machine readable output")
|
|
318
349
|
demo.set_defaults(func=_cmd_demo)
|
|
319
350
|
|
|
351
|
+
prereg = sub.add_parser(
|
|
352
|
+
"prereg",
|
|
353
|
+
help="hash an eval protocol file to commit to it BEFORE the run (--check verifies a receipt)")
|
|
354
|
+
prereg.add_argument("protocol", help="path to the protocol/plan file to hash")
|
|
355
|
+
prereg.add_argument("--check", metavar="RECEIPT",
|
|
356
|
+
help="verify the protocol matches a receipt's prereg_sha256 instead of hashing")
|
|
357
|
+
prereg.add_argument("--json", action="store_true", help="machine readable output")
|
|
358
|
+
prereg.set_defaults(func=_cmd_prereg)
|
|
359
|
+
|
|
320
360
|
return parser
|
|
321
361
|
|
|
322
362
|
|
|
@@ -274,6 +274,15 @@ def decode_eval_claim(bundle, *, expected_context: Optional[str] = None) -> Opti
|
|
|
274
274
|
want = "ed25519:" + base64.b64encode(base64.b64decode(sig_pub_b64)).decode("ascii")
|
|
275
275
|
if claim.get("issuer") != want:
|
|
276
276
|
return None
|
|
277
|
+
# Verify-boundary schema invariants (release-review CRITICAL): emit_eval_receipt signs a hand-built claim
|
|
278
|
+
# WITHOUT build_eval_claim's checks, so a signed claim could carry an out-of-enum comparator or a
|
|
279
|
+
# non-decimal/non-finite threshold ("inf") — either silently collapses a downstream verdict check (e.g. the
|
|
280
|
+
# HF value-vs-verdict guard) into a tautology. Enforce them here, fail-closed, so every decoded claim is sane.
|
|
281
|
+
if claim.get("comparator") not in _COMPARATORS:
|
|
282
|
+
return None
|
|
283
|
+
_thr = claim.get("threshold")
|
|
284
|
+
if not (isinstance(_thr, str) and _DECIMAL_RE.match(_thr)):
|
|
285
|
+
return None
|
|
277
286
|
samples = claim.get("samples")
|
|
278
287
|
if samples is not None:
|
|
279
288
|
if not isinstance(samples, dict) or set(samples) != {"root_b64", "n", "leaf_alg"}:
|
|
@@ -326,7 +335,7 @@ def check_freshness(claim: dict, max_age_seconds: Optional[int] = None, now=None
|
|
|
326
335
|
"""Replay check (v1.1): parse the claim's timestamp and report its age. A receipt carries a timestamp but
|
|
327
336
|
verify never judged it — an old receipt could be replayed as new. Returns
|
|
328
337
|
{"parsed": bool, "age_seconds": int|None, "fresh": bool|None, "reason": str}. ``fresh`` is None when no
|
|
329
|
-
``max_age_seconds`` bound is given (age reported, not judged).
|
|
338
|
+
``max_age_seconds`` bound is given (age reported, not judged). ISO parsing (normalizes a trailing Z)."""
|
|
330
339
|
from datetime import datetime, timezone # noqa: PLC0415
|
|
331
340
|
ts = claim.get("timestamp")
|
|
332
341
|
if not isinstance(ts, str):
|
|
@@ -85,7 +85,8 @@ def to_eval_results_entry(bundle: dict, *, dataset_id: str, task_id: str, value,
|
|
|
85
85
|
date: Optional[str] = None, source_url: Optional[str] = None,
|
|
86
86
|
source_name: Optional[str] = None, source_user: Optional[str] = None,
|
|
87
87
|
notes: Optional[str] = None, include_token: bool = True,
|
|
88
|
-
require_verified: bool = True
|
|
88
|
+
require_verified: bool = True,
|
|
89
|
+
allow_value_mismatch: bool = False) -> dict:
|
|
89
90
|
"""Build one HF `.eval_results/*.yaml` entry for a receipt.
|
|
90
91
|
|
|
91
92
|
``dataset_id``/``task_id`` name the Hub benchmark (per its `eval.yaml`); ``value`` is the
|
|
@@ -96,6 +97,11 @@ def to_eval_results_entry(bundle: dict, *, dataset_id: str, task_id: str, value,
|
|
|
96
97
|
is never generated from a broken receipt. ``include_token=True`` puts the ``pb1.`` token in
|
|
97
98
|
``verifyToken`` (schema-valid, proofbundle-verifiable; NOT the HF-internal badge token — HF's
|
|
98
99
|
"verified" badge is HF's server-side decision, and this module makes no claim about it).
|
|
100
|
+
|
|
101
|
+
v1.8 (external review): if the bundle is an eval receipt whose claim discloses a ``score``,
|
|
102
|
+
the published ``value`` MUST match it (a Hub reader sees the value, not the token) — a
|
|
103
|
+
mismatch raises unless ``allow_value_mismatch=True``. This stops a receipt saying 0.60 from
|
|
104
|
+
being published next to a displayed 0.99.
|
|
99
105
|
"""
|
|
100
106
|
if require_verified:
|
|
101
107
|
result = verify_bundle(bundle)
|
|
@@ -118,6 +124,42 @@ def to_eval_results_entry(bundle: dict, *, dataset_id: str, task_id: str, value,
|
|
|
118
124
|
raise BundleFormatError(
|
|
119
125
|
"value must be a finite number — inf/-inf/nan cannot be represented in eval_results.yaml")
|
|
120
126
|
|
|
127
|
+
# v1.8 (external review): if the receipt is an eval claim, the published value must be
|
|
128
|
+
# CONSISTENT with the signed pass/fail verdict — a Hub reader sees the value, not the token,
|
|
129
|
+
# so publishing a value that contradicts the receipt is exactly the honesty gap to close.
|
|
130
|
+
# The claim minimizes data (it carries threshold/comparator/passed, not the exact score), so
|
|
131
|
+
# we check the strongest thing available: value <comparator> threshold must equal passed.
|
|
132
|
+
if not allow_value_mismatch:
|
|
133
|
+
from .evalclaim import EVAL_CLAIM_SCHEMA, decode_eval_claim # noqa: PLC0415
|
|
134
|
+
claim = decode_eval_claim(bundle)
|
|
135
|
+
if claim is not None:
|
|
136
|
+
# decode_eval_claim now guarantees comparator ∈ the 4-value enum and a decimal (finite) threshold, so the
|
|
137
|
+
# lookup below is total and thr is finite — no "=="/"inf" tautology can silently no-op the check.
|
|
138
|
+
if {"threshold", "comparator", "passed"} <= set(claim):
|
|
139
|
+
thr = float(claim["threshold"])
|
|
140
|
+
cmp_ok = {">=": numeric >= thr, ">": numeric > thr,
|
|
141
|
+
"<=": numeric <= thr, "<": numeric < thr}[claim["comparator"]]
|
|
142
|
+
if cmp_ok != bool(claim["passed"]):
|
|
143
|
+
raise BundleFormatError(
|
|
144
|
+
f"published value {numeric} is inconsistent with the receipt: the signed claim "
|
|
145
|
+
f"says passed={claim['passed']} for {claim['comparator']} {claim['threshold']}, "
|
|
146
|
+
f"but {numeric} {claim['comparator']} {claim['threshold']} is {cmp_ok} — "
|
|
147
|
+
"pass allow_value_mismatch=True only if this is intentional")
|
|
148
|
+
else:
|
|
149
|
+
# decode failed. FAIL-CLOSED (release-review CRITICAL) if the payload IS an eval claim but did not decode
|
|
150
|
+
# (e.g. out-of-enum comparator / non-decimal threshold) — refusing to publish an unchecked value. A
|
|
151
|
+
# genuinely NON-eval bundle (different/absent schema) has no verdict to check → skip. The bundle's signature
|
|
152
|
+
# was already verified above (require_verified), so reading the raw payload's schema label is authentic.
|
|
153
|
+
try:
|
|
154
|
+
_raw = json.loads(base64.b64decode(bundle["payload_b64"]).decode("utf-8"))
|
|
155
|
+
_is_eval = isinstance(_raw, dict) and _raw.get("schema") == EVAL_CLAIM_SCHEMA
|
|
156
|
+
except (ValueError, TypeError, KeyError):
|
|
157
|
+
_is_eval = False
|
|
158
|
+
if _is_eval:
|
|
159
|
+
raise BundleFormatError(
|
|
160
|
+
"cannot verify the published value against the receipt — the eval claim did not decode "
|
|
161
|
+
"(invalid comparator/threshold?); pass allow_value_mismatch=True only if intentional")
|
|
162
|
+
|
|
121
163
|
entry: dict = {"dataset": {"id": dataset_id, "task_id": task_id},
|
|
122
164
|
"value": numeric if isinstance(value, str) else value}
|
|
123
165
|
if include_token:
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Pre-registration helper (v1.8) — commit to an eval protocol BEFORE the run.
|
|
2
|
+
|
|
3
|
+
The single mitigation for best-of-many / cherry-picking that a receipt can carry: hash the
|
|
4
|
+
protocol document (the plan — suite, seeds, decision rule, sampling policy) *before* running the
|
|
5
|
+
eval, put that hash in the claim's ``prereg_sha256``, and sign the receipt. A verifier who is
|
|
6
|
+
later handed the protocol file re-hashes it and checks it matches — so the plan could not have
|
|
7
|
+
been written to fit the result. The signed receipt's own timestamp binds "this hash existed at
|
|
8
|
+
receipt time" without any network dependency.
|
|
9
|
+
|
|
10
|
+
Construction (verified against standards, 2026-07): commit = **sha256 over the RAW file bytes**.
|
|
11
|
+
Document commitments hash raw bytes (git blob addressing, RFC 6962 leaf hashing, in-toto
|
|
12
|
+
``gitBlob``/``sha256`` DigestSet all hash the artifact's own bytes) — NOT a re-normalized form.
|
|
13
|
+
Canonicalization would only add a lossy transform the verifier must reproduce byte-for-byte; a
|
|
14
|
+
trailing-newline or CRLF change breaking the match is tamper-evidence, not a bug. The claim field
|
|
15
|
+
is a bare 64-hex ``prereg_sha256`` (matching the eval-claim schema).
|
|
16
|
+
|
|
17
|
+
Out of scope, stated honestly: this proves the protocol was fixed relative to the receipt's
|
|
18
|
+
signing time; it does NOT prove the run actually followed the protocol, nor timestamp the
|
|
19
|
+
commitment against a third party's clock. An optional RFC 3161 TSA countersignature over the
|
|
20
|
+
hash (e.g. FreeTSA) is the upgrade when the verifier does not trust the issuer's clock — that is
|
|
21
|
+
a deployment choice, not built in here.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import hashlib
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
|
|
29
|
+
__all__ = ["prereg_hash", "verify_prereg"]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def prereg_hash(protocol_path) -> str:
|
|
33
|
+
"""Return the lowercase-hex sha256 over the RAW bytes of the protocol file — the value to
|
|
34
|
+
place in a claim's ``prereg_sha256`` BEFORE running the eval."""
|
|
35
|
+
data = Path(protocol_path).read_bytes()
|
|
36
|
+
return hashlib.sha256(data).hexdigest()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def verify_prereg(protocol_path, claim: dict) -> dict:
|
|
40
|
+
"""Check that ``claim['prereg_sha256']`` matches the sha256 of the protocol file.
|
|
41
|
+
|
|
42
|
+
Returns ``{ok, present, expected, actual, detail}``. ``present`` is False when the claim
|
|
43
|
+
carries no ``prereg_sha256`` (not pre-registered) — the caller decides whether that is
|
|
44
|
+
acceptable; ``ok`` is only True on a present-and-matching hash (fail-closed)."""
|
|
45
|
+
expected = claim.get("prereg_sha256") if isinstance(claim, dict) else None
|
|
46
|
+
result = {"ok": False, "present": expected is not None, "expected": expected,
|
|
47
|
+
"actual": None, "detail": ""}
|
|
48
|
+
if expected is None:
|
|
49
|
+
result["detail"] = "claim carries no prereg_sha256 (not pre-registered)"
|
|
50
|
+
return result
|
|
51
|
+
actual = prereg_hash(protocol_path)
|
|
52
|
+
result["actual"] = actual
|
|
53
|
+
if actual == expected:
|
|
54
|
+
result["ok"] = True
|
|
55
|
+
result["detail"] = "protocol file matches the pre-registered hash"
|
|
56
|
+
else:
|
|
57
|
+
result["detail"] = "protocol file does NOT match the pre-registered hash (plan changed?)"
|
|
58
|
+
return result
|
|
@@ -97,7 +97,9 @@ def parse_tlog_proof(text: str) -> dict:
|
|
|
97
97
|
if pos >= len(lines) or not lines[pos].startswith("index "):
|
|
98
98
|
raise BundleFormatError("tlog-proof is missing the index line")
|
|
99
99
|
index_s = lines[pos][len("index "):]
|
|
100
|
-
|
|
100
|
+
# isascii() before isdigit() (release-review #7): str.isdigit() is True for Unicode digits (e.g. '²' U+00B2,
|
|
101
|
+
# Arabic-Indic), which int() would then reject or mis-parse — mirror checkpoint.py's ASCII-only guard.
|
|
102
|
+
if not (index_s.isascii() and index_s.isdigit()) or (index_s != "0" and index_s.startswith("0")):
|
|
101
103
|
raise BundleFormatError("tlog-proof index must be ASCII decimal with no leading zeros")
|
|
102
104
|
pos += 1
|
|
103
105
|
proof = []
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: proofbundle
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.8.0
|
|
4
4
|
Summary: Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT.
|
|
5
5
|
Author: Konrad Gruszka
|
|
6
6
|
License: MIT
|
|
@@ -84,7 +84,7 @@ file, no server, no network.**
|
|
|
84
84
|
|
|
85
85
|
**At a glance:** `proofbundle emit` signs and anchors a payload; `proofbundle
|
|
86
86
|
verify` checks one self-contained `bundle.json` with three offline cryptographic
|
|
87
|
-
checks → `OK` or `FAILED`. No network, no daemon, no own crypto.
|
|
87
|
+
checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 289 tests + a CI mutation gate.
|
|
88
88
|
|
|
89
89
|
## Contents
|
|
90
90
|
|
|
@@ -19,6 +19,7 @@ src/proofbundle/intoto.py
|
|
|
19
19
|
src/proofbundle/kbjwt.py
|
|
20
20
|
src/proofbundle/merkle.py
|
|
21
21
|
src/proofbundle/persample.py
|
|
22
|
+
src/proofbundle/prereg.py
|
|
22
23
|
src/proofbundle/py.typed
|
|
23
24
|
src/proofbundle/pytest_plugin.py
|
|
24
25
|
src/proofbundle/sdjwt.py
|
|
@@ -33,6 +34,7 @@ src/proofbundle.egg-info/entry_points.txt
|
|
|
33
34
|
src/proofbundle.egg-info/requires.txt
|
|
34
35
|
src/proofbundle.egg-info/top_level.txt
|
|
35
36
|
src/proofbundle/adapters/__init__.py
|
|
37
|
+
src/proofbundle/adapters/_provenance.py
|
|
36
38
|
src/proofbundle/adapters/eee.py
|
|
37
39
|
src/proofbundle/adapters/inspect_ai.py
|
|
38
40
|
src/proofbundle/adapters/lm_eval.py
|
|
@@ -53,6 +55,7 @@ tests/test_emit.py
|
|
|
53
55
|
tests/test_eval_claim_schema.py
|
|
54
56
|
tests/test_evalclaim.py
|
|
55
57
|
tests/test_examples.py
|
|
58
|
+
tests/test_fuzz_parsers.py
|
|
56
59
|
tests/test_hf_evals.py
|
|
57
60
|
tests/test_inspect_hook.py
|
|
58
61
|
tests/test_intoto.py
|
|
@@ -61,7 +64,9 @@ tests/test_kbjwt.py
|
|
|
61
64
|
tests/test_merkle.py
|
|
62
65
|
tests/test_merkle_property.py
|
|
63
66
|
tests/test_persample.py
|
|
67
|
+
tests/test_prereg.py
|
|
64
68
|
tests/test_promptfoo.py
|
|
69
|
+
tests/test_provenance.py
|
|
65
70
|
tests/test_pytest_plugin.py
|
|
66
71
|
tests/test_rekor_interop.py
|
|
67
72
|
tests/test_rfc6962_external_vectors.py
|
|
@@ -39,6 +39,19 @@ class TestEvalClaim(unittest.TestCase):
|
|
|
39
39
|
self.assertEqual(decoded["suite"], "safety-refusal")
|
|
40
40
|
self.assertTrue(decoded["passed"])
|
|
41
41
|
|
|
42
|
+
def test_decode_rejects_bad_comparator_and_threshold(self):
|
|
43
|
+
# release-review CRITICAL: emit_eval_receipt signs a hand-built claim WITHOUT build_eval_claim's checks,
|
|
44
|
+
# so decode_eval_claim must enforce comparator-enum + decimal-threshold at the verify boundary — else a
|
|
45
|
+
# downstream value-consistency check silently no-ops on comparator "==" / non-finite threshold "inf".
|
|
46
|
+
signer = generate_signer()
|
|
47
|
+
for key, bad in (("comparator", "=="), ("comparator", "~="),
|
|
48
|
+
("threshold", "inf"), ("threshold", "nan"), ("threshold", "1e5")):
|
|
49
|
+
claim, _ = _claim(signer)
|
|
50
|
+
claim[key] = bad
|
|
51
|
+
bundle = emit_eval_receipt(claim, signer)
|
|
52
|
+
self.assertTrue(verify_bundle(bundle).ok, f"{key}={bad}: bundle still signs/verifies")
|
|
53
|
+
self.assertIsNone(decode_eval_claim(bundle), f"{key}={bad}: claim must NOT decode")
|
|
54
|
+
|
|
42
55
|
def test_decode_reads_path_once_no_toctou(self):
|
|
43
56
|
# CRITICAL (release review): decode_eval_claim(path) must resolve the path to a dict EXACTLY ONCE and
|
|
44
57
|
# verify + parse the SAME object. A second re-read is a TOCTOU (CWE-367) file-race window that could return
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Property-based fuzzing of the text/JWT parsers (v1.8).
|
|
2
|
+
|
|
3
|
+
The invariant for every attacker-controlled parser: on ANY input it returns a value or raises a
|
|
4
|
+
proofbundle error (BundleFormatError / ProofBundleError / ValueError) — NEVER an uncaught crash
|
|
5
|
+
(AttributeError, IndexError, KeyError, TypeError, UnicodeError, recursion, …) and never a hang.
|
|
6
|
+
This is the "never a raw traceback" contract, checked adversarially with Hypothesis rather than
|
|
7
|
+
by hand-picked cases. Hypothesis is the lowest-friction sound fuzzer for pure-Python parsers
|
|
8
|
+
(no native toolchain); an Atheris coverage-guided driver over the same bodies can be added under
|
|
9
|
+
fuzz/ for Linux CI if deeper coverage is ever wanted. hypothesis is a dev dependency only — this
|
|
10
|
+
module no-ops when it is absent (same pattern as test_merkle_property.py)."""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import unittest
|
|
14
|
+
|
|
15
|
+
try:
|
|
16
|
+
from hypothesis import given, settings
|
|
17
|
+
from hypothesis import strategies as st
|
|
18
|
+
except ImportError: # pragma: no cover - dev-only dependency
|
|
19
|
+
given = None
|
|
20
|
+
|
|
21
|
+
from proofbundle.errors import ProofBundleError
|
|
22
|
+
from proofbundle.tlogproof import parse_tlog_proof, verify_tlog_proof
|
|
23
|
+
from proofbundle.checkpoint import verify_checkpoint, verify_cosignature
|
|
24
|
+
from proofbundle.statuslist import verify_status_snapshot
|
|
25
|
+
from proofbundle.kbjwt import split_key_binding, verify_key_binding
|
|
26
|
+
from proofbundle.sdjwt import verify_sd_jwt
|
|
27
|
+
|
|
28
|
+
_ALLOWED = (ProofBundleError, ValueError) # the documented "malformed input" surface
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _must_not_crash(fn, *args, **kwargs):
|
|
32
|
+
try:
|
|
33
|
+
fn(*args, **kwargs)
|
|
34
|
+
except _ALLOWED:
|
|
35
|
+
pass # documented malformed-input path — fine
|
|
36
|
+
# any other exception propagates and fails the test (the contract violation we hunt)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
if given is not None:
|
|
40
|
+
_texts = st.text(alphabet=st.characters(min_codepoint=1, max_codepoint=0x2FFF), max_size=400)
|
|
41
|
+
|
|
42
|
+
class TestParserRobustness(unittest.TestCase):
|
|
43
|
+
@settings(max_examples=300, deadline=None)
|
|
44
|
+
@given(_texts)
|
|
45
|
+
def test_parse_tlog_proof_never_crashes(self, s):
|
|
46
|
+
_must_not_crash(parse_tlog_proof, s)
|
|
47
|
+
|
|
48
|
+
@settings(max_examples=200, deadline=None)
|
|
49
|
+
@given(_texts, _texts)
|
|
50
|
+
def test_verify_tlog_proof_never_crashes(self, proof, leaf):
|
|
51
|
+
_must_not_crash(verify_tlog_proof, proof, leaf.encode("utf-8", "surrogatepass"),
|
|
52
|
+
"log+00000000+" + "A" * 44)
|
|
53
|
+
|
|
54
|
+
@settings(max_examples=300, deadline=None)
|
|
55
|
+
@given(_texts, _texts)
|
|
56
|
+
def test_verify_checkpoint_never_crashes(self, note, vkey):
|
|
57
|
+
_must_not_crash(verify_checkpoint, note, vkey)
|
|
58
|
+
|
|
59
|
+
@settings(max_examples=300, deadline=None)
|
|
60
|
+
@given(_texts, _texts)
|
|
61
|
+
def test_verify_cosignature_never_crashes(self, note, vkey):
|
|
62
|
+
_must_not_crash(verify_cosignature, note, vkey)
|
|
63
|
+
|
|
64
|
+
@settings(max_examples=300, deadline=None)
|
|
65
|
+
@given(_texts)
|
|
66
|
+
def test_split_key_binding_never_crashes(self, compact):
|
|
67
|
+
sd, kb = split_key_binding(compact) # total by contract → returns a tuple
|
|
68
|
+
self.assertIsInstance(sd, str)
|
|
69
|
+
|
|
70
|
+
@settings(max_examples=300, deadline=None)
|
|
71
|
+
@given(_texts)
|
|
72
|
+
def test_verify_key_binding_never_crashes(self, compact):
|
|
73
|
+
_must_not_crash(verify_key_binding, compact)
|
|
74
|
+
|
|
75
|
+
@settings(max_examples=300, deadline=None)
|
|
76
|
+
@given(_texts)
|
|
77
|
+
def test_verify_sd_jwt_never_crashes(self, compact):
|
|
78
|
+
_must_not_crash(verify_sd_jwt, compact)
|
|
79
|
+
|
|
80
|
+
@settings(max_examples=200, deadline=None)
|
|
81
|
+
@given(_texts)
|
|
82
|
+
def test_verify_status_snapshot_never_crashes(self, token):
|
|
83
|
+
_must_not_crash(verify_status_snapshot, token, expected_uri="u", index=0,
|
|
84
|
+
issuer_pubkey=b"\x00" * 32)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
if __name__ == "__main__":
|
|
88
|
+
unittest.main()
|
|
@@ -150,3 +150,38 @@ class TestEvalResultsEntry(unittest.TestCase):
|
|
|
150
150
|
|
|
151
151
|
if __name__ == "__main__":
|
|
152
152
|
unittest.main()
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
class TestValueConsistency(unittest.TestCase):
|
|
156
|
+
"""v1.8: a published value that contradicts the signed pass/fail verdict is refused."""
|
|
157
|
+
|
|
158
|
+
def _receipt(self, threshold, comparator, score):
|
|
159
|
+
from proofbundle import generate_signer
|
|
160
|
+
from proofbundle.evalclaim import build_eval_claim, emit_eval_receipt
|
|
161
|
+
claim, _ = build_eval_claim(
|
|
162
|
+
suite="s", suite_version="1", metric="acc", comparator=comparator,
|
|
163
|
+
threshold=threshold, score=score, n=10, model_id="m", dataset_id="d",
|
|
164
|
+
issuer="", timestamp="2026-07-02T00:00:00Z")
|
|
165
|
+
return emit_eval_receipt(claim, generate_signer())
|
|
166
|
+
|
|
167
|
+
def test_consistent_value_ok(self):
|
|
168
|
+
r = self._receipt("0.80", ">=", "0.91") # passed=True
|
|
169
|
+
entry = to_eval_results_entry(r, dataset_id="d/x", task_id="t", value=0.91)
|
|
170
|
+
self.assertEqual(entry["value"], 0.91)
|
|
171
|
+
|
|
172
|
+
def test_inconsistent_value_refused(self):
|
|
173
|
+
r = self._receipt("0.80", ">=", "0.60") # passed=False (0.60 < 0.80)
|
|
174
|
+
with self.assertRaises(BundleFormatError) as ctx:
|
|
175
|
+
to_eval_results_entry(r, dataset_id="d/x", task_id="t", value=0.99) # 0.99>=0.80 → True, but claim says False
|
|
176
|
+
self.assertIn("inconsistent", str(ctx.exception))
|
|
177
|
+
|
|
178
|
+
def test_override_allows_mismatch(self):
|
|
179
|
+
r = self._receipt("0.80", ">=", "0.60")
|
|
180
|
+
entry = to_eval_results_entry(r, dataset_id="d/x", task_id="t", value=0.99,
|
|
181
|
+
allow_value_mismatch=True)
|
|
182
|
+
self.assertEqual(entry["value"], 0.99)
|
|
183
|
+
|
|
184
|
+
def test_non_eval_bundle_skips_check(self):
|
|
185
|
+
# a plain emit_bundle (not an eval receipt) has no claim → no cross-check, value accepted
|
|
186
|
+
entry = to_eval_results_entry(_bundle(), dataset_id="d/x", task_id="t", value=0.5)
|
|
187
|
+
self.assertEqual(entry["value"], 0.5)
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""v1.8 pre-registration helper: commit to a protocol before the run, verify after."""
|
|
2
|
+
import hashlib
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import tempfile
|
|
6
|
+
import unittest
|
|
7
|
+
|
|
8
|
+
from proofbundle.cli import main
|
|
9
|
+
from proofbundle.emit import generate_signer
|
|
10
|
+
from proofbundle.evalclaim import build_eval_claim, emit_eval_receipt, issuer_fingerprint
|
|
11
|
+
from proofbundle.prereg import prereg_hash, verify_prereg
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class TestPrereg(unittest.TestCase):
|
|
15
|
+
def _file(self, content: bytes) -> str:
|
|
16
|
+
h = tempfile.NamedTemporaryFile("wb", delete=False)
|
|
17
|
+
h.write(content)
|
|
18
|
+
h.close()
|
|
19
|
+
return h.name
|
|
20
|
+
|
|
21
|
+
def test_hash_is_sha256_of_raw_bytes(self):
|
|
22
|
+
content = b"protocol: fixed seeds 1..5\ndecision: acc >= 0.8\n"
|
|
23
|
+
path = self._file(content)
|
|
24
|
+
try:
|
|
25
|
+
self.assertEqual(prereg_hash(path), hashlib.sha256(content).hexdigest())
|
|
26
|
+
finally:
|
|
27
|
+
os.unlink(path)
|
|
28
|
+
|
|
29
|
+
def test_verify_match(self):
|
|
30
|
+
content = b"the plan"
|
|
31
|
+
path = self._file(content)
|
|
32
|
+
try:
|
|
33
|
+
claim = {"prereg_sha256": hashlib.sha256(content).hexdigest()}
|
|
34
|
+
res = verify_prereg(path, claim)
|
|
35
|
+
self.assertTrue(res["ok"])
|
|
36
|
+
self.assertTrue(res["present"])
|
|
37
|
+
finally:
|
|
38
|
+
os.unlink(path)
|
|
39
|
+
|
|
40
|
+
def test_verify_mismatch_is_caught(self):
|
|
41
|
+
path = self._file(b"the ACTUAL plan")
|
|
42
|
+
try:
|
|
43
|
+
claim = {"prereg_sha256": hashlib.sha256(b"a DIFFERENT plan").hexdigest()}
|
|
44
|
+
res = verify_prereg(path, claim)
|
|
45
|
+
self.assertFalse(res["ok"])
|
|
46
|
+
self.assertIn("does NOT match", res["detail"])
|
|
47
|
+
finally:
|
|
48
|
+
os.unlink(path)
|
|
49
|
+
|
|
50
|
+
def test_not_preregistered_reports_absent(self):
|
|
51
|
+
path = self._file(b"x")
|
|
52
|
+
try:
|
|
53
|
+
res = verify_prereg(path, {}) # no prereg_sha256
|
|
54
|
+
self.assertFalse(res["ok"])
|
|
55
|
+
self.assertFalse(res["present"])
|
|
56
|
+
finally:
|
|
57
|
+
os.unlink(path)
|
|
58
|
+
|
|
59
|
+
def test_trailing_byte_change_breaks_match(self):
|
|
60
|
+
# tamper-evidence: a single appended newline changes the commitment (by design).
|
|
61
|
+
path = self._file(b"plan\n")
|
|
62
|
+
try:
|
|
63
|
+
claim = {"prereg_sha256": hashlib.sha256(b"plan").hexdigest()} # committed without \n
|
|
64
|
+
self.assertFalse(verify_prereg(path, claim)["ok"])
|
|
65
|
+
finally:
|
|
66
|
+
os.unlink(path)
|
|
67
|
+
|
|
68
|
+
def test_cli_roundtrip(self):
|
|
69
|
+
import json
|
|
70
|
+
from proofbundle.cli import main
|
|
71
|
+
from proofbundle import generate_signer
|
|
72
|
+
from proofbundle.evalclaim import build_eval_claim, emit_eval_receipt
|
|
73
|
+
import contextlib
|
|
74
|
+
import io
|
|
75
|
+
proto = self._file(b"suite=mmlu; seeds=1..5; rule=acc>=0.8")
|
|
76
|
+
try:
|
|
77
|
+
h = prereg_hash(proto)
|
|
78
|
+
claim, _ = build_eval_claim(
|
|
79
|
+
suite="s", suite_version="1", metric="acc", comparator=">=", threshold="0.8",
|
|
80
|
+
score="0.9", n=10, model_id="m", dataset_id="d", issuer="",
|
|
81
|
+
timestamp="2026-07-02T00:00:00Z", prereg_sha256=h)
|
|
82
|
+
receipt = emit_eval_receipt(claim, generate_signer())
|
|
83
|
+
rp = tempfile.NamedTemporaryFile("w", suffix=".json", delete=False)
|
|
84
|
+
json.dump(receipt, rp)
|
|
85
|
+
rp.close()
|
|
86
|
+
with contextlib.redirect_stdout(io.StringIO()):
|
|
87
|
+
self.assertEqual(main(["prereg", proto, "--check", rp.name]), 0)
|
|
88
|
+
# a different protocol fails
|
|
89
|
+
other = self._file(b"a different plan")
|
|
90
|
+
self.assertEqual(main(["prereg", other, "--check", rp.name]), 1)
|
|
91
|
+
os.unlink(rp.name)
|
|
92
|
+
os.unlink(other)
|
|
93
|
+
finally:
|
|
94
|
+
os.unlink(proto)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class TestPreregCheckVerifies(unittest.TestCase):
|
|
98
|
+
def test_check_rejects_forged_unsigned_bundle(self):
|
|
99
|
+
# release-review CRITICAL: prereg --check MUST verify the receipt's signature before trusting its
|
|
100
|
+
# prereg_sha256 — a forged/unsigned bundle with a doctored prereg_sha256 must NOT get a PASS.
|
|
101
|
+
with tempfile.TemporaryDirectory() as d:
|
|
102
|
+
proto = os.path.join(d, "protocol.md")
|
|
103
|
+
with open(proto, "wb") as f:
|
|
104
|
+
f.write(b"pre-registered analysis plan v1")
|
|
105
|
+
h = prereg_hash(proto)
|
|
106
|
+
signer = generate_signer()
|
|
107
|
+
claim, _ = build_eval_claim(
|
|
108
|
+
suite="s", suite_version="v1", metric="m", comparator=">=", threshold="0.80",
|
|
109
|
+
score="0.92", n=10, model_id="a/b", dataset_id="c/d",
|
|
110
|
+
issuer=issuer_fingerprint(signer), timestamp="2026-07-01T12:00:00Z",
|
|
111
|
+
model_salt=b"0" * 16, dataset_salt=b"1" * 16, prereg_sha256=h)
|
|
112
|
+
bundle = emit_eval_receipt(claim, signer)
|
|
113
|
+
bundle["signature"]["signature_b64"] = "AA==" * 22 # corrupt the Ed25519 signature
|
|
114
|
+
fp = os.path.join(d, "forged.json")
|
|
115
|
+
with open(fp, "w", encoding="utf-8") as f:
|
|
116
|
+
json.dump(bundle, f)
|
|
117
|
+
rc = main(["prereg", proto, "--check", fp])
|
|
118
|
+
self.assertNotEqual(rc, 0, "a forged/unsigned receipt must FAIL prereg --check")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
if __name__ == "__main__":
|
|
122
|
+
unittest.main()
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""v1.8 provenance hardening: config-hash + run-id + log-native timestamp in adapters."""
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
import tempfile
|
|
5
|
+
import unittest
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from proofbundle.adapters._provenance import add_provenance, config_hash
|
|
9
|
+
from proofbundle.adapters import from_lm_eval_results, from_promptfoo_results
|
|
10
|
+
|
|
11
|
+
FIXTURES = Path(__file__).parent / "fixtures"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class TestConfigHash(unittest.TestCase):
|
|
15
|
+
def test_deterministic_and_labeled(self):
|
|
16
|
+
a = config_hash({"b": 1, "a": 2})
|
|
17
|
+
b = config_hash({"a": 2, "b": 1}) # key order must not matter
|
|
18
|
+
self.assertEqual(a, b)
|
|
19
|
+
self.assertTrue(a.startswith("sha256-jcs:") or a.startswith("sha256-sortkeys:"))
|
|
20
|
+
self.assertEqual(len(a.split(":")[1]), 64)
|
|
21
|
+
|
|
22
|
+
def test_empty_is_none(self):
|
|
23
|
+
self.assertIsNone(config_hash(None))
|
|
24
|
+
self.assertIsNone(config_hash({}))
|
|
25
|
+
self.assertIsNone(config_hash([]))
|
|
26
|
+
|
|
27
|
+
def test_change_changes_hash(self):
|
|
28
|
+
self.assertNotEqual(config_hash({"seed": 1}), config_hash({"seed": 2}))
|
|
29
|
+
|
|
30
|
+
def test_add_provenance_skips_absent(self):
|
|
31
|
+
prov = {}
|
|
32
|
+
add_provenance(prov, run_id=None, config=None, log_timestamp=None)
|
|
33
|
+
self.assertEqual(prov, {})
|
|
34
|
+
add_provenance(prov, run_id="r1", config={"x": 1}, log_timestamp=123)
|
|
35
|
+
self.assertEqual(prov["run_id"], "r1")
|
|
36
|
+
self.assertEqual(prov["run_timestamp"], "123")
|
|
37
|
+
self.assertIn("config_hash", prov)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class TestPromptfooProvenance(unittest.TestCase):
|
|
41
|
+
def test_run_id_and_config_hash_present(self):
|
|
42
|
+
claim, _ = from_promptfoo_results(
|
|
43
|
+
FIXTURES / "promptfoo_results_v3.json", comparator=">=", threshold="0.5",
|
|
44
|
+
timestamp="2026-07-02T00:00:00Z")
|
|
45
|
+
prov = claim["provenance"]
|
|
46
|
+
self.assertEqual(prov["run_id"], "eval-Xa3-2026-07-02T14:03:11")
|
|
47
|
+
self.assertIn("config_hash", prov)
|
|
48
|
+
self.assertIn("run_timestamp", prov)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class TestLmEvalProvenance(unittest.TestCase):
|
|
52
|
+
def _fixture_with_config(self):
|
|
53
|
+
data = json.loads((FIXTURES / "lm_eval_arc_easy_real.json").read_text())
|
|
54
|
+
return data
|
|
55
|
+
|
|
56
|
+
def test_config_hash_and_timestamp(self):
|
|
57
|
+
data = self._fixture_with_config()
|
|
58
|
+
data.setdefault("config", {"model": "hf", "seed": 1234})
|
|
59
|
+
data["date"] = 1780000000.5
|
|
60
|
+
handle = tempfile.NamedTemporaryFile("w", suffix=".json", delete=False)
|
|
61
|
+
json.dump(data, handle)
|
|
62
|
+
handle.close()
|
|
63
|
+
try:
|
|
64
|
+
task = next(iter(data["results"]))
|
|
65
|
+
metric = next(k.split(",")[0] for k in data["results"][task] if "," in k)
|
|
66
|
+
claim, _ = from_lm_eval_results(handle.name, task=task, metric=metric,
|
|
67
|
+
comparator=">=", threshold="0.1",
|
|
68
|
+
timestamp="2026-07-02T00:00:00Z")
|
|
69
|
+
finally:
|
|
70
|
+
os.unlink(handle.name)
|
|
71
|
+
prov = claim["provenance"]
|
|
72
|
+
self.assertIn("config_hash", prov)
|
|
73
|
+
self.assertEqual(prov["run_timestamp"], "1780000000.5") # log-native, not caller ts
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
if __name__ == "__main__":
|
|
77
|
+
unittest.main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|