proofbundle 1.7.0__tar.gz → 1.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. {proofbundle-1.7.0/src/proofbundle.egg-info → proofbundle-1.8.0}/PKG-INFO +2 -2
  2. {proofbundle-1.7.0 → proofbundle-1.8.0}/README.md +1 -1
  3. {proofbundle-1.7.0 → proofbundle-1.8.0}/pyproject.toml +1 -1
  4. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/__init__.py +6 -1
  5. proofbundle-1.8.0/src/proofbundle/adapters/_provenance.py +63 -0
  6. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/inspect_ai.py +9 -0
  7. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/lm_eval.py +10 -0
  8. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/promptfoo.py +7 -0
  9. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/cli.py +40 -0
  10. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/evalclaim.py +10 -1
  11. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/hf_evals.py +43 -1
  12. proofbundle-1.8.0/src/proofbundle/prereg.py +58 -0
  13. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/tlogproof.py +3 -1
  14. {proofbundle-1.7.0 → proofbundle-1.8.0/src/proofbundle.egg-info}/PKG-INFO +2 -2
  15. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/SOURCES.txt +5 -0
  16. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_evalclaim.py +13 -0
  17. proofbundle-1.8.0/tests/test_fuzz_parsers.py +88 -0
  18. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_hf_evals.py +35 -0
  19. proofbundle-1.8.0/tests/test_prereg.py +122 -0
  20. proofbundle-1.8.0/tests/test_provenance.py +77 -0
  21. {proofbundle-1.7.0 → proofbundle-1.8.0}/LICENSE +0 -0
  22. {proofbundle-1.7.0 → proofbundle-1.8.0}/setup.cfg +0 -0
  23. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/_inspect_registry.py +0 -0
  24. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/_integration.py +0 -0
  25. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/__init__.py +0 -0
  26. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/eee.py +0 -0
  27. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/adapters/samples.py +0 -0
  28. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/bundle.py +0 -0
  29. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/checkpoint.py +0 -0
  30. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/demo.py +0 -0
  31. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/dsse.py +0 -0
  32. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/eee_eval_schema.json +0 -0
  33. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/emit.py +0 -0
  34. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/errors.py +0 -0
  35. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/inspect_hook.py +0 -0
  36. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/intoto.py +0 -0
  37. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/kbjwt.py +0 -0
  38. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/merkle.py +0 -0
  39. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/persample.py +0 -0
  40. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/py.typed +0 -0
  41. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/pytest_plugin.py +0 -0
  42. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/sdjwt.py +0 -0
  43. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/sdjwt_issue.py +0 -0
  44. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/signature.py +0 -0
  45. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle/statuslist.py +0 -0
  46. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/dependency_links.txt +0 -0
  47. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/entry_points.txt +0 -0
  48. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/requires.txt +0 -0
  49. {proofbundle-1.7.0 → proofbundle-1.8.0}/src/proofbundle.egg-info/top_level.txt +0 -0
  50. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_adapters.py +0 -0
  51. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_adversarial.py +0 -0
  52. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_bundle.py +0 -0
  53. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_bundle_robustness.py +0 -0
  54. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_checkpoint.py +0 -0
  55. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_cli.py +0 -0
  56. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_cli_eval.py +0 -0
  57. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_cosignature.py +0 -0
  58. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_cosignature_mldsa.py +0 -0
  59. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_demo.py +0 -0
  60. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_eee.py +0 -0
  61. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_emit.py +0 -0
  62. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_eval_claim_schema.py +0 -0
  63. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_examples.py +0 -0
  64. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_inspect_hook.py +0 -0
  65. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_intoto.py +0 -0
  66. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_intoto_dsse.py +0 -0
  67. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_kbjwt.py +0 -0
  68. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_merkle.py +0 -0
  69. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_merkle_property.py +0 -0
  70. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_persample.py +0 -0
  71. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_promptfoo.py +0 -0
  72. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_pytest_plugin.py +0 -0
  73. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_rekor_interop.py +0 -0
  74. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_rfc6962_external_vectors.py +0 -0
  75. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_schema.py +0 -0
  76. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_sdjwt_issue.py +0 -0
  77. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_sdjwt_reference.py +0 -0
  78. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_signature.py +0 -0
  79. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_statuslist.py +0 -0
  80. {proofbundle-1.7.0 → proofbundle-1.8.0}/tests/test_tlogproof.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proofbundle
3
- Version: 1.7.0
3
+ Version: 1.8.0
4
4
  Summary: Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT.
5
5
  Author: Konrad Gruszka
6
6
  License: MIT
@@ -84,7 +84,7 @@ file, no server, no network.**
84
84
 
85
85
  **At a glance:** `proofbundle emit` signs and anchors a payload; `proofbundle
86
86
  verify` checks one self-contained `bundle.json` with three offline cryptographic
87
- checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 257 tests + a CI mutation gate.
87
+ checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 289 tests + a CI mutation gate.
88
88
 
89
89
  ## Contents
90
90
 
@@ -36,7 +36,7 @@ file, no server, no network.**
36
36
 
37
37
  **At a glance:** `proofbundle emit` signs and anchors a payload; `proofbundle
38
38
  verify` checks one self-contained `bundle.json` with three offline cryptographic
39
- checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 257 tests + a CI mutation gate.
39
+ checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 289 tests + a CI mutation gate.
40
40
 
41
41
  ## Contents
42
42
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "proofbundle"
7
- version = "1.7.0"
7
+ version = "1.8.0"
8
8
  description = "Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -13,7 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  from typing import TYPE_CHECKING
15
15
 
16
- __version__ = "1.7.0"
16
+ __version__ = "1.8.0"
17
17
 
18
18
  __all__ = [
19
19
  "__version__",
@@ -36,6 +36,8 @@ __all__ = [
36
36
  "sample_opening",
37
37
  "verify_sample_opening",
38
38
  "audit_challenge",
39
+ "prereg_hash",
40
+ "verify_prereg",
39
41
  "VerificationResult",
40
42
  "Check",
41
43
  "ProofBundleError",
@@ -59,6 +61,8 @@ _LAZY = {
59
61
  "sample_opening": ".persample",
60
62
  "verify_sample_opening": ".persample",
61
63
  "audit_challenge": ".persample",
64
+ "prereg_hash": ".prereg",
65
+ "verify_prereg": ".prereg",
62
66
  }
63
67
 
64
68
  if TYPE_CHECKING: # static analysers + IDEs see the real names/types; runtime stays lazy
@@ -70,6 +74,7 @@ if TYPE_CHECKING: # static analysers + IDEs see the real names/types; runtime s
70
74
  from .hf_evals import receipt_token, verify_receipt_token
71
75
  from .persample import (audit_challenge, build_sample_tree, sample_opening,
72
76
  verify_sample_opening)
77
+ from .prereg import prereg_hash, verify_prereg
73
78
  from .statuslist import verify_status_snapshot
74
79
  from .tlogproof import verify_tlog_proof
75
80
  from .merkle import verify_consistency, verify_inclusion
@@ -0,0 +1,63 @@
1
+ """Shared provenance helpers for adapters (v1.8).
2
+
3
+ The external review found that run-id and config-hash were missing from nearly every adapter and
4
+ that the two flagship adapters took the timestamp from the caller rather than the eval log. These
5
+ helpers close that: each adapter now records, where the framework exposes it, a stable RUN id, a
6
+ CONFIG hash, and the LOG-NATIVE timestamp — so a receipt is traceable back to the exact run.
7
+
8
+ Design notes (verified against framework source, 2026-07):
9
+ - No framework ships a canonical config hash, so we compute our own. Config is an in-memory
10
+ object re-serialized non-deterministically, so it MUST be canonicalized before hashing —
11
+ RFC 8785 JCS via the same `rfc8785` extra the emit path already needs. If that extra is
12
+ absent we fall back to a deterministic `json.dumps(sort_keys=True)` and LABEL the hash
13
+ algorithm accordingly, so a verifier is never misled about how the hash was formed.
14
+ - The hash is over the config's JSON, prefixed with a domain tag, hex sha256. It is provenance
15
+ metadata (traceability), NOT a security commitment — it is not salted and reveals structure;
16
+ it exists so two receipts from the same config are linkable and a changed config is visible.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import hashlib
22
+ import json
23
+ from typing import Optional
24
+
25
+ _CONFIG_DOMAIN = b"proofbundle/v1.8/config-hash\x00"
26
+
27
+
28
+ def config_hash(config) -> Optional[str]:
29
+ """Return ``"<alg>:<hex>"`` over the canonical JSON of a config object, or None if it is
30
+ empty/None. ``<alg>`` is ``sha256-jcs`` when RFC 8785 is available, else ``sha256-sortkeys``
31
+ (both deterministic; the label tells a verifier which normalization produced the hex)."""
32
+ if config is None or config == {} or config == []:
33
+ return None
34
+ try:
35
+ import rfc8785 # noqa: PLC0415 — same optional dep as the emit path
36
+ canonical = rfc8785.dumps(config)
37
+ alg = "sha256-jcs"
38
+ except (ImportError, ValueError, TypeError):
39
+ # rfc8785 rejects non-JCS-able values (e.g. floats it deems unsafe); fall back to a
40
+ # deterministic stdlib serialization and label it so the difference is never hidden.
41
+ try:
42
+ canonical = json.dumps(config, sort_keys=True, separators=(",", ":"),
43
+ ensure_ascii=False).encode("utf-8")
44
+ except (TypeError, ValueError):
45
+ return None
46
+ alg = "sha256-sortkeys"
47
+ return f"{alg}:{hashlib.sha256(_CONFIG_DOMAIN + canonical).hexdigest()}"
48
+
49
+
50
+ def add_provenance(provenance: dict, *, run_id=None, config=None, log_timestamp=None,
51
+ config_hash_value: Optional[str] = None) -> dict:
52
+ """Merge the standard traceability fields into a provenance dict, skipping absent ones.
53
+
54
+ ``config_hash_value`` lets a caller pass a precomputed hash (e.g. over already-canonical
55
+ material) instead of a config object; otherwise ``config`` is hashed here."""
56
+ if run_id:
57
+ provenance["run_id"] = str(run_id)
58
+ if log_timestamp is not None:
59
+ provenance["run_timestamp"] = str(log_timestamp)
60
+ ch = config_hash_value if config_hash_value is not None else config_hash(config)
61
+ if ch:
62
+ provenance["config_hash"] = ch
63
+ return provenance
@@ -90,6 +90,15 @@ def from_inspect_ai_log(path, metric: str, *, comparator: str, threshold: str, t
90
90
  if tv is not None:
91
91
  provenance["task_version"] = str(tv)
92
92
 
93
+ # v1.8 (external review): run-id + config-hash + LOG-NATIVE timestamp so a receipt traces back
94
+ # to the exact run. inspect_ai: eval.run_id (unique run id), eval.created (UTC datetime string),
95
+ # eval.task_args (the config material — no native config hash exists, so we compute one).
96
+ from ._provenance import add_provenance # noqa: PLC0415
97
+ task_args = getattr(ev, "task_args", None)
98
+ add_provenance(provenance, run_id=getattr(ev, "run_id", None),
99
+ config=task_args if isinstance(task_args, dict) else None,
100
+ log_timestamp=getattr(ev, "created", None))
101
+
93
102
  return build_eval_claim(
94
103
  suite=suite, suite_version=str(getattr(ev, "task_version", "1")),
95
104
  metric=metric, comparator=comparator, threshold=threshold, score=_score_str(value),
@@ -71,6 +71,16 @@ def from_lm_eval_results(path, task: str, metric: str, *, comparator: str, thres
71
71
  if stderr is not None:
72
72
  provenance["stderr"] = repr(stderr) if not isinstance(stderr, str) else stderr
73
73
 
74
+ # v1.8 (external review): config-hash + LOG-NATIVE timestamp. lm-eval has no dedicated run-id;
75
+ # its `date` is a Unix float (distinct from the ISO filename stamp). `config` is the run config
76
+ # block (model/args/seeds/gen_kwargs); no native hash exists, so we compute one.
77
+ from ._provenance import add_provenance # noqa: PLC0415
78
+ add_provenance(provenance, config=cfg if isinstance(cfg, dict) and cfg else None,
79
+ log_timestamp=data.get("date"))
80
+ task_hashes = data.get("task_hashes", {})
81
+ if isinstance(task_hashes, dict) and task_hashes.get(task):
82
+ provenance["task_hash"] = str(task_hashes[task]) # lm-eval's native per-task sample hash
83
+
74
84
  return build_eval_claim(
75
85
  suite=task, suite_version=str(data.get("versions", {}).get(task, "lm-eval")),
76
86
  metric=metric, comparator=comparator, threshold=threshold, score=str(score), n=n,
@@ -124,6 +124,13 @@ def from_promptfoo_results(path, *, comparator: str, threshold: str, timestamp:
124
124
  if summary.get("timestamp"):
125
125
  provenance["run_timestamp"] = str(summary["timestamp"])
126
126
 
127
+ # v1.8 (external review): a uniform run_id key across adapters + a config-hash over the FULL
128
+ # resolved suite config (providers/prompts/tests/…), not just the tests-derived dataset id.
129
+ from ._provenance import add_provenance # noqa: PLC0415
130
+ add_provenance(provenance, run_id=eval_id,
131
+ config=config if isinstance(config, dict) and config else None,
132
+ log_timestamp=metadata.get("evaluationCreatedAt"))
133
+
127
134
  return build_eval_claim(
128
135
  suite=suite, suite_version=f"promptfoo-summary-v{version}",
129
136
  metric="pass_rate", comparator=comparator, threshold=threshold, score=score, n=n,
@@ -229,6 +229,37 @@ def _cmd_demo(args: argparse.Namespace) -> int:
229
229
  return run_demo(as_json=args.json)
230
230
 
231
231
 
232
+ def _cmd_prereg(args: argparse.Namespace) -> int:
233
+ from .prereg import prereg_hash, verify_prereg # noqa: PLC0415
234
+ try:
235
+ if args.check is not None:
236
+ from .evalclaim import decode_eval_claim # noqa: PLC0415
237
+ # Release-review CRITICAL: --check MUST verify the receipt (Ed25519 + Merkle) BEFORE trusting its
238
+ # prereg_sha256 — the old load_bundle+manual-decode read an UNAUTHENTICATED claim, so a forged/unsigned
239
+ # bundle with a doctored prereg_sha256 got a false PASS (the exact anti-cherry-picking bypass this guards).
240
+ claim = decode_eval_claim(args.check)
241
+ if claim is None:
242
+ print("=> FAILED: not a valid, issuer-bound eval receipt", file=sys.stderr)
243
+ return 1
244
+ res = verify_prereg(args.protocol, claim)
245
+ if args.json:
246
+ print(json.dumps(res))
247
+ else:
248
+ print(f"[{'PASS' if res['ok'] else 'FAIL'}] prereg: {res['detail']}")
249
+ return 0 if res["ok"] else 1
250
+ h = prereg_hash(args.protocol)
251
+ if args.json:
252
+ print(json.dumps({"prereg_sha256": h}))
253
+ else:
254
+ print(h)
255
+ print("place this in the eval claim's prereg_sha256 BEFORE running the eval",
256
+ file=sys.stderr)
257
+ return 0
258
+ except (ProofBundleError, OSError, ValueError, KeyError) as exc:
259
+ print(f"ERROR: {exc}", file=sys.stderr)
260
+ return 2
261
+
262
+
232
263
  def build_parser() -> argparse.ArgumentParser:
233
264
  parser = argparse.ArgumentParser(
234
265
  prog="proofbundle",
@@ -317,6 +348,15 @@ def build_parser() -> argparse.ArgumentParser:
317
348
  demo.add_argument("--json", action="store_true", help="machine readable output")
318
349
  demo.set_defaults(func=_cmd_demo)
319
350
 
351
+ prereg = sub.add_parser(
352
+ "prereg",
353
+ help="hash an eval protocol file to commit to it BEFORE the run (--check verifies a receipt)")
354
+ prereg.add_argument("protocol", help="path to the protocol/plan file to hash")
355
+ prereg.add_argument("--check", metavar="RECEIPT",
356
+ help="verify the protocol matches a receipt's prereg_sha256 instead of hashing")
357
+ prereg.add_argument("--json", action="store_true", help="machine readable output")
358
+ prereg.set_defaults(func=_cmd_prereg)
359
+
320
360
  return parser
321
361
 
322
362
 
@@ -274,6 +274,15 @@ def decode_eval_claim(bundle, *, expected_context: Optional[str] = None) -> Opti
274
274
  want = "ed25519:" + base64.b64encode(base64.b64decode(sig_pub_b64)).decode("ascii")
275
275
  if claim.get("issuer") != want:
276
276
  return None
277
+ # Verify-boundary schema invariants (release-review CRITICAL): emit_eval_receipt signs a hand-built claim
278
+ # WITHOUT build_eval_claim's checks, so a signed claim could carry an out-of-enum comparator or a
279
+ # non-decimal/non-finite threshold ("inf") — either silently collapses a downstream verdict check (e.g. the
280
+ # HF value-vs-verdict guard) into a tautology. Enforce them here, fail-closed, so every decoded claim is sane.
281
+ if claim.get("comparator") not in _COMPARATORS:
282
+ return None
283
+ _thr = claim.get("threshold")
284
+ if not (isinstance(_thr, str) and _DECIMAL_RE.match(_thr)):
285
+ return None
277
286
  samples = claim.get("samples")
278
287
  if samples is not None:
279
288
  if not isinstance(samples, dict) or set(samples) != {"root_b64", "n", "leaf_alg"}:
@@ -326,7 +335,7 @@ def check_freshness(claim: dict, max_age_seconds: Optional[int] = None, now=None
326
335
  """Replay check (v1.1): parse the claim's timestamp and report its age. A receipt carries a timestamp but
327
336
  verify never judged it — an old receipt could be replayed as new. Returns
328
337
  {"parsed": bool, "age_seconds": int|None, "fresh": bool|None, "reason": str}. ``fresh`` is None when no
329
- ``max_age_seconds`` bound is given (age reported, not judged). 3.9-safe ISO parsing (normalizes a 'Z')."""
338
+ ``max_age_seconds`` bound is given (age reported, not judged). ISO parsing (normalizes a trailing Z)."""
330
339
  from datetime import datetime, timezone # noqa: PLC0415
331
340
  ts = claim.get("timestamp")
332
341
  if not isinstance(ts, str):
@@ -85,7 +85,8 @@ def to_eval_results_entry(bundle: dict, *, dataset_id: str, task_id: str, value,
85
85
  date: Optional[str] = None, source_url: Optional[str] = None,
86
86
  source_name: Optional[str] = None, source_user: Optional[str] = None,
87
87
  notes: Optional[str] = None, include_token: bool = True,
88
- require_verified: bool = True) -> dict:
88
+ require_verified: bool = True,
89
+ allow_value_mismatch: bool = False) -> dict:
89
90
  """Build one HF `.eval_results/*.yaml` entry for a receipt.
90
91
 
91
92
  ``dataset_id``/``task_id`` name the Hub benchmark (per its `eval.yaml`); ``value`` is the
@@ -96,6 +97,11 @@ def to_eval_results_entry(bundle: dict, *, dataset_id: str, task_id: str, value,
96
97
  is never generated from a broken receipt. ``include_token=True`` puts the ``pb1.`` token in
97
98
  ``verifyToken`` (schema-valid, proofbundle-verifiable; NOT the HF-internal badge token — HF's
98
99
  "verified" badge is HF's server-side decision, and this module makes no claim about it).
100
+
101
+ v1.8 (external review): if the bundle is an eval receipt whose claim discloses a ``score``,
102
+ the published ``value`` MUST match it (a Hub reader sees the value, not the token) — a
103
+ mismatch raises unless ``allow_value_mismatch=True``. This stops a receipt saying 0.60 from
104
+ being published next to a displayed 0.99.
99
105
  """
100
106
  if require_verified:
101
107
  result = verify_bundle(bundle)
@@ -118,6 +124,42 @@ def to_eval_results_entry(bundle: dict, *, dataset_id: str, task_id: str, value,
118
124
  raise BundleFormatError(
119
125
  "value must be a finite number — inf/-inf/nan cannot be represented in eval_results.yaml")
120
126
 
127
+ # v1.8 (external review): if the receipt is an eval claim, the published value must be
128
+ # CONSISTENT with the signed pass/fail verdict — a Hub reader sees the value, not the token,
129
+ # so publishing a value that contradicts the receipt is exactly the honesty gap to close.
130
+ # The claim minimizes data (it carries threshold/comparator/passed, not the exact score), so
131
+ # we check the strongest thing available: value <comparator> threshold must equal passed.
132
+ if not allow_value_mismatch:
133
+ from .evalclaim import EVAL_CLAIM_SCHEMA, decode_eval_claim # noqa: PLC0415
134
+ claim = decode_eval_claim(bundle)
135
+ if claim is not None:
136
+ # decode_eval_claim now guarantees comparator ∈ the 4-value enum and a decimal (finite) threshold, so the
137
+ # lookup below is total and thr is finite — no "=="/"inf" tautology can silently no-op the check.
138
+ if {"threshold", "comparator", "passed"} <= set(claim):
139
+ thr = float(claim["threshold"])
140
+ cmp_ok = {">=": numeric >= thr, ">": numeric > thr,
141
+ "<=": numeric <= thr, "<": numeric < thr}[claim["comparator"]]
142
+ if cmp_ok != bool(claim["passed"]):
143
+ raise BundleFormatError(
144
+ f"published value {numeric} is inconsistent with the receipt: the signed claim "
145
+ f"says passed={claim['passed']} for {claim['comparator']} {claim['threshold']}, "
146
+ f"but {numeric} {claim['comparator']} {claim['threshold']} is {cmp_ok} — "
147
+ "pass allow_value_mismatch=True only if this is intentional")
148
+ else:
149
+ # decode failed. FAIL-CLOSED (release-review CRITICAL) if the payload IS an eval claim but did not decode
150
+ # (e.g. out-of-enum comparator / non-decimal threshold) — refusing to publish an unchecked value. A
151
+ # genuinely NON-eval bundle (different/absent schema) has no verdict to check → skip. The bundle's signature
152
+ # was already verified above (require_verified), so reading the raw payload's schema label is authentic.
153
+ try:
154
+ _raw = json.loads(base64.b64decode(bundle["payload_b64"]).decode("utf-8"))
155
+ _is_eval = isinstance(_raw, dict) and _raw.get("schema") == EVAL_CLAIM_SCHEMA
156
+ except (ValueError, TypeError, KeyError):
157
+ _is_eval = False
158
+ if _is_eval:
159
+ raise BundleFormatError(
160
+ "cannot verify the published value against the receipt — the eval claim did not decode "
161
+ "(invalid comparator/threshold?); pass allow_value_mismatch=True only if intentional")
162
+
121
163
  entry: dict = {"dataset": {"id": dataset_id, "task_id": task_id},
122
164
  "value": numeric if isinstance(value, str) else value}
123
165
  if include_token:
@@ -0,0 +1,58 @@
1
+ """Pre-registration helper (v1.8) — commit to an eval protocol BEFORE the run.
2
+
3
+ The single mitigation for best-of-many / cherry-picking that a receipt can carry: hash the
4
+ protocol document (the plan — suite, seeds, decision rule, sampling policy) *before* running the
5
+ eval, put that hash in the claim's ``prereg_sha256``, and sign the receipt. A verifier who is
6
+ later handed the protocol file re-hashes it and checks it matches — so the plan could not have
7
+ been written to fit the result. The signed receipt's own timestamp binds "this hash existed at
8
+ receipt time" without any network dependency.
9
+
10
+ Construction (verified against standards, 2026-07): commit = **sha256 over the RAW file bytes**.
11
+ Document commitments hash raw bytes (git blob addressing, RFC 6962 leaf hashing, in-toto
12
+ ``gitBlob``/``sha256`` DigestSet all hash the artifact's own bytes) — NOT a re-normalized form.
13
+ Canonicalization would only add a lossy transform the verifier must reproduce byte-for-byte; a
14
+ trailing-newline or CRLF change breaking the match is tamper-evidence, not a bug. The claim field
15
+ is a bare 64-hex ``prereg_sha256`` (matching the eval-claim schema).
16
+
17
+ Out of scope, stated honestly: this proves the protocol was fixed relative to the receipt's
18
+ signing time; it does NOT prove the run actually followed the protocol, nor timestamp the
19
+ commitment against a third party's clock. An optional RFC 3161 TSA countersignature over the
20
+ hash (e.g. FreeTSA) is the upgrade when the verifier does not trust the issuer's clock — that is
21
+ a deployment choice, not built in here.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import hashlib
27
+ from pathlib import Path
28
+
29
+ __all__ = ["prereg_hash", "verify_prereg"]
30
+
31
+
32
+ def prereg_hash(protocol_path) -> str:
33
+ """Return the lowercase-hex sha256 over the RAW bytes of the protocol file — the value to
34
+ place in a claim's ``prereg_sha256`` BEFORE running the eval."""
35
+ data = Path(protocol_path).read_bytes()
36
+ return hashlib.sha256(data).hexdigest()
37
+
38
+
39
+ def verify_prereg(protocol_path, claim: dict) -> dict:
40
+ """Check that ``claim['prereg_sha256']`` matches the sha256 of the protocol file.
41
+
42
+ Returns ``{ok, present, expected, actual, detail}``. ``present`` is False when the claim
43
+ carries no ``prereg_sha256`` (not pre-registered) — the caller decides whether that is
44
+ acceptable; ``ok`` is only True on a present-and-matching hash (fail-closed)."""
45
+ expected = claim.get("prereg_sha256") if isinstance(claim, dict) else None
46
+ result = {"ok": False, "present": expected is not None, "expected": expected,
47
+ "actual": None, "detail": ""}
48
+ if expected is None:
49
+ result["detail"] = "claim carries no prereg_sha256 (not pre-registered)"
50
+ return result
51
+ actual = prereg_hash(protocol_path)
52
+ result["actual"] = actual
53
+ if actual == expected:
54
+ result["ok"] = True
55
+ result["detail"] = "protocol file matches the pre-registered hash"
56
+ else:
57
+ result["detail"] = "protocol file does NOT match the pre-registered hash (plan changed?)"
58
+ return result
@@ -97,7 +97,9 @@ def parse_tlog_proof(text: str) -> dict:
97
97
  if pos >= len(lines) or not lines[pos].startswith("index "):
98
98
  raise BundleFormatError("tlog-proof is missing the index line")
99
99
  index_s = lines[pos][len("index "):]
100
- if not index_s.isdigit() or (index_s != "0" and index_s.startswith("0")):
100
+ # isascii() before isdigit() (release-review #7): str.isdigit() is True for Unicode digits (e.g. '²' U+00B2,
101
+ # Arabic-Indic), which int() would then reject or mis-parse — mirror checkpoint.py's ASCII-only guard.
102
+ if not (index_s.isascii() and index_s.isdigit()) or (index_s != "0" and index_s.startswith("0")):
101
103
  raise BundleFormatError("tlog-proof index must be ASCII decimal with no leading zeros")
102
104
  pos += 1
103
105
  proof = []
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proofbundle
3
- Version: 1.7.0
3
+ Version: 1.8.0
4
4
  Summary: Emit and verify portable cryptographic evidence bundles, offline: Ed25519 + RFC 6962 Merkle + optional SD-JWT.
5
5
  Author: Konrad Gruszka
6
6
  License: MIT
@@ -84,7 +84,7 @@ file, no server, no network.**
84
84
 
85
85
  **At a glance:** `proofbundle emit` signs and anchors a payload; `proofbundle
86
86
  verify` checks one self-contained `bundle.json` with three offline cryptographic
87
- checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 257 tests + a CI mutation gate.
87
+ checks → `OK` or `FAILED`. No network, no daemon, no own crypto. 289 tests + a CI mutation gate.
88
88
 
89
89
  ## Contents
90
90
 
@@ -19,6 +19,7 @@ src/proofbundle/intoto.py
19
19
  src/proofbundle/kbjwt.py
20
20
  src/proofbundle/merkle.py
21
21
  src/proofbundle/persample.py
22
+ src/proofbundle/prereg.py
22
23
  src/proofbundle/py.typed
23
24
  src/proofbundle/pytest_plugin.py
24
25
  src/proofbundle/sdjwt.py
@@ -33,6 +34,7 @@ src/proofbundle.egg-info/entry_points.txt
33
34
  src/proofbundle.egg-info/requires.txt
34
35
  src/proofbundle.egg-info/top_level.txt
35
36
  src/proofbundle/adapters/__init__.py
37
+ src/proofbundle/adapters/_provenance.py
36
38
  src/proofbundle/adapters/eee.py
37
39
  src/proofbundle/adapters/inspect_ai.py
38
40
  src/proofbundle/adapters/lm_eval.py
@@ -53,6 +55,7 @@ tests/test_emit.py
53
55
  tests/test_eval_claim_schema.py
54
56
  tests/test_evalclaim.py
55
57
  tests/test_examples.py
58
+ tests/test_fuzz_parsers.py
56
59
  tests/test_hf_evals.py
57
60
  tests/test_inspect_hook.py
58
61
  tests/test_intoto.py
@@ -61,7 +64,9 @@ tests/test_kbjwt.py
61
64
  tests/test_merkle.py
62
65
  tests/test_merkle_property.py
63
66
  tests/test_persample.py
67
+ tests/test_prereg.py
64
68
  tests/test_promptfoo.py
69
+ tests/test_provenance.py
65
70
  tests/test_pytest_plugin.py
66
71
  tests/test_rekor_interop.py
67
72
  tests/test_rfc6962_external_vectors.py
@@ -39,6 +39,19 @@ class TestEvalClaim(unittest.TestCase):
39
39
  self.assertEqual(decoded["suite"], "safety-refusal")
40
40
  self.assertTrue(decoded["passed"])
41
41
 
42
+ def test_decode_rejects_bad_comparator_and_threshold(self):
43
+ # release-review CRITICAL: emit_eval_receipt signs a hand-built claim WITHOUT build_eval_claim's checks,
44
+ # so decode_eval_claim must enforce comparator-enum + decimal-threshold at the verify boundary — else a
45
+ # downstream value-consistency check silently no-ops on comparator "==" / non-finite threshold "inf".
46
+ signer = generate_signer()
47
+ for key, bad in (("comparator", "=="), ("comparator", "~="),
48
+ ("threshold", "inf"), ("threshold", "nan"), ("threshold", "1e5")):
49
+ claim, _ = _claim(signer)
50
+ claim[key] = bad
51
+ bundle = emit_eval_receipt(claim, signer)
52
+ self.assertTrue(verify_bundle(bundle).ok, f"{key}={bad}: bundle still signs/verifies")
53
+ self.assertIsNone(decode_eval_claim(bundle), f"{key}={bad}: claim must NOT decode")
54
+
42
55
  def test_decode_reads_path_once_no_toctou(self):
43
56
  # CRITICAL (release review): decode_eval_claim(path) must resolve the path to a dict EXACTLY ONCE and
44
57
  # verify + parse the SAME object. A second re-read is a TOCTOU (CWE-367) file-race window that could return
@@ -0,0 +1,88 @@
1
+ """Property-based fuzzing of the text/JWT parsers (v1.8).
2
+
3
+ The invariant for every attacker-controlled parser: on ANY input it returns a value or raises a
4
+ proofbundle error (BundleFormatError / ProofBundleError / ValueError) — NEVER an uncaught crash
5
+ (AttributeError, IndexError, KeyError, TypeError, UnicodeError, recursion, …) and never a hang.
6
+ This is the "never a raw traceback" contract, checked adversarially with Hypothesis rather than
7
+ by hand-picked cases. Hypothesis is the lowest-friction sound fuzzer for pure-Python parsers
8
+ (no native toolchain); an Atheris coverage-guided driver over the same bodies can be added under
9
+ fuzz/ for Linux CI if deeper coverage is ever wanted. hypothesis is a dev dependency only — this
10
+ module no-ops when it is absent (same pattern as test_merkle_property.py)."""
11
+ from __future__ import annotations
12
+
13
+ import unittest
14
+
15
+ try:
16
+ from hypothesis import given, settings
17
+ from hypothesis import strategies as st
18
+ except ImportError: # pragma: no cover - dev-only dependency
19
+ given = None
20
+
21
+ from proofbundle.errors import ProofBundleError
22
+ from proofbundle.tlogproof import parse_tlog_proof, verify_tlog_proof
23
+ from proofbundle.checkpoint import verify_checkpoint, verify_cosignature
24
+ from proofbundle.statuslist import verify_status_snapshot
25
+ from proofbundle.kbjwt import split_key_binding, verify_key_binding
26
+ from proofbundle.sdjwt import verify_sd_jwt
27
+
28
+ _ALLOWED = (ProofBundleError, ValueError) # the documented "malformed input" surface
29
+
30
+
31
+ def _must_not_crash(fn, *args, **kwargs):
32
+ try:
33
+ fn(*args, **kwargs)
34
+ except _ALLOWED:
35
+ pass # documented malformed-input path — fine
36
+ # any other exception propagates and fails the test (the contract violation we hunt)
37
+
38
+
39
+ if given is not None:
40
+ _texts = st.text(alphabet=st.characters(min_codepoint=1, max_codepoint=0x2FFF), max_size=400)
41
+
42
+ class TestParserRobustness(unittest.TestCase):
43
+ @settings(max_examples=300, deadline=None)
44
+ @given(_texts)
45
+ def test_parse_tlog_proof_never_crashes(self, s):
46
+ _must_not_crash(parse_tlog_proof, s)
47
+
48
+ @settings(max_examples=200, deadline=None)
49
+ @given(_texts, _texts)
50
+ def test_verify_tlog_proof_never_crashes(self, proof, leaf):
51
+ _must_not_crash(verify_tlog_proof, proof, leaf.encode("utf-8", "surrogatepass"),
52
+ "log+00000000+" + "A" * 44)
53
+
54
+ @settings(max_examples=300, deadline=None)
55
+ @given(_texts, _texts)
56
+ def test_verify_checkpoint_never_crashes(self, note, vkey):
57
+ _must_not_crash(verify_checkpoint, note, vkey)
58
+
59
+ @settings(max_examples=300, deadline=None)
60
+ @given(_texts, _texts)
61
+ def test_verify_cosignature_never_crashes(self, note, vkey):
62
+ _must_not_crash(verify_cosignature, note, vkey)
63
+
64
+ @settings(max_examples=300, deadline=None)
65
+ @given(_texts)
66
+ def test_split_key_binding_never_crashes(self, compact):
67
+ sd, kb = split_key_binding(compact) # total by contract → returns a tuple
68
+ self.assertIsInstance(sd, str)
69
+
70
+ @settings(max_examples=300, deadline=None)
71
+ @given(_texts)
72
+ def test_verify_key_binding_never_crashes(self, compact):
73
+ _must_not_crash(verify_key_binding, compact)
74
+
75
+ @settings(max_examples=300, deadline=None)
76
+ @given(_texts)
77
+ def test_verify_sd_jwt_never_crashes(self, compact):
78
+ _must_not_crash(verify_sd_jwt, compact)
79
+
80
+ @settings(max_examples=200, deadline=None)
81
+ @given(_texts)
82
+ def test_verify_status_snapshot_never_crashes(self, token):
83
+ _must_not_crash(verify_status_snapshot, token, expected_uri="u", index=0,
84
+ issuer_pubkey=b"\x00" * 32)
85
+
86
+
87
+ if __name__ == "__main__":
88
+ unittest.main()
@@ -150,3 +150,38 @@ class TestEvalResultsEntry(unittest.TestCase):
150
150
 
151
151
  if __name__ == "__main__":
152
152
  unittest.main()
153
+
154
+
155
+ class TestValueConsistency(unittest.TestCase):
156
+ """v1.8: a published value that contradicts the signed pass/fail verdict is refused."""
157
+
158
+ def _receipt(self, threshold, comparator, score):
159
+ from proofbundle import generate_signer
160
+ from proofbundle.evalclaim import build_eval_claim, emit_eval_receipt
161
+ claim, _ = build_eval_claim(
162
+ suite="s", suite_version="1", metric="acc", comparator=comparator,
163
+ threshold=threshold, score=score, n=10, model_id="m", dataset_id="d",
164
+ issuer="", timestamp="2026-07-02T00:00:00Z")
165
+ return emit_eval_receipt(claim, generate_signer())
166
+
167
+ def test_consistent_value_ok(self):
168
+ r = self._receipt("0.80", ">=", "0.91") # passed=True
169
+ entry = to_eval_results_entry(r, dataset_id="d/x", task_id="t", value=0.91)
170
+ self.assertEqual(entry["value"], 0.91)
171
+
172
+ def test_inconsistent_value_refused(self):
173
+ r = self._receipt("0.80", ">=", "0.60") # passed=False (0.60 < 0.80)
174
+ with self.assertRaises(BundleFormatError) as ctx:
175
+ to_eval_results_entry(r, dataset_id="d/x", task_id="t", value=0.99) # 0.99>=0.80 → True, but claim says False
176
+ self.assertIn("inconsistent", str(ctx.exception))
177
+
178
+ def test_override_allows_mismatch(self):
179
+ r = self._receipt("0.80", ">=", "0.60")
180
+ entry = to_eval_results_entry(r, dataset_id="d/x", task_id="t", value=0.99,
181
+ allow_value_mismatch=True)
182
+ self.assertEqual(entry["value"], 0.99)
183
+
184
+ def test_non_eval_bundle_skips_check(self):
185
+ # a plain emit_bundle (not an eval receipt) has no claim → no cross-check, value accepted
186
+ entry = to_eval_results_entry(_bundle(), dataset_id="d/x", task_id="t", value=0.5)
187
+ self.assertEqual(entry["value"], 0.5)
@@ -0,0 +1,122 @@
1
+ """v1.8 pre-registration helper: commit to a protocol before the run, verify after."""
2
+ import hashlib
3
+ import json
4
+ import os
5
+ import tempfile
6
+ import unittest
7
+
8
+ from proofbundle.cli import main
9
+ from proofbundle.emit import generate_signer
10
+ from proofbundle.evalclaim import build_eval_claim, emit_eval_receipt, issuer_fingerprint
11
+ from proofbundle.prereg import prereg_hash, verify_prereg
12
+
13
+
14
+ class TestPrereg(unittest.TestCase):
15
+ def _file(self, content: bytes) -> str:
16
+ h = tempfile.NamedTemporaryFile("wb", delete=False)
17
+ h.write(content)
18
+ h.close()
19
+ return h.name
20
+
21
+ def test_hash_is_sha256_of_raw_bytes(self):
22
+ content = b"protocol: fixed seeds 1..5\ndecision: acc >= 0.8\n"
23
+ path = self._file(content)
24
+ try:
25
+ self.assertEqual(prereg_hash(path), hashlib.sha256(content).hexdigest())
26
+ finally:
27
+ os.unlink(path)
28
+
29
+ def test_verify_match(self):
30
+ content = b"the plan"
31
+ path = self._file(content)
32
+ try:
33
+ claim = {"prereg_sha256": hashlib.sha256(content).hexdigest()}
34
+ res = verify_prereg(path, claim)
35
+ self.assertTrue(res["ok"])
36
+ self.assertTrue(res["present"])
37
+ finally:
38
+ os.unlink(path)
39
+
40
+ def test_verify_mismatch_is_caught(self):
41
+ path = self._file(b"the ACTUAL plan")
42
+ try:
43
+ claim = {"prereg_sha256": hashlib.sha256(b"a DIFFERENT plan").hexdigest()}
44
+ res = verify_prereg(path, claim)
45
+ self.assertFalse(res["ok"])
46
+ self.assertIn("does NOT match", res["detail"])
47
+ finally:
48
+ os.unlink(path)
49
+
50
+ def test_not_preregistered_reports_absent(self):
51
+ path = self._file(b"x")
52
+ try:
53
+ res = verify_prereg(path, {}) # no prereg_sha256
54
+ self.assertFalse(res["ok"])
55
+ self.assertFalse(res["present"])
56
+ finally:
57
+ os.unlink(path)
58
+
59
+ def test_trailing_byte_change_breaks_match(self):
60
+ # tamper-evidence: a single appended newline changes the commitment (by design).
61
+ path = self._file(b"plan\n")
62
+ try:
63
+ claim = {"prereg_sha256": hashlib.sha256(b"plan").hexdigest()} # committed without \n
64
+ self.assertFalse(verify_prereg(path, claim)["ok"])
65
+ finally:
66
+ os.unlink(path)
67
+
68
+ def test_cli_roundtrip(self):
69
+ import json
70
+ from proofbundle.cli import main
71
+ from proofbundle import generate_signer
72
+ from proofbundle.evalclaim import build_eval_claim, emit_eval_receipt
73
+ import contextlib
74
+ import io
75
+ proto = self._file(b"suite=mmlu; seeds=1..5; rule=acc>=0.8")
76
+ try:
77
+ h = prereg_hash(proto)
78
+ claim, _ = build_eval_claim(
79
+ suite="s", suite_version="1", metric="acc", comparator=">=", threshold="0.8",
80
+ score="0.9", n=10, model_id="m", dataset_id="d", issuer="",
81
+ timestamp="2026-07-02T00:00:00Z", prereg_sha256=h)
82
+ receipt = emit_eval_receipt(claim, generate_signer())
83
+ rp = tempfile.NamedTemporaryFile("w", suffix=".json", delete=False)
84
+ json.dump(receipt, rp)
85
+ rp.close()
86
+ with contextlib.redirect_stdout(io.StringIO()):
87
+ self.assertEqual(main(["prereg", proto, "--check", rp.name]), 0)
88
+ # a different protocol fails
89
+ other = self._file(b"a different plan")
90
+ self.assertEqual(main(["prereg", other, "--check", rp.name]), 1)
91
+ os.unlink(rp.name)
92
+ os.unlink(other)
93
+ finally:
94
+ os.unlink(proto)
95
+
96
+
97
+ class TestPreregCheckVerifies(unittest.TestCase):
98
+ def test_check_rejects_forged_unsigned_bundle(self):
99
+ # release-review CRITICAL: prereg --check MUST verify the receipt's signature before trusting its
100
+ # prereg_sha256 — a forged/unsigned bundle with a doctored prereg_sha256 must NOT get a PASS.
101
+ with tempfile.TemporaryDirectory() as d:
102
+ proto = os.path.join(d, "protocol.md")
103
+ with open(proto, "wb") as f:
104
+ f.write(b"pre-registered analysis plan v1")
105
+ h = prereg_hash(proto)
106
+ signer = generate_signer()
107
+ claim, _ = build_eval_claim(
108
+ suite="s", suite_version="v1", metric="m", comparator=">=", threshold="0.80",
109
+ score="0.92", n=10, model_id="a/b", dataset_id="c/d",
110
+ issuer=issuer_fingerprint(signer), timestamp="2026-07-01T12:00:00Z",
111
+ model_salt=b"0" * 16, dataset_salt=b"1" * 16, prereg_sha256=h)
112
+ bundle = emit_eval_receipt(claim, signer)
113
+ bundle["signature"]["signature_b64"] = "AA==" * 22 # corrupt the Ed25519 signature
114
+ fp = os.path.join(d, "forged.json")
115
+ with open(fp, "w", encoding="utf-8") as f:
116
+ json.dump(bundle, f)
117
+ rc = main(["prereg", proto, "--check", fp])
118
+ self.assertNotEqual(rc, 0, "a forged/unsigned receipt must FAIL prereg --check")
119
+
120
+
121
+ if __name__ == "__main__":
122
+ unittest.main()
@@ -0,0 +1,77 @@
1
+ """v1.8 provenance hardening: config-hash + run-id + log-native timestamp in adapters."""
2
+ import json
3
+ import os
4
+ import tempfile
5
+ import unittest
6
+ from pathlib import Path
7
+
8
+ from proofbundle.adapters._provenance import add_provenance, config_hash
9
+ from proofbundle.adapters import from_lm_eval_results, from_promptfoo_results
10
+
11
+ FIXTURES = Path(__file__).parent / "fixtures"
12
+
13
+
14
+ class TestConfigHash(unittest.TestCase):
15
+ def test_deterministic_and_labeled(self):
16
+ a = config_hash({"b": 1, "a": 2})
17
+ b = config_hash({"a": 2, "b": 1}) # key order must not matter
18
+ self.assertEqual(a, b)
19
+ self.assertTrue(a.startswith("sha256-jcs:") or a.startswith("sha256-sortkeys:"))
20
+ self.assertEqual(len(a.split(":")[1]), 64)
21
+
22
+ def test_empty_is_none(self):
23
+ self.assertIsNone(config_hash(None))
24
+ self.assertIsNone(config_hash({}))
25
+ self.assertIsNone(config_hash([]))
26
+
27
+ def test_change_changes_hash(self):
28
+ self.assertNotEqual(config_hash({"seed": 1}), config_hash({"seed": 2}))
29
+
30
+ def test_add_provenance_skips_absent(self):
31
+ prov = {}
32
+ add_provenance(prov, run_id=None, config=None, log_timestamp=None)
33
+ self.assertEqual(prov, {})
34
+ add_provenance(prov, run_id="r1", config={"x": 1}, log_timestamp=123)
35
+ self.assertEqual(prov["run_id"], "r1")
36
+ self.assertEqual(prov["run_timestamp"], "123")
37
+ self.assertIn("config_hash", prov)
38
+
39
+
40
+ class TestPromptfooProvenance(unittest.TestCase):
41
+ def test_run_id_and_config_hash_present(self):
42
+ claim, _ = from_promptfoo_results(
43
+ FIXTURES / "promptfoo_results_v3.json", comparator=">=", threshold="0.5",
44
+ timestamp="2026-07-02T00:00:00Z")
45
+ prov = claim["provenance"]
46
+ self.assertEqual(prov["run_id"], "eval-Xa3-2026-07-02T14:03:11")
47
+ self.assertIn("config_hash", prov)
48
+ self.assertIn("run_timestamp", prov)
49
+
50
+
51
+ class TestLmEvalProvenance(unittest.TestCase):
52
+ def _fixture_with_config(self):
53
+ data = json.loads((FIXTURES / "lm_eval_arc_easy_real.json").read_text())
54
+ return data
55
+
56
+ def test_config_hash_and_timestamp(self):
57
+ data = self._fixture_with_config()
58
+ data.setdefault("config", {"model": "hf", "seed": 1234})
59
+ data["date"] = 1780000000.5
60
+ handle = tempfile.NamedTemporaryFile("w", suffix=".json", delete=False)
61
+ json.dump(data, handle)
62
+ handle.close()
63
+ try:
64
+ task = next(iter(data["results"]))
65
+ metric = next(k.split(",")[0] for k in data["results"][task] if "," in k)
66
+ claim, _ = from_lm_eval_results(handle.name, task=task, metric=metric,
67
+ comparator=">=", threshold="0.1",
68
+ timestamp="2026-07-02T00:00:00Z")
69
+ finally:
70
+ os.unlink(handle.name)
71
+ prov = claim["provenance"]
72
+ self.assertIn("config_hash", prov)
73
+ self.assertEqual(prov["run_timestamp"], "1780000000.5") # log-native, not caller ts
74
+
75
+
76
+ if __name__ == "__main__":
77
+ unittest.main()
File without changes
File without changes