falsification-ledger 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- falsification_ledger/__init__.py +40 -0
- falsification_ledger/__main__.py +4 -0
- falsification_ledger/cli.py +146 -0
- falsification_ledger/contracts.py +107 -0
- falsification_ledger/ledger.py +363 -0
- falsification_ledger-0.1.0.dist-info/METADATA +208 -0
- falsification_ledger-0.1.0.dist-info/RECORD +11 -0
- falsification_ledger-0.1.0.dist-info/WHEEL +5 -0
- falsification_ledger-0.1.0.dist-info/entry_points.txt +3 -0
- falsification_ledger-0.1.0.dist-info/licenses/LICENSE +21 -0
- falsification_ledger-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""falsification-ledger: pre-registration and falsification ledger.
|
|
2
|
+
|
|
3
|
+
A hash-chained, append-only JSONL ledger for research claims: register the
|
|
4
|
+
hypothesis and the evidence that would kill it *before* running the study,
|
|
5
|
+
submit evidence, adjudicate, and report hit-rate with a Wilson confidence
|
|
6
|
+
interval against a random baseline. Every event is content-addressed and
|
|
7
|
+
chain-verified; nothing here trades, prices, or decides.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .contracts import (
|
|
11
|
+
evidence_status,
|
|
12
|
+
falsification_object_id,
|
|
13
|
+
load_falsification_schema,
|
|
14
|
+
validate_falsification_report,
|
|
15
|
+
)
|
|
16
|
+
from .ledger import (
|
|
17
|
+
conclude_prediction,
|
|
18
|
+
ensure_prediction_registered,
|
|
19
|
+
has_prediction_registered,
|
|
20
|
+
ledger_path,
|
|
21
|
+
register_prediction,
|
|
22
|
+
report_prediction_hitrate,
|
|
23
|
+
verify_chain,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
__version__ = "0.1.0"
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
"conclude_prediction",
|
|
30
|
+
"ensure_prediction_registered",
|
|
31
|
+
"evidence_status",
|
|
32
|
+
"falsification_object_id",
|
|
33
|
+
"has_prediction_registered",
|
|
34
|
+
"ledger_path",
|
|
35
|
+
"load_falsification_schema",
|
|
36
|
+
"register_prediction",
|
|
37
|
+
"report_prediction_hitrate",
|
|
38
|
+
"validate_falsification_report",
|
|
39
|
+
"verify_chain",
|
|
40
|
+
]
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Command-line interface for falsification-ledger.
|
|
2
|
+
|
|
3
|
+
Subcommands:
|
|
4
|
+
|
|
5
|
+
- ``init`` create the ledger state directory
|
|
6
|
+
- ``preregister`` register a claim (verdict + reason) before evidence;
|
|
7
|
+
optionally attach a falsification contract JSON
|
|
8
|
+
- ``submit`` validate a falsification report and print its
|
|
9
|
+
content ID and evidence status (read-only)
|
|
10
|
+
- ``adjudicate`` backfill the actual verdict for a case
|
|
11
|
+
- ``report`` hit-rate report (Wilson 95% CI vs random baseline)
|
|
12
|
+
- ``verify`` hash-chain integrity check of the whole ledger
|
|
13
|
+
- ``version`` print version
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import argparse
|
|
19
|
+
import json
|
|
20
|
+
import sys
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any
|
|
23
|
+
|
|
24
|
+
from . import __version__
|
|
25
|
+
from .contracts import evidence_status, falsification_object_id, validate_falsification_report
|
|
26
|
+
from .ledger import (
|
|
27
|
+
conclude_prediction,
|
|
28
|
+
ledger_path,
|
|
29
|
+
register_prediction,
|
|
30
|
+
report_prediction_hitrate,
|
|
31
|
+
verify_chain,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _print_json(body: dict[str, Any]) -> None:
|
|
36
|
+
print(json.dumps(body, ensure_ascii=False, indent=2))
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _load_json_file(path: str) -> dict[str, Any]:
|
|
40
|
+
value = json.loads(Path(path).read_text(encoding="utf-8"))
|
|
41
|
+
if not isinstance(value, dict):
|
|
42
|
+
raise ValueError(f"expected a JSON object: {path}")
|
|
43
|
+
return value
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
47
|
+
parser = argparse.ArgumentParser(
|
|
48
|
+
prog="fl",
|
|
49
|
+
description="Pre-registration and falsification ledger for research.",
|
|
50
|
+
)
|
|
51
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
52
|
+
|
|
53
|
+
init = sub.add_parser("init", help="create the ledger state directory")
|
|
54
|
+
init.add_argument("--state-dir", required=True, help="ledger state directory")
|
|
55
|
+
|
|
56
|
+
prereg = sub.add_parser("preregister", help="register a claim before evidence")
|
|
57
|
+
prereg.add_argument("--state-dir", required=True)
|
|
58
|
+
prereg.add_argument("--case-id", required=True)
|
|
59
|
+
prereg.add_argument(
|
|
60
|
+
"--verdict", required=True, choices=["support", "against", "uncertain"]
|
|
61
|
+
)
|
|
62
|
+
prereg.add_argument("--reason", required=True)
|
|
63
|
+
prereg.add_argument("--source-type", default="other",
|
|
64
|
+
choices=["paper", "business", "cross_domain", "pipeline", "other"])
|
|
65
|
+
prereg.add_argument("--contract", default=None,
|
|
66
|
+
help="falsification contract JSON: what evidence would kill the claim")
|
|
67
|
+
|
|
68
|
+
submit = sub.add_parser("submit", help="validate a falsification report (read-only)")
|
|
69
|
+
submit.add_argument("--report", required=True, help="falsification report JSON path")
|
|
70
|
+
submit.add_argument("--schema", default=None, help="override schema JSON path")
|
|
71
|
+
|
|
72
|
+
adjudicate = sub.add_parser("adjudicate", help="backfill the actual verdict")
|
|
73
|
+
adjudicate.add_argument("--state-dir", required=True)
|
|
74
|
+
adjudicate.add_argument("--case-id", required=True)
|
|
75
|
+
adjudicate.add_argument(
|
|
76
|
+
"--verdict", required=True, choices=["support", "against", "uncertain"]
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
report = sub.add_parser("report", help="hit-rate report")
|
|
80
|
+
report.add_argument("--state-dir", required=True)
|
|
81
|
+
report.add_argument("--min-cases", type=int, default=20)
|
|
82
|
+
|
|
83
|
+
verify = sub.add_parser("verify", help="hash-chain integrity check")
|
|
84
|
+
verify.add_argument("--state-dir", required=True)
|
|
85
|
+
|
|
86
|
+
sub.add_parser("version", help="print version")
|
|
87
|
+
return parser
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def main(argv: list[str] | None = None) -> int:
|
|
91
|
+
parser = build_parser()
|
|
92
|
+
args = parser.parse_args(argv)
|
|
93
|
+
|
|
94
|
+
if args.command == "version":
|
|
95
|
+
print(__version__)
|
|
96
|
+
return 0
|
|
97
|
+
|
|
98
|
+
if args.command == "init":
|
|
99
|
+
path = ledger_path(args.state_dir)
|
|
100
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
101
|
+
print(f"init: ledger ready at {path}")
|
|
102
|
+
return 0
|
|
103
|
+
|
|
104
|
+
if args.command == "preregister":
|
|
105
|
+
contract = _load_json_file(args.contract) if args.contract else None
|
|
106
|
+
result = register_prediction(
|
|
107
|
+
args.state_dir,
|
|
108
|
+
args.case_id,
|
|
109
|
+
args.verdict,
|
|
110
|
+
args.reason,
|
|
111
|
+
args.source_type,
|
|
112
|
+
falsification_contract=contract,
|
|
113
|
+
)
|
|
114
|
+
_print_json(result)
|
|
115
|
+
return 0
|
|
116
|
+
|
|
117
|
+
if args.command == "submit":
|
|
118
|
+
report = _load_json_file(args.report)
|
|
119
|
+
blockers = validate_falsification_report(report, args.schema)
|
|
120
|
+
body = {
|
|
121
|
+
"content_id": falsification_object_id(report),
|
|
122
|
+
"evidence_status": evidence_status(report, args.schema),
|
|
123
|
+
"blockers": blockers,
|
|
124
|
+
}
|
|
125
|
+
_print_json(body)
|
|
126
|
+
return 0 if not blockers else 1
|
|
127
|
+
|
|
128
|
+
if args.command == "adjudicate":
|
|
129
|
+
_print_json(conclude_prediction(args.state_dir, args.case_id, args.verdict))
|
|
130
|
+
return 0
|
|
131
|
+
|
|
132
|
+
if args.command == "report":
|
|
133
|
+
_print_json(report_prediction_hitrate(args.state_dir, min_cases=args.min_cases))
|
|
134
|
+
return 0
|
|
135
|
+
|
|
136
|
+
if args.command == "verify":
|
|
137
|
+
body = verify_chain(args.state_dir)
|
|
138
|
+
_print_json(body)
|
|
139
|
+
return 0 if body.get("ok") else 1
|
|
140
|
+
|
|
141
|
+
parser.error(f"unknown command: {args.command}")
|
|
142
|
+
return 2
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
if __name__ == "__main__":
|
|
146
|
+
sys.exit(main())
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Falsification report contract: schema validation, content IDs, evidence status.
|
|
2
|
+
|
|
3
|
+
A falsification report is the machine-readable evidence that a pre-registered
|
|
4
|
+
claim survived (or did not survive) an independent check. The contract:
|
|
5
|
+
|
|
6
|
+
- validates reports against ``schema/falsification-report.schema.json``
|
|
7
|
+
(fail-closed: blockers are returned, never silently tolerated),
|
|
8
|
+
- computes a content ID (``sha256:<digest>``) over the canonical bytes so
|
|
9
|
+
reports are addressable and tamper-evident,
|
|
10
|
+
- classifies evidence status for gates: ``valid`` / ``invalid`` /
|
|
11
|
+
``missing`` (a falsified or inconsistent report counts as *missing*
|
|
12
|
+
evidence — absence of proof is not proof of absence, but it is not
|
|
13
|
+
support either).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import hashlib
|
|
19
|
+
import json
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
import jsonschema
|
|
24
|
+
|
|
25
|
+
SCHEMA_FILE = "falsification-report.schema.json"
|
|
26
|
+
SCHEMA_VERSION = "falsification_ledger.falsification_report.v1"
|
|
27
|
+
DOMAIN_PREFIX = "falsification-ledger/falsification-report.v1"
|
|
28
|
+
CONSISTENCY_TOLERANCE = 0.005
|
|
29
|
+
|
|
30
|
+
DEFAULT_SCHEMA_PATH = Path(__file__).resolve().parents[2] / "schema" / SCHEMA_FILE
|
|
31
|
+
|
|
32
|
+
_schema_cache: dict[str, Any] | None = None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def load_falsification_schema(schema_path: Path | str | None = None) -> dict[str, Any]:
|
|
36
|
+
"""Load and sanity-check the falsification report schema (cached)."""
|
|
37
|
+
global _schema_cache
|
|
38
|
+
if _schema_cache is None or schema_path is not None:
|
|
39
|
+
path = Path(schema_path) if schema_path is not None else DEFAULT_SCHEMA_PATH
|
|
40
|
+
schema = json.loads(path.read_text(encoding="utf-8"))
|
|
41
|
+
jsonschema.Draft202012Validator.check_schema(schema)
|
|
42
|
+
_schema_cache = schema
|
|
43
|
+
return _schema_cache
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def falsification_object_id(value: dict[str, Any]) -> str:
|
|
47
|
+
"""Content ID: sha256(domain_prefix || 0x00 || canonical bytes)."""
|
|
48
|
+
payload = json.dumps(
|
|
49
|
+
value,
|
|
50
|
+
ensure_ascii=False,
|
|
51
|
+
sort_keys=True,
|
|
52
|
+
separators=(",", ":"),
|
|
53
|
+
allow_nan=False,
|
|
54
|
+
).encode("utf-8")
|
|
55
|
+
digest = hashlib.sha256(
|
|
56
|
+
DOMAIN_PREFIX.encode("utf-8") + b"\x00" + payload
|
|
57
|
+
).hexdigest()
|
|
58
|
+
return f"sha256:{digest}"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def validate_falsification_report(
|
|
62
|
+
value: dict[str, Any],
|
|
63
|
+
schema_path: Path | str | None = None,
|
|
64
|
+
) -> list[str]:
|
|
65
|
+
"""Return blockers; an empty list means the report conforms to the contract."""
|
|
66
|
+
schema = load_falsification_schema(schema_path)
|
|
67
|
+
validator = jsonschema.Draft202012Validator(schema)
|
|
68
|
+
blockers: list[str] = []
|
|
69
|
+
for error in sorted(validator.iter_errors(value), key=lambda e: list(e.path)):
|
|
70
|
+
path = "/".join(str(p) for p in error.path) or "$"
|
|
71
|
+
blockers.append(f"{path}: {error.message}")
|
|
72
|
+
if not blockers and value.get("schema_version") != SCHEMA_VERSION:
|
|
73
|
+
blockers.append(f"schema_version mismatch: {value.get('schema_version')!r}")
|
|
74
|
+
if not blockers:
|
|
75
|
+
consistency = value.get("consistency")
|
|
76
|
+
if consistency is not None:
|
|
77
|
+
if consistency.get("tolerance") != CONSISTENCY_TOLERANCE:
|
|
78
|
+
blockers.append(
|
|
79
|
+
f"consistency.tolerance must be {CONSISTENCY_TOLERANCE}, "
|
|
80
|
+
f"got {consistency.get('tolerance')!r}"
|
|
81
|
+
)
|
|
82
|
+
return blockers
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def evidence_status(
|
|
86
|
+
value: dict[str, Any],
|
|
87
|
+
schema_path: Path | str | None = None,
|
|
88
|
+
) -> str:
|
|
89
|
+
"""Evidence status for gates (fail-closed).
|
|
90
|
+
|
|
91
|
+
- ``invalid`` — report does not conform to the contract, or the
|
|
92
|
+
conclusion is falsified, or consistency is explicitly broken.
|
|
93
|
+
- ``missing`` — inconclusive: treated as absent evidence.
|
|
94
|
+
- ``valid`` — conformant, not falsified, consistent.
|
|
95
|
+
"""
|
|
96
|
+
blockers = validate_falsification_report(value, schema_path)
|
|
97
|
+
if blockers:
|
|
98
|
+
return "invalid"
|
|
99
|
+
conclusion = value.get("conclusion")
|
|
100
|
+
if conclusion == "falsified":
|
|
101
|
+
return "invalid"
|
|
102
|
+
if conclusion == "inconclusive":
|
|
103
|
+
return "missing"
|
|
104
|
+
consistency = value.get("consistency")
|
|
105
|
+
if consistency is not None and consistency.get("consistent") is False:
|
|
106
|
+
return "invalid"
|
|
107
|
+
return "valid"
|
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
"""Append-only, hash-chained prediction ledger core.
|
|
2
|
+
|
|
3
|
+
The ledger answers one question honestly: *did your pre-registered beliefs
|
|
4
|
+
hit?* Events are appended to a JSONL file; every event carries the hash of
|
|
5
|
+
the previous event, so the file is a hash chain and any edit is detectable
|
|
6
|
+
by ``verify_chain``.
|
|
7
|
+
|
|
8
|
+
Event kinds:
|
|
9
|
+
|
|
10
|
+
- ``register`` — a claim about a case: expected verdict + reason, recorded
|
|
11
|
+
before evidence is seen (one per case, duplicate rejected).
|
|
12
|
+
- ``conclude`` — the actual verdict, backfilled at study end (requires a
|
|
13
|
+
prior register; once per case).
|
|
14
|
+
|
|
15
|
+
Reports aggregate resolved cases into a hit rate with a Wilson 95% CI and
|
|
16
|
+
compare it against the random baseline (most common actual verdict): if the
|
|
17
|
+
baseline is inside the CI, there is no systematic signal.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import datetime
|
|
23
|
+
import hashlib
|
|
24
|
+
import json
|
|
25
|
+
import math
|
|
26
|
+
import uuid
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from typing import Any
|
|
29
|
+
|
|
30
|
+
SCHEMA_VERSION = "falsification_ledger.prediction_event.v1"
|
|
31
|
+
REPORT_SCHEMA_VERSION = "falsification_ledger.prediction_report.v1"
|
|
32
|
+
VERDICTS = ("support", "against", "uncertain")
|
|
33
|
+
SOURCE_TYPES = ("paper", "business", "cross_domain", "pipeline", "other")
|
|
34
|
+
DEFAULT_MIN_CASES = 20
|
|
35
|
+
SAFETY = {
|
|
36
|
+
"production_effect": False,
|
|
37
|
+
"changes_probability": False,
|
|
38
|
+
"allow_real_trade": False,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _now_text() -> str:
|
|
43
|
+
return datetime.datetime.now().astimezone().isoformat(timespec="seconds")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _canonical_bytes(value: dict[str, Any]) -> bytes:
|
|
47
|
+
return json.dumps(
|
|
48
|
+
value, ensure_ascii=False, sort_keys=True, separators=(",", ":"), allow_nan=False
|
|
49
|
+
).encode("utf-8")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _event_hash(prev_hash: str | None, payload: dict[str, Any]) -> str:
|
|
53
|
+
"""sha256(prev_hash || 0x00 || canonical payload)."""
|
|
54
|
+
body = (prev_hash or "").encode("utf-8") + b"\x00" + _canonical_bytes(payload)
|
|
55
|
+
return hashlib.sha256(body).hexdigest()
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def ledger_path(state_dir: Path | str) -> Path:
|
|
59
|
+
return Path(state_dir) / "ledger.jsonl"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _load_events(path: Path) -> list[dict[str, Any]]:
|
|
63
|
+
if not path.exists():
|
|
64
|
+
return []
|
|
65
|
+
events: list[dict[str, Any]] = []
|
|
66
|
+
for line_no, line in enumerate(
|
|
67
|
+
path.read_text(encoding="utf-8").splitlines(), start=1
|
|
68
|
+
):
|
|
69
|
+
line = line.strip()
|
|
70
|
+
if not line:
|
|
71
|
+
continue
|
|
72
|
+
try:
|
|
73
|
+
events.append(json.loads(line))
|
|
74
|
+
except json.JSONDecodeError as exc:
|
|
75
|
+
raise ValueError(f"ledger_corrupt:{path}:{line_no}") from exc
|
|
76
|
+
return events
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _append_event(path: Path, event: dict[str, Any]) -> None:
|
|
80
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
81
|
+
with path.open("a", encoding="utf-8") as handle:
|
|
82
|
+
handle.write(json.dumps(event, ensure_ascii=False) + "\n")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _last_hash(events: list[dict[str, Any]]) -> str | None:
|
|
86
|
+
return events[-1].get("event_hash") if events else None
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def register_prediction(
|
|
90
|
+
state_dir: Path | str,
|
|
91
|
+
case_id: str,
|
|
92
|
+
expected_verdict: str,
|
|
93
|
+
expected_reason: str,
|
|
94
|
+
source_type: str = "other",
|
|
95
|
+
*,
|
|
96
|
+
falsification_contract: dict[str, Any] | None = None,
|
|
97
|
+
) -> dict[str, Any]:
|
|
98
|
+
"""Register a claim about a case (once per case; duplicates rejected).
|
|
99
|
+
|
|
100
|
+
``falsification_contract`` is optional free-form JSON describing what
|
|
101
|
+
evidence would kill the claim — written down *before* the study runs.
|
|
102
|
+
"""
|
|
103
|
+
if expected_verdict not in VERDICTS:
|
|
104
|
+
raise ValueError(f"invalid_verdict:{expected_verdict}")
|
|
105
|
+
if source_type not in SOURCE_TYPES:
|
|
106
|
+
raise ValueError(f"invalid_source_type:{source_type}")
|
|
107
|
+
if not expected_reason.strip():
|
|
108
|
+
raise ValueError("expected_reason_required")
|
|
109
|
+
path = ledger_path(state_dir)
|
|
110
|
+
events = _load_events(path)
|
|
111
|
+
if any(
|
|
112
|
+
e.get("case_id") == case_id and e.get("event") == "register"
|
|
113
|
+
for e in events
|
|
114
|
+
):
|
|
115
|
+
raise ValueError(f"duplicate_register:{case_id}")
|
|
116
|
+
payload = {
|
|
117
|
+
"schema_version": SCHEMA_VERSION,
|
|
118
|
+
"event": "register",
|
|
119
|
+
"record_id": str(uuid.uuid4()),
|
|
120
|
+
"case_id": case_id,
|
|
121
|
+
"expected_verdict": expected_verdict,
|
|
122
|
+
"expected_reason": expected_reason.strip(),
|
|
123
|
+
"source_type": source_type,
|
|
124
|
+
"falsification_contract": falsification_contract,
|
|
125
|
+
"actual_verdict": None,
|
|
126
|
+
"recorded_at": _now_text(),
|
|
127
|
+
"concluded_at": None,
|
|
128
|
+
}
|
|
129
|
+
event = {**payload, "prev_hash": _last_hash(events)}
|
|
130
|
+
event["event_hash"] = _event_hash(event["prev_hash"], payload)
|
|
131
|
+
_append_event(path, event)
|
|
132
|
+
return {"event": "register", "case_id": case_id, "recorded_at": event["recorded_at"]}
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def has_prediction_registered(state_dir: Path | str, case_id: str) -> bool:
|
|
136
|
+
"""Whether the case already has a registration (required before conclude)."""
|
|
137
|
+
path = ledger_path(state_dir)
|
|
138
|
+
return any(
|
|
139
|
+
e.get("case_id") == case_id and e.get("event") == "register"
|
|
140
|
+
for e in _load_events(path)
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def ensure_prediction_registered(
|
|
145
|
+
state_dir: Path | str,
|
|
146
|
+
case_id: str,
|
|
147
|
+
expected_verdict: str,
|
|
148
|
+
expected_reason: str,
|
|
149
|
+
source_type: str = "other",
|
|
150
|
+
) -> dict[str, Any]:
|
|
151
|
+
"""Idempotent register: skip when already registered (for automation)."""
|
|
152
|
+
if has_prediction_registered(state_dir, case_id):
|
|
153
|
+
return {"event": "register", "case_id": case_id, "already_registered": True}
|
|
154
|
+
return {
|
|
155
|
+
**register_prediction(
|
|
156
|
+
state_dir, case_id, expected_verdict, expected_reason, source_type
|
|
157
|
+
),
|
|
158
|
+
"already_registered": False,
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def conclude_prediction(
|
|
163
|
+
state_dir: Path | str,
|
|
164
|
+
case_id: str,
|
|
165
|
+
actual_verdict: str,
|
|
166
|
+
) -> dict[str, Any]:
|
|
167
|
+
"""Backfill the actual verdict (register required; once per case)."""
|
|
168
|
+
if actual_verdict not in VERDICTS:
|
|
169
|
+
raise ValueError(f"invalid_verdict:{actual_verdict}")
|
|
170
|
+
path = ledger_path(state_dir)
|
|
171
|
+
events = _load_events(path)
|
|
172
|
+
if not any(
|
|
173
|
+
e.get("case_id") == case_id and e.get("event") == "register"
|
|
174
|
+
for e in events
|
|
175
|
+
):
|
|
176
|
+
raise ValueError(f"register_required:{case_id}")
|
|
177
|
+
if any(
|
|
178
|
+
e.get("case_id") == case_id and e.get("event") == "conclude"
|
|
179
|
+
for e in events
|
|
180
|
+
):
|
|
181
|
+
raise ValueError(f"already_concluded:{case_id}")
|
|
182
|
+
payload = {
|
|
183
|
+
"schema_version": SCHEMA_VERSION,
|
|
184
|
+
"event": "conclude",
|
|
185
|
+
"record_id": str(uuid.uuid4()),
|
|
186
|
+
"case_id": case_id,
|
|
187
|
+
"actual_verdict": actual_verdict,
|
|
188
|
+
"recorded_at": _now_text(),
|
|
189
|
+
"concluded_at": _now_text(),
|
|
190
|
+
}
|
|
191
|
+
event = {**payload, "prev_hash": _last_hash(events)}
|
|
192
|
+
event["event_hash"] = _event_hash(event["prev_hash"], payload)
|
|
193
|
+
_append_event(path, event)
|
|
194
|
+
return {"event": "conclude", "case_id": case_id, "concluded_at": event["concluded_at"]}
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def verify_chain(state_dir: Path | str) -> dict[str, Any]:
|
|
198
|
+
"""Recompute and check the hash chain of the whole ledger.
|
|
199
|
+
|
|
200
|
+
Detects any edit, insertion, or reordering after the fact. Read-only.
|
|
201
|
+
"""
|
|
202
|
+
path = ledger_path(state_dir)
|
|
203
|
+
if not path.exists():
|
|
204
|
+
return {"exists": False, "ok": True, "events": 0, "first_bad_line": None}
|
|
205
|
+
problems: list[dict[str, Any]] = []
|
|
206
|
+
prev_hash: str | None = None
|
|
207
|
+
events = 0
|
|
208
|
+
for line_no, line in enumerate(
|
|
209
|
+
path.read_text(encoding="utf-8").splitlines(), start=1
|
|
210
|
+
):
|
|
211
|
+
line = line.strip()
|
|
212
|
+
if not line:
|
|
213
|
+
continue
|
|
214
|
+
try:
|
|
215
|
+
event = json.loads(line)
|
|
216
|
+
except json.JSONDecodeError as exc:
|
|
217
|
+
problems.append({"line": line_no, "reason": f"invalid_json:{exc.msg}"})
|
|
218
|
+
continue
|
|
219
|
+
if not isinstance(event, dict):
|
|
220
|
+
problems.append({"line": line_no, "reason": "not_object"})
|
|
221
|
+
continue
|
|
222
|
+
events += 1
|
|
223
|
+
stored_prev = event.get("prev_hash") or None
|
|
224
|
+
if stored_prev != prev_hash:
|
|
225
|
+
problems.append(
|
|
226
|
+
{
|
|
227
|
+
"line": line_no,
|
|
228
|
+
"reason": "prev_hash_mismatch",
|
|
229
|
+
"expected": prev_hash,
|
|
230
|
+
"found": stored_prev,
|
|
231
|
+
}
|
|
232
|
+
)
|
|
233
|
+
payload = {
|
|
234
|
+
key: value
|
|
235
|
+
for key, value in event.items()
|
|
236
|
+
if key not in ("event_hash", "prev_hash")
|
|
237
|
+
}
|
|
238
|
+
expected = _event_hash(stored_prev, payload)
|
|
239
|
+
if event.get("event_hash") != expected:
|
|
240
|
+
problems.append(
|
|
241
|
+
{"line": line_no, "reason": "event_hash_mismatch", "expected": expected}
|
|
242
|
+
)
|
|
243
|
+
prev_hash = event.get("event_hash")
|
|
244
|
+
return {
|
|
245
|
+
"exists": True,
|
|
246
|
+
"ok": not problems,
|
|
247
|
+
"events": events,
|
|
248
|
+
"first_bad_line": problems[0]["line"] if problems else None,
|
|
249
|
+
"problems": problems[:20],
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _wilson_ci(hits: int, n: int, z: float = 1.96) -> tuple[float, float]:
|
|
254
|
+
if n == 0:
|
|
255
|
+
return (0.0, 0.0)
|
|
256
|
+
p = hits / n
|
|
257
|
+
denom = 1 + z * z / n
|
|
258
|
+
center = (p + z * z / (2 * n)) / denom
|
|
259
|
+
half = z * math.sqrt((p * (1 - p) + z * z / (4 * n)) / n) / denom
|
|
260
|
+
return (max(0.0, center - half), min(1.0, center + half))
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def report_prediction_hitrate(
|
|
264
|
+
state_dir: Path | str,
|
|
265
|
+
min_cases: int = DEFAULT_MIN_CASES,
|
|
266
|
+
) -> dict[str, Any]:
|
|
267
|
+
"""Read-only hit-rate report (Wilson 95% CI, compared against random baseline).
|
|
268
|
+
|
|
269
|
+
Rule: a hit is ``expected_verdict == actual_verdict`` over *certain*
|
|
270
|
+
expectations; ``uncertain`` expectations count into participation only.
|
|
271
|
+
The report is verdict-ready when ``resolved >= min_cases``,
|
|
272
|
+
``completeness >= 0.9`` and ``participation >= 0.7``. If the Wilson CI
|
|
273
|
+
contains the random baseline, there is no systematic signal.
|
|
274
|
+
"""
|
|
275
|
+
events = _load_events(ledger_path(state_dir))
|
|
276
|
+
registers = {
|
|
277
|
+
e["case_id"]: e for e in events if e.get("event") == "register"
|
|
278
|
+
}
|
|
279
|
+
concludes = {
|
|
280
|
+
e["case_id"]: e for e in events if e.get("event") == "conclude"
|
|
281
|
+
}
|
|
282
|
+
resolved = sorted(set(registers) & set(concludes))
|
|
283
|
+
unresolved = sorted(set(registers) - set(concludes))
|
|
284
|
+
registered_total = len(registers)
|
|
285
|
+
completeness = (
|
|
286
|
+
len(resolved) / registered_total if registered_total else 0.0
|
|
287
|
+
)
|
|
288
|
+
hits = 0
|
|
289
|
+
certain_cases = 0
|
|
290
|
+
uncertain_cases = 0
|
|
291
|
+
for case in resolved:
|
|
292
|
+
expected = registers[case]["expected_verdict"]
|
|
293
|
+
if expected == "uncertain":
|
|
294
|
+
uncertain_cases += 1
|
|
295
|
+
continue
|
|
296
|
+
certain_cases += 1
|
|
297
|
+
if expected == concludes[case]["actual_verdict"]:
|
|
298
|
+
hits += 1
|
|
299
|
+
n = len(resolved)
|
|
300
|
+
hit_rate = hits / certain_cases if certain_cases else None
|
|
301
|
+
low, high = _wilson_ci(hits, certain_cases) if certain_cases else (None, None)
|
|
302
|
+
participation = certain_cases / n if n else 0.0
|
|
303
|
+
actual_distribution: dict[str, int] = {}
|
|
304
|
+
for case in resolved:
|
|
305
|
+
if registers[case]["expected_verdict"] == "uncertain":
|
|
306
|
+
continue
|
|
307
|
+
verdict = concludes[case]["actual_verdict"]
|
|
308
|
+
actual_distribution[verdict] = actual_distribution.get(verdict, 0) + 1
|
|
309
|
+
baseline = (
|
|
310
|
+
max(actual_distribution.values()) / certain_cases if certain_cases else None
|
|
311
|
+
)
|
|
312
|
+
baseline_in_ci = (
|
|
313
|
+
low <= baseline <= high if (low is not None and baseline is not None) else None
|
|
314
|
+
)
|
|
315
|
+
by_source_type: dict[str, dict[str, Any]] = {}
|
|
316
|
+
for case in resolved:
|
|
317
|
+
source = registers[case].get("source_type", "other")
|
|
318
|
+
bucket = by_source_type.setdefault(
|
|
319
|
+
source, {"resolved": 0, "certain": 0, "hits": 0, "hit_rate": None}
|
|
320
|
+
)
|
|
321
|
+
bucket["resolved"] += 1
|
|
322
|
+
expected = registers[case]["expected_verdict"]
|
|
323
|
+
if expected == "uncertain":
|
|
324
|
+
continue
|
|
325
|
+
bucket["certain"] += 1
|
|
326
|
+
if expected == concludes[case]["actual_verdict"]:
|
|
327
|
+
bucket["hits"] += 1
|
|
328
|
+
for source, bucket in by_source_type.items():
|
|
329
|
+
bucket["hit_rate"] = (
|
|
330
|
+
bucket["hits"] / bucket["certain"] if bucket["certain"] else None
|
|
331
|
+
)
|
|
332
|
+
verdict_ready = (
|
|
333
|
+
n >= min_cases and completeness >= 0.9 and participation >= 0.7
|
|
334
|
+
)
|
|
335
|
+
return {
|
|
336
|
+
"schema_version": REPORT_SCHEMA_VERSION,
|
|
337
|
+
"generated_at": _now_text(),
|
|
338
|
+
"min_cases": min_cases,
|
|
339
|
+
"resolved_cases": n,
|
|
340
|
+
"unresolved_cases": len(unresolved),
|
|
341
|
+
"registered_total": registered_total,
|
|
342
|
+
"completeness": round(completeness, 4),
|
|
343
|
+
"certain_cases": certain_cases,
|
|
344
|
+
"uncertain_cases": uncertain_cases,
|
|
345
|
+
"participation": round(participation, 4),
|
|
346
|
+
"hit_rate": round(hit_rate, 4) if hit_rate is not None else None,
|
|
347
|
+
"wilson_ci_95": [round(low, 4), round(high, 4)]
|
|
348
|
+
if low is not None
|
|
349
|
+
else None,
|
|
350
|
+
"random_baseline": round(baseline, 4) if baseline is not None else None,
|
|
351
|
+
"baseline_in_ci": baseline_in_ci,
|
|
352
|
+
"actual_verdict_distribution": actual_distribution,
|
|
353
|
+
"by_source_type": by_source_type,
|
|
354
|
+
"verdict_ready": verdict_ready,
|
|
355
|
+
"verdict_rule": (
|
|
356
|
+
"hit = expected_verdict == actual_verdict, only over certain "
|
|
357
|
+
"expectations; uncertain expectations count into participation; "
|
|
358
|
+
"ready when resolved>=min_cases and completeness>=0.9 and "
|
|
359
|
+
"participation>=0.7; "
|
|
360
|
+
"if wilson_ci_95 contains random_baseline -> no systematic signal"
|
|
361
|
+
),
|
|
362
|
+
"safety": SAFETY,
|
|
363
|
+
}
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: falsification-ledger
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Pre-registration and falsification ledger for research: hash-chained, append-only, with Wilson-CI hit-rate reporting and fail-closed evidence contracts.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Keywords: preregistration,falsification,reproducibility,ledger,research,quant
|
|
7
|
+
Classifier: Development Status :: 3 - Alpha
|
|
8
|
+
Classifier: Environment :: Console
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering
|
|
16
|
+
Requires-Python: >=3.11
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Requires-Dist: jsonschema>=4.18
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
|
|
22
|
+
# falsification-ledger
|
|
23
|
+
|
|
24
|
+
A hash-chained, append-only ledger for research claims: **pre-register the
|
|
25
|
+
hypothesis and the evidence that would kill it *before* the study runs**,
|
|
26
|
+
then adjudicate honestly and measure your hit rate against a random
|
|
27
|
+
baseline. Python 3.11+, one dependency (`jsonschema`), Windows / Linux /
|
|
28
|
+
macOS.
|
|
29
|
+
|
|
30
|
+
**Status:** v0.1 — alpha. The ledger semantics are distilled from a
|
|
31
|
+
production research pipeline, but this standalone package is new: expect the
|
|
32
|
+
CLI and schemas to shift before v1.0.
|
|
33
|
+
|
|
34
|
+
## Why this exists
|
|
35
|
+
|
|
36
|
+
Quantitative research has a self-deception problem: you test 500 factor
|
|
37
|
+
ideas, remember the 3 that worked, and forget the 497 that died. By the time
|
|
38
|
+
you "validate" the lucky survivors, the evidence is already contaminated by
|
|
39
|
+
what you saw. Every backtest-hygiene tool on the market attacks the
|
|
40
|
+
*statistics* of this problem (deflated Sharpe, PBO, multiple-testing
|
|
41
|
+
corrections). `falsification-ledger` attacks the *process*: it makes you
|
|
42
|
+
write down, before seeing evidence:
|
|
43
|
+
|
|
44
|
+
- what you expect (`support` / `against` / `uncertain`), and
|
|
45
|
+
- what evidence would **kill** your claim (the falsification contract).
|
|
46
|
+
|
|
47
|
+
Then it keeps the receipts. Every event lands in an append-only JSONL
|
|
48
|
+
**hash chain** — any edit after the fact is detected by `fl verify` — and
|
|
49
|
+
the report answers the only question that matters: *do your pre-registered
|
|
50
|
+
beliefs actually hit, or is your hit rate indistinguishable from a random
|
|
51
|
+
baseline?* (Wilson 95% CI vs the most common actual verdict.)
|
|
52
|
+
|
|
53
|
+
## Philosophy
|
|
54
|
+
|
|
55
|
+
**Research is a promise; the ledger keeps it.**
|
|
56
|
+
|
|
57
|
+
- **Falsifiability is the default, not the exception.** Popper's criterion
|
|
58
|
+
— a claim is scientific only if something could count against it — is
|
|
59
|
+
usually invoked as a lecture. Here it is a required JSON field
|
|
60
|
+
(`falsification_contract` on `preregister`).
|
|
61
|
+
- **Pre-analysis plans have known costs and benefits.** [Olken (2015),
|
|
62
|
+
"Promises and Perils of Pre-Analysis Plans"](https://www.aeaweb.org/articles?id=10.1257/jep.29.3.61)
|
|
63
|
+
(JEP 29(3)) documents both; this tool implements the benefits (frozen
|
|
64
|
+
expectations, audit trail) while keeping the costs explicit (`uncertain`
|
|
65
|
+
verdicts and exploratory source types are first-class, so you can register
|
|
66
|
+
what you genuinely do not know).
|
|
67
|
+
- **Moderation beats total freezing.** [Banerjee & Duflo, "In Praise of
|
|
68
|
+
Moderation"](https://www.semanticscholar.org/paper/05ecf99a05419f0a268fe885be11a2cf4a8dbd46)
|
|
69
|
+
argue for layered pre-registration; `source_type` (paper / business /
|
|
70
|
+
cross_domain / pipeline / other) exists so confirmatory and exploratory
|
|
71
|
+
claims are never mixed in the same bucket.
|
|
72
|
+
- **Finance can become scientific.** [López de Prado (2023), *Causal Factor
|
|
73
|
+
Investing*](https://www.cambridge.org/core/elements/causal-factor-investing/9AFE270D7099B787B8FD4F4CBADE0C6E)
|
|
74
|
+
asks whether factor investing can become a science; this ledger is one
|
|
75
|
+
concrete answer — evidence with a chain of custody, adjudicated against a
|
|
76
|
+
pre-registered expectation.
|
|
77
|
+
- **Automated research needs machine-checkable evidence.** [EviBound
|
|
78
|
+
(arXiv:2511.05524)](https://ar5iv.labs.arxiv.org/html/2511.05524) and
|
|
79
|
+
[ECLIPSE v2.0](https://ideas.repec.org/p/osf/metaar/z3fke_v1.html) argue
|
|
80
|
+
that agentic research pipelines must eliminate false claims through
|
|
81
|
+
verifiable evidence; `fl submit` validates falsification reports against a
|
|
82
|
+
JSON Schema and computes content IDs, so gates can trust the evidence
|
|
83
|
+
without trusting the messenger.
|
|
84
|
+
|
|
85
|
+
## Quick start
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
# install from PyPI (once published)
|
|
89
|
+
pip install falsification-ledger
|
|
90
|
+
|
|
91
|
+
# or run without installing anything:
|
|
92
|
+
# PYTHONPATH=src python -m falsification_ledger --help
|
|
93
|
+
|
|
94
|
+
# try the full loop on a scratch ledger (creates files under a temp dir)
|
|
95
|
+
python examples/demo.py
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
The manual loop:
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
fl init --state-dir ~/.research-ledger
|
|
102
|
+
|
|
103
|
+
# 1. BEFORE running the study: register what you expect,
|
|
104
|
+
# and what evidence would kill the claim.
|
|
105
|
+
fl preregister --state-dir ~/.research-ledger \
|
|
106
|
+
--case-id MOMENTUM-OOS-2026Q3 \
|
|
107
|
+
--verdict support \
|
|
108
|
+
--reason "momentum rank IC stays positive OOS" \
|
|
109
|
+
--source-type paper \
|
|
110
|
+
--contract kill-criteria.json
|
|
111
|
+
|
|
112
|
+
# 2. When an independent check produces evidence, submit it:
|
|
113
|
+
fl submit --report falsification-report.json
|
|
114
|
+
# -> {"content_id": "sha256:...", "evidence_status": "valid", ...}
|
|
115
|
+
|
|
116
|
+
# 3. AFTER the study: adjudicate honestly.
|
|
117
|
+
fl adjudicate --state-dir ~/.research-ledger \
|
|
118
|
+
--case-id MOMENTUM-OOS-2026Q3 --verdict support
|
|
119
|
+
|
|
120
|
+
# 4. Measure whether you are better than a coin flip.
|
|
121
|
+
fl report --state-dir ~/.research-ledger --min-cases 20
|
|
122
|
+
|
|
123
|
+
# 5. Any time: prove nobody rewrote history.
|
|
124
|
+
fl verify --state-dir ~/.research-ledger
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
## Commands
|
|
128
|
+
|
|
129
|
+
| Command | What it does |
|
|
130
|
+
| --- | --- |
|
|
131
|
+
| `init` | Create the ledger state directory |
|
|
132
|
+
| `preregister` | Register a claim: `--case-id`, `--verdict` (support/against/uncertain), `--reason`, optional `--source-type`, optional `--contract` (falsification contract JSON). Duplicate registration for the same case is rejected |
|
|
133
|
+
| `submit` | Validate a falsification report against the contract schema; print its content ID (`sha256:...`) and evidence status (`valid` / `invalid` / `missing`). Read-only; exits non-zero on blockers |
|
|
134
|
+
| `adjudicate` | Backfill the actual verdict for a registered case (register required; once per case) |
|
|
135
|
+
| `report` | Hit-rate report: resolved cases, completeness, participation, hit rate with **Wilson 95% CI**, random baseline, per-source-type breakdown, `verdict_ready` gate |
|
|
136
|
+
| `verify` | Recompute the hash chain of the whole ledger; detects any edit, insertion, or reordering |
|
|
137
|
+
| `version` | Print version |
|
|
138
|
+
|
|
139
|
+
Global flag: `--state-dir` on every stateful command (default: none — the
|
|
140
|
+
ledger path is always explicit, so a `git add .` can never sweep it into
|
|
141
|
+
version control).
|
|
142
|
+
|
|
143
|
+
## Ledger format
|
|
144
|
+
|
|
145
|
+
The ledger is a JSONL file at `<state-dir>/ledger.jsonl`. Every line is one
|
|
146
|
+
event:
|
|
147
|
+
|
|
148
|
+
```json
|
|
149
|
+
{"schema_version": "falsification_ledger.prediction_event.v1",
|
|
150
|
+
"event": "register", "record_id": "...", "case_id": "CASE-1",
|
|
151
|
+
"expected_verdict": "support", "expected_reason": "...",
|
|
152
|
+
"source_type": "paper", "falsification_contract": {...},
|
|
153
|
+
"actual_verdict": null, "recorded_at": "...", "concluded_at": null,
|
|
154
|
+
"prev_hash": null,
|
|
155
|
+
"event_hash": "sha256(prev_hash || 0x00 || canonical payload)"}
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
`verify` recomputes every `event_hash` and checks each `prev_hash` link.
|
|
159
|
+
**Any tampering — editing a reason, deleting a line, reordering events —
|
|
160
|
+
breaks the chain at a specific line number.**
|
|
161
|
+
|
|
162
|
+
## Falsification reports
|
|
163
|
+
|
|
164
|
+
A falsification report is the machine-readable evidence produced by an
|
|
165
|
+
independent check (null-model randomization, OOS rank IC, FDR correction,
|
|
166
|
+
protocol deviation, effect CI, cost sensitivity, ...). The contract:
|
|
167
|
+
|
|
168
|
+
- schema: [`schema/falsification-report.schema.json`](schema/falsification-report.schema.json)
|
|
169
|
+
(draft 2020-12, `additionalProperties: false`, fail-closed);
|
|
170
|
+
- content ID: `sha256:` over `domain-prefix || 0x00 || canonical JSON` —
|
|
171
|
+
the same report always yields the same ID, a one-field change yields a
|
|
172
|
+
different ID;
|
|
173
|
+
- evidence status (fail-closed for gates):
|
|
174
|
+
- `valid` — conformant, conclusion `not_falsified`, consistency intact;
|
|
175
|
+
- `invalid` — non-conformant, or conclusion `falsified`, or explicitly
|
|
176
|
+
inconsistent;
|
|
177
|
+
- `missing` — conclusion `inconclusive`: treated as *absent* evidence.
|
|
178
|
+
|
|
179
|
+
## Verification model
|
|
180
|
+
|
|
181
|
+
`fl verify` is the tamper-evidence layer: it re-derives the entire chain
|
|
182
|
+
from the file bytes and reports the first bad line. Combined with
|
|
183
|
+
`preregister` (frozen expectations) and `submit` (content-addressed
|
|
184
|
+
evidence), a research pipeline can prove to itself — and to reviewers —
|
|
185
|
+
that the expectation existed before the evidence did. Nothing here trades,
|
|
186
|
+
prices, or decides.
|
|
187
|
+
|
|
188
|
+
## Development
|
|
189
|
+
|
|
190
|
+
```bash
|
|
191
|
+
python -m pip install -e . pytest
|
|
192
|
+
python -m pytest
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
CI runs the full test suite on Ubuntu, Windows and macOS with Python 3.11
|
|
196
|
+
and 3.12. Issues are handled on weekends; pull requests are welcome.
|
|
197
|
+
|
|
198
|
+
## Related work
|
|
199
|
+
|
|
200
|
+
- [Olken (2015), Promises and Perils of Pre-Analysis Plans](https://www.aeaweb.org/articles?id=10.1257/jep.29.3.61) — the economics of freezing expectations
|
|
201
|
+
- [Banerjee & Duflo, In Praise of Moderation](https://www.semanticscholar.org/paper/05ecf99a05419f0a268fe885be11a2cf4a8dbd46) — layered pre-registration
|
|
202
|
+
- [López de Prado (2023), Causal Factor Investing](https://www.cambridge.org/core/elements/causal-factor-investing/9AFE270D7099B787B8FD4F4CBADE0C6E) — can factor investing become scientific?
|
|
203
|
+
- [EviBound: Evidence-Bound Autonomous Research (arXiv:2511.05524)](https://ar5iv.labs.arxiv.org/html/2511.05524) — governance for agentic research
|
|
204
|
+
- [ECLIPSE v2.0: A Systematic Falsification Framework](https://ideas.repec.org/p/osf/metaar/z3fke_v1.html) — enforce falsifiability integrity
|
|
205
|
+
|
|
206
|
+
## License
|
|
207
|
+
|
|
208
|
+
MIT
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
falsification_ledger/__init__.py,sha256=Kp8_0uUlig5EEv9tYW4Kv5JswY4_rM4rG2nP7Nsj0rs,1133
|
|
2
|
+
falsification_ledger/__main__.py,sha256=MHKZ_ae3fSLGTLUUMOx15fWdeOnJSHhq-zslRP5F5Lc,79
|
|
3
|
+
falsification_ledger/cli.py,sha256=LLgAvyiuJXQec_4fGEC803E0aUUUx8A3uotMMHG_i0s,5152
|
|
4
|
+
falsification_ledger/contracts.py,sha256=WFS-_x3xSvAGbs-dfeyUebdHMR2dDTr_rsZtLILtahY,4075
|
|
5
|
+
falsification_ledger/ledger.py,sha256=oWm-Kpgek5UQSJB8paTMYXjqqh6cqq1n4e4etsUyJdo,13145
|
|
6
|
+
falsification_ledger-0.1.0.dist-info/licenses/LICENSE,sha256=fuOqBeL8dZoJ9gqMC8N389Mw-Y-J_WlfkVMAJ6a5gfY,1090
|
|
7
|
+
falsification_ledger-0.1.0.dist-info/METADATA,sha256=nnk8FiR2fSCku4JXXOPpCHOCrjxtTQjePqXUojVt-DM,9878
|
|
8
|
+
falsification_ledger-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
9
|
+
falsification_ledger-0.1.0.dist-info/entry_points.txt,sha256=m6ONL2FaDJ65GHZa-IZl_-KSJhWgOzmT7nP7rmvySRU,106
|
|
10
|
+
falsification_ledger-0.1.0.dist-info/top_level.txt,sha256=zAoTdWSre6QyNE_kRcCPKARqg4Jl-ghMlSrOcw6bh-k,21
|
|
11
|
+
falsification_ledger-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Falsification Ledger contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
falsification_ledger
|