jevaudit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
jevaudit/__init__.py ADDED
@@ -0,0 +1,17 @@
1
+ """jevaudit — 通用校准审计器(audit ANY decision model, not just Jev)。
2
+
3
+ decide → 三指纹账本 → Brier + 基线 + 校准曲线报告。
4
+ 三轮实战沉淀的 gotchas 内建:真值语义先证伪 / price 语义修正 / 基线 0.25 常数。
5
+ """
6
+ from jevaudit.ledger import canonical, input_hash, output_hash, code_hash, add_record, load_ledger
7
+ from jevaudit.gating import gate2, GateResult
8
+ from jevaudit.accuracy_report import brier_score, calibration_curve, report
9
+ from jevaudit.price_sem import implied_price, residual_check, match_tick
10
+
11
+ __all__ = [
12
+ "canonical", "input_hash", "output_hash", "code_hash", "add_record", "load_ledger",
13
+ "gate2", "GateResult",
14
+ "brier_score", "calibration_curve", "report",
15
+ "implied_price", "residual_check", "match_tick",
16
+ ]
17
+ __version__ = "0.1.0"
@@ -0,0 +1,73 @@
1
+ """accuracy_report.py — Brier Score + 校准曲线 -> calibration_report.md。
2
+
3
+ Brier = (1/N) * sum((p_i - o_i)^2),仅统计 outcome in {0,1}(skipped 不计分)。
4
+ 目标:验证置信度 p 与真实概率 o 的匹配程度,置信度漂移率 <= 0.01。
5
+ """
6
+ import statistics
7
+
8
+ from .ledger import load_ledger
9
+
10
+ REPORT = "calibration_report.md"
11
+
12
+
13
+ def scorable(rows: list) -> list:
14
+ """可计分条目:p 存在且 outcome in {0,1}。"""
15
+ return [r for r in rows
16
+ if isinstance(r.get("p"), (int, float))
17
+ and r.get("outcome") in (0, 1)]
18
+
19
+
20
+ def brier_score(p_list: list, o_list: list) -> float:
21
+ """Brier Score。p_list/o_list 等长;outcome=None 表示 skipped 不计分。"""
22
+ pairs = [(p, o) for p, o in zip(p_list, o_list) if o in (0, 1)]
23
+ if not pairs:
24
+ raise ValueError("no scorable samples")
25
+ return sum((p - o) ** 2 for p, o in pairs) / len(pairs)
26
+
27
+
28
+ def calibration_curve(p_list: list, o_list: list, buckets: int = 5) -> list:
29
+ """校准曲线:按 p 值等宽分桶,每桶返回 {lo, hi, n, mean_p, mean_o}。"""
30
+ pairs = sorted((p, o) for p, o in zip(p_list, o_list) if o in (0, 1))
31
+ if not pairs:
32
+ return []
33
+ width = 1.0 / buckets
34
+ curve = []
35
+ for i in range(buckets):
36
+ lo, hi = i * width, (i + 1) * width
37
+ seg = [(p, o) for p, o in pairs if lo <= p < hi or (i == buckets - 1 and p == 1.0)]
38
+ if not seg:
39
+ continue
40
+ curve.append({
41
+ "lo": round(lo, 2), "hi": round(hi, 2), "n": len(seg),
42
+ "mean_p": round(statistics.mean(p for p, _ in seg), 4),
43
+ "mean_o": round(statistics.mean(o for _, o in seg), 4),
44
+ })
45
+ return curve
46
+
47
+
48
+ def report(rows: list, out: str = REPORT) -> str:
49
+ """产出 Markdown 报告(难听话逻辑内置:样本不足不下结论)。"""
50
+ scor = scorable(rows)
51
+ lines = ["# Calibration Report (mm-audit-poc)", ""]
52
+ if len(scor) < 3:
53
+ lines.append(f"**样本不足(n={len(scor)} < 3),不下结论。**")
54
+ else:
55
+ b = brier_score([r["p"] for r in scor], [r["outcome"] for r in scor])
56
+ # 对照基线:全猜 0.5 的 Brier(常数 0.25,与结果分布无关)
57
+ base = sum((0.5 - o) ** 2 for o in (r["outcome"] for r in scor)) / len(scor)
58
+ curve = calibration_curve([r["p"] for r in scor], [r["outcome"] for r in scor])
59
+ drift = abs(statistics.mean(r["p"] for r in scor)
60
+ - statistics.mean(r["outcome"] for r in scor))
61
+ lines += [
62
+ f"- n={len(scor)} Brier={b:.4f}",
63
+ f"- 对照基线(全猜 0.5)Brier={base:.4f} {'✅ Jev 优于瞎猜' if b < base else '❌ Jev 不优于瞎猜(≈随机)'}",
64
+ f"- 漂移率 |mean_p - mean_o| = {drift:.4f} {'✅ <=0.01 达标' if drift <= 0.01 else '❌ >0.01 未达标'}",
65
+ "",
66
+ "| p 桶 | n | mean_p | mean_o |",
67
+ "|---|---|---|---|",
68
+ ]
69
+ for c in curve:
70
+ lines.append(f"| [{c['lo']}, {c['hi']}) | {c['n']} | {c['mean_p']} | {c['mean_o']} |")
71
+ with open(out, "w", encoding="utf-8") as f:
72
+ f.write("\n".join(lines) + "\n")
73
+ return out
jevaudit/gating.py ADDED
@@ -0,0 +1,32 @@
1
+ """gating.py — 三段式置信度门控(用户令门控带)。
2
+
3
+ Score >= 0.70 -> KEEP (自动采用决策,直接出单/调开执行层)
4
+ 0.30 <= Score < 0.70 -> CONFIRM(存 fallback log,后置复核)
5
+ Score < 0.30 -> DROP (直接抛弃,节省算力开销)
6
+
7
+ 注意:与 jevkit 现成 gate()(act/confirm/escalate)不同,本模块是
8
+ 做市 PoC 专用门控带,不改旧件、不互相污染。
9
+ """
10
+ import enum
11
+
12
+
13
+ class GateResult(enum.Enum):
14
+ KEEP = "KEEP"
15
+ CONFIRM = "CONFIRM"
16
+ DROP = "DROP"
17
+
18
+
19
+ ACT = 0.70
20
+ DROP_BELOW = 0.30
21
+
22
+
23
+ def gate2(p: float, act: float = ACT, drop_below: float = DROP_BELOW) -> GateResult:
24
+ """置信度门控。p 必须在 [0,1]。"""
25
+ p = float(p)
26
+ if not 0.0 <= p <= 1.0:
27
+ raise ValueError(f"p out of range: {p}")
28
+ if p >= act:
29
+ return GateResult.KEEP
30
+ if p >= drop_below:
31
+ return GateResult.CONFIRM
32
+ return GateResult.DROP
jevaudit/ledger.py ADDED
@@ -0,0 +1,61 @@
1
+ """ledger.py — 追加写 JSONL 对账账本(原子落盘 + 坏行容忍 + 三指纹)。"""
2
+ import hashlib
3
+ import json
4
+ import os
5
+
6
+
7
+ def canonical(obj) -> str:
8
+ """规范 JSON:ensure_ascii=False + sort_keys + 紧凑分隔符。"""
9
+ return json.dumps(obj, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
10
+
11
+
12
+ def _sha(s: str) -> str:
13
+ return hashlib.sha256(s.encode("utf-8")).hexdigest()
14
+
15
+
16
+ def input_hash(state: str, questions: dict) -> str:
17
+ """指纹 1:sha256(规范化的 state+questions)[:32]。输入给偏了首查它。"""
18
+ return _sha(canonical({"state": state, "questions": questions}))[:32]
19
+
20
+
21
+ def output_hash(response) -> str:
22
+ """指纹 2:sha256(模型原始响应)[:32]。同输入重跑结果变了没(模型漂移)。"""
23
+ return _sha(canonical(response))[:32]
24
+
25
+
26
+ def code_hash(script_path: str = None) -> str:
27
+ """指纹 3:sha256(判定脚本字节)[:16],src- 前缀。是不是我们改了逻辑。"""
28
+ p = script_path or os.path.join(os.path.dirname(os.path.abspath(__file__)), "mm_poc.py")
29
+ try:
30
+ with open(p, "rb") as f:
31
+ return "src-" + hashlib.sha256(f.read()).hexdigest()[:12]
32
+ except OSError:
33
+ return "src-unavailable"
34
+
35
+
36
+ def add_record(path: str, record: dict) -> None:
37
+ """原子追加一行 JSONL。幂等无害:只追加不重写。"""
38
+ line = canonical(record) + "\n"
39
+ fd = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o644)
40
+ try:
41
+ os.write(fd, line.encode("utf-8"))
42
+ os.fsync(fd)
43
+ finally:
44
+ os.close(fd)
45
+
46
+
47
+ def load_ledger(path: str) -> list:
48
+ """读回账本,坏行容忍(跳过不可解析行,不抛异常)。"""
49
+ rows = []
50
+ if not os.path.exists(path):
51
+ return rows
52
+ with open(path, encoding="utf-8") as f:
53
+ for line in f:
54
+ line = line.strip()
55
+ if not line:
56
+ continue
57
+ try:
58
+ rows.append(json.loads(line))
59
+ except json.JSONDecodeError:
60
+ continue # 坏行容忍
61
+ return rows
jevaudit/price_sem.py ADDED
@@ -0,0 +1,28 @@
1
+ """price_sem.py — price 列语义修正(动手派根因报告落地)。
2
+
3
+ 核心规则:price 列是截断/舍入后的展示价(残差恒 < 0.0125),
4
+ usdcSize 是精确执行金额。**price 列严禁精确校验**:
5
+ - 真执行价 = usdcSize / size(implied price)
6
+ - 校验用 tick 级容差(0.01/0.001)或残差结构检查
7
+ """
8
+ BAND = 0.0125 # 实测截断带(残差峰值钉在 +0.0125)
9
+
10
+
11
+ def implied_price(size: float, usdc_size: float) -> float:
12
+ """真执行价 = usdcSize / size。别再用展示价当真值。"""
13
+ return usdc_size / size
14
+
15
+
16
+ def residual_check(price: float, implied: float, band: float = BAND) -> str:
17
+ """残差结构三态:ok / reverse(SELL 侧,标注不判死)/ out_of_band。"""
18
+ d = implied - price
19
+ if d < 0:
20
+ return "reverse"
21
+ if d < band:
22
+ return "ok"
23
+ return "out_of_band"
24
+
25
+
26
+ def match_tick(price: float, implied: float, tick: float = 0.01) -> bool:
27
+ """tick 级容差校验选项。"""
28
+ return abs(implied - price) <= tick
@@ -0,0 +1,87 @@
1
+ Metadata-Version: 2.4
2
+ Name: jevaudit
3
+ Version: 0.1.0
4
+ Summary: Universal calibration auditor for decision models: decide -> three-fingerprint ledger -> Brier + baselines + calibration curve. Audits ANY decision model, not just Jev.
5
+ Author: jevaudit contributors
6
+ License: MIT
7
+ Keywords: calibration,audit,brier,decision,ledger,llm,agent
8
+ Classifier: Development Status :: 4 - Beta
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Topic :: Software Development :: Quality Assurance
13
+ Requires-Python: >=3.9
14
+ Description-Content-Type: text/markdown
15
+
16
+ # jevaudit — 通用校准审计器
17
+
18
+ > Audit ANY decision model, not just Jev.
19
+ > decide → 三指纹账本 → Brier + 基线 + 校准曲线报告。
20
+
21
+ ## Why
22
+
23
+ 任何"模型给概率、事后有真值"的决策系统都需要审计:置信度 p 和真实结果 o
24
+ 到底差多少?是不是瞎猜?基线是多少?jevaudit 把审计变成三件套:
25
+ **账本(可回溯)+ Brier(可比较)+ 校准曲线(可诊断)**。
26
+
27
+ - **三指纹账本**:input_hash / output_hash / code_hash——输入给偏了、模型漂移了、
28
+ 判定脚本改了,首查指纹
29
+ - **基线内置**:全猜 0.5 的 Brier = **0.25 常数**(与结果分布无关),模型低于
30
+ 0.25 才是有用——报告自动对比,难听话自动打(>0.4 直接写"接近随机")
31
+ - **前视防护**:审计时只看决策时点可得的信息,结算后的信息严禁进输入
32
+
33
+ ## Install
34
+
35
+ ```bash
36
+ pip install jevaudit # (after publish — for now: copy the jevaudit/ directory)
37
+ ```
38
+
39
+ ## Quickstart
40
+
41
+ ```python
42
+ from jevaudit import add_record, load_ledger, input_hash, output_hash, code_hash
43
+ from jevaudit import gate2, brier_score, calibration_curve, report
44
+
45
+ # 1) 每次决策记一条(三指纹 + check_spec 必填)
46
+ add_record("ledger.jsonl", {
47
+ "id": "dec-001",
48
+ "input_hash": input_hash(state, questions), # sha256(规范输入)[:32]
49
+ "output_hash": output_hash(response), # sha256(响应)[:32]
50
+ "code_hash": code_hash(), # sha256(判定脚本)[:16]
51
+ "p": 0.95, # 模型给的置信度
52
+ "outcome": 1, # 真实回填结果 (1/0)
53
+ "check_spec": {"baseline": "implied_price", "tick": 0.01}, # 用的什么基准
54
+ })
55
+
56
+ # 2) 审计:Brier + 基线 + 校准曲线
57
+ rows = load_ledger("ledger.jsonl")
58
+ b = brier_score([r["p"] for r in rows], [r["outcome"] for r in rows])
59
+ curve = calibration_curve([r["p"] for r in rows], [r["outcome"] for r in rows])
60
+ report(rows, "calibration_report.md") # 产出 Markdown(含 0.25 基线对比行)
61
+
62
+ # 3) 门控:KEEP / CONFIRM / DROP
63
+ action = gate2(0.75) # KEEP (>=0.7) / CONFIRM (0.3-0.7) / DROP (<0.3)
64
+ ```
65
+
66
+ ## Gotchas(三轮实战沉淀,踩不到的坑)
67
+
68
+ 1. **Brier 难看时,第一步永远是证伪校验基准**(真值语义/价格语义/结算语义),
69
+ 第二步才怀疑模型——两次实战:真值生成器语义反了 → Brier 必然反向
70
+ 2. **price 列严禁精确校验**:展示价是截断/舍入价(残差恒 < 0.0125),
71
+ 真执行价 = `implied_price(usdcSize/size)`;校验用 tick 级容差(0.01/0.001)
72
+ 3. **check_spec 必填**:记录用的什么基准/容差,让"度量分歧"可回溯,
73
+ 不然下个人又把基准当真值查一遍
74
+ 4. **["0","0"] 是退化态**:已结算二元市场必须恰好一边=1,两边全 0 = 未真结算,拒
75
+ 5. **outcomePrices 可能是字符串**(JSON 编码数组),先 json.loads 再算
76
+ 6. **残差 <0 的反例(~1.4%)疑似 SELL 侧语义**:标注 `reverse`,不判死不混入
77
+ 7. **样本 <3 不下结论**;漂移率 |mean_p − mean_o| > 0.01 如实打 ❌,不圆场
78
+
79
+ ## 配套
80
+
81
+ - **jevkit**(PyPI):OpenRouter Decisions API (Jev) 的第一个开源第三方客户端
82
+ —— client + CLI + gate 模式
83
+ - 实战战绩:n=100 黑箱审计(Polymarket 真实交易 vs Jev,前视防护 100/100)
84
+
85
+ ## License
86
+
87
+ MIT
@@ -0,0 +1,9 @@
1
+ jevaudit/__init__.py,sha256=BwAiptQBaAW2a24fksBCZE3iR0eazJigLrJuGXyVqm8,815
2
+ jevaudit/accuracy_report.py,sha256=ceFl9Z2mxIc27Wz_NhYxIV77vl0pES7HwNznjdk-ha0,3170
3
+ jevaudit/gating.py,sha256=JRQX-kT1aHyazjGdy2H1AwI_vKOKQumTXzEO2AOtXdQ,945
4
+ jevaudit/ledger.py,sha256=69obBQRSiujZdYEeScbkvMklqsmQ3gRfW5_AUD2dwG0,2058
5
+ jevaudit/price_sem.py,sha256=PQ8XYbk38TOAsafB0ac82eZ3sK65mFsoAcJMEBZsjks,1030
6
+ jevaudit-0.1.0.dist-info/METADATA,sha256=yBTvE13DdV2ftFlJGtPPW61aDVtF5E7amUeHBsqIp_M,4047
7
+ jevaudit-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
8
+ jevaudit-0.1.0.dist-info/top_level.txt,sha256=9KO-Wk54zcpGc00y5klBAqBlc1BH_GxydEbwXtdi5a0,9
9
+ jevaudit-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1 @@
1
+ jevaudit