sothstan 0.1.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sothstan/__init__.py +29 -0
- sothstan/_version.py +3 -0
- sothstan/audit.py +210 -0
- sothstan/baseline.py +165 -0
- sothstan/baselines/_demo_aqua-70b.json +163 -0
- sothstan/baselines/_demo_breeze-8b.json +163 -0
- sothstan/cli.py +307 -0
- sothstan/compare.py +124 -0
- sothstan/http.py +284 -0
- sothstan/mockserver.py +298 -0
- sothstan/probes/__init__.py +22 -0
- sothstan/probes/adversarial.py +86 -0
- sothstan/probes/base.py +53 -0
- sothstan/probes/canon.py +48 -0
- sothstan/probes/errors.py +48 -0
- sothstan/probes/limits.py +74 -0
- sothstan/probes/reasoning.py +63 -0
- sothstan/probes/template.py +67 -0
- sothstan/probes/token_count.py +99 -0
- sothstan/py.typed +0 -0
- sothstan/report.py +102 -0
- sothstan/runner.py +253 -0
- sothstan/types.py +51 -0
- sothstan/verdict.py +155 -0
- sothstan-0.1.0rc1.dist-info/METADATA +195 -0
- sothstan-0.1.0rc1.dist-info/RECORD +30 -0
- sothstan-0.1.0rc1.dist-info/WHEEL +5 -0
- sothstan-0.1.0rc1.dist-info/entry_points.txt +2 -0
- sothstan-0.1.0rc1.dist-info/licenses/LICENSE +21 -0
- sothstan-0.1.0rc1.dist-info/top_level.txt +1 -0
sothstan/__init__.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Sothstan — LLM API 模型验真。
|
|
2
|
+
|
|
3
|
+
一个端点声称自己在服务模型 X,它真的在服务 X 吗?
|
|
4
|
+
sothstan 用探测(probe)提取指纹信号(signal),与官方模型指纹基线(baseline)
|
|
5
|
+
做混淆集似然比比较(confusion-set likelihood ratio),给出统计判决(verdict)。
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from sothstan._version import __version__
|
|
9
|
+
from sothstan.baseline import Baseline, load_baseline, save_baseline
|
|
10
|
+
from sothstan.http import ApiClient, ApiError, BudgetExceeded
|
|
11
|
+
from sothstan.runner import collect_baseline_signals, verify
|
|
12
|
+
from sothstan.types import Signal, VerdictLabel
|
|
13
|
+
from sothstan.verdict import Verdict, decide
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"ApiError",
|
|
17
|
+
"ApiClient",
|
|
18
|
+
"Baseline",
|
|
19
|
+
"BudgetExceeded",
|
|
20
|
+
"Signal",
|
|
21
|
+
"Verdict",
|
|
22
|
+
"VerdictLabel",
|
|
23
|
+
"collect_baseline_signals",
|
|
24
|
+
"decide",
|
|
25
|
+
"load_baseline",
|
|
26
|
+
"save_baseline",
|
|
27
|
+
"verify",
|
|
28
|
+
"__version__",
|
|
29
|
+
]
|
sothstan/_version.py
ADDED
sothstan/audit.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""批量审计:一个 CSV 目标清单 → 受控并发验真 → 可发布的汇总报告。
|
|
2
|
+
|
|
3
|
+
CSV 格式(utf-8-sig,含表头):name,base_url,model,key_env
|
|
4
|
+
- key_env 可为空(匿名请求)
|
|
5
|
+
- 同一 model 的多行共享注册表基线;混淆集 = 注册表中除该模型外的全部基线
|
|
6
|
+
(与 `sothstan check` 同规则:demo 基线默认排除)
|
|
7
|
+
|
|
8
|
+
退出码约定(cmd_audit):0=全部 AUTHENTIC;1=含 SUSPICIOUS;2=含 MISMATCH;
|
|
9
|
+
3=全部 INCONCLUSIVE/失败;4=基础设施错误(文件/表头/基线全缺)。
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import csv
|
|
15
|
+
import os
|
|
16
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from datetime import datetime, timezone
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from sothstan._version import __version__
|
|
22
|
+
from sothstan.baseline import Baseline, load_registry
|
|
23
|
+
from sothstan.report import HONESTY_APPENDIX, result_to_dict
|
|
24
|
+
from sothstan.runner import verify
|
|
25
|
+
from sothstan.verdict import VerdictLabel
|
|
26
|
+
|
|
27
|
+
REQUIRED_COLUMNS = {"name", "base_url", "model", "key_env"}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class AuditTarget:
|
|
32
|
+
name: str
|
|
33
|
+
base_url: str
|
|
34
|
+
model: str
|
|
35
|
+
key_env: str = ""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class AuditReport:
|
|
40
|
+
rows: list[dict] = field(default_factory=list)
|
|
41
|
+
created_at: str = ""
|
|
42
|
+
seed: int = 0
|
|
43
|
+
summary: dict = field(default_factory=dict)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def load_targets(path: str | Path) -> tuple[list[AuditTarget], list[str]]:
|
|
47
|
+
"""解析目标 CSV,返回 (targets, problems)。缺列/空行/缺值如实收集。"""
|
|
48
|
+
problems: list[str] = []
|
|
49
|
+
targets: list[AuditTarget] = []
|
|
50
|
+
with open(path, encoding="utf-8-sig", newline="") as f:
|
|
51
|
+
reader = csv.DictReader(f)
|
|
52
|
+
if reader.fieldnames is None or not REQUIRED_COLUMNS.issubset(set(reader.fieldnames)):
|
|
53
|
+
return [], [f"表头缺列:需要 {sorted(REQUIRED_COLUMNS)},实际 {reader.fieldnames}"]
|
|
54
|
+
for i, row in enumerate(reader, start=2):
|
|
55
|
+
name = (row.get("name") or "").strip()
|
|
56
|
+
base_url = (row.get("base_url") or "").strip()
|
|
57
|
+
model = (row.get("model") or "").strip()
|
|
58
|
+
key_env = (row.get("key_env") or "").strip()
|
|
59
|
+
if not name and not base_url and not model:
|
|
60
|
+
continue # 整行空白,跳过
|
|
61
|
+
if not name or not base_url or not model:
|
|
62
|
+
problems.append(f"第 {i} 行:name/base_url/model 均不能为空")
|
|
63
|
+
continue
|
|
64
|
+
targets.append(AuditTarget(name=name, base_url=base_url, model=model, key_env=key_env))
|
|
65
|
+
if not targets and not problems:
|
|
66
|
+
problems.append("清单为空:没有任何目标")
|
|
67
|
+
return targets, problems
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _rivals_for(model: str, registry: dict[str, Baseline], include_demo: bool) -> list[Baseline]:
|
|
71
|
+
return [
|
|
72
|
+
b for mid, b in registry.items()
|
|
73
|
+
if mid != model and (include_demo or not b.is_demo)
|
|
74
|
+
]
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def audit_one(
|
|
78
|
+
target: AuditTarget,
|
|
79
|
+
registry: dict[str, Baseline],
|
|
80
|
+
include_demo: bool,
|
|
81
|
+
seed: int,
|
|
82
|
+
timeout: float,
|
|
83
|
+
retries: int,
|
|
84
|
+
max_requests: int,
|
|
85
|
+
max_prompt_tokens: int,
|
|
86
|
+
adversarial: bool,
|
|
87
|
+
) -> dict:
|
|
88
|
+
"""验真单个目标,返回一行报告数据(不抛异常——失败也是一行)。"""
|
|
89
|
+
row: dict = {
|
|
90
|
+
"name": target.name,
|
|
91
|
+
"base_url": target.base_url,
|
|
92
|
+
"claimed_model": target.model,
|
|
93
|
+
"error": None,
|
|
94
|
+
"report": None,
|
|
95
|
+
}
|
|
96
|
+
baseline = registry.get(target.model)
|
|
97
|
+
if baseline is None:
|
|
98
|
+
row["error"] = f"注册表中没有 {target.model} 的基线"
|
|
99
|
+
return row
|
|
100
|
+
api_key = os.environ.get(target.key_env) if target.key_env else None
|
|
101
|
+
result = verify(
|
|
102
|
+
base_url=target.base_url,
|
|
103
|
+
model=target.model,
|
|
104
|
+
baseline=baseline,
|
|
105
|
+
rivals=_rivals_for(target.model, registry, include_demo),
|
|
106
|
+
api_key=api_key,
|
|
107
|
+
seed=seed,
|
|
108
|
+
timeout=timeout,
|
|
109
|
+
retries=retries,
|
|
110
|
+
max_requests=max_requests,
|
|
111
|
+
max_prompt_tokens=max_prompt_tokens,
|
|
112
|
+
options={"adversarial": True} if adversarial else None,
|
|
113
|
+
)
|
|
114
|
+
row["report"] = result_to_dict(result, target.base_url, target.model, seed)
|
|
115
|
+
return row
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def run_audit(
|
|
119
|
+
targets: list[AuditTarget],
|
|
120
|
+
registry_dir: str | Path | None = None,
|
|
121
|
+
include_demo: bool = False,
|
|
122
|
+
seed: int = 20260928,
|
|
123
|
+
max_workers: int = 4,
|
|
124
|
+
timeout: float = 30.0,
|
|
125
|
+
retries: int = 2,
|
|
126
|
+
max_requests: int = 64,
|
|
127
|
+
max_prompt_tokens: int = 200_000,
|
|
128
|
+
adversarial: bool = False,
|
|
129
|
+
) -> AuditReport:
|
|
130
|
+
"""受控并发跑批。注册表只加载一次,逐目标共享。"""
|
|
131
|
+
registry = load_registry(registry_dir)
|
|
132
|
+
report = AuditReport(
|
|
133
|
+
created_at=datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
|
134
|
+
seed=seed,
|
|
135
|
+
)
|
|
136
|
+
if not registry:
|
|
137
|
+
report.summary = {"error": "注册表为空:没有任何可用基线"}
|
|
138
|
+
return report
|
|
139
|
+
|
|
140
|
+
def _one(t: AuditTarget) -> dict:
|
|
141
|
+
return audit_one(
|
|
142
|
+
t, registry, include_demo, seed, timeout, retries,
|
|
143
|
+
max_requests, max_prompt_tokens, adversarial,
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
with ThreadPoolExecutor(max_workers=max(1, max_workers)) as pool:
|
|
147
|
+
report.rows = list(pool.map(_one, targets))
|
|
148
|
+
|
|
149
|
+
labels = [
|
|
150
|
+
r["report"]["verdict"]["label"] for r in report.rows if r["report"] is not None
|
|
151
|
+
]
|
|
152
|
+
report.summary = {
|
|
153
|
+
"total": len(report.rows),
|
|
154
|
+
"AUTHENTIC": labels.count(VerdictLabel.AUTHENTIC.value),
|
|
155
|
+
"SUSPICIOUS": labels.count(VerdictLabel.SUSPICIOUS.value),
|
|
156
|
+
"MISMATCH": labels.count(VerdictLabel.MISMATCH.value),
|
|
157
|
+
"INCONCLUSIVE": labels.count(VerdictLabel.INCONCLUSIVE.value),
|
|
158
|
+
"errors": sum(1 for r in report.rows if r["error"]),
|
|
159
|
+
"registry_models": sorted(registry),
|
|
160
|
+
}
|
|
161
|
+
return report
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def render_audit_markdown(report: AuditReport) -> str:
|
|
165
|
+
"""可发布的汇总报告:摘要表 + 声明。逐端点完整证据在 JSON。"""
|
|
166
|
+
s = report.summary
|
|
167
|
+
lines = [
|
|
168
|
+
"# Sothstan批量审计报告",
|
|
169
|
+
"",
|
|
170
|
+
f"- 生成时间:{report.created_at}(sothstan {__version__})",
|
|
171
|
+
f"- seed:{report.seed}(同 seed 同探测集,证据链可复现)",
|
|
172
|
+
f"- 目标:{s.get('total', 0)} 个|"
|
|
173
|
+
f"AUTHENTIC {s.get('AUTHENTIC', 0)}|SUSPICIOUS {s.get('SUSPICIOUS', 0)}|"
|
|
174
|
+
f"MISMATCH {s.get('MISMATCH', 0)}|INCONCLUSIVE {s.get('INCONCLUSIVE', 0)}|"
|
|
175
|
+
f"失败 {s.get('errors', 0)}",
|
|
176
|
+
f"- 混淆集基线:{', '.join(s.get('registry_models', [])) or '(无)'}",
|
|
177
|
+
"",
|
|
178
|
+
"| 目标 | 声称模型 | 判决 | margin | 最强竞争解释 | 证据链 |",
|
|
179
|
+
"|---|---|---|---|---|---|",
|
|
180
|
+
]
|
|
181
|
+
for r in report.rows:
|
|
182
|
+
rep = r["report"]
|
|
183
|
+
if rep is None:
|
|
184
|
+
lines.append(f"| {r['name']} | {r['claimed_model']} | 失败 | — | — | {r['error']} |")
|
|
185
|
+
continue
|
|
186
|
+
v = rep["verdict"]
|
|
187
|
+
lines.append(
|
|
188
|
+
f"| {r['name']} | {v['claimed_model']} | **{v['label']}** "
|
|
189
|
+
f"| {v['margin_total']:+.1f} | {v['runner_up'] or '—'} "
|
|
190
|
+
f"| `{rep['transcript_sha256'][:8]}…` |"
|
|
191
|
+
)
|
|
192
|
+
lines += [
|
|
193
|
+
"",
|
|
194
|
+
"> 逐端点完整信号证据表见同报告的 JSON(含每个信号的观测值/基线值/竞争者得分)。",
|
|
195
|
+
"",
|
|
196
|
+
HONESTY_APPENDIX,
|
|
197
|
+
"",
|
|
198
|
+
]
|
|
199
|
+
return "\n".join(lines)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def audit_exit_code(report: AuditReport) -> int:
|
|
203
|
+
s = report.summary
|
|
204
|
+
if s.get("MISMATCH"):
|
|
205
|
+
return 2
|
|
206
|
+
if s.get("SUSPICIOUS"):
|
|
207
|
+
return 1
|
|
208
|
+
if s.get("total") and s.get("AUTHENTIC"):
|
|
209
|
+
return 0
|
|
210
|
+
return 3 # 全部 INCONCLUSIVE/失败/空
|
sothstan/baseline.py
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""基线:官方模型指纹的版本化存储。
|
|
2
|
+
|
|
3
|
+
一个 Baseline = 一个官方模型在某一时刻的完整指纹快照。
|
|
4
|
+
注册表(registry 目录)里的全部基线互为"混淆集"候选。
|
|
5
|
+
|
|
6
|
+
诚实边界:基线必须用 `sothstan collect` 对**官方端点**采集;
|
|
7
|
+
仓库内置的 `_demo_*.json` 描述的是本库自带 mock 人格,仅用于离线
|
|
8
|
+
测试与 selftest,绝不可当作任何真实模型的指纹使用。
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from sothstan.types import SIGNAL_KINDS
|
|
18
|
+
|
|
19
|
+
SCHEMA_VERSION = 1
|
|
20
|
+
PACKAGE_REGISTRY = Path(__file__).parent / "baselines"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class SignalSpec:
|
|
25
|
+
name: str
|
|
26
|
+
kind: str
|
|
27
|
+
value: object
|
|
28
|
+
weight: float = 1.0
|
|
29
|
+
meta: dict = field(default_factory=dict)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class Baseline:
|
|
34
|
+
model_id: str
|
|
35
|
+
family: str
|
|
36
|
+
signals: dict[str, SignalSpec]
|
|
37
|
+
collected_at: str = ""
|
|
38
|
+
collector: str = ""
|
|
39
|
+
endpoint_hint: str = ""
|
|
40
|
+
tags: list[str] = field(default_factory=list)
|
|
41
|
+
honesty_note: str = ""
|
|
42
|
+
|
|
43
|
+
def get_signal(self, name: str) -> SignalSpec | None:
|
|
44
|
+
return self.signals.get(name)
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def is_demo(self) -> bool:
|
|
48
|
+
return "demo" in self.tags
|
|
49
|
+
|
|
50
|
+
def to_dict(self) -> dict:
|
|
51
|
+
return {
|
|
52
|
+
"schema_version": SCHEMA_VERSION,
|
|
53
|
+
"model_id": self.model_id,
|
|
54
|
+
"family": self.family,
|
|
55
|
+
"tags": self.tags,
|
|
56
|
+
"collected_at": self.collected_at,
|
|
57
|
+
"collector": self.collector,
|
|
58
|
+
"endpoint_hint": self.endpoint_hint,
|
|
59
|
+
"honesty_note": self.honesty_note,
|
|
60
|
+
"signals": {
|
|
61
|
+
name: {
|
|
62
|
+
"kind": s.kind,
|
|
63
|
+
"value": s.value,
|
|
64
|
+
"weight": s.weight,
|
|
65
|
+
"meta": s.meta,
|
|
66
|
+
}
|
|
67
|
+
for name, s in sorted(self.signals.items())
|
|
68
|
+
},
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def validate_baseline_dict(data: dict) -> list[str]:
|
|
73
|
+
"""返回问题列表;空列表 = 合法。"""
|
|
74
|
+
problems: list[str] = []
|
|
75
|
+
if not isinstance(data, dict):
|
|
76
|
+
return ["顶层必须是对象"]
|
|
77
|
+
if data.get("schema_version") != SCHEMA_VERSION:
|
|
78
|
+
problems.append(f"schema_version 必须为 {SCHEMA_VERSION}")
|
|
79
|
+
if not data.get("model_id") or not isinstance(data.get("model_id"), str):
|
|
80
|
+
problems.append("model_id 缺失或非字符串")
|
|
81
|
+
if not isinstance(data.get("family"), str) or not data.get("family"):
|
|
82
|
+
problems.append("family 缺失")
|
|
83
|
+
signals = data.get("signals")
|
|
84
|
+
if not isinstance(signals, dict) or not signals:
|
|
85
|
+
problems.append("signals 必须为非空对象")
|
|
86
|
+
return problems
|
|
87
|
+
for name, spec in signals.items():
|
|
88
|
+
if not isinstance(spec, dict):
|
|
89
|
+
problems.append(f"{name}: 必须为对象")
|
|
90
|
+
continue
|
|
91
|
+
weight = spec.get("weight", 1.0)
|
|
92
|
+
if not isinstance(weight, (int, float)) or isinstance(weight, bool):
|
|
93
|
+
problems.append(f"{name}: weight 必须为数值")
|
|
94
|
+
kind = spec.get("kind")
|
|
95
|
+
if kind not in SIGNAL_KINDS:
|
|
96
|
+
problems.append(f"{name}: 非法 kind {kind!r}")
|
|
97
|
+
continue
|
|
98
|
+
value = spec.get("value")
|
|
99
|
+
if kind == "int_vector" and not (isinstance(value, list) and value):
|
|
100
|
+
problems.append(f"{name}: int_vector 需要非空数组")
|
|
101
|
+
if kind == "int" and not isinstance(value, int):
|
|
102
|
+
problems.append(f"{name}: int 需要整数")
|
|
103
|
+
if kind == "categorical" and not isinstance(value, str):
|
|
104
|
+
problems.append(f"{name}: categorical 需要字符串")
|
|
105
|
+
if kind == "bool" and not isinstance(value, bool):
|
|
106
|
+
problems.append(f"{name}: bool 需要布尔值")
|
|
107
|
+
return problems
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def baseline_from_dict(data: dict) -> Baseline:
|
|
111
|
+
problems = validate_baseline_dict(data)
|
|
112
|
+
if problems:
|
|
113
|
+
raise ValueError("基线不合法: " + "; ".join(problems))
|
|
114
|
+
signals = {
|
|
115
|
+
name: SignalSpec(
|
|
116
|
+
name=name,
|
|
117
|
+
kind=spec["kind"],
|
|
118
|
+
value=spec["value"],
|
|
119
|
+
weight=float(spec.get("weight", 1.0)),
|
|
120
|
+
meta=spec.get("meta", {}),
|
|
121
|
+
)
|
|
122
|
+
for name, spec in data["signals"].items()
|
|
123
|
+
}
|
|
124
|
+
return Baseline(
|
|
125
|
+
model_id=data["model_id"],
|
|
126
|
+
family=data["family"],
|
|
127
|
+
signals=signals,
|
|
128
|
+
collected_at=data.get("collected_at", ""),
|
|
129
|
+
collector=data.get("collector", ""),
|
|
130
|
+
endpoint_hint=data.get("endpoint_hint", ""),
|
|
131
|
+
tags=list(data.get("tags", [])),
|
|
132
|
+
honesty_note=data.get("honesty_note", ""),
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def load_baseline(path: str | Path) -> Baseline:
|
|
137
|
+
p = Path(path)
|
|
138
|
+
data = json.loads(p.read_text(encoding="utf-8"))
|
|
139
|
+
return baseline_from_dict(data)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def save_baseline(baseline: Baseline, path: str | Path) -> Path:
|
|
143
|
+
p = Path(path)
|
|
144
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
145
|
+
p.write_text(
|
|
146
|
+
json.dumps(baseline.to_dict(), ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
147
|
+
)
|
|
148
|
+
return p
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def load_registry(directory: str | Path | None = None) -> dict[str, Baseline]:
|
|
152
|
+
"""加载目录下全部 *.json 基线,按 model_id 索引。坏文件如实跳过并收集错误。"""
|
|
153
|
+
out: dict[str, Baseline] = {}
|
|
154
|
+
errors: list[str] = []
|
|
155
|
+
d = Path(directory) if directory is not None else PACKAGE_REGISTRY
|
|
156
|
+
if not d.exists():
|
|
157
|
+
return out
|
|
158
|
+
for f in sorted(d.glob("*.json")):
|
|
159
|
+
try:
|
|
160
|
+
b = baseline_from_dict(json.loads(f.read_text(encoding="utf-8")))
|
|
161
|
+
except (ValueError, json.JSONDecodeError) as exc:
|
|
162
|
+
errors.append(f"{f.name}: {exc}")
|
|
163
|
+
continue
|
|
164
|
+
out[b.model_id] = b
|
|
165
|
+
return out
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schema_version": 1,
|
|
3
|
+
"model_id": "aqua-70b",
|
|
4
|
+
"family": "aqua",
|
|
5
|
+
"tags": [
|
|
6
|
+
"demo"
|
|
7
|
+
],
|
|
8
|
+
"collected_at": "2026-09-28T13:34:11+00:00",
|
|
9
|
+
"collector": "sothstan 0.1.0rc1",
|
|
10
|
+
"endpoint_hint": "",
|
|
11
|
+
"honesty_note": "内置 mock 人格的指纹,仅用于离线测试与 selftest,不代表任何真实模型。",
|
|
12
|
+
"signals": {
|
|
13
|
+
"errors.empty_messages": {
|
|
14
|
+
"kind": "categorical",
|
|
15
|
+
"value": "400|invalid_parameter__messages_must",
|
|
16
|
+
"weight": 1.0,
|
|
17
|
+
"meta": {}
|
|
18
|
+
},
|
|
19
|
+
"errors.invalid_role": {
|
|
20
|
+
"kind": "categorical",
|
|
21
|
+
"value": "400|invalid_parameter__invalid_role",
|
|
22
|
+
"weight": 1.0,
|
|
23
|
+
"meta": {}
|
|
24
|
+
},
|
|
25
|
+
"errors.temperature_out_of_range": {
|
|
26
|
+
"kind": "categorical",
|
|
27
|
+
"value": "400|invalid_parameter__temperature_m",
|
|
28
|
+
"weight": 1.0,
|
|
29
|
+
"meta": {}
|
|
30
|
+
},
|
|
31
|
+
"errors.unknown_param": {
|
|
32
|
+
"kind": "categorical",
|
|
33
|
+
"value": "400|invalid_parameter__unknown_param",
|
|
34
|
+
"weight": 1.0,
|
|
35
|
+
"meta": {}
|
|
36
|
+
},
|
|
37
|
+
"limits.logprobs_support": {
|
|
38
|
+
"kind": "bool",
|
|
39
|
+
"value": true,
|
|
40
|
+
"weight": 1.0,
|
|
41
|
+
"meta": {}
|
|
42
|
+
},
|
|
43
|
+
"limits.max_tokens_huge": {
|
|
44
|
+
"kind": "categorical",
|
|
45
|
+
"value": "accepted",
|
|
46
|
+
"weight": 1.0,
|
|
47
|
+
"meta": {}
|
|
48
|
+
},
|
|
49
|
+
"limits.model_field_echo": {
|
|
50
|
+
"kind": "categorical",
|
|
51
|
+
"value": "aqua-70b",
|
|
52
|
+
"weight": 0.0,
|
|
53
|
+
"meta": {}
|
|
54
|
+
},
|
|
55
|
+
"limits.n_param_support": {
|
|
56
|
+
"kind": "bool",
|
|
57
|
+
"value": true,
|
|
58
|
+
"weight": 1.0,
|
|
59
|
+
"meta": {}
|
|
60
|
+
},
|
|
61
|
+
"reasoning.completion_per_char": {
|
|
62
|
+
"kind": "int",
|
|
63
|
+
"value": 493,
|
|
64
|
+
"weight": 1.0,
|
|
65
|
+
"meta": {}
|
|
66
|
+
},
|
|
67
|
+
"reasoning.completion_tokens": {
|
|
68
|
+
"kind": "int",
|
|
69
|
+
"value": 39,
|
|
70
|
+
"weight": 1.0,
|
|
71
|
+
"meta": {}
|
|
72
|
+
},
|
|
73
|
+
"reasoning.reasoning_field": {
|
|
74
|
+
"kind": "bool",
|
|
75
|
+
"value": false,
|
|
76
|
+
"weight": 1.0,
|
|
77
|
+
"meta": {}
|
|
78
|
+
},
|
|
79
|
+
"reasoning.style_marker": {
|
|
80
|
+
"kind": "categorical",
|
|
81
|
+
"value": "steps_with_final",
|
|
82
|
+
"weight": 1.0,
|
|
83
|
+
"meta": {}
|
|
84
|
+
},
|
|
85
|
+
"template.empty_user_total": {
|
|
86
|
+
"kind": "int",
|
|
87
|
+
"value": 4,
|
|
88
|
+
"weight": 1.5,
|
|
89
|
+
"meta": {}
|
|
90
|
+
},
|
|
91
|
+
"template.pair_growth": {
|
|
92
|
+
"kind": "int",
|
|
93
|
+
"value": 48,
|
|
94
|
+
"weight": 1.0,
|
|
95
|
+
"meta": {}
|
|
96
|
+
},
|
|
97
|
+
"template.system_overhead": {
|
|
98
|
+
"kind": "int",
|
|
99
|
+
"value": 17,
|
|
100
|
+
"weight": 1.0,
|
|
101
|
+
"meta": {}
|
|
102
|
+
},
|
|
103
|
+
"token_count.completion_int": {
|
|
104
|
+
"kind": "int",
|
|
105
|
+
"value": 7,
|
|
106
|
+
"weight": 1.0,
|
|
107
|
+
"meta": {
|
|
108
|
+
"finish_reason": "stop"
|
|
109
|
+
}
|
|
110
|
+
},
|
|
111
|
+
"token_count.prompt_curve": {
|
|
112
|
+
"kind": "int_vector",
|
|
113
|
+
"value": [
|
|
114
|
+
11,
|
|
115
|
+
17,
|
|
116
|
+
41,
|
|
117
|
+
48,
|
|
118
|
+
103,
|
|
119
|
+
91,
|
|
120
|
+
28,
|
|
121
|
+
31,
|
|
122
|
+
26,
|
|
123
|
+
22,
|
|
124
|
+
33,
|
|
125
|
+
33,
|
|
126
|
+
32,
|
|
127
|
+
9,
|
|
128
|
+
23,
|
|
129
|
+
34,
|
|
130
|
+
33,
|
|
131
|
+
27,
|
|
132
|
+
41,
|
|
133
|
+
43
|
|
134
|
+
],
|
|
135
|
+
"weight": 2.0,
|
|
136
|
+
"meta": {
|
|
137
|
+
"canon_ids": [
|
|
138
|
+
"zh_short_1",
|
|
139
|
+
"zh_short_2",
|
|
140
|
+
"zh_mid_1",
|
|
141
|
+
"zh_mid_2",
|
|
142
|
+
"zh_long_1",
|
|
143
|
+
"zh_long_2",
|
|
144
|
+
"mix_zh_en_1",
|
|
145
|
+
"mix_zh_en_2",
|
|
146
|
+
"numbers_1",
|
|
147
|
+
"numbers_2",
|
|
148
|
+
"fullwidth_1",
|
|
149
|
+
"fullwidth_2",
|
|
150
|
+
"emoji_1",
|
|
151
|
+
"emoji_2",
|
|
152
|
+
"rare_cjk_1",
|
|
153
|
+
"rare_cjk_2",
|
|
154
|
+
"code_1",
|
|
155
|
+
"code_2",
|
|
156
|
+
"repeat_1",
|
|
157
|
+
"mixed_all"
|
|
158
|
+
],
|
|
159
|
+
"mode": "canon"
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
}
|