decisio 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- decisio/__init__.py +13 -0
- decisio/bench/__init__.py +3 -0
- decisio/bench/banking77_labels.json +81 -0
- decisio/bench/di_cal.py +87 -0
- decisio/bench/di_report.py +68 -0
- decisio/bench/di_rows.py +48 -0
- decisio/bench/jevbench_v15.py +158 -0
- decisio/bench/latency.py +268 -0
- decisio/names.py +68 -0
- decisio/readout/__init__.py +3 -0
- decisio/readout/calibration.py +123 -0
- decisio/readout/debias.py +158 -0
- decisio/readout/intent_head.py +191 -0
- decisio/readout/letters.py +174 -0
- decisio/serve/__init__.py +3 -0
- decisio/serve/abstention.py +144 -0
- decisio/serve/client.py +65 -0
- decisio/serve/hf_letters.py +159 -0
- decisio/serve/hidden_engine.py +310 -0
- decisio/serve/image_engine.py +148 -0
- decisio/serve/make_text_only.py +63 -0
- decisio/serve/systemone.py +943 -0
- decisio/serve/systemone_conformance.py +332 -0
- decisio/serve/tasks.py +206 -0
- decisio/serve/temperature.py +29 -0
- decisio/serve/vllm_engine.py +987 -0
- decisio/vllm_plugin/__init__.py +94 -0
- decisio/vllm_plugin/hidden.py +87 -0
- decisio/vllm_plugin/models.py +40 -0
- decisio/vllm_plugin/weights.py +17 -0
- decisio-0.1.1.dist-info/METADATA +190 -0
- decisio-0.1.1.dist-info/RECORD +37 -0
- decisio-0.1.1.dist-info/WHEEL +5 -0
- decisio-0.1.1.dist-info/entry_points.txt +2 -0
- decisio-0.1.1.dist-info/licenses/LICENSE +202 -0
- decisio-0.1.1.dist-info/licenses/NOTICE +2 -0
- decisio-0.1.1.dist-info/top_level.txt +1 -0
decisio/__init__.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# SPDX-FileCopyrightText: Copyright contributors to the decisio project
|
|
3
|
+
"""decisio: a prefill-only decision server on vLLM.
|
|
4
|
+
|
|
5
|
+
decisio.serve the HTTP server (`python -m decisio.serve.vllm_engine`), its routes, tasks and engines
|
|
6
|
+
decisio.readout the letters readout and the per-task corrections (calibration, intent head)
|
|
7
|
+
decisio.vllm_plugin the model classes vLLM loads through the `vllm.general_plugins` entry point
|
|
8
|
+
decisio.bench the benchmark stage: JevBench v1.5 open-set scoring, Decision Index reports, latency
|
|
9
|
+
|
|
10
|
+
Importing `decisio` pulls in nothing beyond the standard library.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
__version__ = "0.1.1"
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
{
|
|
2
|
+
"banking77": [
|
|
3
|
+
"activate my card",
|
|
4
|
+
"age limit",
|
|
5
|
+
"apple pay or google pay",
|
|
6
|
+
"atm support",
|
|
7
|
+
"automatic top up",
|
|
8
|
+
"balance not updated after bank transfer",
|
|
9
|
+
"balance not updated after cheque or cash deposit",
|
|
10
|
+
"beneficiary not allowed",
|
|
11
|
+
"cancel transfer",
|
|
12
|
+
"card about to expire",
|
|
13
|
+
"card acceptance",
|
|
14
|
+
"card arrival",
|
|
15
|
+
"card delivery estimate",
|
|
16
|
+
"card linking",
|
|
17
|
+
"card not working",
|
|
18
|
+
"card payment fee charged",
|
|
19
|
+
"card payment not recognised",
|
|
20
|
+
"card payment wrong exchange rate",
|
|
21
|
+
"card swallowed",
|
|
22
|
+
"cash withdrawal charge",
|
|
23
|
+
"cash withdrawal not recognised",
|
|
24
|
+
"change pin",
|
|
25
|
+
"compromised card",
|
|
26
|
+
"contactless not working",
|
|
27
|
+
"country support",
|
|
28
|
+
"declined card payment",
|
|
29
|
+
"declined cash withdrawal",
|
|
30
|
+
"declined transfer",
|
|
31
|
+
"direct debit payment not recognised",
|
|
32
|
+
"disposable card limits",
|
|
33
|
+
"edit personal details",
|
|
34
|
+
"exchange charge",
|
|
35
|
+
"exchange rate",
|
|
36
|
+
"exchange via app",
|
|
37
|
+
"extra charge on statement",
|
|
38
|
+
"failed transfer",
|
|
39
|
+
"fiat currency support",
|
|
40
|
+
"get disposable virtual card",
|
|
41
|
+
"get physical card",
|
|
42
|
+
"getting spare card",
|
|
43
|
+
"getting virtual card",
|
|
44
|
+
"lost or stolen card",
|
|
45
|
+
"lost or stolen phone",
|
|
46
|
+
"order physical card",
|
|
47
|
+
"passcode forgotten",
|
|
48
|
+
"pending card payment",
|
|
49
|
+
"pending cash withdrawal",
|
|
50
|
+
"pending top up",
|
|
51
|
+
"pending transfer",
|
|
52
|
+
"pin blocked",
|
|
53
|
+
"receiving money",
|
|
54
|
+
"Refund not showing up",
|
|
55
|
+
"request refund",
|
|
56
|
+
"reverted card payment?",
|
|
57
|
+
"supported cards and currencies",
|
|
58
|
+
"terminate account",
|
|
59
|
+
"top up by bank transfer charge",
|
|
60
|
+
"top up by card charge",
|
|
61
|
+
"top up by cash or cheque",
|
|
62
|
+
"top up failed",
|
|
63
|
+
"top up limits",
|
|
64
|
+
"top up reverted",
|
|
65
|
+
"topping up by card",
|
|
66
|
+
"transaction charged twice",
|
|
67
|
+
"transfer fee charged",
|
|
68
|
+
"transfer into account",
|
|
69
|
+
"transfer not received by recipient",
|
|
70
|
+
"transfer timing",
|
|
71
|
+
"unable to verify identity",
|
|
72
|
+
"verify my identity",
|
|
73
|
+
"verify source of funds",
|
|
74
|
+
"verify top up",
|
|
75
|
+
"virtual card not working",
|
|
76
|
+
"visa or mastercard",
|
|
77
|
+
"why verify identity",
|
|
78
|
+
"wrong amount of cash received",
|
|
79
|
+
"wrong exchange rate for cash withdrawal"
|
|
80
|
+
]
|
|
81
|
+
}
|
decisio/bench/di_cal.py
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# SPDX-FileCopyrightText: Copyright contributors to the decisio project
|
|
3
|
+
"""Accuracy, ECE and Brier per Decision Index benchmark from the kit's own results.jsonl joined with its rows' gold
|
|
4
|
+
(the kit scores accuracy and macro-F1; it reports no calibration). ECE: 10 equal-mass bins on the top probability.
|
|
5
|
+
|
|
6
|
+
python -m decisio.bench.di_cal <run dir with results.jsonl> --rows <rows.jsonl.gz> [--out di_cal.json]
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import ast
|
|
11
|
+
import gzip
|
|
12
|
+
import json
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def ece_equal_mass(conf, correct, bins=10):
|
|
19
|
+
"""Expected calibration error over `bins` bins of equal item count, sorted by confidence."""
|
|
20
|
+
order = np.argsort(conf)
|
|
21
|
+
e = 0.0
|
|
22
|
+
for chunk in np.array_split(order, bins):
|
|
23
|
+
if len(chunk):
|
|
24
|
+
e += len(chunk) / len(conf) * abs(correct[chunk].mean() - conf[chunk].mean())
|
|
25
|
+
return float(e)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def as_dict(x):
|
|
29
|
+
return x if isinstance(x, dict) else ast.literal_eval(x)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def main():
|
|
33
|
+
ap = argparse.ArgumentParser()
|
|
34
|
+
ap.add_argument("run")
|
|
35
|
+
ap.add_argument("--rows", required=True)
|
|
36
|
+
ap.add_argument("--out", default=None)
|
|
37
|
+
a = ap.parse_args()
|
|
38
|
+
gold = {}
|
|
39
|
+
for line in gzip.open(a.rows, "rt"):
|
|
40
|
+
r = json.loads(line)
|
|
41
|
+
gold[r["id"]] = (r["family"], as_dict(r["expected"]))
|
|
42
|
+
per, missing, n = {}, 0, 0
|
|
43
|
+
for line in open(Path(a.run) / "results.jsonl"):
|
|
44
|
+
r = json.loads(line)
|
|
45
|
+
n += 1
|
|
46
|
+
key = (
|
|
47
|
+
r["group_id"]
|
|
48
|
+
if r["group_id"] in gold
|
|
49
|
+
else next((k for k in (r["run_id"].split(":", 2)[-1],) if k in gold), None)
|
|
50
|
+
)
|
|
51
|
+
if key is None or r.get("status") != "ok":
|
|
52
|
+
missing += 1
|
|
53
|
+
continue
|
|
54
|
+
fam, exp = gold[key]
|
|
55
|
+
for qn, g in exp.items():
|
|
56
|
+
ans = as_dict(r["response"])["answers"][qn]
|
|
57
|
+
p = np.array(list(ans["probabilities"].values()), dtype=np.float64)
|
|
58
|
+
keys = list(ans["probabilities"])
|
|
59
|
+
y = keys.index(str(g))
|
|
60
|
+
d = per.setdefault(fam, {"ok": [], "conf": [], "brier": [], "nll": []})
|
|
61
|
+
d["ok"].append(float(ans["choice"] == str(g)))
|
|
62
|
+
d["conf"].append(float(p.max()))
|
|
63
|
+
d["brier"].append(float(((p - np.eye(len(p))[y]) ** 2).sum()))
|
|
64
|
+
d["nll"].append(float(-np.log(max(p[y], 1e-300))))
|
|
65
|
+
out = {"rows": n, "not_joined_or_failed": missing, "benchmarks": {}}
|
|
66
|
+
for fam, d in sorted(per.items()):
|
|
67
|
+
ok, conf = np.array(d["ok"]), np.array(d["conf"])
|
|
68
|
+
out["benchmarks"][fam] = {
|
|
69
|
+
"n": len(ok),
|
|
70
|
+
"accuracy": float(ok.mean()),
|
|
71
|
+
"ece": ece_equal_mass(conf, ok),
|
|
72
|
+
"brier": float(np.mean(d["brier"])),
|
|
73
|
+
"nll": float(np.mean(d["nll"])),
|
|
74
|
+
"mean_confidence": float(conf.mean()),
|
|
75
|
+
}
|
|
76
|
+
b = out["benchmarks"][fam]
|
|
77
|
+
print(
|
|
78
|
+
f"{fam:14s} n {b['n']:5d} accuracy {b['accuracy']:.4f} ECE {b['ece']:.4f} Brier {b['brier']:.4f} "
|
|
79
|
+
f"NLL {b['nll']:.4f}"
|
|
80
|
+
)
|
|
81
|
+
json.dump(out, open(a.out or Path(a.run) / "di_cal.json", "w"), indent=1)
|
|
82
|
+
if missing:
|
|
83
|
+
raise SystemExit(f"{missing} of {n} rows not joined or failed")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
if __name__ == "__main__":
|
|
87
|
+
main()
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# SPDX-FileCopyrightText: Copyright contributors to the decisio project
|
|
3
|
+
"""The Decision Index 0.2.1 values of the four benchmarks run here, computed by the kit's own index02 functions from the
|
|
4
|
+
kit's own `benchmark-summary.json` (its `score` writes that file first; the index step after it needs all 38 benchmarks
|
|
5
|
+
and stops on the ones not run, so the four values are computed here with the same functions, chance levels and
|
|
6
|
+
coverage rule the board uses).
|
|
7
|
+
|
|
8
|
+
python -m decisio.bench.di_report <run dir with benchmark-summary.json and results.jsonl> --suite-dir <suite> \
|
|
9
|
+
[--out ...]
|
|
10
|
+
GPQA Diamond is track-scored (the 0.1 panel's rule: unanswered groups score zero inside the metric), so the kit's
|
|
11
|
+
`score_panel` runs over the results first, exactly as its index step does.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import argparse
|
|
15
|
+
import json
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from decision_index.scoring import index02 as X
|
|
19
|
+
from decision_index.scoring.index import score_panel
|
|
20
|
+
from decision_index.scoring.report import load_results
|
|
21
|
+
from decision_index.suite.io import Suite
|
|
22
|
+
|
|
23
|
+
IDS = {4: "BANKING77", 5: "CLINC150+OOS", 25: "GPQA Diamond", 57: "MMLU-Pro"}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def main():
|
|
27
|
+
ap = argparse.ArgumentParser()
|
|
28
|
+
ap.add_argument("run")
|
|
29
|
+
ap.add_argument("--suite-dir", required=True)
|
|
30
|
+
ap.add_argument("--out", default=None)
|
|
31
|
+
a = ap.parse_args()
|
|
32
|
+
s = json.load(open(Path(a.run) / "benchmark-summary.json"))
|
|
33
|
+
spec = X.spec("0.2.1")
|
|
34
|
+
by_id = {b["catalog_id"]: b for b in s["benchmarks"]}
|
|
35
|
+
scored = score_panel(Suite(Path(a.suite_dir), "0.2.1"), load_results(Path(a.run) / "results.jsonl"))
|
|
36
|
+
out = {
|
|
37
|
+
"edition": "0.2.1",
|
|
38
|
+
"engine": s.get("engine"),
|
|
39
|
+
"latency_ms": s.get("successful_request_latency_ms"),
|
|
40
|
+
"benchmarks": {},
|
|
41
|
+
}
|
|
42
|
+
for n, name in IDS.items():
|
|
43
|
+
b = by_id.get(n, {})
|
|
44
|
+
v = X.benchmark_value(n, spec, scored.get(n), b)
|
|
45
|
+
out["benchmarks"][name] = {
|
|
46
|
+
"catalog_id": n,
|
|
47
|
+
"metric": b.get("metric"),
|
|
48
|
+
"native_score": b.get("score"),
|
|
49
|
+
"requests": b.get("requests"),
|
|
50
|
+
"answered": b.get("answered"),
|
|
51
|
+
"errors": b.get("errors"),
|
|
52
|
+
"unsupported": b.get("unsupported"),
|
|
53
|
+
"median_ms": b.get("median_ms"),
|
|
54
|
+
"chance": X.chance_of(n, spec),
|
|
55
|
+
"raw": v.get("raw"),
|
|
56
|
+
"skill": v.get("skill"),
|
|
57
|
+
"coverage": v.get("coverage"),
|
|
58
|
+
}
|
|
59
|
+
r = out["benchmarks"][name]
|
|
60
|
+
print(
|
|
61
|
+
f"{name:14s} {str(r['metric']):9s} native {r['native_score']} answered {r['answered']}/{r['requests']} "
|
|
62
|
+
f"raw {r['raw']} skill {r['skill']} median {r['median_ms']} ms"
|
|
63
|
+
)
|
|
64
|
+
json.dump(out, open(a.out or Path(a.run) / "di_report.json", "w"), indent=1)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
if __name__ == "__main__":
|
|
68
|
+
main()
|
decisio/bench/di_rows.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# SPDX-FileCopyrightText: Copyright contributors to the decisio project
|
|
3
|
+
"""The rows file of the four Decision Index 0.2.1 benchmarks the runs in runs/ measured, taken from the kit's rebuilt
|
|
4
|
+
suite directory: its selected rows of BANKING77, CLINC150+OOS and GPQA Diamond, then its added rows of MMLU-Pro, in
|
|
5
|
+
the suite's order. Checks the counts per benchmark and, by default, the uncompressed sha256 the runs used.
|
|
6
|
+
|
|
7
|
+
python -m decisio.bench.di_rows <suite dir> <out rows.jsonl.gz> [--no-hash-check]
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import collections
|
|
12
|
+
import gzip
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
FAMILIES = {"BANKING77": 3080, "CLINC150+OOS": 5500, "GPQA-Diamond": 198, "MMLU-Pro": 12032}
|
|
18
|
+
RUNS_SHA256 = "9b8537423f834be743373e3b033ea367b7c4e196b61be1b8480cddab0338ce80" # uncompressed, runs/ 2026-09-27, -30
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def rows(suite: Path) -> list[str]:
|
|
22
|
+
out = []
|
|
23
|
+
for name in ("selected-rows.jsonl.gz", "added-rows.jsonl.gz"):
|
|
24
|
+
with gzip.open(suite / name, "rt") as f:
|
|
25
|
+
out += [line for line in f if json.loads(line)["family"] in FAMILIES]
|
|
26
|
+
return out
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def main():
|
|
30
|
+
ap = argparse.ArgumentParser()
|
|
31
|
+
ap.add_argument("suite", type=Path)
|
|
32
|
+
ap.add_argument("out", type=Path)
|
|
33
|
+
ap.add_argument("--no-hash-check", action="store_true")
|
|
34
|
+
a = ap.parse_args()
|
|
35
|
+
got = rows(a.suite)
|
|
36
|
+
counts = collections.Counter(json.loads(line)["family"] for line in got)
|
|
37
|
+
if dict(counts) != FAMILIES:
|
|
38
|
+
raise SystemExit(f"row counts {dict(counts)} differ from the runs' {FAMILIES}")
|
|
39
|
+
digest = hashlib.sha256("".join(got).encode()).hexdigest()
|
|
40
|
+
if digest != RUNS_SHA256 and not a.no_hash_check:
|
|
41
|
+
raise SystemExit(f"rows sha256 {digest} differs from the runs' {RUNS_SHA256} (--no-hash-check to go on)")
|
|
42
|
+
with gzip.GzipFile(a.out, "wb", mtime=0) as f:
|
|
43
|
+
f.write("".join(got).encode())
|
|
44
|
+
print(f"{sum(counts.values())} rows {dict(counts)} sha256 {digest} -> {a.out}")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
if __name__ == "__main__":
|
|
48
|
+
main()
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
# SPDX-FileCopyrightText: Copyright contributors to the decisio project
|
|
3
|
+
"""JevBench v1.5 open-set Intelligence, scored privately from the harness's own per-item results
|
|
4
|
+
(github.com/fstandhartinger/jevbench, MIT; the frozen method: its docs/METHOD-v1.5.md sections 3.1-3.2, and the v1.5.0
|
|
5
|
+
headline amendment A, equal request types).
|
|
6
|
+
|
|
7
|
+
The v1.5 harness and scorer are not published (the public repository holds the v1.4.2.2 harness, the 231 published v1.2
|
|
8
|
+
items, typed choice / noul / score, and the frozen v1.5 method), so this implements the method's open-set rules on the
|
|
9
|
+
harness's results for the published items. It is a private reading, not a board number: the board's open set has 904
|
|
10
|
+
items (601 published, of which the 370 added in v1.5 are not in the repository; 303 never published), and half of
|
|
11
|
+
Intelligence is the sealed set, which only the evaluator runs.
|
|
12
|
+
|
|
13
|
+
Choice CC = 100 (acc - c) / (1 - c), c the mean over the tier's items of 1 / options; invalid answers wrong
|
|
14
|
+
Noul P(yes) <= 0.20 is No, >= 0.80 is Yes, anything between is an abstention, counted wrong;
|
|
15
|
+
CC = 100 (acc - 0.5) / 0.5
|
|
16
|
+
Score CC = 100 (1 - mean nMAE / mean nMAE_chance); nMAE = |expected position - gold| / (K - 1); an item's
|
|
17
|
+
nMAE_chance is the mean of |l - gold| / (K - 1) over its K levels; an invalid answer takes the item's
|
|
18
|
+
largest possible error
|
|
19
|
+
Tiers easy 0.10, standard 0.20, judge 0.30, hard 0.40, renormalised over the tiers that have items
|
|
20
|
+
Types A (the v1.5.x headline): choice, noul, score a third each; B (the frozen method): 0.50 / 0.25 / 0.25
|
|
21
|
+
Per-tier values are not clipped.
|
|
22
|
+
|
|
23
|
+
python -m decisio.bench.jevbench_v15 --run easy=<tasks.jsonl>:<results.jsonl> --run standard=... --run hard=... \
|
|
24
|
+
--out v15.json
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import argparse
|
|
30
|
+
import json
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
|
|
33
|
+
TIER_WEIGHTS = {"easy": 0.10, "standard": 0.20, "judge": 0.30, "hard": 0.40}
|
|
34
|
+
TYPE_WEIGHTS = {
|
|
35
|
+
"A": {"choice": 1 / 3, "noul": 1 / 3, "score": 1 / 3},
|
|
36
|
+
"B": {"choice": 0.50, "noul": 0.25, "score": 0.25},
|
|
37
|
+
}
|
|
38
|
+
NOUL_NO, NOUL_YES = 0.20, 0.80
|
|
39
|
+
PUBLIC_FILE_TIER = {"easy": "easy", "original": "standard", "hard": "hard"} # the published v1.2 files' tiers
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def item_score(task: dict, rec: dict | None) -> dict:
|
|
43
|
+
"""One item under the v1.5 rules: {"type", "unit"} plus the type's fields. `rec` is the harness's result row for
|
|
44
|
+
the task (None: not answered, scored as invalid)."""
|
|
45
|
+
qtype = task["question"]["type"]
|
|
46
|
+
probs = rec.get("probs") if rec and rec.get("valid") else None
|
|
47
|
+
if qtype == "choice":
|
|
48
|
+
k = len(task["labels"])
|
|
49
|
+
return {
|
|
50
|
+
"type": "choice",
|
|
51
|
+
"correct": bool(probs is not None and rec.get("predicted") == str(task["expected"])),
|
|
52
|
+
"chance": 1.0 / k,
|
|
53
|
+
"valid": probs is not None,
|
|
54
|
+
}
|
|
55
|
+
if qtype == "noul":
|
|
56
|
+
if probs is None:
|
|
57
|
+
return {"type": "noul", "correct": False, "abstained": False, "valid": False}
|
|
58
|
+
p = float(probs["yes"])
|
|
59
|
+
said = "yes" if p >= NOUL_YES else ("no" if p <= NOUL_NO else None)
|
|
60
|
+
return {
|
|
61
|
+
"type": "noul",
|
|
62
|
+
"correct": said == str(task["expected"]),
|
|
63
|
+
"abstained": said is None,
|
|
64
|
+
"valid": True,
|
|
65
|
+
"p_yes": p,
|
|
66
|
+
}
|
|
67
|
+
k, gold = len(task["labels"]), int(task["expected"])
|
|
68
|
+
chance = sum(abs(lv - gold) for lv in range(k)) / k / (k - 1)
|
|
69
|
+
if probs is None:
|
|
70
|
+
return {"type": "score", "nmae": max(gold, k - 1 - gold) / (k - 1), "nmae_chance": chance, "valid": False}
|
|
71
|
+
pred = sum(int(lv) * float(p) for lv, p in probs.items())
|
|
72
|
+
return {"type": "score", "nmae": abs(pred - gold) / (k - 1), "nmae_chance": chance, "valid": True}
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def cell_cc(qtype: str, items: list[dict]) -> float:
|
|
76
|
+
"""Chance-corrected competence of one type x tier cell (not clipped)."""
|
|
77
|
+
n = len(items)
|
|
78
|
+
if qtype == "choice":
|
|
79
|
+
acc, c = sum(x["correct"] for x in items) / n, sum(x["chance"] for x in items) / n
|
|
80
|
+
return 100.0 * (acc - c) / (1.0 - c)
|
|
81
|
+
if qtype == "noul":
|
|
82
|
+
return 100.0 * (sum(x["correct"] for x in items) / n - 0.5) / 0.5
|
|
83
|
+
return 100.0 * (1.0 - sum(x["nmae"] for x in items) / sum(x["nmae_chance"] for x in items))
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def score_runs(runs: list[tuple[str, list[dict], list[dict]]]) -> dict:
|
|
87
|
+
"""runs: [(tier, tasks, harness result rows)]. Returns the per-type per-tier competence, CC per type and I_open
|
|
88
|
+
under type weightings A and B, with the item counts."""
|
|
89
|
+
cells: dict[tuple[str, str], list[dict]] = {}
|
|
90
|
+
for tier, tasks, rows in runs:
|
|
91
|
+
if tier not in TIER_WEIGHTS:
|
|
92
|
+
raise ValueError(f"unknown tier {tier!r}")
|
|
93
|
+
by = {r["task_id"]: r for r in rows}
|
|
94
|
+
for t in tasks:
|
|
95
|
+
s = item_score(t, by.get(t["id"]))
|
|
96
|
+
cells.setdefault((s["type"], tier), []).append(s)
|
|
97
|
+
out = {
|
|
98
|
+
"per_type": {},
|
|
99
|
+
"n_items": sum(len(v) for v in cells.values()),
|
|
100
|
+
"invalid": sum(not x["valid"] for v in cells.values() for x in v),
|
|
101
|
+
}
|
|
102
|
+
for qtype in ("choice", "noul", "score"):
|
|
103
|
+
tiers = {tier: cell_cc(qtype, cells[(qtype, tier)]) for tier in TIER_WEIGHTS if (qtype, tier) in cells}
|
|
104
|
+
if not tiers:
|
|
105
|
+
continue
|
|
106
|
+
w = sum(TIER_WEIGHTS[t] for t in tiers)
|
|
107
|
+
rec = {
|
|
108
|
+
"cc": sum(TIER_WEIGHTS[t] * v for t, v in tiers.items()) / w,
|
|
109
|
+
"tiers": tiers,
|
|
110
|
+
"n": {t: len(cells[(qtype, t)]) for t in tiers},
|
|
111
|
+
}
|
|
112
|
+
if qtype == "noul":
|
|
113
|
+
its = [x for t in tiers for x in cells[(qtype, t)]]
|
|
114
|
+
rec["abstention_rate"] = sum(x["abstained"] for x in its) / len(its)
|
|
115
|
+
out["per_type"][qtype] = rec
|
|
116
|
+
for name, W in TYPE_WEIGHTS.items():
|
|
117
|
+
sup = {t: W[t] for t in out["per_type"]}
|
|
118
|
+
out[f"I_open_{name}"] = sum(sup[t] * out["per_type"][t]["cc"] for t in sup) / sum(sup.values())
|
|
119
|
+
out["tiers_missing"] = sorted(set(TIER_WEIGHTS) - {tier for _, tier in cells})
|
|
120
|
+
return out
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def load_jsonl(path):
|
|
124
|
+
return [json.loads(line) for line in open(path) if line.strip()]
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def main():
|
|
128
|
+
ap = argparse.ArgumentParser()
|
|
129
|
+
ap.add_argument("--run", action="append", required=True, help="<tier>=<tasks.jsonl>:<results.jsonl>")
|
|
130
|
+
ap.add_argument("--out", required=True)
|
|
131
|
+
a = ap.parse_args()
|
|
132
|
+
runs = []
|
|
133
|
+
for spec in a.run:
|
|
134
|
+
tier, paths = spec.split("=", 1)
|
|
135
|
+
tasks, results = paths.split(":", 1)
|
|
136
|
+
runs.append((tier, load_jsonl(tasks), load_jsonl(results)))
|
|
137
|
+
rep = score_runs(runs)
|
|
138
|
+
rep["note"] = (
|
|
139
|
+
"private reading of JevBench v1.5 open-set Intelligence on the published v1.2 items; not a board "
|
|
140
|
+
"number (the board's open set has 904 items and half of Intelligence is sealed)"
|
|
141
|
+
)
|
|
142
|
+
Path(a.out).write_text(json.dumps(rep, indent=1))
|
|
143
|
+
for qtype, r in rep["per_type"].items():
|
|
144
|
+
print(
|
|
145
|
+
f"{qtype:6s} CC {r['cc']:6.2f} "
|
|
146
|
+
+ " ".join(f"{t} {v:.1f} (n={r['n'][t]})" for t, v in r["tiers"].items())
|
|
147
|
+
+ (f" abstained {r['abstention_rate']:.3f}" if qtype == "noul" else "")
|
|
148
|
+
)
|
|
149
|
+
print(
|
|
150
|
+
f"I_open A (equal types) {rep['I_open_A']:.2f} B (50/25/25) {rep['I_open_B']:.2f} "
|
|
151
|
+
f"items {rep['n_items']}, "
|
|
152
|
+
f"invalid {rep['invalid']}, tiers without items {rep['tiers_missing']}"
|
|
153
|
+
)
|
|
154
|
+
print("JEVBENCH_V15_DONE")
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
if __name__ == "__main__":
|
|
158
|
+
main()
|