sys1bench 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sys1bench-0.3.2/PKG-INFO +116 -0
- sys1bench-0.3.2/README.md +65 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/pyproject.toml +2 -3
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/__init__.py +1 -1
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/jev_typesafe.py +42 -7
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/cli.py +54 -15
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/data/__init__.py +21 -4
- sys1bench-0.3.2/src/sys1bench/data/configs/example_future_vendor.yaml +18 -0
- sys1bench-0.3.2/src/sys1bench/data/configs/hybrid_mock.yaml +4 -0
- sys1bench-0.3.2/src/sys1bench/data/configs/jev_1.13.yaml +4 -0
- sys1bench-0.3.2/src/sys1bench/data/configs/jev_1.13_openrouter.yaml +3 -0
- sys1bench-0.3.2/src/sys1bench/data/configs/kev_0.8b.yaml +12 -0
- sys1bench-0.3.2/src/sys1bench/data/configs/kev_4b.yaml +12 -0
- sys1bench-0.3.2/src/sys1bench/data/configs/kev_9b.yaml +12 -0
- sys1bench-0.3.2/src/sys1bench/data/configs/laya_en.yaml +6 -0
- sys1bench-0.3.2/src/sys1bench/data/configs/laya_typed_decisions.yaml +6 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/report/plots.py +26 -1
- sys1bench-0.3.2/src/sys1bench.egg-info/PKG-INFO +116 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench.egg-info/SOURCES.txt +9 -0
- sys1bench-0.3.0/PKG-INFO +0 -195
- sys1bench-0.3.0/README.md +0 -143
- sys1bench-0.3.0/src/sys1bench.egg-info/PKG-INFO +0 -195
- {sys1bench-0.3.0 → sys1bench-0.3.2}/LICENSE +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/setup.cfg +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/__init__.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/base.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/embed_knn.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/encoder_finetuned.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/generic_http.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/hybrid_router.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/jev_openrouter.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/laya_local.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/llm_constrained.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/majority_prior.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/mock.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/nli_zeroshot.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/adapters/regex_keyword.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/analysis/__init__.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/analysis/arms.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/analysis/decision_value.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/analysis/decomposition.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/analysis/meta_eval.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/analysis/stats.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/data/canary/canary.jsonl +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/data/framings/guardrail_intent.yaml +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/data/framings/log_triage.yaml +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/data/framings/phishing_email.yaml +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/data/framings/policy_compliance.yaml +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/data/framings/support_tickets.yaml +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/framing/__init__.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/framing/expand.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/framing/perturb.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/__init__.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/base.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/guardrail_intent.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/log_triage.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/multilingual_tickets.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/phishing_email.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/policy_compliance.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/rag_relevance.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/generators/support_tickets.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/metrics/__init__.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/metrics/calibration.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/metrics/consistency.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/metrics/decision_value.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/metrics/efficiency.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/metrics/ordinal.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/metrics/robustness.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/metrics/selective.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/report/__init__.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/report/dashboard.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/report/latex.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/report/results_doc.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/report/scorecard.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/__init__.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/benchmark_runner.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/cache.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/canary.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/hybrid_sweep.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/noul_consistency.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/ordinal_probes.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/robustness.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/suite.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/runners/sweeps.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench/schemas.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench.egg-info/dependency_links.txt +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench.egg-info/entry_points.txt +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench.egg-info/requires.txt +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/src/sys1bench.egg-info/top_level.txt +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/tests/test_generators_framing.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/tests/test_metrics.py +0 -0
- {sys1bench-0.3.0 → sys1bench-0.3.2}/tests/test_runner_end_to_end.py +0 -0
sys1bench-0.3.2/PKG-INFO
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sys1bench
|
|
3
|
+
Version: 0.3.2
|
|
4
|
+
Summary: Benchmark harness for typed System One decision models (choice / score / noul): calibration vs noise floor, framing sensitivity, selective prediction, ordinal fidelity, interference, robustness.
|
|
5
|
+
Author-email: Rahul Sharma <rsharma@rptu.de>
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/rssr25/system-one-bench
|
|
8
|
+
Project-URL: Documentation, https://github.com/rssr25/system-one-bench/blob/main/docs/SPEC_v2.md
|
|
9
|
+
Project-URL: Issues, https://github.com/rssr25/system-one-bench/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/rssr25/system-one-bench/blob/main/CHANGELOG.md
|
|
11
|
+
Keywords: benchmark,calibration,system-one,decision-models,jev,laya,evaluation,llm
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Software Development :: Testing
|
|
21
|
+
Classifier: Operating System :: OS Independent
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.11
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: numpy>=1.26
|
|
27
|
+
Requires-Dist: scipy>=1.11
|
|
28
|
+
Requires-Dist: pandas>=2.0
|
|
29
|
+
Requires-Dist: pydantic>=2.5
|
|
30
|
+
Requires-Dist: httpx>=0.27
|
|
31
|
+
Requires-Dist: pyyaml>=6.0
|
|
32
|
+
Requires-Dist: typer>=0.12
|
|
33
|
+
Provides-Extra: plots
|
|
34
|
+
Requires-Dist: matplotlib>=3.8; extra == "plots"
|
|
35
|
+
Provides-Extra: local
|
|
36
|
+
Requires-Dist: torch>=2.4; extra == "local"
|
|
37
|
+
Requires-Dist: transformers>=4.45; extra == "local"
|
|
38
|
+
Requires-Dist: sentence-transformers>=3.0; extra == "local"
|
|
39
|
+
Provides-Extra: laya
|
|
40
|
+
Requires-Dist: laya>=0.3; extra == "laya"
|
|
41
|
+
Provides-Extra: llm
|
|
42
|
+
Requires-Dist: vllm>=0.6; extra == "llm"
|
|
43
|
+
Requires-Dist: outlines>=0.1; extra == "llm"
|
|
44
|
+
Provides-Extra: dev
|
|
45
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
46
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
47
|
+
Requires-Dist: matplotlib>=3.8; extra == "dev"
|
|
48
|
+
Requires-Dist: build>=1.2; extra == "dev"
|
|
49
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
50
|
+
Dynamic: license-file
|
|
51
|
+
|
|
52
|
+
<p align="center">
|
|
53
|
+
<img src="docs/assets/logo.svg" alt="sys1bench" width="480">
|
|
54
|
+
</p>
|
|
55
|
+
|
|
56
|
+
<p align="center">
|
|
57
|
+
<a href="https://pypi.org/project/sys1bench/"><img alt="PyPI" src="https://img.shields.io/pypi/v/sys1bench?color=4F46E5&label=PyPI&logo=pypi&logoColor=white&cacheSeconds=3600"></a>
|
|
58
|
+
<a href="https://pypi.org/project/sys1bench/"><img alt="Python" src="https://img.shields.io/pypi/pyversions/sys1bench?color=06B6D4&logo=python&logoColor=white&cacheSeconds=3600"></a>
|
|
59
|
+
<a href="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml"><img alt="CI" src="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml/badge.svg"></a>
|
|
60
|
+
<a href="LICENSE"><img alt="License" src="https://img.shields.io/badge/license-Apache--2.0-blue.svg"></a>
|
|
61
|
+
</p>
|
|
62
|
+
|
|
63
|
+
<p align="center"><b>A benchmark for typed System One decision models</b><br>
|
|
64
|
+
Jev · Laya · Kev · and whatever comes next</p>
|
|
65
|
+
|
|
66
|
+
---
|
|
67
|
+
|
|
68
|
+
System One models read a block of state plus typed questions and return a probability distribution over the answers you defined — one of N options, a level on an ordinal scale, or yes/no — in a single forward pass. No text is generated, so nothing has to be parsed. **sys1bench** measures what that buys: whether the probabilities are calibrated (against a noise floor), how much the answer depends on how the question is worded, how confidence behaves out of scope, how accuracy scales with options and state length, whether batched questions interfere, and what it all costs per decision. Headline numbers come from generated items whose labels follow a stated policy, so they cannot be memorised.
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
pip install sys1bench
|
|
72
|
+
sys1bench suite all results/my-model --config my_model.yaml # A–I on generated, policy-labelled data
|
|
73
|
+
sys1bench report results --html results/dashboard.html # tables + interactive dashboard
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## Results at a glance
|
|
77
|
+
|
|
78
|
+
<p align="center">
|
|
79
|
+
<img src="docs/figures/framing_tickets_priority.png" alt="Accuracy across question wordings on the policy-following priority score" width="49%">
|
|
80
|
+
<img src="docs/figures/reliability_tickets_priority.png" alt="Reliability diagram, priority score, all models" width="41%">
|
|
81
|
+
</p>
|
|
82
|
+
<p align="center">
|
|
83
|
+
<img src="docs/figures/cardinality_accuracy.png" alt="Exact accuracy versus number of options" width="45%">
|
|
84
|
+
<img src="docs/figures/latency_vs_questions.png" alt="Median latency versus number of questions per request" width="45%">
|
|
85
|
+
</p>
|
|
86
|
+
|
|
87
|
+
<p align="center"><sub>Jev 1.13.0 (hosted), Laya 0.3.4 and Kev 0.8B / 4B / 9B (local, GB10) on identical generated manifests, n = 500, September 2026. Left to right: accuracy across five paraphrases and criteria variants of the same question (red = canonical wording); reliability on the same question; accuracy as the option count grows from 2 to 255; median latency as more questions share one request. Hosted latency includes the network path and is not comparable to local compute time.</sub></p>
|
|
88
|
+
|
|
89
|
+
## Why this benchmark
|
|
90
|
+
|
|
91
|
+
| | |
|
|
92
|
+
|---|---|
|
|
93
|
+
| **Policy-labelled data** | Labels follow a rule stated in the question (e.g. *urgency + 1 if angry + 1 if premium tier, cap 4*). A regex cannot solve it; neither can memorised public sets. Control arms are paired item-for-item: unknowable, label-noise, distractor, none-of-the-above. |
|
|
94
|
+
| **Calibration relative to noise** | Every ECE is divided by the ECE a perfectly calibrated model would show on that sample. Brier decomposition, clipped NLL, per-primitive temperature refit, quantisation report. |
|
|
95
|
+
| **Wording as a factor** | Accuracy is the median over five paraphrases and three criteria variants, with the range shown; adversarial wordings reported as a worst case. |
|
|
96
|
+
| **Selective prediction and cost** | Risk–coverage, coverage at 5 % risk, abstention with and without an explicit option, and realised cost per 10 k decisions under argmax, Bayes and escalate policies from per-question cost matrices. |
|
|
97
|
+
| **Ordinal, not categorical** | MAE, weighted kappa, ranked probability score, monotonicity along controlled severity ladders, 3/5/10-level scale invariance. |
|
|
98
|
+
| **Honest infrastructure** | Hosted and local latency never share an axis; the serving device is recorded on every row; a local model that lands on CPU when CUDA was requested aborts the run; every response is cached and re-derivable offline. |
|
|
99
|
+
|
|
100
|
+
## Add your model
|
|
101
|
+
|
|
102
|
+
A hosted model with a conventional JSON decisions API needs a config file, no code:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
sys1bench configs example_future_vendor > my_model.yaml # fill in url, auth env var, field names
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Anything else subclasses `BaseAdapter` (declare capabilities, implement `decide`) and registers with `@register("my_model")` or the `sys1bench.adapters` entry-point group from your own package. Kev, which speaks the TypeSafe API, runs with `--config kev_4b` and no code at all.
|
|
109
|
+
|
|
110
|
+
## Documentation
|
|
111
|
+
|
|
112
|
+
- [Specification](docs/SPEC_v2.md) — contract, data tiers, suites A–I, statistics, audits, rules for future models
|
|
113
|
+
- [Design review](docs/REVIEW_v1.md) — how the design was derived from the public state of the art
|
|
114
|
+
- [Changelog](CHANGELOG.md) · [Issues](https://github.com/rssr25/system-one-bench/issues)
|
|
115
|
+
|
|
116
|
+
<p align="center"><sub>Apache 2.0 · Rahul Sharma · numbers change with model versions, so every row carries the version string the provider returned</sub></p>
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="docs/assets/logo.svg" alt="sys1bench" width="480">
|
|
3
|
+
</p>
|
|
4
|
+
|
|
5
|
+
<p align="center">
|
|
6
|
+
<a href="https://pypi.org/project/sys1bench/"><img alt="PyPI" src="https://img.shields.io/pypi/v/sys1bench?color=4F46E5&label=PyPI&logo=pypi&logoColor=white&cacheSeconds=3600"></a>
|
|
7
|
+
<a href="https://pypi.org/project/sys1bench/"><img alt="Python" src="https://img.shields.io/pypi/pyversions/sys1bench?color=06B6D4&logo=python&logoColor=white&cacheSeconds=3600"></a>
|
|
8
|
+
<a href="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml"><img alt="CI" src="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml/badge.svg"></a>
|
|
9
|
+
<a href="LICENSE"><img alt="License" src="https://img.shields.io/badge/license-Apache--2.0-blue.svg"></a>
|
|
10
|
+
</p>
|
|
11
|
+
|
|
12
|
+
<p align="center"><b>A benchmark for typed System One decision models</b><br>
|
|
13
|
+
Jev · Laya · Kev · and whatever comes next</p>
|
|
14
|
+
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
System One models read a block of state plus typed questions and return a probability distribution over the answers you defined — one of N options, a level on an ordinal scale, or yes/no — in a single forward pass. No text is generated, so nothing has to be parsed. **sys1bench** measures what that buys: whether the probabilities are calibrated (against a noise floor), how much the answer depends on how the question is worded, how confidence behaves out of scope, how accuracy scales with options and state length, whether batched questions interfere, and what it all costs per decision. Headline numbers come from generated items whose labels follow a stated policy, so they cannot be memorised.
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install sys1bench
|
|
21
|
+
sys1bench suite all results/my-model --config my_model.yaml # A–I on generated, policy-labelled data
|
|
22
|
+
sys1bench report results --html results/dashboard.html # tables + interactive dashboard
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## Results at a glance
|
|
26
|
+
|
|
27
|
+
<p align="center">
|
|
28
|
+
<img src="docs/figures/framing_tickets_priority.png" alt="Accuracy across question wordings on the policy-following priority score" width="49%">
|
|
29
|
+
<img src="docs/figures/reliability_tickets_priority.png" alt="Reliability diagram, priority score, all models" width="41%">
|
|
30
|
+
</p>
|
|
31
|
+
<p align="center">
|
|
32
|
+
<img src="docs/figures/cardinality_accuracy.png" alt="Exact accuracy versus number of options" width="45%">
|
|
33
|
+
<img src="docs/figures/latency_vs_questions.png" alt="Median latency versus number of questions per request" width="45%">
|
|
34
|
+
</p>
|
|
35
|
+
|
|
36
|
+
<p align="center"><sub>Jev 1.13.0 (hosted), Laya 0.3.4 and Kev 0.8B / 4B / 9B (local, GB10) on identical generated manifests, n = 500, September 2026. Left to right: accuracy across five paraphrases and criteria variants of the same question (red = canonical wording); reliability on the same question; accuracy as the option count grows from 2 to 255; median latency as more questions share one request. Hosted latency includes the network path and is not comparable to local compute time.</sub></p>
|
|
37
|
+
|
|
38
|
+
## Why this benchmark
|
|
39
|
+
|
|
40
|
+
| | |
|
|
41
|
+
|---|---|
|
|
42
|
+
| **Policy-labelled data** | Labels follow a rule stated in the question (e.g. *urgency + 1 if angry + 1 if premium tier, cap 4*). A regex cannot solve it; neither can memorised public sets. Control arms are paired item-for-item: unknowable, label-noise, distractor, none-of-the-above. |
|
|
43
|
+
| **Calibration relative to noise** | Every ECE is divided by the ECE a perfectly calibrated model would show on that sample. Brier decomposition, clipped NLL, per-primitive temperature refit, quantisation report. |
|
|
44
|
+
| **Wording as a factor** | Accuracy is the median over five paraphrases and three criteria variants, with the range shown; adversarial wordings reported as a worst case. |
|
|
45
|
+
| **Selective prediction and cost** | Risk–coverage, coverage at 5 % risk, abstention with and without an explicit option, and realised cost per 10 k decisions under argmax, Bayes and escalate policies from per-question cost matrices. |
|
|
46
|
+
| **Ordinal, not categorical** | MAE, weighted kappa, ranked probability score, monotonicity along controlled severity ladders, 3/5/10-level scale invariance. |
|
|
47
|
+
| **Honest infrastructure** | Hosted and local latency never share an axis; the serving device is recorded on every row; a local model that lands on CPU when CUDA was requested aborts the run; every response is cached and re-derivable offline. |
|
|
48
|
+
|
|
49
|
+
## Add your model
|
|
50
|
+
|
|
51
|
+
A hosted model with a conventional JSON decisions API needs a config file, no code:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
sys1bench configs example_future_vendor > my_model.yaml # fill in url, auth env var, field names
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Anything else subclasses `BaseAdapter` (declare capabilities, implement `decide`) and registers with `@register("my_model")` or the `sys1bench.adapters` entry-point group from your own package. Kev, which speaks the TypeSafe API, runs with `--config kev_4b` and no code at all.
|
|
58
|
+
|
|
59
|
+
## Documentation
|
|
60
|
+
|
|
61
|
+
- [Specification](docs/SPEC_v2.md) — contract, data tiers, suites A–I, statistics, audits, rules for future models
|
|
62
|
+
- [Design review](docs/REVIEW_v1.md) — how the design was derived from the public state of the art
|
|
63
|
+
- [Changelog](CHANGELOG.md) · [Issues](https://github.com/rssr25/system-one-bench/issues)
|
|
64
|
+
|
|
65
|
+
<p align="center"><sub>Apache 2.0 · Rahul Sharma · numbers change with model versions, so every row carries the version string the provider returned</sub></p>
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "sys1bench"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.2"
|
|
4
4
|
description = "Benchmark harness for typed System One decision models (choice / score / noul): calibration vs noise floor, framing sensitivity, selective prediction, ordinal fidelity, interference, robustness."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11"
|
|
@@ -41,7 +41,6 @@ dev = ["pytest>=8.0", "ruff>=0.6", "matplotlib>=3.8", "build>=1.2", "twine>=5.0"
|
|
|
41
41
|
[project.urls]
|
|
42
42
|
Homepage = "https://github.com/rssr25/system-one-bench"
|
|
43
43
|
Documentation = "https://github.com/rssr25/system-one-bench/blob/main/docs/SPEC_v2.md"
|
|
44
|
-
Results = "https://github.com/rssr25/system-one-bench/blob/main/docs/RESULTS_2026-09-22_narrative.md"
|
|
45
44
|
Issues = "https://github.com/rssr25/system-one-bench/issues"
|
|
46
45
|
Changelog = "https://github.com/rssr25/system-one-bench/blob/main/CHANGELOG.md"
|
|
47
46
|
|
|
@@ -59,7 +58,7 @@ build-backend = "setuptools.build_meta"
|
|
|
59
58
|
where = ["src"]
|
|
60
59
|
|
|
61
60
|
[tool.setuptools.package-data]
|
|
62
|
-
"sys1bench.data" = ["framings/*.yaml", "canary/*.jsonl"]
|
|
61
|
+
"sys1bench.data" = ["framings/*.yaml", "canary/*.jsonl", "configs/*.yaml"]
|
|
63
62
|
|
|
64
63
|
[tool.pytest.ini_options]
|
|
65
64
|
testpaths = ["tests"]
|
|
@@ -27,7 +27,7 @@ from typing import Any
|
|
|
27
27
|
import httpx
|
|
28
28
|
|
|
29
29
|
from ..schemas import Answer, DecisionRequest, DecisionResponse, LatencyRecord, ModelCapabilities, ProviderRecord, Question
|
|
30
|
-
from .base import BaseAdapter, finalize_answer, register
|
|
30
|
+
from .base import BaseAdapter, FatalAdapterError, finalize_answer, register
|
|
31
31
|
|
|
32
32
|
DEFAULT_URL = "https://api.typesafe.ai/v1/systemone"
|
|
33
33
|
PRICE_PER_M_INPUT = 0.042
|
|
@@ -38,7 +38,7 @@ def _load_dotenv_key() -> str | None:
|
|
|
38
38
|
for k in ENV_KEYS:
|
|
39
39
|
if os.environ.get(k):
|
|
40
40
|
return os.environ[k]
|
|
41
|
-
for candidate in (Path.cwd() / ".env",
|
|
41
|
+
for candidate in (Path.cwd() / ".env",):
|
|
42
42
|
if candidate.exists():
|
|
43
43
|
for line in candidate.read_text().splitlines():
|
|
44
44
|
line = line.strip()
|
|
@@ -89,17 +89,46 @@ def parse_answer(q: Question, payload: dict[str, Any]) -> Answer:
|
|
|
89
89
|
@register("jev_typesafe")
|
|
90
90
|
class JevTypeSafeAdapter(BaseAdapter):
|
|
91
91
|
def __init__(self, model_id: str = "jev-1.13.0", url: str = DEFAULT_URL, api_key: str | None = None,
|
|
92
|
-
timeout_s: float = 30.0, max_retries: int = 6, allow_alias: bool = False,
|
|
92
|
+
timeout_s: float = 30.0, max_retries: int = 6, allow_alias: bool = False, version_note: str | None = None,
|
|
93
|
+
probe_models: bool = False, deployment: str = "hosted", hardware: str | None = None, **tunables) -> None:
|
|
94
|
+
"""Also serves any TypeSafe-API-compatible server (e.g. Kev at http://127.0.0.1:8009/v1/systemone): set `url`,
|
|
95
|
+
`api_key: local`, `allow_alias: true`, `deployment: local`, and `probe_models: true` to record what the server
|
|
96
|
+
reports at /v1/models (backend, dtype, revision) as the version hash."""
|
|
93
97
|
super().__init__(model_id, **tunables)
|
|
94
98
|
if (model_id.endswith("latest") or model_id.endswith("preview")) and not allow_alias:
|
|
95
99
|
raise ValueError("pin a versioned Jev id (e.g. jev-1.13.0); aliases move silently. Pass allow_alias=True to override.")
|
|
96
100
|
self.url, self.timeout_s, self.max_retries = url, timeout_s, max_retries
|
|
101
|
+
self.deployment, self.hardware = deployment, hardware
|
|
97
102
|
self.api_key = api_key or _load_dotenv_key()
|
|
103
|
+
if not self.api_key:
|
|
104
|
+
raise FatalAdapterError("No TypeSafe API key found. Set TypeSafe_API_KEY (or TYPESAFE_API_KEY) in the environment or in a .env file "
|
|
105
|
+
"in the working directory. Keys: https://typesafe.ai")
|
|
98
106
|
self._client = httpx.Client(timeout=timeout_s, http2=False)
|
|
107
|
+
self.version_note = version_note
|
|
108
|
+
self.model_label: str | None = None
|
|
109
|
+
if probe_models:
|
|
110
|
+
try:
|
|
111
|
+
base = url.split("/v1/")[0]
|
|
112
|
+
r = self._client.get(f"{base}/v1/models", headers={"Authorization": f"Bearer {self.api_key}"})
|
|
113
|
+
r.raise_for_status()
|
|
114
|
+
data = r.json()
|
|
115
|
+
models = data.get("models") or data.get("data") or []
|
|
116
|
+
mine = next((m for m in models if m.get("name") == model_id or m.get("id") == model_id), models[0] if models else {})
|
|
117
|
+
bits = [str(mine.get(k)) for k in ("name", "id", "revision", "run", "backend", "dtype", "release_date") if mine.get(k)]
|
|
118
|
+
bits += [f"{k}={v}" for k, v in data.items() if k not in ("models", "data") and isinstance(v, (str, int, float))]
|
|
119
|
+
self.version_note = " ".join(bits)[:200] or version_note
|
|
120
|
+
self.tunables["server_models"] = self.version_note
|
|
121
|
+
# a server that only exposes an alias (Kev: "kev-latest") is labelled by the run it reports, so that
|
|
122
|
+
# several checkpoints behind the same alias do not merge in reports
|
|
123
|
+
run = mine.get("run") or mine.get("revision")
|
|
124
|
+
if run and (model_id.endswith("latest") or model_id.endswith("preview")):
|
|
125
|
+
self.model_label = str(run).split("/")[-1]
|
|
126
|
+
except Exception as e: # probing is best effort
|
|
127
|
+
self.tunables["server_models_error"] = str(e)[:120]
|
|
99
128
|
|
|
100
129
|
@property
|
|
101
130
|
def capabilities(self) -> ModelCapabilities:
|
|
102
|
-
return ModelCapabilities(name=self.model_id, deployment=
|
|
131
|
+
return ModelCapabilities(name=self.model_id, deployment=self.deployment, max_options=255, max_score_levels=10, # type: ignore[arg-type]
|
|
103
132
|
max_state_tokens=32_000, emits_confidence=True, supports_native_abstain=False,
|
|
104
133
|
supports_batching=True, languages={"en"})
|
|
105
134
|
|
|
@@ -121,6 +150,8 @@ class JevTypeSafeAdapter(BaseAdapter):
|
|
|
121
150
|
except ValueError:
|
|
122
151
|
pass
|
|
123
152
|
raise httpx.HTTPStatusError(f"{r.status_code}: {r.text[:200]}", request=r.request, response=r)
|
|
153
|
+
if r.status_code in (401, 403):
|
|
154
|
+
raise FatalAdapterError(f"TypeSafe API rejected the key (HTTP {r.status_code}). Check TypeSafe_API_KEY.")
|
|
124
155
|
if 400 <= r.status_code < 500 and r.status_code != 429:
|
|
125
156
|
# request rejected (400 over the 32k state limit, 413, 422 malformed): never retry, classify as vendor rejection
|
|
126
157
|
raise ValueError(f"{r.status_code} rejected: {r.text[:300]}")
|
|
@@ -140,6 +171,8 @@ class JevTypeSafeAdapter(BaseAdapter):
|
|
|
140
171
|
"questions": {k: to_vendor_question(q) for k, q in request.questions.items()}}
|
|
141
172
|
try:
|
|
142
173
|
data, ms, headers = self._post(body)
|
|
174
|
+
except FatalAdapterError:
|
|
175
|
+
raise
|
|
143
176
|
except Exception as e:
|
|
144
177
|
kind = "rejected_by_vendor" if isinstance(e, ValueError) else "transport"
|
|
145
178
|
return DecisionResponse(
|
|
@@ -156,9 +189,11 @@ class JevTypeSafeAdapter(BaseAdapter):
|
|
|
156
189
|
return DecisionResponse(
|
|
157
190
|
answers=answers,
|
|
158
191
|
latency=LatencyRecord(client_ms=ms, route="typesafe", timestamp=time.time()),
|
|
159
|
-
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
|
|
160
|
-
|
|
161
|
-
|
|
192
|
+
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
|
|
193
|
+
model_id_returned=self.model_label or data.get("model"),
|
|
194
|
+
version_hash=(str(data.get("model") or self.model_id) + (f" | {self.version_note}" if self.version_note else "")),
|
|
195
|
+
generation_id=headers.get("x-request-id") or headers.get("x-typesafe-request-id") or headers.get("request-id"),
|
|
196
|
+
route="typesafe" if self.url == DEFAULT_URL else self.url, hardware=self.hardware,
|
|
162
197
|
billed_input_tokens=in_tok, billed_output_tokens=usage.get("output_tokens"),
|
|
163
198
|
cost_usd=(in_tok / 1e6 * PRICE_PER_M_INPUT) if in_tok is not None else None),
|
|
164
199
|
raw=data,
|
|
@@ -17,7 +17,7 @@ import yaml
|
|
|
17
17
|
|
|
18
18
|
from . import __version__
|
|
19
19
|
from .adapters import get_adapter, list_adapters
|
|
20
|
-
from .data import framings_path
|
|
20
|
+
from .data import canary_path, config_path, framings_path, list_configs
|
|
21
21
|
from .framing import apply_corruption, expand_framings, permute_options, strip_options, strip_state
|
|
22
22
|
from .framing.expand import load_framings
|
|
23
23
|
from .generators import GENERATORS, get_generator
|
|
@@ -38,14 +38,31 @@ def _main(ctx: typer.Context, version: bool = typer.Option(False, "--version", "
|
|
|
38
38
|
typer.echo(ctx.get_help())
|
|
39
39
|
|
|
40
40
|
|
|
41
|
+
def _fail(msg: str, code: int = 2) -> None:
|
|
42
|
+
typer.secho(f"error: {msg}", fg=typer.colors.RED, err=True)
|
|
43
|
+
raise typer.Exit(code)
|
|
44
|
+
|
|
45
|
+
|
|
41
46
|
def _adapter_from(adapter: str | None, model: str | None, config: str | None):
|
|
47
|
+
from .adapters.base import FatalAdapterError
|
|
48
|
+
|
|
42
49
|
cfg = {}
|
|
43
50
|
if config:
|
|
44
|
-
|
|
45
|
-
|
|
51
|
+
try:
|
|
52
|
+
cfg = yaml.safe_load(config_path(config).read_text()) or {}
|
|
53
|
+
except FileNotFoundError as e:
|
|
54
|
+
_fail(str(e))
|
|
55
|
+
adapter = adapter or cfg.pop("adapter", None)
|
|
56
|
+
if not adapter:
|
|
57
|
+
_fail("no adapter given: pass --adapter <id> or --config <yaml with an `adapter:` key>. Known: " + ", ".join(list_adapters()))
|
|
46
58
|
if model:
|
|
47
59
|
cfg["model_id"] = model
|
|
48
|
-
|
|
60
|
+
try:
|
|
61
|
+
return get_adapter(adapter, **cfg)
|
|
62
|
+
except KeyError as e:
|
|
63
|
+
_fail(str(e).strip('"'))
|
|
64
|
+
except (ValueError, FatalAdapterError) as e:
|
|
65
|
+
_fail(str(e))
|
|
49
66
|
|
|
50
67
|
|
|
51
68
|
@app.command()
|
|
@@ -55,8 +72,19 @@ def adapters():
|
|
|
55
72
|
typer.echo(a)
|
|
56
73
|
|
|
57
74
|
|
|
75
|
+
@app.command()
|
|
76
|
+
def configs(show: Optional[str] = typer.Argument(None, help="print a packaged config by name")):
|
|
77
|
+
"""List packaged example model configs (use with --config <name>), or print one to copy and edit."""
|
|
78
|
+
if show:
|
|
79
|
+
typer.echo(config_path(show).read_text())
|
|
80
|
+
return
|
|
81
|
+
for c in sorted(list_configs()):
|
|
82
|
+
typer.echo(c)
|
|
83
|
+
|
|
84
|
+
|
|
58
85
|
@app.command()
|
|
59
86
|
def generators():
|
|
87
|
+
"""List Tier G generators and their versions."""
|
|
60
88
|
for g in sorted(GENERATORS):
|
|
61
89
|
typer.echo(f"{g}@{GENERATORS[g].version}")
|
|
62
90
|
|
|
@@ -93,14 +121,22 @@ def run(manifest: Path, out: Path, adapter: str | None = None, model: str | None
|
|
|
93
121
|
if permutations:
|
|
94
122
|
items = permute_options(items, permutations)
|
|
95
123
|
c = ResponseCache(cache)
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
rows_all += run_items(
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
124
|
+
from .adapters.base import FatalAdapterError
|
|
125
|
+
|
|
126
|
+
try:
|
|
127
|
+
rows_all += run_items(items, ad, c, suite=suite, arm="main", concurrency=concurrency, progress=True)
|
|
128
|
+
base = [i for i in items if i.permutation_id == "p0" and all(q.framing_id == "f0" for q in i.questions.values())]
|
|
129
|
+
if corruption:
|
|
130
|
+
rows_all += run_items(apply_corruption(base), ad, c, suite=suite, arm="corruption", concurrency=concurrency)
|
|
131
|
+
if short_circuit:
|
|
132
|
+
rows_all += run_items(strip_state(base), ad, c, suite=suite, arm="state_only", concurrency=concurrency)
|
|
133
|
+
rows_all += run_items(strip_options(base), ad, c, suite=suite, arm="options_only", concurrency=concurrency)
|
|
134
|
+
except FatalAdapterError as e:
|
|
135
|
+
_fail(str(e), code=3)
|
|
103
136
|
write_predictions(rows_all, out)
|
|
137
|
+
n_err = sum(1 for r in rows_all if r.error)
|
|
138
|
+
if n_err:
|
|
139
|
+
typer.secho(f"warning: {n_err}/{len(rows_all)} rows have errors (see the `error` field); first: {next(r.error for r in rows_all if r.error)}", fg=typer.colors.YELLOW, err=True)
|
|
104
140
|
typer.echo(f"wrote {len(rows_all)} prediction rows to {out}; cache size {len(c)}")
|
|
105
141
|
|
|
106
142
|
|
|
@@ -124,11 +160,11 @@ def score(predictions: list[Path], out: Path = Path("results/summary.json"), tab
|
|
|
124
160
|
|
|
125
161
|
|
|
126
162
|
@app.command()
|
|
127
|
-
def canary(adapter: str, model: str | None = None, manifest: Path =
|
|
163
|
+
def canary(adapter: str, model: str | None = None, manifest: Path | None = None,
|
|
128
164
|
store: Path = Path("results/canary"), config: Path | None = None):
|
|
129
|
-
"""Run the drift canary for a hosted model."""
|
|
165
|
+
"""Run the drift canary for a hosted model (default: the packaged 200-item set)."""
|
|
130
166
|
ad = _adapter_from(adapter, model, str(config) if config else None)
|
|
131
|
-
res = run_canary(load_manifest(manifest), ad, store)
|
|
167
|
+
res = run_canary(load_manifest(manifest or canary_path()), ad, store)
|
|
132
168
|
typer.echo(json.dumps(res, indent=1))
|
|
133
169
|
if res["drift_suspected"]:
|
|
134
170
|
raise typer.Exit(code=2)
|
|
@@ -259,7 +295,8 @@ def robustness(out: Path, adapter: Optional[str] = None, model: Optional[str] =
|
|
|
259
295
|
|
|
260
296
|
|
|
261
297
|
@app.command()
|
|
262
|
-
def plots(results_dir: Path, out: Optional[Path] = None, questions: str = "tickets.queue,tickets.is_angry,tickets.priority,phish.is_phishing,phish.attack_class,phish.urgency"
|
|
298
|
+
def plots(results_dir: Path, out: Optional[Path] = None, questions: str = "tickets.queue,tickets.is_angry,tickets.priority,phish.is_phishing,phish.attack_class,phish.urgency",
|
|
299
|
+
include: Optional[str] = typer.Option(None, help="comma-separated substrings; only model ids matching one of them are drawn")):
|
|
263
300
|
"""Render reliability diagrams, risk-coverage curves and framing-range charts for every model dir under results_dir."""
|
|
264
301
|
from .report.plots import framing_range_plot, reliability_diagram, risk_coverage_plot
|
|
265
302
|
from .report.scorecard import framing_scorecard
|
|
@@ -279,6 +316,8 @@ def plots(results_dir: Path, out: Optional[Path] = None, questions: str = "ticke
|
|
|
279
316
|
if not rows:
|
|
280
317
|
continue
|
|
281
318
|
label = rows[0].model_id
|
|
319
|
+
if include and not any(tok.strip().lower() in label.lower() for tok in include.split(",")):
|
|
320
|
+
continue
|
|
282
321
|
by_model[label] = [r for r in rows if r.permutation_id == "p0" and r.framing_id == "f0"]
|
|
283
322
|
acc_by_model[label] = framing_scorecard(rows)["accuracy_by_framing"]
|
|
284
323
|
if not by_model:
|
|
@@ -22,10 +22,10 @@ def framings_path(name_or_path: str | Path) -> Path:
|
|
|
22
22
|
p = Path(name_or_path)
|
|
23
23
|
if p.exists():
|
|
24
24
|
return p
|
|
25
|
-
stem
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
25
|
+
for stem in (str(name_or_path), p.stem): # names may contain dots (jev_1.13), so try the full name first
|
|
26
|
+
cand = _pkg_file("framings", f"{stem}.yaml")
|
|
27
|
+
if cand.exists():
|
|
28
|
+
return cand
|
|
29
29
|
raise FileNotFoundError(f"no framing set {name_or_path!r}; packaged: {sorted(list_framings())}")
|
|
30
30
|
|
|
31
31
|
|
|
@@ -35,3 +35,20 @@ def list_framings() -> list[str]:
|
|
|
35
35
|
|
|
36
36
|
def canary_path() -> Path:
|
|
37
37
|
return _pkg_file("canary", "canary.jsonl")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def config_path(name_or_path: str | Path) -> Path:
|
|
41
|
+
"""Resolve a model config: an existing path, or a packaged example in `configs/<name>.yaml`
|
|
42
|
+
(jev_1.13, jev_1.13_openrouter, laya_en, laya_typed_decisions, example_future_vendor, hybrid_mock)."""
|
|
43
|
+
p = Path(name_or_path)
|
|
44
|
+
if p.exists():
|
|
45
|
+
return p
|
|
46
|
+
for stem in (str(name_or_path), p.stem):
|
|
47
|
+
cand = _pkg_file("configs", f"{stem}.yaml")
|
|
48
|
+
if cand.exists():
|
|
49
|
+
return cand
|
|
50
|
+
raise FileNotFoundError(f"no config {name_or_path!r}; packaged: {sorted(list_configs())}")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def list_configs() -> list[str]:
|
|
54
|
+
return [p.stem for p in Path(str(resources.files(_PKG).joinpath("configs"))).glob("*.yaml")]
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Template for a new hosted System One model: no code needed if its JSON shape is conventional.
|
|
2
|
+
adapter: generic_http
|
|
3
|
+
model_id: vendor/decider-1.0
|
|
4
|
+
url: https://api.vendor.example/v1/decisions
|
|
5
|
+
auth_env: VENDOR_API_KEY
|
|
6
|
+
type_map: {choice: choice, score: score, noul: boolean}
|
|
7
|
+
body_template: {"model": "{model_id}", "state": "{state}", "questions": "{questions}"}
|
|
8
|
+
answers_path: [answers]
|
|
9
|
+
fields:
|
|
10
|
+
probs: [[probabilities], [distribution]]
|
|
11
|
+
p_true: [[probability]]
|
|
12
|
+
confidence: [[confidence]]
|
|
13
|
+
model_returned: [[model]]
|
|
14
|
+
generation_id: [[id]]
|
|
15
|
+
input_tokens: [[usage, input_tokens]]
|
|
16
|
+
price_per_m_input: 0.05
|
|
17
|
+
max_options: 255
|
|
18
|
+
max_state_tokens: 32000
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# Kev (github.com/jaredpalmer/kev): a Jev-compatible decision model served locally.
|
|
2
|
+
# uv run --extra serve python -m kev.serve --run jaredpalmer/kev-0.8b --port 8009
|
|
3
|
+
adapter: jev_typesafe
|
|
4
|
+
model_id: kev-latest # Kev's only served name; the revision is recorded from /v1/models
|
|
5
|
+
url: http://127.0.0.1:8009/v1/systemone
|
|
6
|
+
api_key: local
|
|
7
|
+
allow_alias: true
|
|
8
|
+
probe_models: true
|
|
9
|
+
deployment: local
|
|
10
|
+
version_note: jaredpalmer/kev-0.8b
|
|
11
|
+
timeout_s: 120
|
|
12
|
+
max_retries: 3
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# Kev (github.com/jaredpalmer/kev): a Jev-compatible decision model served locally.
|
|
2
|
+
# uv run --extra serve python -m kev.serve --run jaredpalmer/kev-4b --port 8009
|
|
3
|
+
adapter: jev_typesafe
|
|
4
|
+
model_id: kev-latest # Kev's only served name; the revision is recorded from /v1/models
|
|
5
|
+
url: http://127.0.0.1:8009/v1/systemone
|
|
6
|
+
api_key: local
|
|
7
|
+
allow_alias: true
|
|
8
|
+
probe_models: true
|
|
9
|
+
deployment: local
|
|
10
|
+
version_note: jaredpalmer/kev-4b
|
|
11
|
+
timeout_s: 120
|
|
12
|
+
max_retries: 3
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# Kev (github.com/jaredpalmer/kev): a Jev-compatible decision model served locally.
|
|
2
|
+
# uv run --extra serve python -m kev.serve --run jaredpalmer/kev-9b --port 8009
|
|
3
|
+
adapter: jev_typesafe
|
|
4
|
+
model_id: kev-latest # Kev's only served name; the revision is recorded from /v1/models
|
|
5
|
+
url: http://127.0.0.1:8009/v1/systemone
|
|
6
|
+
api_key: local
|
|
7
|
+
allow_alias: true
|
|
8
|
+
probe_models: true
|
|
9
|
+
deployment: local
|
|
10
|
+
version_note: jaredpalmer/kev-9b
|
|
11
|
+
timeout_s: 120
|
|
12
|
+
max_retries: 3
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
adapter: laya_local
|
|
2
|
+
model_id: convaiinnovations/laya
|
|
3
|
+
checkpoint: english # english | multilingual | typed-decisions | router
|
|
4
|
+
device: cuda # falls back to CPU if CUDA is unavailable; the device used is recorded per row
|
|
5
|
+
head_max_len: 192 # Suite C sweeps [64,128,192,256,384,512]
|
|
6
|
+
max_len: 512
|
|
@@ -92,7 +92,7 @@ def framing_range_plot(acc_by_model_framing: dict[str, dict[str, float]], out: P
|
|
|
92
92
|
ax.plot(v, i, marker="o" if f == "f0" else ".", color="#c33" if f == "f0" else "#333", ms=7 if f == "f0" else 5)
|
|
93
93
|
ax.set_yticks(range(len(models)))
|
|
94
94
|
ax.set_yticklabels(models, fontsize=8)
|
|
95
|
-
ax.set_xlabel("accuracy
|
|
95
|
+
ax.set_xlabel("accuracy (red: canonical wording, dots: other framings)", fontsize=8)
|
|
96
96
|
ax.set_title(title, fontsize=10)
|
|
97
97
|
ax.set_xlim(0, 1.02)
|
|
98
98
|
fig.tight_layout()
|
|
@@ -141,3 +141,28 @@ def interference_heatmap(res: dict[str, Any], out: Path, metric: str = "mean_jsd
|
|
|
141
141
|
fig.savefig(out, dpi=160)
|
|
142
142
|
plt.close(fig)
|
|
143
143
|
return out
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def multi_line_plot(series: dict[str, list[tuple[float, float]]], out: Path, xlabel: str, ylabel: str, title: str = "",
|
|
147
|
+
xlog: bool = False, ylim: tuple[float, float] | None = None) -> Path:
|
|
148
|
+
"""Several named series on one axis; used for cross-model sweep comparisons in the README."""
|
|
149
|
+
plt = _plt()
|
|
150
|
+
fig, ax = plt.subplots(figsize=(5.2, 3.4))
|
|
151
|
+
for name, pts in series.items():
|
|
152
|
+
pts = sorted(p for p in pts if p[1] == p[1])
|
|
153
|
+
if pts:
|
|
154
|
+
ax.plot([p[0] for p in pts], [p[1] for p in pts], marker="o", ms=4, lw=1.6, label=name)
|
|
155
|
+
if xlog:
|
|
156
|
+
ax.set_xscale("log")
|
|
157
|
+
if ylim:
|
|
158
|
+
ax.set_ylim(*ylim)
|
|
159
|
+
ax.set_xlabel(xlabel)
|
|
160
|
+
ax.set_ylabel(ylabel)
|
|
161
|
+
ax.set_title(title, fontsize=10)
|
|
162
|
+
ax.grid(alpha=0.25)
|
|
163
|
+
ax.legend(fontsize=7, frameon=False)
|
|
164
|
+
fig.tight_layout()
|
|
165
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
166
|
+
fig.savefig(out, dpi=170)
|
|
167
|
+
plt.close(fig)
|
|
168
|
+
return out
|