sys1bench 0.3.1__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. sys1bench-0.3.2/PKG-INFO +116 -0
  2. sys1bench-0.3.2/README.md +65 -0
  3. {sys1bench-0.3.1 → sys1bench-0.3.2}/pyproject.toml +1 -2
  4. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/__init__.py +1 -1
  5. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/jev_typesafe.py +33 -5
  6. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/cli.py +4 -1
  7. sys1bench-0.3.2/src/sys1bench/data/configs/kev_0.8b.yaml +12 -0
  8. sys1bench-0.3.2/src/sys1bench/data/configs/kev_4b.yaml +12 -0
  9. sys1bench-0.3.2/src/sys1bench/data/configs/kev_9b.yaml +12 -0
  10. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/report/plots.py +26 -1
  11. sys1bench-0.3.2/src/sys1bench.egg-info/PKG-INFO +116 -0
  12. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench.egg-info/SOURCES.txt +3 -0
  13. sys1bench-0.3.1/PKG-INFO +0 -195
  14. sys1bench-0.3.1/README.md +0 -143
  15. sys1bench-0.3.1/src/sys1bench.egg-info/PKG-INFO +0 -195
  16. {sys1bench-0.3.1 → sys1bench-0.3.2}/LICENSE +0 -0
  17. {sys1bench-0.3.1 → sys1bench-0.3.2}/setup.cfg +0 -0
  18. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/__init__.py +0 -0
  19. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/base.py +0 -0
  20. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/embed_knn.py +0 -0
  21. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/encoder_finetuned.py +0 -0
  22. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/generic_http.py +0 -0
  23. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/hybrid_router.py +0 -0
  24. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/jev_openrouter.py +0 -0
  25. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/laya_local.py +0 -0
  26. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/llm_constrained.py +0 -0
  27. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/majority_prior.py +0 -0
  28. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/mock.py +0 -0
  29. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/nli_zeroshot.py +0 -0
  30. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/adapters/regex_keyword.py +0 -0
  31. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/analysis/__init__.py +0 -0
  32. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/analysis/arms.py +0 -0
  33. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/analysis/decision_value.py +0 -0
  34. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/analysis/decomposition.py +0 -0
  35. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/analysis/meta_eval.py +0 -0
  36. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/analysis/stats.py +0 -0
  37. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/__init__.py +0 -0
  38. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/canary/canary.jsonl +0 -0
  39. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/configs/example_future_vendor.yaml +0 -0
  40. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/configs/hybrid_mock.yaml +0 -0
  41. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/configs/jev_1.13.yaml +0 -0
  42. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/configs/jev_1.13_openrouter.yaml +0 -0
  43. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/configs/laya_en.yaml +0 -0
  44. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/configs/laya_typed_decisions.yaml +0 -0
  45. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/framings/guardrail_intent.yaml +0 -0
  46. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/framings/log_triage.yaml +0 -0
  47. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/framings/phishing_email.yaml +0 -0
  48. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/framings/policy_compliance.yaml +0 -0
  49. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/data/framings/support_tickets.yaml +0 -0
  50. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/framing/__init__.py +0 -0
  51. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/framing/expand.py +0 -0
  52. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/framing/perturb.py +0 -0
  53. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/__init__.py +0 -0
  54. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/base.py +0 -0
  55. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/guardrail_intent.py +0 -0
  56. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/log_triage.py +0 -0
  57. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/multilingual_tickets.py +0 -0
  58. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/phishing_email.py +0 -0
  59. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/policy_compliance.py +0 -0
  60. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/rag_relevance.py +0 -0
  61. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/generators/support_tickets.py +0 -0
  62. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/metrics/__init__.py +0 -0
  63. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/metrics/calibration.py +0 -0
  64. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/metrics/consistency.py +0 -0
  65. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/metrics/decision_value.py +0 -0
  66. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/metrics/efficiency.py +0 -0
  67. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/metrics/ordinal.py +0 -0
  68. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/metrics/robustness.py +0 -0
  69. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/metrics/selective.py +0 -0
  70. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/report/__init__.py +0 -0
  71. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/report/dashboard.py +0 -0
  72. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/report/latex.py +0 -0
  73. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/report/results_doc.py +0 -0
  74. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/report/scorecard.py +0 -0
  75. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/__init__.py +0 -0
  76. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/benchmark_runner.py +0 -0
  77. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/cache.py +0 -0
  78. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/canary.py +0 -0
  79. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/hybrid_sweep.py +0 -0
  80. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/noul_consistency.py +0 -0
  81. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/ordinal_probes.py +0 -0
  82. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/robustness.py +0 -0
  83. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/suite.py +0 -0
  84. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/runners/sweeps.py +0 -0
  85. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench/schemas.py +0 -0
  86. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench.egg-info/dependency_links.txt +0 -0
  87. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench.egg-info/entry_points.txt +0 -0
  88. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench.egg-info/requires.txt +0 -0
  89. {sys1bench-0.3.1 → sys1bench-0.3.2}/src/sys1bench.egg-info/top_level.txt +0 -0
  90. {sys1bench-0.3.1 → sys1bench-0.3.2}/tests/test_generators_framing.py +0 -0
  91. {sys1bench-0.3.1 → sys1bench-0.3.2}/tests/test_metrics.py +0 -0
  92. {sys1bench-0.3.1 → sys1bench-0.3.2}/tests/test_runner_end_to_end.py +0 -0
@@ -0,0 +1,116 @@
1
+ Metadata-Version: 2.4
2
+ Name: sys1bench
3
+ Version: 0.3.2
4
+ Summary: Benchmark harness for typed System One decision models (choice / score / noul): calibration vs noise floor, framing sensitivity, selective prediction, ordinal fidelity, interference, robustness.
5
+ Author-email: Rahul Sharma <rsharma@rptu.de>
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Homepage, https://github.com/rssr25/system-one-bench
8
+ Project-URL: Documentation, https://github.com/rssr25/system-one-bench/blob/main/docs/SPEC_v2.md
9
+ Project-URL: Issues, https://github.com/rssr25/system-one-bench/issues
10
+ Project-URL: Changelog, https://github.com/rssr25/system-one-bench/blob/main/CHANGELOG.md
11
+ Keywords: benchmark,calibration,system-one,decision-models,jev,laya,evaluation,llm
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Software Development :: Testing
21
+ Classifier: Operating System :: OS Independent
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.11
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: numpy>=1.26
27
+ Requires-Dist: scipy>=1.11
28
+ Requires-Dist: pandas>=2.0
29
+ Requires-Dist: pydantic>=2.5
30
+ Requires-Dist: httpx>=0.27
31
+ Requires-Dist: pyyaml>=6.0
32
+ Requires-Dist: typer>=0.12
33
+ Provides-Extra: plots
34
+ Requires-Dist: matplotlib>=3.8; extra == "plots"
35
+ Provides-Extra: local
36
+ Requires-Dist: torch>=2.4; extra == "local"
37
+ Requires-Dist: transformers>=4.45; extra == "local"
38
+ Requires-Dist: sentence-transformers>=3.0; extra == "local"
39
+ Provides-Extra: laya
40
+ Requires-Dist: laya>=0.3; extra == "laya"
41
+ Provides-Extra: llm
42
+ Requires-Dist: vllm>=0.6; extra == "llm"
43
+ Requires-Dist: outlines>=0.1; extra == "llm"
44
+ Provides-Extra: dev
45
+ Requires-Dist: pytest>=8.0; extra == "dev"
46
+ Requires-Dist: ruff>=0.6; extra == "dev"
47
+ Requires-Dist: matplotlib>=3.8; extra == "dev"
48
+ Requires-Dist: build>=1.2; extra == "dev"
49
+ Requires-Dist: twine>=5.0; extra == "dev"
50
+ Dynamic: license-file
51
+
52
+ <p align="center">
53
+ <img src="docs/assets/logo.svg" alt="sys1bench" width="480">
54
+ </p>
55
+
56
+ <p align="center">
57
+ <a href="https://pypi.org/project/sys1bench/"><img alt="PyPI" src="https://img.shields.io/pypi/v/sys1bench?color=4F46E5&label=PyPI&logo=pypi&logoColor=white&cacheSeconds=3600"></a>
58
+ <a href="https://pypi.org/project/sys1bench/"><img alt="Python" src="https://img.shields.io/pypi/pyversions/sys1bench?color=06B6D4&logo=python&logoColor=white&cacheSeconds=3600"></a>
59
+ <a href="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml"><img alt="CI" src="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml/badge.svg"></a>
60
+ <a href="LICENSE"><img alt="License" src="https://img.shields.io/badge/license-Apache--2.0-blue.svg"></a>
61
+ </p>
62
+
63
+ <p align="center"><b>A benchmark for typed System One decision models</b><br>
64
+ Jev · Laya · Kev · and whatever comes next</p>
65
+
66
+ ---
67
+
68
+ System One models read a block of state plus typed questions and return a probability distribution over the answers you defined — one of N options, a level on an ordinal scale, or yes/no — in a single forward pass. No text is generated, so nothing has to be parsed. **sys1bench** measures what that buys: whether the probabilities are calibrated (against a noise floor), how much the answer depends on how the question is worded, how confidence behaves out of scope, how accuracy scales with options and state length, whether batched questions interfere, and what it all costs per decision. Headline numbers come from generated items whose labels follow a stated policy, so they cannot be memorised.
69
+
70
+ ```bash
71
+ pip install sys1bench
72
+ sys1bench suite all results/my-model --config my_model.yaml # A–I on generated, policy-labelled data
73
+ sys1bench report results --html results/dashboard.html # tables + interactive dashboard
74
+ ```
75
+
76
+ ## Results at a glance
77
+
78
+ <p align="center">
79
+ <img src="docs/figures/framing_tickets_priority.png" alt="Accuracy across question wordings on the policy-following priority score" width="49%">
80
+ <img src="docs/figures/reliability_tickets_priority.png" alt="Reliability diagram, priority score, all models" width="41%">
81
+ </p>
82
+ <p align="center">
83
+ <img src="docs/figures/cardinality_accuracy.png" alt="Exact accuracy versus number of options" width="45%">
84
+ <img src="docs/figures/latency_vs_questions.png" alt="Median latency versus number of questions per request" width="45%">
85
+ </p>
86
+
87
+ <p align="center"><sub>Jev 1.13.0 (hosted), Laya 0.3.4 and Kev 0.8B / 4B / 9B (local, GB10) on identical generated manifests, n = 500, September 2026. Left to right: accuracy across five paraphrases and criteria variants of the same question (red = canonical wording); reliability on the same question; accuracy as the option count grows from 2 to 255; median latency as more questions share one request. Hosted latency includes the network path and is not comparable to local compute time.</sub></p>
88
+
89
+ ## Why this benchmark
90
+
91
+ | | |
92
+ |---|---|
93
+ | **Policy-labelled data** | Labels follow a rule stated in the question (e.g. *urgency + 1 if angry + 1 if premium tier, cap 4*). A regex cannot solve it; neither can memorised public sets. Control arms are paired item-for-item: unknowable, label-noise, distractor, none-of-the-above. |
94
+ | **Calibration relative to noise** | Every ECE is divided by the ECE a perfectly calibrated model would show on that sample. Brier decomposition, clipped NLL, per-primitive temperature refit, quantisation report. |
95
+ | **Wording as a factor** | Accuracy is the median over five paraphrases and three criteria variants, with the range shown; adversarial wordings reported as a worst case. |
96
+ | **Selective prediction and cost** | Risk–coverage, coverage at 5 % risk, abstention with and without an explicit option, and realised cost per 10 k decisions under argmax, Bayes and escalate policies from per-question cost matrices. |
97
+ | **Ordinal, not categorical** | MAE, weighted kappa, ranked probability score, monotonicity along controlled severity ladders, 3/5/10-level scale invariance. |
98
+ | **Honest infrastructure** | Hosted and local latency never share an axis; the serving device is recorded on every row; a local model that lands on CPU when CUDA was requested aborts the run; every response is cached and re-derivable offline. |
99
+
100
+ ## Add your model
101
+
102
+ A hosted model with a conventional JSON decisions API needs a config file, no code:
103
+
104
+ ```bash
105
+ sys1bench configs example_future_vendor > my_model.yaml # fill in url, auth env var, field names
106
+ ```
107
+
108
+ Anything else subclasses `BaseAdapter` (declare capabilities, implement `decide`) and registers with `@register("my_model")` or the `sys1bench.adapters` entry-point group from your own package. Kev, which speaks the TypeSafe API, runs with `--config kev_4b` and no code at all.
109
+
110
+ ## Documentation
111
+
112
+ - [Specification](docs/SPEC_v2.md) — contract, data tiers, suites A–I, statistics, audits, rules for future models
113
+ - [Design review](docs/REVIEW_v1.md) — how the design was derived from the public state of the art
114
+ - [Changelog](CHANGELOG.md) · [Issues](https://github.com/rssr25/system-one-bench/issues)
115
+
116
+ <p align="center"><sub>Apache 2.0 · Rahul Sharma · numbers change with model versions, so every row carries the version string the provider returned</sub></p>
@@ -0,0 +1,65 @@
1
+ <p align="center">
2
+ <img src="docs/assets/logo.svg" alt="sys1bench" width="480">
3
+ </p>
4
+
5
+ <p align="center">
6
+ <a href="https://pypi.org/project/sys1bench/"><img alt="PyPI" src="https://img.shields.io/pypi/v/sys1bench?color=4F46E5&label=PyPI&logo=pypi&logoColor=white&cacheSeconds=3600"></a>
7
+ <a href="https://pypi.org/project/sys1bench/"><img alt="Python" src="https://img.shields.io/pypi/pyversions/sys1bench?color=06B6D4&logo=python&logoColor=white&cacheSeconds=3600"></a>
8
+ <a href="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml"><img alt="CI" src="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml/badge.svg"></a>
9
+ <a href="LICENSE"><img alt="License" src="https://img.shields.io/badge/license-Apache--2.0-blue.svg"></a>
10
+ </p>
11
+
12
+ <p align="center"><b>A benchmark for typed System One decision models</b><br>
13
+ Jev · Laya · Kev · and whatever comes next</p>
14
+
15
+ ---
16
+
17
+ System One models read a block of state plus typed questions and return a probability distribution over the answers you defined — one of N options, a level on an ordinal scale, or yes/no — in a single forward pass. No text is generated, so nothing has to be parsed. **sys1bench** measures what that buys: whether the probabilities are calibrated (against a noise floor), how much the answer depends on how the question is worded, how confidence behaves out of scope, how accuracy scales with options and state length, whether batched questions interfere, and what it all costs per decision. Headline numbers come from generated items whose labels follow a stated policy, so they cannot be memorised.
18
+
19
+ ```bash
20
+ pip install sys1bench
21
+ sys1bench suite all results/my-model --config my_model.yaml # A–I on generated, policy-labelled data
22
+ sys1bench report results --html results/dashboard.html # tables + interactive dashboard
23
+ ```
24
+
25
+ ## Results at a glance
26
+
27
+ <p align="center">
28
+ <img src="docs/figures/framing_tickets_priority.png" alt="Accuracy across question wordings on the policy-following priority score" width="49%">
29
+ <img src="docs/figures/reliability_tickets_priority.png" alt="Reliability diagram, priority score, all models" width="41%">
30
+ </p>
31
+ <p align="center">
32
+ <img src="docs/figures/cardinality_accuracy.png" alt="Exact accuracy versus number of options" width="45%">
33
+ <img src="docs/figures/latency_vs_questions.png" alt="Median latency versus number of questions per request" width="45%">
34
+ </p>
35
+
36
+ <p align="center"><sub>Jev 1.13.0 (hosted), Laya 0.3.4 and Kev 0.8B / 4B / 9B (local, GB10) on identical generated manifests, n = 500, September 2026. Left to right: accuracy across five paraphrases and criteria variants of the same question (red = canonical wording); reliability on the same question; accuracy as the option count grows from 2 to 255; median latency as more questions share one request. Hosted latency includes the network path and is not comparable to local compute time.</sub></p>
37
+
38
+ ## Why this benchmark
39
+
40
+ | | |
41
+ |---|---|
42
+ | **Policy-labelled data** | Labels follow a rule stated in the question (e.g. *urgency + 1 if angry + 1 if premium tier, cap 4*). A regex cannot solve it; neither can memorised public sets. Control arms are paired item-for-item: unknowable, label-noise, distractor, none-of-the-above. |
43
+ | **Calibration relative to noise** | Every ECE is divided by the ECE a perfectly calibrated model would show on that sample. Brier decomposition, clipped NLL, per-primitive temperature refit, quantisation report. |
44
+ | **Wording as a factor** | Accuracy is the median over five paraphrases and three criteria variants, with the range shown; adversarial wordings reported as a worst case. |
45
+ | **Selective prediction and cost** | Risk–coverage, coverage at 5 % risk, abstention with and without an explicit option, and realised cost per 10 k decisions under argmax, Bayes and escalate policies from per-question cost matrices. |
46
+ | **Ordinal, not categorical** | MAE, weighted kappa, ranked probability score, monotonicity along controlled severity ladders, 3/5/10-level scale invariance. |
47
+ | **Honest infrastructure** | Hosted and local latency never share an axis; the serving device is recorded on every row; a local model that lands on CPU when CUDA was requested aborts the run; every response is cached and re-derivable offline. |
48
+
49
+ ## Add your model
50
+
51
+ A hosted model with a conventional JSON decisions API needs a config file, no code:
52
+
53
+ ```bash
54
+ sys1bench configs example_future_vendor > my_model.yaml # fill in url, auth env var, field names
55
+ ```
56
+
57
+ Anything else subclasses `BaseAdapter` (declare capabilities, implement `decide`) and registers with `@register("my_model")` or the `sys1bench.adapters` entry-point group from your own package. Kev, which speaks the TypeSafe API, runs with `--config kev_4b` and no code at all.
58
+
59
+ ## Documentation
60
+
61
+ - [Specification](docs/SPEC_v2.md) — contract, data tiers, suites A–I, statistics, audits, rules for future models
62
+ - [Design review](docs/REVIEW_v1.md) — how the design was derived from the public state of the art
63
+ - [Changelog](CHANGELOG.md) · [Issues](https://github.com/rssr25/system-one-bench/issues)
64
+
65
+ <p align="center"><sub>Apache 2.0 · Rahul Sharma · numbers change with model versions, so every row carries the version string the provider returned</sub></p>
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sys1bench"
3
- version = "0.3.1"
3
+ version = "0.3.2"
4
4
  description = "Benchmark harness for typed System One decision models (choice / score / noul): calibration vs noise floor, framing sensitivity, selective prediction, ordinal fidelity, interference, robustness."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -41,7 +41,6 @@ dev = ["pytest>=8.0", "ruff>=0.6", "matplotlib>=3.8", "build>=1.2", "twine>=5.0"
41
41
  [project.urls]
42
42
  Homepage = "https://github.com/rssr25/system-one-bench"
43
43
  Documentation = "https://github.com/rssr25/system-one-bench/blob/main/docs/SPEC_v2.md"
44
- Results = "https://github.com/rssr25/system-one-bench/blob/main/docs/RESULTS_2026-09-22_narrative.md"
45
44
  Issues = "https://github.com/rssr25/system-one-bench/issues"
46
45
  Changelog = "https://github.com/rssr25/system-one-bench/blob/main/CHANGELOG.md"
47
46
 
@@ -1,3 +1,3 @@
1
1
  """sys1bench: benchmark harness for typed System One decision models."""
2
2
 
3
- __version__ = "0.3.1"
3
+ __version__ = "0.3.2"
@@ -89,20 +89,46 @@ def parse_answer(q: Question, payload: dict[str, Any]) -> Answer:
89
89
  @register("jev_typesafe")
90
90
  class JevTypeSafeAdapter(BaseAdapter):
91
91
  def __init__(self, model_id: str = "jev-1.13.0", url: str = DEFAULT_URL, api_key: str | None = None,
92
- timeout_s: float = 30.0, max_retries: int = 6, allow_alias: bool = False, **tunables) -> None:
92
+ timeout_s: float = 30.0, max_retries: int = 6, allow_alias: bool = False, version_note: str | None = None,
93
+ probe_models: bool = False, deployment: str = "hosted", hardware: str | None = None, **tunables) -> None:
94
+ """Also serves any TypeSafe-API-compatible server (e.g. Kev at http://127.0.0.1:8009/v1/systemone): set `url`,
95
+ `api_key: local`, `allow_alias: true`, `deployment: local`, and `probe_models: true` to record what the server
96
+ reports at /v1/models (backend, dtype, revision) as the version hash."""
93
97
  super().__init__(model_id, **tunables)
94
98
  if (model_id.endswith("latest") or model_id.endswith("preview")) and not allow_alias:
95
99
  raise ValueError("pin a versioned Jev id (e.g. jev-1.13.0); aliases move silently. Pass allow_alias=True to override.")
96
100
  self.url, self.timeout_s, self.max_retries = url, timeout_s, max_retries
101
+ self.deployment, self.hardware = deployment, hardware
97
102
  self.api_key = api_key or _load_dotenv_key()
98
103
  if not self.api_key:
99
104
  raise FatalAdapterError("No TypeSafe API key found. Set TypeSafe_API_KEY (or TYPESAFE_API_KEY) in the environment or in a .env file "
100
105
  "in the working directory. Keys: https://typesafe.ai")
101
106
  self._client = httpx.Client(timeout=timeout_s, http2=False)
107
+ self.version_note = version_note
108
+ self.model_label: str | None = None
109
+ if probe_models:
110
+ try:
111
+ base = url.split("/v1/")[0]
112
+ r = self._client.get(f"{base}/v1/models", headers={"Authorization": f"Bearer {self.api_key}"})
113
+ r.raise_for_status()
114
+ data = r.json()
115
+ models = data.get("models") or data.get("data") or []
116
+ mine = next((m for m in models if m.get("name") == model_id or m.get("id") == model_id), models[0] if models else {})
117
+ bits = [str(mine.get(k)) for k in ("name", "id", "revision", "run", "backend", "dtype", "release_date") if mine.get(k)]
118
+ bits += [f"{k}={v}" for k, v in data.items() if k not in ("models", "data") and isinstance(v, (str, int, float))]
119
+ self.version_note = " ".join(bits)[:200] or version_note
120
+ self.tunables["server_models"] = self.version_note
121
+ # a server that only exposes an alias (Kev: "kev-latest") is labelled by the run it reports, so that
122
+ # several checkpoints behind the same alias do not merge in reports
123
+ run = mine.get("run") or mine.get("revision")
124
+ if run and (model_id.endswith("latest") or model_id.endswith("preview")):
125
+ self.model_label = str(run).split("/")[-1]
126
+ except Exception as e: # probing is best effort
127
+ self.tunables["server_models_error"] = str(e)[:120]
102
128
 
103
129
  @property
104
130
  def capabilities(self) -> ModelCapabilities:
105
- return ModelCapabilities(name=self.model_id, deployment="hosted", max_options=255, max_score_levels=10,
131
+ return ModelCapabilities(name=self.model_id, deployment=self.deployment, max_options=255, max_score_levels=10, # type: ignore[arg-type]
106
132
  max_state_tokens=32_000, emits_confidence=True, supports_native_abstain=False,
107
133
  supports_batching=True, languages={"en"})
108
134
 
@@ -163,9 +189,11 @@ class JevTypeSafeAdapter(BaseAdapter):
163
189
  return DecisionResponse(
164
190
  answers=answers,
165
191
  latency=LatencyRecord(client_ms=ms, route="typesafe", timestamp=time.time()),
166
- provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, model_id_returned=data.get("model"),
167
- version_hash=str(data.get("model") or self.model_id),
168
- generation_id=headers.get("x-request-id") or headers.get("request-id"), route="typesafe",
192
+ provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
193
+ model_id_returned=self.model_label or data.get("model"),
194
+ version_hash=(str(data.get("model") or self.model_id) + (f" | {self.version_note}" if self.version_note else "")),
195
+ generation_id=headers.get("x-request-id") or headers.get("x-typesafe-request-id") or headers.get("request-id"),
196
+ route="typesafe" if self.url == DEFAULT_URL else self.url, hardware=self.hardware,
169
197
  billed_input_tokens=in_tok, billed_output_tokens=usage.get("output_tokens"),
170
198
  cost_usd=(in_tok / 1e6 * PRICE_PER_M_INPUT) if in_tok is not None else None),
171
199
  raw=data,
@@ -295,7 +295,8 @@ def robustness(out: Path, adapter: Optional[str] = None, model: Optional[str] =
295
295
 
296
296
 
297
297
  @app.command()
298
- def plots(results_dir: Path, out: Optional[Path] = None, questions: str = "tickets.queue,tickets.is_angry,tickets.priority,phish.is_phishing,phish.attack_class,phish.urgency"):
298
+ def plots(results_dir: Path, out: Optional[Path] = None, questions: str = "tickets.queue,tickets.is_angry,tickets.priority,phish.is_phishing,phish.attack_class,phish.urgency",
299
+ include: Optional[str] = typer.Option(None, help="comma-separated substrings; only model ids matching one of them are drawn")):
299
300
  """Render reliability diagrams, risk-coverage curves and framing-range charts for every model dir under results_dir."""
300
301
  from .report.plots import framing_range_plot, reliability_diagram, risk_coverage_plot
301
302
  from .report.scorecard import framing_scorecard
@@ -315,6 +316,8 @@ def plots(results_dir: Path, out: Optional[Path] = None, questions: str = "ticke
315
316
  if not rows:
316
317
  continue
317
318
  label = rows[0].model_id
319
+ if include and not any(tok.strip().lower() in label.lower() for tok in include.split(",")):
320
+ continue
318
321
  by_model[label] = [r for r in rows if r.permutation_id == "p0" and r.framing_id == "f0"]
319
322
  acc_by_model[label] = framing_scorecard(rows)["accuracy_by_framing"]
320
323
  if not by_model:
@@ -0,0 +1,12 @@
1
+ # Kev (github.com/jaredpalmer/kev): a Jev-compatible decision model served locally.
2
+ # uv run --extra serve python -m kev.serve --run jaredpalmer/kev-0.8b --port 8009
3
+ adapter: jev_typesafe
4
+ model_id: kev-latest # Kev's only served name; the revision is recorded from /v1/models
5
+ url: http://127.0.0.1:8009/v1/systemone
6
+ api_key: local
7
+ allow_alias: true
8
+ probe_models: true
9
+ deployment: local
10
+ version_note: jaredpalmer/kev-0.8b
11
+ timeout_s: 120
12
+ max_retries: 3
@@ -0,0 +1,12 @@
1
+ # Kev (github.com/jaredpalmer/kev): a Jev-compatible decision model served locally.
2
+ # uv run --extra serve python -m kev.serve --run jaredpalmer/kev-4b --port 8009
3
+ adapter: jev_typesafe
4
+ model_id: kev-latest # Kev's only served name; the revision is recorded from /v1/models
5
+ url: http://127.0.0.1:8009/v1/systemone
6
+ api_key: local
7
+ allow_alias: true
8
+ probe_models: true
9
+ deployment: local
10
+ version_note: jaredpalmer/kev-4b
11
+ timeout_s: 120
12
+ max_retries: 3
@@ -0,0 +1,12 @@
1
+ # Kev (github.com/jaredpalmer/kev): a Jev-compatible decision model served locally.
2
+ # uv run --extra serve python -m kev.serve --run jaredpalmer/kev-9b --port 8009
3
+ adapter: jev_typesafe
4
+ model_id: kev-latest # Kev's only served name; the revision is recorded from /v1/models
5
+ url: http://127.0.0.1:8009/v1/systemone
6
+ api_key: local
7
+ allow_alias: true
8
+ probe_models: true
9
+ deployment: local
10
+ version_note: jaredpalmer/kev-9b
11
+ timeout_s: 120
12
+ max_retries: 3
@@ -92,7 +92,7 @@ def framing_range_plot(acc_by_model_framing: dict[str, dict[str, float]], out: P
92
92
  ax.plot(v, i, marker="o" if f == "f0" else ".", color="#c33" if f == "f0" else "#333", ms=7 if f == "f0" else 5)
93
93
  ax.set_yticks(range(len(models)))
94
94
  ax.set_yticklabels(models, fontsize=8)
95
- ax.set_xlabel("accuracy (red = canonical framing, dots = paraphrases / criteria variants)")
95
+ ax.set_xlabel("accuracy (red: canonical wording, dots: other framings)", fontsize=8)
96
96
  ax.set_title(title, fontsize=10)
97
97
  ax.set_xlim(0, 1.02)
98
98
  fig.tight_layout()
@@ -141,3 +141,28 @@ def interference_heatmap(res: dict[str, Any], out: Path, metric: str = "mean_jsd
141
141
  fig.savefig(out, dpi=160)
142
142
  plt.close(fig)
143
143
  return out
144
+
145
+
146
+ def multi_line_plot(series: dict[str, list[tuple[float, float]]], out: Path, xlabel: str, ylabel: str, title: str = "",
147
+ xlog: bool = False, ylim: tuple[float, float] | None = None) -> Path:
148
+ """Several named series on one axis; used for cross-model sweep comparisons in the README."""
149
+ plt = _plt()
150
+ fig, ax = plt.subplots(figsize=(5.2, 3.4))
151
+ for name, pts in series.items():
152
+ pts = sorted(p for p in pts if p[1] == p[1])
153
+ if pts:
154
+ ax.plot([p[0] for p in pts], [p[1] for p in pts], marker="o", ms=4, lw=1.6, label=name)
155
+ if xlog:
156
+ ax.set_xscale("log")
157
+ if ylim:
158
+ ax.set_ylim(*ylim)
159
+ ax.set_xlabel(xlabel)
160
+ ax.set_ylabel(ylabel)
161
+ ax.set_title(title, fontsize=10)
162
+ ax.grid(alpha=0.25)
163
+ ax.legend(fontsize=7, frameon=False)
164
+ fig.tight_layout()
165
+ out.parent.mkdir(parents=True, exist_ok=True)
166
+ fig.savefig(out, dpi=170)
167
+ plt.close(fig)
168
+ return out
@@ -0,0 +1,116 @@
1
+ Metadata-Version: 2.4
2
+ Name: sys1bench
3
+ Version: 0.3.2
4
+ Summary: Benchmark harness for typed System One decision models (choice / score / noul): calibration vs noise floor, framing sensitivity, selective prediction, ordinal fidelity, interference, robustness.
5
+ Author-email: Rahul Sharma <rsharma@rptu.de>
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Homepage, https://github.com/rssr25/system-one-bench
8
+ Project-URL: Documentation, https://github.com/rssr25/system-one-bench/blob/main/docs/SPEC_v2.md
9
+ Project-URL: Issues, https://github.com/rssr25/system-one-bench/issues
10
+ Project-URL: Changelog, https://github.com/rssr25/system-one-bench/blob/main/CHANGELOG.md
11
+ Keywords: benchmark,calibration,system-one,decision-models,jev,laya,evaluation,llm
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Software Development :: Testing
21
+ Classifier: Operating System :: OS Independent
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.11
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: numpy>=1.26
27
+ Requires-Dist: scipy>=1.11
28
+ Requires-Dist: pandas>=2.0
29
+ Requires-Dist: pydantic>=2.5
30
+ Requires-Dist: httpx>=0.27
31
+ Requires-Dist: pyyaml>=6.0
32
+ Requires-Dist: typer>=0.12
33
+ Provides-Extra: plots
34
+ Requires-Dist: matplotlib>=3.8; extra == "plots"
35
+ Provides-Extra: local
36
+ Requires-Dist: torch>=2.4; extra == "local"
37
+ Requires-Dist: transformers>=4.45; extra == "local"
38
+ Requires-Dist: sentence-transformers>=3.0; extra == "local"
39
+ Provides-Extra: laya
40
+ Requires-Dist: laya>=0.3; extra == "laya"
41
+ Provides-Extra: llm
42
+ Requires-Dist: vllm>=0.6; extra == "llm"
43
+ Requires-Dist: outlines>=0.1; extra == "llm"
44
+ Provides-Extra: dev
45
+ Requires-Dist: pytest>=8.0; extra == "dev"
46
+ Requires-Dist: ruff>=0.6; extra == "dev"
47
+ Requires-Dist: matplotlib>=3.8; extra == "dev"
48
+ Requires-Dist: build>=1.2; extra == "dev"
49
+ Requires-Dist: twine>=5.0; extra == "dev"
50
+ Dynamic: license-file
51
+
52
+ <p align="center">
53
+ <img src="docs/assets/logo.svg" alt="sys1bench" width="480">
54
+ </p>
55
+
56
+ <p align="center">
57
+ <a href="https://pypi.org/project/sys1bench/"><img alt="PyPI" src="https://img.shields.io/pypi/v/sys1bench?color=4F46E5&label=PyPI&logo=pypi&logoColor=white&cacheSeconds=3600"></a>
58
+ <a href="https://pypi.org/project/sys1bench/"><img alt="Python" src="https://img.shields.io/pypi/pyversions/sys1bench?color=06B6D4&logo=python&logoColor=white&cacheSeconds=3600"></a>
59
+ <a href="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml"><img alt="CI" src="https://github.com/rssr25/system-one-bench/actions/workflows/ci.yml/badge.svg"></a>
60
+ <a href="LICENSE"><img alt="License" src="https://img.shields.io/badge/license-Apache--2.0-blue.svg"></a>
61
+ </p>
62
+
63
+ <p align="center"><b>A benchmark for typed System One decision models</b><br>
64
+ Jev · Laya · Kev · and whatever comes next</p>
65
+
66
+ ---
67
+
68
+ System One models read a block of state plus typed questions and return a probability distribution over the answers you defined — one of N options, a level on an ordinal scale, or yes/no — in a single forward pass. No text is generated, so nothing has to be parsed. **sys1bench** measures what that buys: whether the probabilities are calibrated (against a noise floor), how much the answer depends on how the question is worded, how confidence behaves out of scope, how accuracy scales with options and state length, whether batched questions interfere, and what it all costs per decision. Headline numbers come from generated items whose labels follow a stated policy, so they cannot be memorised.
69
+
70
+ ```bash
71
+ pip install sys1bench
72
+ sys1bench suite all results/my-model --config my_model.yaml # A–I on generated, policy-labelled data
73
+ sys1bench report results --html results/dashboard.html # tables + interactive dashboard
74
+ ```
75
+
76
+ ## Results at a glance
77
+
78
+ <p align="center">
79
+ <img src="docs/figures/framing_tickets_priority.png" alt="Accuracy across question wordings on the policy-following priority score" width="49%">
80
+ <img src="docs/figures/reliability_tickets_priority.png" alt="Reliability diagram, priority score, all models" width="41%">
81
+ </p>
82
+ <p align="center">
83
+ <img src="docs/figures/cardinality_accuracy.png" alt="Exact accuracy versus number of options" width="45%">
84
+ <img src="docs/figures/latency_vs_questions.png" alt="Median latency versus number of questions per request" width="45%">
85
+ </p>
86
+
87
+ <p align="center"><sub>Jev 1.13.0 (hosted), Laya 0.3.4 and Kev 0.8B / 4B / 9B (local, GB10) on identical generated manifests, n = 500, September 2026. Left to right: accuracy across five paraphrases and criteria variants of the same question (red = canonical wording); reliability on the same question; accuracy as the option count grows from 2 to 255; median latency as more questions share one request. Hosted latency includes the network path and is not comparable to local compute time.</sub></p>
88
+
89
+ ## Why this benchmark
90
+
91
+ | | |
92
+ |---|---|
93
+ | **Policy-labelled data** | Labels follow a rule stated in the question (e.g. *urgency + 1 if angry + 1 if premium tier, cap 4*). A regex cannot solve it; neither can memorised public sets. Control arms are paired item-for-item: unknowable, label-noise, distractor, none-of-the-above. |
94
+ | **Calibration relative to noise** | Every ECE is divided by the ECE a perfectly calibrated model would show on that sample. Brier decomposition, clipped NLL, per-primitive temperature refit, quantisation report. |
95
+ | **Wording as a factor** | Accuracy is the median over five paraphrases and three criteria variants, with the range shown; adversarial wordings reported as a worst case. |
96
+ | **Selective prediction and cost** | Risk–coverage, coverage at 5 % risk, abstention with and without an explicit option, and realised cost per 10 k decisions under argmax, Bayes and escalate policies from per-question cost matrices. |
97
+ | **Ordinal, not categorical** | MAE, weighted kappa, ranked probability score, monotonicity along controlled severity ladders, 3/5/10-level scale invariance. |
98
+ | **Honest infrastructure** | Hosted and local latency never share an axis; the serving device is recorded on every row; a local model that lands on CPU when CUDA was requested aborts the run; every response is cached and re-derivable offline. |
99
+
100
+ ## Add your model
101
+
102
+ A hosted model with a conventional JSON decisions API needs a config file, no code:
103
+
104
+ ```bash
105
+ sys1bench configs example_future_vendor > my_model.yaml # fill in url, auth env var, field names
106
+ ```
107
+
108
+ Anything else subclasses `BaseAdapter` (declare capabilities, implement `decide`) and registers with `@register("my_model")` or the `sys1bench.adapters` entry-point group from your own package. Kev, which speaks the TypeSafe API, runs with `--config kev_4b` and no code at all.
109
+
110
+ ## Documentation
111
+
112
+ - [Specification](docs/SPEC_v2.md) — contract, data tiers, suites A–I, statistics, audits, rules for future models
113
+ - [Design review](docs/REVIEW_v1.md) — how the design was derived from the public state of the art
114
+ - [Changelog](CHANGELOG.md) · [Issues](https://github.com/rssr25/system-one-bench/issues)
115
+
116
+ <p align="center"><sub>Apache 2.0 · Rahul Sharma · numbers change with model versions, so every row carries the version string the provider returned</sub></p>
@@ -36,6 +36,9 @@ src/sys1bench/data/configs/example_future_vendor.yaml
36
36
  src/sys1bench/data/configs/hybrid_mock.yaml
37
37
  src/sys1bench/data/configs/jev_1.13.yaml
38
38
  src/sys1bench/data/configs/jev_1.13_openrouter.yaml
39
+ src/sys1bench/data/configs/kev_0.8b.yaml
40
+ src/sys1bench/data/configs/kev_4b.yaml
41
+ src/sys1bench/data/configs/kev_9b.yaml
39
42
  src/sys1bench/data/configs/laya_en.yaml
40
43
  src/sys1bench/data/configs/laya_typed_decisions.yaml
41
44
  src/sys1bench/data/framings/guardrail_intent.yaml