eyewright 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eyewright-0.3.0/.gitignore +7 -0
- eyewright-0.3.0/PKG-INFO +7 -0
- eyewright-0.3.0/README.md +145 -0
- eyewright-0.3.0/bench/README.md +34 -0
- eyewright-0.3.0/bench/RIGOR.md +72 -0
- eyewright-0.3.0/bench/SCREENING-2026-10-09.md +52 -0
- eyewright-0.3.0/bench/build_train_cases.py +45 -0
- eyewright-0.3.0/bench/build_von_labels.py +43 -0
- eyewright-0.3.0/bench/calibrate_consults.py +162 -0
- eyewright-0.3.0/bench/cases-train-fit.jsonl +1200 -0
- eyewright-0.3.0/bench/cases-train-holdout.jsonl +600 -0
- eyewright-0.3.0/bench/cases-train.jsonl +1800 -0
- eyewright-0.3.0/bench/cases.jsonl +400 -0
- eyewright-0.3.0/bench/compare_consults.py +245 -0
- eyewright-0.3.0/bench/ensemble.py +193 -0
- eyewright-0.3.0/bench/ensemble_bins.py +66 -0
- eyewright-0.3.0/bench/export_cases.py +21 -0
- eyewright-0.3.0/bench/gate_analysis.py +108 -0
- eyewright-0.3.0/bench/responses-clef-flash_latest.jsonl +3 -0
- eyewright-0.3.0/bench/responses-gemma12b.jsonl +400 -0
- eyewright-0.3.0/bench/responses-holdout-laya-td.jsonl +600 -0
- eyewright-0.3.0/bench/responses-holdout-tev1.jsonl +600 -0
- eyewright-0.3.0/bench/responses-jev-113.jsonl +400 -0
- eyewright-0.3.0/bench/responses-jev-router.jsonl +438 -0
- eyewright-0.3.0/bench/responses-jevify-gemma.jsonl +400 -0
- eyewright-0.3.0/bench/responses-laya-laya-english.jsonl +400 -0
- eyewright-0.3.0/bench/responses-laya-laya-multilingual.jsonl +400 -0
- eyewright-0.3.0/bench/responses-laya-laya-typed-decisions.jsonl +400 -0
- eyewright-0.3.0/bench/responses-tev1_0.8b.jsonl +400 -0
- eyewright-0.3.0/bench/responses-tev1_4b.jsonl +400 -0
- eyewright-0.3.0/bench/responses-train-gemma12b.jsonl +1442 -0
- eyewright-0.3.0/bench/responses-train-jev-113.jsonl +1800 -0
- eyewright-0.3.0/bench/responses-train-jev-router.jsonl +1800 -0
- eyewright-0.3.0/bench/responses-von.jsonl +400 -0
- eyewright-0.3.0/bench/run_calibration_pipeline.ps1 +45 -0
- eyewright-0.3.0/bench/run_generative_noul.py +236 -0
- eyewright-0.3.0/bench/run_laya.py +50 -0
- eyewright-0.3.0/bench/run_native_noul.py +100 -0
- eyewright-0.3.0/bench/run_screening.ps1 +39 -0
- eyewright-0.3.0/bench/run_systemone.py +44 -0
- eyewright-0.3.0/bench/run_von_noul.py +62 -0
- eyewright-0.3.0/bench/score.py +127 -0
- eyewright-0.3.0/bench/stats_ci.py +110 -0
- eyewright-0.3.0/bench/three_eye_analysis.py +110 -0
- eyewright-0.3.0/bench/von-calibration.json +54 -0
- eyewright-0.3.0/bench/von-labels-train.jsonl +1800 -0
- eyewright-0.3.0/bench/von_calibration_check.py +131 -0
- eyewright-0.3.0/docs/ENSEMBLE.md +76 -0
- eyewright-0.3.0/docs/GATE-ANALYSIS.md +65 -0
- eyewright-0.3.0/docs/RESULTS.md +81 -0
- eyewright-0.3.0/docs/THREE-EYE.md +103 -0
- eyewright-0.3.0/examples/gate-jev.json +29 -0
- eyewright-0.3.0/examples/gate.json +28 -0
- eyewright-0.3.0/eyewright/__init__.py +6 -0
- eyewright-0.3.0/eyewright/cli.py +69 -0
- eyewright-0.3.0/eyewright/decide.py +132 -0
- eyewright-0.3.0/eyewright/gate.py +177 -0
- eyewright-0.3.0/eyewright/generative.py +117 -0
- eyewright-0.3.0/eyewright/mcp_server.py +45 -0
- eyewright-0.3.0/eyewright/server.py +68 -0
- eyewright-0.3.0/publish/RESERVE.md +104 -0
- eyewright-0.3.0/publish/npm/README.md +14 -0
- eyewright-0.3.0/publish/npm/index.js +10 -0
- eyewright-0.3.0/publish/npm/package.json +27 -0
- eyewright-0.3.0/pyproject.toml +19 -0
- eyewright-0.3.0/tests/mcp_smoke.py +27 -0
- eyewright-0.3.0/uv.lock +706 -0
eyewright-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
# eyewright
|
|
2
|
+
|
|
3
|
+
**eyewright** — the decision core of the local-agent stack. A committee of local
|
|
4
|
+
decision models behind one callable: used at the *decide* step of
|
|
5
|
+
[Orbit](../orbit), and available to any program or AI harness.
|
|
6
|
+
|
|
7
|
+
> **Name:** `eyewright` throughout — distribution, import, CLI, and MCP tools.
|
|
8
|
+
> Decision + registry status: **ADR-0005** (`local-ai-system/docs/decisions/`).
|
|
9
|
+
|
|
10
|
+
- **Third eye (main):** `laya-typed-decisions` — local specialist, 0.766 acc /
|
|
11
|
+
Brier 0.019 on the typed-decisions bench.
|
|
12
|
+
- **Two sensory eyes (consults):** `tev1:4b` + `laya-multilingual`. In the
|
|
13
|
+
0.25–0.75 uncertain band both consult; `>= 2` confident (`>= 0.9`) and
|
|
14
|
+
agreeing → consensus; one confident + another within 0.05 of main →
|
|
15
|
+
review-conflict → the caller gathers more evidence.
|
|
16
|
+
|
|
17
|
+
## Local-first, with an optional upgrade
|
|
18
|
+
|
|
19
|
+
The default gate is **fully local** — Ollama (`gemma4:12b`, `tev1:4b`) plus a local
|
|
20
|
+
Laya checkpoint. No prompts leave the machine; the numbers below are from the same
|
|
21
|
+
600-noul gate proxy used to pick the local committee.
|
|
22
|
+
|
|
23
|
+
Individual eyes can optionally be upgraded to a **cloud model with ZDR routing**:
|
|
24
|
+
`typesafe/jev-1.13` (Typesafe's System One decision model) is reachable through
|
|
25
|
+
OpenRouter's native decisions API and plugs in as a drop-in consult
|
|
26
|
+
(`"type": "openrouter-decisions"`, `"zdr": true`, key from `OPENROUTER_API_KEY`) —
|
|
27
|
+
see `examples/gate-jev.json`.
|
|
28
|
+
|
|
29
|
+
Screening numbers (main = `laya-typed-decisions`; `bench/SCREENING-2026-10-09.md`;
|
|
30
|
+
screening, not confirmatory — see `bench/RIGOR.md`):
|
|
31
|
+
|
|
32
|
+
| gate | acted % | acted accuracy | network |
|
|
33
|
+
|---|---|---|---|
|
|
34
|
+
| local default: `tev1:4b + laya-multilingual` | 23.7% | 90.8% | none |
|
|
35
|
+
| local System-2 add: `tev1:4b + gemma4:12b` | 34.3% | 94.7% | none |
|
|
36
|
+
| cloud consult (precision): `tev1:4b + jev-1.13` | 10.0% | 100% | ZDR |
|
|
37
|
+
| cloud main (automation): `jev-1.13 + tev1:4b + laya-multilingual` | 32.5% | 91.3% | ZDR |
|
|
38
|
+
|
|
39
|
+
ZDR here is **request-level routing** (`provider.zdr`), not a cryptographic
|
|
40
|
+
guarantee: it keeps prompts away from providers that retain data, but only
|
|
41
|
+
self-hosting is independently verifiable. That trade — more automation at the cost
|
|
42
|
+
of "nothing leaves the box" — is the point of making it a switch.
|
|
43
|
+
|
|
44
|
+
```powershell
|
|
45
|
+
$env:OPENROUTER_API_KEY = "<key>"
|
|
46
|
+
uv run eyewright gate --config examples/gate-jev.json --state "..." --question "Is the task complete?"
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Four ways to call it
|
|
50
|
+
|
|
51
|
+
**Python**
|
|
52
|
+
```python
|
|
53
|
+
from eyewright import ask
|
|
54
|
+
|
|
55
|
+
result = ask(cfg, "a = 1, b = 2", "Which relation holds?",
|
|
56
|
+
choices={"a_bigger": "a is bigger", "b_bigger": "b is bigger"})
|
|
57
|
+
# -> {"kind": "choice", "answer": "b_bigger", "confidence": 0.0038, ...}
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
**CLI**
|
|
61
|
+
```powershell
|
|
62
|
+
uv run eyewright ask --state "The sky is blue." --question "Is the statement true?"
|
|
63
|
+
uv run eyewright ask --state "a = 1, b = 2" --question "Which relation holds?" --choices '{"a_bigger":"a is bigger","b_bigger":"b is bigger"}'
|
|
64
|
+
uv run eyewright gate --state "..." --question "Is the task complete?"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
**HTTP** (for any language)
|
|
68
|
+
```powershell
|
|
69
|
+
uv run eyewright serve --port 8090
|
|
70
|
+
# GET /health
|
|
71
|
+
# POST /v1/gate {"state": "...", "question": "..."} -> {p_done, decision, trace}
|
|
72
|
+
# POST /v1/ask {"state": "...", "question": "...", "choices"?|"levels"?, "engine"?: "generative"|"auto"} -> {kind, answer, confidence, ...}
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
**MCP** (for any AI harness; needs the `mcp` extra)
|
|
76
|
+
```powershell
|
|
77
|
+
uv run --extra mcp eyewright mcp
|
|
78
|
+
# tools: eyewright_gate(state, question), eyewright_ask(state, question, choices?, engine?)
|
|
79
|
+
```
|
|
80
|
+
Smoke test: `uv run --extra mcp python tests/mcp_smoke.py`.
|
|
81
|
+
|
|
82
|
+
Wired into OpenCode + Cursor on this machine. Use the **venv exe directly** so a
|
|
83
|
+
harness reload never triggers a `uv sync`/lock cycle:
|
|
84
|
+
|
|
85
|
+
```jsonc
|
|
86
|
+
"eyewright": {
|
|
87
|
+
"type": "local",
|
|
88
|
+
"command": ["C:/Users/kcpat/projects/oracle/.venv/Scripts/eyewright.exe", "mcp",
|
|
89
|
+
"--config", "C:/Users/kcpat/projects/oracle/examples/gate.json"],
|
|
90
|
+
"cwd": "C:/Users/kcpat/projects/oracle"
|
|
91
|
+
}
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Setup: `uv sync --extra mcp` once so the venv has the `mcp` extra.
|
|
95
|
+
|
|
96
|
+
Config for all four faces: `examples/gate.json` (primary + consults +
|
|
97
|
+
thresholds). Pass `--config <path>` elsewhere.
|
|
98
|
+
|
|
99
|
+
## Engines behind `ask`
|
|
100
|
+
|
|
101
|
+
| engine | what runs | notes |
|
|
102
|
+
|---|---|---|
|
|
103
|
+
| `typed` (default) | yes/no: the three-eye committee; choose/rate: the primary decision model | specialist domain; committee trace |
|
|
104
|
+
| `generative` | a local LLM (`gemma4:12b` via the config's `generative` block) with a strict JSON contract | returns answer + reason; confidence is the model's own estimate — **not calibrated yet** (`"calibrated": false`) |
|
|
105
|
+
| `auto` | typed first; generative fallback when the typed path errors or has no signal | needs a `generative` block |
|
|
106
|
+
|
|
107
|
+
```powershell
|
|
108
|
+
uv run eyewright ask --engine generative --state "a = 1, b = 2" --question "Is b greater than a?"
|
|
109
|
+
# -> {"kind": "noul", "answer": "yes", "confidence": 1.0, "reason": "...", "calibrated": false}
|
|
110
|
+
uv run eyewright ask --engine auto --state "..." --question "..."
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Where a deterministic check exists (`a > b`), call the check — eyewright is for
|
|
114
|
+
judgment, not arithmetic.
|
|
115
|
+
|
|
116
|
+
Gate confidence on any von map must read the pre-band posterior (`noul_raw`);
|
|
117
|
+
von is retired from the consult roster (2026-10-06). See `docs/THREE-EYE.md`.
|
|
118
|
+
|
|
119
|
+
## Layout
|
|
120
|
+
|
|
121
|
+
- `eyewright/gate.py` — three-eye policy (`decide_gate`) + systemone clients.
|
|
122
|
+
- `eyewright/decide.py` — `ask()`: one call, any typed decision.
|
|
123
|
+
- `eyewright/server.py` — `eyewright serve` HTTP service.
|
|
124
|
+
- `eyewright/mcp_server.py` — `eyewright mcp` stdio MCP tools.
|
|
125
|
+
- `eyewright/cli.py` — `eyewright gate|ask|serve|mcp`.
|
|
126
|
+
- `bench/` — validation bench + saved responses; `docs/` — THREE-EYE,
|
|
127
|
+
GATE-ANALYSIS, RESULTS, ENSEMBLE.
|
|
128
|
+
|
|
129
|
+
## Services it talks to
|
|
130
|
+
|
|
131
|
+
| service | address | notes |
|
|
132
|
+
|---|---|---|
|
|
133
|
+
| laya-serve | `127.0.0.1:8000` | key file in `model-lab/decision-bench/`; `"model":"multilingual"` forces that checkpoint |
|
|
134
|
+
| Ollama | `127.0.0.1:11434` | `tev1:4b` via `/v1/systemone` |
|
|
135
|
+
| von | `127.0.0.1:8001` | `von-sdk`; retired as a consult |
|
|
136
|
+
|
|
137
|
+
## Roadmap
|
|
138
|
+
|
|
139
|
+
1. ✅ `eyewright ask` (package + CLI) and `eyewright serve` (HTTP).
|
|
140
|
+
2. ✅ `eyewright mcp` (stdio MCP: `eyewright_gate`, `eyewright_ask`) — wired into the
|
|
141
|
+
OpenCode and Cursor configs.
|
|
142
|
+
3. ✅ Generative engine behind `ask` (`--engine generative` / `auto`).
|
|
143
|
+
4. Next: post-hoc calibration of the generative engine (ticket path D); wire as
|
|
144
|
+
a verifier provider into ctxd's Audit plane / control-plane's verifier
|
|
145
|
+
family (advisory, never the truth).
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# bench
|
|
2
|
+
|
|
3
|
+
Validation bench for the Oracle's gate policy. Scripts run **from this
|
|
4
|
+
directory** (they read the saved response files by name).
|
|
5
|
+
|
|
6
|
+
- `three_eye_analysis.py` — consult-pair selection on the 600-noul gate proxy
|
|
7
|
+
(writes nothing; prints the THREE-EYE table).
|
|
8
|
+
- `gate_analysis.py` — committee vs laya-TD-alone at matched coverage
|
|
9
|
+
(2,000-decision replay; GATE-ANALYSIS.md).
|
|
10
|
+
- `ensemble.py` / `ensemble_bins.py` — main-vs-trio committee analysis.
|
|
11
|
+
- `von_calibration_check.py` — von shipped vs calibrated on the gate proxy
|
|
12
|
+
(reads the calibrated run from `C:\Users\kcpat\ai-cache\von-calibrated\` when
|
|
13
|
+
present; gates on `noul_raw`).
|
|
14
|
+
- `run_von_noul.py` — re-asks von only the noul questions (writes to the
|
|
15
|
+
ai-cache dir; unique temp + atomic replace).
|
|
16
|
+
- `build_von_labels.py` — rebuild `von-labels-train.jsonl` from the
|
|
17
|
+
typed-decisions train parquet (`D:\datasets\typed-decisions`).
|
|
18
|
+
- `score.py`, `run_laya.py`, `run_systemone.py`, `export_cases.py` — the
|
|
19
|
+
original decision-model bench harness.
|
|
20
|
+
|
|
21
|
+
Data: `cases.jsonl` (400 test cases), `responses-*.jsonl` (saved per model),
|
|
22
|
+
`von-labels-train.jsonl`, `von-calibration.json`.
|
|
23
|
+
|
|
24
|
+
Reproduce (from `bench/`):
|
|
25
|
+
|
|
26
|
+
```powershell
|
|
27
|
+
uv run python three_eye_analysis.py
|
|
28
|
+
uv run python gate_analysis.py
|
|
29
|
+
uv run python von_calibration_check.py
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Keys live outside the repo (secrets): `model-lab/decision-bench/.laya-api-key`,
|
|
33
|
+
`.von-api-key`. The calibrated von run is written to
|
|
34
|
+
`C:\Users\kcpat\ai-cache\von-calibrated\` (outside git).
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# Rigor notes: where the 600-noul bench stands (2026-10-09)
|
|
2
|
+
|
|
3
|
+
**Verdict: this is a screening bench, not a confirmatory benchmark.** It is good
|
|
4
|
+
for ranking candidates cheaply under a fixed protocol; it should not carry
|
|
5
|
+
adoption claims ("X is better") without the upgrades below. Keep using it for
|
|
6
|
+
fast comparisons — label results *exploratory*.
|
|
7
|
+
|
|
8
|
+
## What is already right
|
|
9
|
+
|
|
10
|
+
- Decision models are called through their **native decision APIs** (`/v1/systemone`),
|
|
11
|
+
not chat-prompted.
|
|
12
|
+
- **Raw responses are committed** per candidate (`responses-*.jsonl`) — re-scorable.
|
|
13
|
+
- Calibration discipline: refits are fitted on the **train** split, scored on test.
|
|
14
|
+
- Local models are versioned by **digest** (model-lab snapshots); protocol written
|
|
15
|
+
down (THREE-EYE.md).
|
|
16
|
+
- The bench targets the gate's exact shape (noul confidence), with cost/latency
|
|
17
|
+
considered in GATE-ANALYSIS.
|
|
18
|
+
|
|
19
|
+
## Known weaknesses
|
|
20
|
+
|
|
21
|
+
1. **Test-set reuse (selection leakage).** The 600 rows selected the consult pair,
|
|
22
|
+
the thresholds, the von verdict, and now the Jev/Gemma roster. Every round turns
|
|
23
|
+
test into dev. No frozen holdout remains.
|
|
24
|
+
2. **No uncertainty reported.** Example bootstrap on the incumbent row
|
|
25
|
+
(`stats_ci.py`): acted coverage 23.7% [20.3, 27.2]; acted accuracy 90.8%
|
|
26
|
+
[85.8, 95.3]; consensus 87 firings [70, 104] at 85.1% [77.2, 92.2]. The
|
|
27
|
+
"+0.5 pt committee vs single gate" (GATE-ANALYSIS) is ~7 decisions out of
|
|
28
|
+
1,478 — noise.
|
|
29
|
+
3. **Proxy labels, unaudited.** Gold is teacher-generated (3 samples); no human
|
|
30
|
+
recheck, no inter-rater agreement. The ticket's own "loop-artifact holdout"
|
|
31
|
+
is the missing external-validity piece.
|
|
32
|
+
4. **Single operating point.** No coverage-risk curves, no reliability diagrams;
|
|
33
|
+
Brier/ECE reported ad hoc.
|
|
34
|
+
5. **Mixed confidence kinds.** Decision models emit posteriors; generative
|
|
35
|
+
candidates (gemma, Jev-over-API, Jevify) emit **self-reported** confidence.
|
|
36
|
+
Thresholding both at 0.75/0.9 compares different objects until the generative
|
|
37
|
+
side is calibrated (tagged `"calibrated": false` in `eyewright.ask`).
|
|
38
|
+
6. **Prompt / API drift.** LLM candidates are prompt-sensitive; OpenRouter models
|
|
39
|
+
change under the same id (`typesafe/jev-router` even routes to varying upstream
|
|
40
|
+
models, dynamic pricing). No paraphrase ablation, seeds, or version pinning.
|
|
41
|
+
7. **Field/scope traps already hit once.**
|
|
42
|
+
- Gate on `noul_raw` (pre-band), not band `noul` (clamped to [0.80,0.85] /
|
|
43
|
+
[0.15,0.20] — cannot pass 0.9).
|
|
44
|
+
- Scope rows explicitly: the "0.413" von row is the full 2,000-decision average;
|
|
45
|
+
the gate table is noul-only (0.613).
|
|
46
|
+
|
|
47
|
+
## Upgrade path (ordered by ROI)
|
|
48
|
+
|
|
49
|
+
1. **Protocol freeze** — `BENCH.md`: field definitions, prompts, decode params,
|
|
50
|
+
thresholds, primary metric, decision rule; roles (screening vs confirmatory).
|
|
51
|
+
2. **Role separation** — keep the 600 as screening/dev; iterate on train-derived
|
|
52
|
+
rows; reserve a frozen holdout; build the **loop-artifact holdout** (real
|
|
53
|
+
artifacts, done/not-done + defects).
|
|
54
|
+
3. **Stats as standard** — bootstrap CIs; paired McNemar vs incumbent; Holm
|
|
55
|
+
correction across candidates; report n and intervals everywhere.
|
|
56
|
+
4. **Label audit** — human re-adjudicate ~100 rows; report agreement; bound
|
|
57
|
+
conclusions by label noise.
|
|
58
|
+
5. **Curves, not cells** — coverage-risk (accuracy vs acted %) as the headline;
|
|
59
|
+
reliability diagrams; cost/latency columns; adopt on dominance, not one cell.
|
|
60
|
+
6. **Robustness** — 3 prompt paraphrases x 2 seeds for generative candidates;
|
|
61
|
+
option-order shuffle; pin API model versions + dates per run.
|
|
62
|
+
|
|
63
|
+
## Candidate-specific notes (2026-10-09)
|
|
64
|
+
|
|
65
|
+
- `clef-flash` is blocked twice: "non-finite logit" on Ollama 0.35.1 **and** 0.40.0,
|
|
66
|
+
and the `9b-mxfp8` tag requires an MLX runtime (unavailable on Windows). Excluded.
|
|
67
|
+
- `typesafe/jev-router` (OpenRouter) is a **router**, not a fixed model: it picks
|
|
68
|
+
an upstream model per request (a ping returned `openai/gpt-6-luna`) with dynamic
|
|
69
|
+
pricing. Results measure the router product; upstream hits are tallied in the
|
|
70
|
+
screening output.
|
|
71
|
+
- `jevify-gemma4-26b-a4b` is a local Gemma-4 MoE (4B active) fine-tune; via chat,
|
|
72
|
+
JSON-contract prompt, `think=false`.
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# Screening run: Jev / Gemma / Jevify vs the Oracle on the 600-noul gate proxy (2026-10-09)
|
|
2
|
+
|
|
3
|
+
Same rows and policy as the original Oracle-vs-laya comparison (THREE-EYE.md):
|
|
4
|
+
main accept >= 0.75, reject <= 0.25, uncertain band consults; consensus = >= 2
|
|
5
|
+
confident consults (>= 0.9) agreeing (majority for triples); conflict = one confident
|
|
6
|
+
+ another consult within 0.05 of main.
|
|
7
|
+
|
|
8
|
+
**Screening, not confirmatory** — see `RIGOR.md`. Generative candidates' confidences
|
|
9
|
+
are self-reported (uncalibrated); `jev-router` answers are routed to upstream models;
|
|
10
|
+
`jev-1.13` is called through the native OpenRouter decisions API (`--zdr`).
|
|
11
|
+
|
|
12
|
+
## Solo stats (600 noul decisions)
|
|
13
|
+
|
|
14
|
+
| model | n | acc | >=0.9 | Brier | ECE |
|
|
15
|
+
|---|---|---|---|---|---|
|
|
16
|
+
| laya-typed-decisions | 600 | 0.8567 | 0.0 | 0.1436 | 0.1921 |
|
|
17
|
+
| tev1:4b | 600 | 0.7883 | 0.45 | 0.1528 | 0.0495 |
|
|
18
|
+
| laya-multilingual | 600 | 0.4967 | 0.598 | 0.4117 | 0.3757 |
|
|
19
|
+
| von (retired) | 600 | 0.6133 | 0.0 | 0.2473 | 0.1246 |
|
|
20
|
+
| jev-1.13 * | 600 | 0.7833 | 0.122 | 0.139 | 0.0805 |
|
|
21
|
+
| jev-router * | 598 | 0.8194 | 0.554 | 0.1353 | 0.0703 |
|
|
22
|
+
| jevify-gemma * | 598 | 0.8177 | 0.908 | 0.1687 | 0.1506 |
|
|
23
|
+
| gemma4:12b * | 600 | 0.8233 | 0.858 | 0.1615 | 0.132 |
|
|
24
|
+
|
|
25
|
+
\* generative candidate (or native decisions API): confidence is the model's own, not a calibrated posterior.
|
|
26
|
+
|
|
27
|
+
## Gate table (same policy)
|
|
28
|
+
|
|
29
|
+
| config | acted n | acted % (95% CI) | acted acc (95% CI) | consensus n/acc | conflict | not-acted\* |
|
|
30
|
+
|---|---|---|---|---|---|---|
|
|
31
|
+
| incumbent: laya / tev1 + laya-ml | 142 | 23.7% [20.3–27.2] | 90.8% [85.7–95.1] | 87/85.1% | 24 | 434 |
|
|
32
|
+
| laya / tev1 + jev-1.13 | 60 | 10.0% [7.7–12.5] | 100.0% [100.0–100.0] | 5/100.0% | 47 | 493 |
|
|
33
|
+
| laya / tev1 + jev-router | 142 | 23.7% [20.2–27.2] | 99.3% [97.7–100.0] | 87/98.9% | 4 | 454 |
|
|
34
|
+
| laya / tev1 + jevify-gemma | 211 | 35.2% [31.3–39.0] | 93.8% [90.5–97.0] | 156/91.7% | 23 | 366 |
|
|
35
|
+
| laya / tev1 + gemma4:12b | 206 | 34.3% [30.3–38.2] | 94.7% [91.4–97.7] | 151/92.7% | 20 | 374 |
|
|
36
|
+
| TRIPLE laya / tev1 + jev-1.13 + gemma4:12b | 207 | 34.5% [30.5–38.3] | 94.7% [91.4–97.7] | 152/92.8% | 55 | 338 |
|
|
37
|
+
| TRIPLE laya / tev1 + jev-router + gemma4:12b | 245 | 40.8% [36.8–44.8] | 93.5% [90.5–96.4] | 190/91.6% | 45 | 310 |
|
|
38
|
+
| laya / jev-1.13 + gemma4:12b (no tev1) | 61 | 10.2% [7.8–12.7] | 100.0% [100.0–100.0] | 6/100.0% | 59 | 480 |
|
|
39
|
+
| jev-1.13 main / tev1 + laya-ml | 195 | 32.5% [28.7–36.2] | 91.3% [87.1–94.9] | 61/78.7% | 15 | 390 |
|
|
40
|
+
| jev-router main / tev1 + laya-ml | 364 | 60.9% [56.9–64.7] | 78.0% [73.8–82.3] | 6/50.0% | 2 | 232 |
|
|
41
|
+
| jevify-gemma main / tev1 + laya-ml | 326 | 54.5% [50.5–58.4] | 80.1% [75.8–84.3] | 0/- | 0 | 272 |
|
|
42
|
+
| gemma4:12b main / tev1 + laya-ml | 323 | 53.8% [49.7–57.8] | 80.8% [76.4–85.0] | 0/- | 0 | 277 |
|
|
43
|
+
| reference: laya / tev1 + von | 55 | 9.2% [7.0–11.7] | 100.0% [100.0–100.0] | 0/- | 19 | 526 |
|
|
44
|
+
|
|
45
|
+
\* not-acted = main-rejects (p <= 0.25) + review paths (conflicts counted separately).
|
|
46
|
+
|
|
47
|
+
## jev-router upstream models (what it routed to)
|
|
48
|
+
|
|
49
|
+
- deepseek/deepseek-v4.1-flash: 520 calls
|
|
50
|
+
- google/gemini-3.8-flash: 100 calls
|
|
51
|
+
- openai/gpt-6.1-sol: 2 calls
|
|
52
|
+
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Build cases-train.jsonl (1,800 noul) from von-labels-train.jsonl and a
|
|
2
|
+
deterministic fit/holdout split (seed 0).
|
|
3
|
+
|
|
4
|
+
cases-train.jsonl all 1,800
|
|
5
|
+
cases-train-fit.jsonl 1,200 (calibration fitting only)
|
|
6
|
+
cases-train-holdout.jsonl 600 (frozen for evaluation; never fit on)
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import random
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
SRC = Path("von-labels-train.jsonl")
|
|
14
|
+
HOLDOUT_N = 600
|
|
15
|
+
|
|
16
|
+
rows = []
|
|
17
|
+
for i, line in enumerate(open(SRC, encoding="utf-8"), 1):
|
|
18
|
+
r = json.loads(line)
|
|
19
|
+
q = r["question"]
|
|
20
|
+
gold = str(r["gold"]).strip().lower()
|
|
21
|
+
label = "true" if gold in ("yes", "true", "1") else "false"
|
|
22
|
+
rows.append({
|
|
23
|
+
"id": f"train_{i:06d}",
|
|
24
|
+
"workflow": "train",
|
|
25
|
+
"state": r["state"],
|
|
26
|
+
"questions": {"q": {"type": "noul",
|
|
27
|
+
"instructions": q.get("instructions", ""),
|
|
28
|
+
"criteria": q.get("criteria") or {}}},
|
|
29
|
+
"gold": {"q": {"type": "noul", "label": label}},
|
|
30
|
+
})
|
|
31
|
+
|
|
32
|
+
idx = list(range(len(rows)))
|
|
33
|
+
random.Random(0).shuffle(idx)
|
|
34
|
+
hold = set(idx[:HOLDOUT_N])
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def dump(path, items):
|
|
38
|
+
Path(path).write_text(
|
|
39
|
+
"\n".join(json.dumps(r, ensure_ascii=False) for r in items) + "\n", encoding="utf-8")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
dump("cases-train.jsonl", rows)
|
|
43
|
+
dump("cases-train-fit.jsonl", [r for i, r in enumerate(rows) if i not in hold])
|
|
44
|
+
dump("cases-train-holdout.jsonl", [r for i, r in enumerate(rows) if i in hold])
|
|
45
|
+
print(f"train={len(rows)} fit={len(rows) - len(hold)} holdout={len(hold)}")
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Build von calibration labels from the typed-decisions TRAIN split (noul questions only).
|
|
2
|
+
|
|
3
|
+
Output: von-labels-train.jsonl, one wire question per line:
|
|
4
|
+
{"state": <state>, "instructions": <noul instructions>, "choices": {"true":..., "false":...}, "gold": "true"|"false"}
|
|
5
|
+
"""
|
|
6
|
+
import json
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import pandas as pd
|
|
10
|
+
|
|
11
|
+
SRC = Path(r"D:\datasets\typed-decisions\all\train-00000-of-00001.parquet")
|
|
12
|
+
OUT = Path(__file__).parent / "von-labels-train.jsonl"
|
|
13
|
+
|
|
14
|
+
df = pd.read_parquet(SRC)
|
|
15
|
+
n = 0
|
|
16
|
+
with OUT.open("w", encoding="utf-8") as f:
|
|
17
|
+
for _, row in df.iterrows():
|
|
18
|
+
state = json.loads(row["state"])
|
|
19
|
+
questions = json.loads(row["questions"])
|
|
20
|
+
gold = json.loads(row["gold"])
|
|
21
|
+
for qname, q in questions.items():
|
|
22
|
+
if q.get("type") != "noul":
|
|
23
|
+
continue
|
|
24
|
+
g = gold.get(qname) or {}
|
|
25
|
+
label = g.get("label")
|
|
26
|
+
if label is None:
|
|
27
|
+
continue
|
|
28
|
+
choices = q.get("criteria") or {
|
|
29
|
+
"true": "The statement holds.",
|
|
30
|
+
"false": "The statement is false.",
|
|
31
|
+
}
|
|
32
|
+
f.write(json.dumps({
|
|
33
|
+
"state": state,
|
|
34
|
+
"question": {
|
|
35
|
+
"type": "noul",
|
|
36
|
+
"instructions": q.get("instructions", ""),
|
|
37
|
+
"criteria": choices,
|
|
38
|
+
},
|
|
39
|
+
"gold": "yes" if str(label).lower() == "true" else "no",
|
|
40
|
+
}, ensure_ascii=False) + "\n")
|
|
41
|
+
n += 1
|
|
42
|
+
|
|
43
|
+
print(f"wrote {n} noul calibration rows -> {OUT}")
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""Fit a confidence calibration for a consult on the train split, apply to the
|
|
2
|
+
frozen holdout + test, and report ECE/Brier before/after.
|
|
3
|
+
|
|
4
|
+
Fit split: cases-train-fit.jsonl (1,200 rows). Evaluation: cases-train-holdout.jsonl
|
|
5
|
+
(600, never fit on) and cases.jsonl (the original test 600). Two maps are fit and
|
|
6
|
+
compared: isotonic (PAV) and a Platt-style scalar on the logit.
|
|
7
|
+
|
|
8
|
+
uv run python calibrate_consults.py --name gemma4:12b \
|
|
9
|
+
--raw-train responses-train-gemma12b.jsonl --raw-test responses-gemma12b.jsonl \
|
|
10
|
+
--out-train-cal responses-train-gemma12b-cal.jsonl \
|
|
11
|
+
--out-test-cal responses-gemma12b-cal.jsonl
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import argparse
|
|
15
|
+
import json
|
|
16
|
+
import math
|
|
17
|
+
|
|
18
|
+
ap = argparse.ArgumentParser()
|
|
19
|
+
ap.add_argument("--name", required=True)
|
|
20
|
+
ap.add_argument("--raw-train", required=True)
|
|
21
|
+
ap.add_argument("--raw-test", required=True)
|
|
22
|
+
ap.add_argument("--out-train-cal", required=True)
|
|
23
|
+
ap.add_argument("--out-test-cal", required=True)
|
|
24
|
+
args = ap.parse_args()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def load_gold(path):
|
|
28
|
+
g = {}
|
|
29
|
+
for line in open(path, encoding="utf-8"):
|
|
30
|
+
c = json.loads(line)
|
|
31
|
+
for q, gold in c["gold"].items():
|
|
32
|
+
if gold.get("type") == "noul":
|
|
33
|
+
g[(c["id"], q)] = 1.0 if str(gold.get("label")).lower() == "true" else 0.0
|
|
34
|
+
return g
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def load_p(path):
|
|
38
|
+
out = {}
|
|
39
|
+
for line in open(path, encoding="utf-8"):
|
|
40
|
+
r = json.loads(line)
|
|
41
|
+
if not r.get("ok"):
|
|
42
|
+
continue
|
|
43
|
+
for q, a in (r.get("resp") or {}).get("answers", {}).items():
|
|
44
|
+
p = a.get("noul_raw", a.get("noul"))
|
|
45
|
+
if p is not None:
|
|
46
|
+
out[(r["id"], q)] = float(p)
|
|
47
|
+
return out
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def fit_isotonic(points):
|
|
51
|
+
blocks = []
|
|
52
|
+
for x, y in sorted(points):
|
|
53
|
+
blocks.append([1.0, y, x])
|
|
54
|
+
while len(blocks) >= 2 and blocks[-2][1] / blocks[-2][0] > blocks[-1][1] / blocks[-1][0]:
|
|
55
|
+
w2, wy2, _ = blocks.pop()
|
|
56
|
+
w1, wy1, _ = blocks.pop()
|
|
57
|
+
blocks.append([w1 + w2, wy1 + wy2, x])
|
|
58
|
+
xs = [b[2] for b in blocks]
|
|
59
|
+
ys = [b[1] / b[0] for b in blocks]
|
|
60
|
+
return xs, ys
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def apply_isotonic(x, xs, ys):
|
|
64
|
+
if x <= xs[0]:
|
|
65
|
+
return ys[0]
|
|
66
|
+
if x >= xs[-1]:
|
|
67
|
+
return ys[-1]
|
|
68
|
+
for i in range(1, len(xs)):
|
|
69
|
+
if x <= xs[i]:
|
|
70
|
+
x0, x1, y0, y1 = xs[i - 1], xs[i], ys[i - 1], ys[i]
|
|
71
|
+
return y1 if x1 == x0 else y0 + (y1 - y0) * (x - x0) / (x1 - x0)
|
|
72
|
+
return ys[-1]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _logit(p, eps=1e-6):
|
|
76
|
+
p = min(1 - eps, max(eps, p))
|
|
77
|
+
return math.log(p / (1 - p))
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def fit_platt(points):
|
|
81
|
+
a, b = 1.0, 0.0
|
|
82
|
+
for _ in range(500):
|
|
83
|
+
ga = gb = 0.0
|
|
84
|
+
for x, y in points:
|
|
85
|
+
z = a * _logit(x) + b
|
|
86
|
+
q = 1 / (1 + math.exp(-z))
|
|
87
|
+
ga += (q - y) * _logit(x)
|
|
88
|
+
gb += (q - y)
|
|
89
|
+
a -= 0.1 * ga / len(points)
|
|
90
|
+
b -= 0.1 * gb / len(points)
|
|
91
|
+
return a, b
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def apply_platt(x, a, b):
|
|
95
|
+
return 1 / (1 + math.exp(-(a * _logit(x) + b)))
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def ece_brier(points, fn):
|
|
99
|
+
xs = [(fn(x), y) for x, y in points]
|
|
100
|
+
brier = sum((q - y) ** 2 for q, y in xs) / len(xs)
|
|
101
|
+
buckets = {}
|
|
102
|
+
for q, y in xs:
|
|
103
|
+
conf = q if q >= 0.5 else 1 - q
|
|
104
|
+
hit = (q >= 0.5) == (y == 1.0)
|
|
105
|
+
b = min(int(conf * 10), 9)
|
|
106
|
+
d = buckets.setdefault(b, [0.0, 0, 0])
|
|
107
|
+
d[0] += conf
|
|
108
|
+
d[1] += hit
|
|
109
|
+
d[2] += 1
|
|
110
|
+
ece = sum(v[2] / len(xs) * abs(v[0] / v[2] - v[1] / v[2]) for v in buckets.values())
|
|
111
|
+
return round(ece, 4), round(brier, 4)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
gold_train = load_gold("cases-train.jsonl")
|
|
115
|
+
gold_test = load_gold("cases.jsonl")
|
|
116
|
+
fit_ids = {json.loads(l)["id"] for l in open("cases-train-fit.jsonl", encoding="utf-8")}
|
|
117
|
+
hold_ids = {json.loads(l)["id"] for l in open("cases-train-holdout.jsonl", encoding="utf-8")}
|
|
118
|
+
raw_train = load_p(args.raw_train)
|
|
119
|
+
raw_test = load_p(args.raw_test)
|
|
120
|
+
|
|
121
|
+
fit_pts = [(p, gold_train[k]) for k, p in raw_train.items() if k in gold_train and k[0] in fit_ids]
|
|
122
|
+
hold_pts = [(p, gold_train[k]) for k, p in raw_train.items() if k in gold_train and k[0] in hold_ids]
|
|
123
|
+
test_pts = [(p, gold_test[k]) for k, p in raw_test.items() if k in gold_test]
|
|
124
|
+
|
|
125
|
+
xs, ys = fit_isotonic(fit_pts)
|
|
126
|
+
a, b = fit_platt(fit_pts)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def m_iso(p):
|
|
130
|
+
return apply_isotonic(p, xs, ys)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def m_platt(p):
|
|
134
|
+
return apply_platt(p, a, b)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
print(f"# Calibration for {args.name}")
|
|
138
|
+
print(f"fit={len(fit_pts)} holdout={len(hold_pts)} test={len(test_pts)} platt a={a:.3f} b={b:.3f}")
|
|
139
|
+
print(f"{'split':<10} {'raw ECE/Brier':<20} {'isotonic':<20} {'platt':<20}")
|
|
140
|
+
for label, pts in (("fit", fit_pts), ("holdout", hold_pts), ("test", test_pts)):
|
|
141
|
+
print(f"{label:<10} {str(ece_brier(pts, lambda x: x)):<20} "
|
|
142
|
+
f"{str(ece_brier(pts, m_iso)):<20} {str(ece_brier(pts, m_platt)):<20}")
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def write_cal(src, dst, fn):
|
|
146
|
+
with open(dst, "w", encoding="utf-8") as fo:
|
|
147
|
+
for line in open(src, encoding="utf-8"):
|
|
148
|
+
r = json.loads(line)
|
|
149
|
+
for q, a in (r.get("resp") or {}).get("answers", {}).items():
|
|
150
|
+
p = a.get("noul_raw", a.get("noul"))
|
|
151
|
+
if p is not None:
|
|
152
|
+
c = round(fn(float(p)), 4)
|
|
153
|
+
a["noul"] = c
|
|
154
|
+
a["noul_raw"] = c
|
|
155
|
+
a["calibrated_from"] = round(float(p), 4)
|
|
156
|
+
a["calibration"] = "isotonic"
|
|
157
|
+
fo.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
write_cal(args.raw_train, args.out_train_cal, m_iso)
|
|
161
|
+
write_cal(args.raw_test, args.out_test_cal, m_iso)
|
|
162
|
+
print(f"\nwrote {args.out_train_cal} and {args.out_test_cal}")
|