eyewright 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. eyewright-0.3.0/.gitignore +7 -0
  2. eyewright-0.3.0/PKG-INFO +7 -0
  3. eyewright-0.3.0/README.md +145 -0
  4. eyewright-0.3.0/bench/README.md +34 -0
  5. eyewright-0.3.0/bench/RIGOR.md +72 -0
  6. eyewright-0.3.0/bench/SCREENING-2026-10-09.md +52 -0
  7. eyewright-0.3.0/bench/build_train_cases.py +45 -0
  8. eyewright-0.3.0/bench/build_von_labels.py +43 -0
  9. eyewright-0.3.0/bench/calibrate_consults.py +162 -0
  10. eyewright-0.3.0/bench/cases-train-fit.jsonl +1200 -0
  11. eyewright-0.3.0/bench/cases-train-holdout.jsonl +600 -0
  12. eyewright-0.3.0/bench/cases-train.jsonl +1800 -0
  13. eyewright-0.3.0/bench/cases.jsonl +400 -0
  14. eyewright-0.3.0/bench/compare_consults.py +245 -0
  15. eyewright-0.3.0/bench/ensemble.py +193 -0
  16. eyewright-0.3.0/bench/ensemble_bins.py +66 -0
  17. eyewright-0.3.0/bench/export_cases.py +21 -0
  18. eyewright-0.3.0/bench/gate_analysis.py +108 -0
  19. eyewright-0.3.0/bench/responses-clef-flash_latest.jsonl +3 -0
  20. eyewright-0.3.0/bench/responses-gemma12b.jsonl +400 -0
  21. eyewright-0.3.0/bench/responses-holdout-laya-td.jsonl +600 -0
  22. eyewright-0.3.0/bench/responses-holdout-tev1.jsonl +600 -0
  23. eyewright-0.3.0/bench/responses-jev-113.jsonl +400 -0
  24. eyewright-0.3.0/bench/responses-jev-router.jsonl +438 -0
  25. eyewright-0.3.0/bench/responses-jevify-gemma.jsonl +400 -0
  26. eyewright-0.3.0/bench/responses-laya-laya-english.jsonl +400 -0
  27. eyewright-0.3.0/bench/responses-laya-laya-multilingual.jsonl +400 -0
  28. eyewright-0.3.0/bench/responses-laya-laya-typed-decisions.jsonl +400 -0
  29. eyewright-0.3.0/bench/responses-tev1_0.8b.jsonl +400 -0
  30. eyewright-0.3.0/bench/responses-tev1_4b.jsonl +400 -0
  31. eyewright-0.3.0/bench/responses-train-gemma12b.jsonl +1442 -0
  32. eyewright-0.3.0/bench/responses-train-jev-113.jsonl +1800 -0
  33. eyewright-0.3.0/bench/responses-train-jev-router.jsonl +1800 -0
  34. eyewright-0.3.0/bench/responses-von.jsonl +400 -0
  35. eyewright-0.3.0/bench/run_calibration_pipeline.ps1 +45 -0
  36. eyewright-0.3.0/bench/run_generative_noul.py +236 -0
  37. eyewright-0.3.0/bench/run_laya.py +50 -0
  38. eyewright-0.3.0/bench/run_native_noul.py +100 -0
  39. eyewright-0.3.0/bench/run_screening.ps1 +39 -0
  40. eyewright-0.3.0/bench/run_systemone.py +44 -0
  41. eyewright-0.3.0/bench/run_von_noul.py +62 -0
  42. eyewright-0.3.0/bench/score.py +127 -0
  43. eyewright-0.3.0/bench/stats_ci.py +110 -0
  44. eyewright-0.3.0/bench/three_eye_analysis.py +110 -0
  45. eyewright-0.3.0/bench/von-calibration.json +54 -0
  46. eyewright-0.3.0/bench/von-labels-train.jsonl +1800 -0
  47. eyewright-0.3.0/bench/von_calibration_check.py +131 -0
  48. eyewright-0.3.0/docs/ENSEMBLE.md +76 -0
  49. eyewright-0.3.0/docs/GATE-ANALYSIS.md +65 -0
  50. eyewright-0.3.0/docs/RESULTS.md +81 -0
  51. eyewright-0.3.0/docs/THREE-EYE.md +103 -0
  52. eyewright-0.3.0/examples/gate-jev.json +29 -0
  53. eyewright-0.3.0/examples/gate.json +28 -0
  54. eyewright-0.3.0/eyewright/__init__.py +6 -0
  55. eyewright-0.3.0/eyewright/cli.py +69 -0
  56. eyewright-0.3.0/eyewright/decide.py +132 -0
  57. eyewright-0.3.0/eyewright/gate.py +177 -0
  58. eyewright-0.3.0/eyewright/generative.py +117 -0
  59. eyewright-0.3.0/eyewright/mcp_server.py +45 -0
  60. eyewright-0.3.0/eyewright/server.py +68 -0
  61. eyewright-0.3.0/publish/RESERVE.md +104 -0
  62. eyewright-0.3.0/publish/npm/README.md +14 -0
  63. eyewright-0.3.0/publish/npm/index.js +10 -0
  64. eyewright-0.3.0/publish/npm/package.json +27 -0
  65. eyewright-0.3.0/pyproject.toml +19 -0
  66. eyewright-0.3.0/tests/mcp_smoke.py +27 -0
  67. eyewright-0.3.0/uv.lock +706 -0
@@ -0,0 +1,7 @@
1
+ __pycache__/
2
+ *.pyc
3
+ .venv/
4
+ *api-key*
5
+ *.key
6
+ .env
7
+ *.log
@@ -0,0 +1,7 @@
1
+ Metadata-Version: 2.5
2
+ Name: eyewright
3
+ Version: 0.3.0
4
+ Summary: eyewright - local decision core: three-eye gate + one-call typed decisions
5
+ Requires-Python: >=3.12
6
+ Provides-Extra: mcp
7
+ Requires-Dist: mcp>=2; extra == 'mcp'
@@ -0,0 +1,145 @@
1
+ # eyewright
2
+
3
+ **eyewright** — the decision core of the local-agent stack. A committee of local
4
+ decision models behind one callable: used at the *decide* step of
5
+ [Orbit](../orbit), and available to any program or AI harness.
6
+
7
+ > **Name:** `eyewright` throughout — distribution, import, CLI, and MCP tools.
8
+ > Decision + registry status: **ADR-0005** (`local-ai-system/docs/decisions/`).
9
+
10
+ - **Third eye (main):** `laya-typed-decisions` — local specialist, 0.766 acc /
11
+ Brier 0.019 on the typed-decisions bench.
12
+ - **Two sensory eyes (consults):** `tev1:4b` + `laya-multilingual`. In the
13
+ 0.25–0.75 uncertain band both consult; `>= 2` confident (`>= 0.9`) and
14
+ agreeing → consensus; one confident + another within 0.05 of main →
15
+ review-conflict → the caller gathers more evidence.
16
+
17
+ ## Local-first, with an optional upgrade
18
+
19
+ The default gate is **fully local** — Ollama (`gemma4:12b`, `tev1:4b`) plus a local
20
+ Laya checkpoint. No prompts leave the machine; the numbers below are from the same
21
+ 600-noul gate proxy used to pick the local committee.
22
+
23
+ Individual eyes can optionally be upgraded to a **cloud model with ZDR routing**:
24
+ `typesafe/jev-1.13` (Typesafe's System One decision model) is reachable through
25
+ OpenRouter's native decisions API and plugs in as a drop-in consult
26
+ (`"type": "openrouter-decisions"`, `"zdr": true`, key from `OPENROUTER_API_KEY`) —
27
+ see `examples/gate-jev.json`.
28
+
29
+ Screening numbers (main = `laya-typed-decisions`; `bench/SCREENING-2026-10-09.md`;
30
+ screening, not confirmatory — see `bench/RIGOR.md`):
31
+
32
+ | gate | acted % | acted accuracy | network |
33
+ |---|---|---|---|
34
+ | local default: `tev1:4b + laya-multilingual` | 23.7% | 90.8% | none |
35
+ | local System-2 add: `tev1:4b + gemma4:12b` | 34.3% | 94.7% | none |
36
+ | cloud consult (precision): `tev1:4b + jev-1.13` | 10.0% | 100% | ZDR |
37
+ | cloud main (automation): `jev-1.13 + tev1:4b + laya-multilingual` | 32.5% | 91.3% | ZDR |
38
+
39
+ ZDR here is **request-level routing** (`provider.zdr`), not a cryptographic
40
+ guarantee: it keeps prompts away from providers that retain data, but only
41
+ self-hosting is independently verifiable. That trade — more automation at the cost
42
+ of "nothing leaves the box" — is the point of making it a switch.
43
+
44
+ ```powershell
45
+ $env:OPENROUTER_API_KEY = "<key>"
46
+ uv run eyewright gate --config examples/gate-jev.json --state "..." --question "Is the task complete?"
47
+ ```
48
+
49
+ ## Four ways to call it
50
+
51
+ **Python**
52
+ ```python
53
+ from eyewright import ask
54
+
55
+ result = ask(cfg, "a = 1, b = 2", "Which relation holds?",
56
+ choices={"a_bigger": "a is bigger", "b_bigger": "b is bigger"})
57
+ # -> {"kind": "choice", "answer": "b_bigger", "confidence": 0.0038, ...}
58
+ ```
59
+
60
+ **CLI**
61
+ ```powershell
62
+ uv run eyewright ask --state "The sky is blue." --question "Is the statement true?"
63
+ uv run eyewright ask --state "a = 1, b = 2" --question "Which relation holds?" --choices '{"a_bigger":"a is bigger","b_bigger":"b is bigger"}'
64
+ uv run eyewright gate --state "..." --question "Is the task complete?"
65
+ ```
66
+
67
+ **HTTP** (for any language)
68
+ ```powershell
69
+ uv run eyewright serve --port 8090
70
+ # GET /health
71
+ # POST /v1/gate {"state": "...", "question": "..."} -> {p_done, decision, trace}
72
+ # POST /v1/ask {"state": "...", "question": "...", "choices"?|"levels"?, "engine"?: "generative"|"auto"} -> {kind, answer, confidence, ...}
73
+ ```
74
+
75
+ **MCP** (for any AI harness; needs the `mcp` extra)
76
+ ```powershell
77
+ uv run --extra mcp eyewright mcp
78
+ # tools: eyewright_gate(state, question), eyewright_ask(state, question, choices?, engine?)
79
+ ```
80
+ Smoke test: `uv run --extra mcp python tests/mcp_smoke.py`.
81
+
82
+ Wired into OpenCode + Cursor on this machine. Use the **venv exe directly** so a
83
+ harness reload never triggers a `uv sync`/lock cycle:
84
+
85
+ ```jsonc
86
+ "eyewright": {
87
+ "type": "local",
88
+ "command": ["C:/Users/kcpat/projects/oracle/.venv/Scripts/eyewright.exe", "mcp",
89
+ "--config", "C:/Users/kcpat/projects/oracle/examples/gate.json"],
90
+ "cwd": "C:/Users/kcpat/projects/oracle"
91
+ }
92
+ ```
93
+
94
+ Setup: `uv sync --extra mcp` once so the venv has the `mcp` extra.
95
+
96
+ Config for all four faces: `examples/gate.json` (primary + consults +
97
+ thresholds). Pass `--config <path>` elsewhere.
98
+
99
+ ## Engines behind `ask`
100
+
101
+ | engine | what runs | notes |
102
+ |---|---|---|
103
+ | `typed` (default) | yes/no: the three-eye committee; choose/rate: the primary decision model | specialist domain; committee trace |
104
+ | `generative` | a local LLM (`gemma4:12b` via the config's `generative` block) with a strict JSON contract | returns answer + reason; confidence is the model's own estimate — **not calibrated yet** (`"calibrated": false`) |
105
+ | `auto` | typed first; generative fallback when the typed path errors or has no signal | needs a `generative` block |
106
+
107
+ ```powershell
108
+ uv run eyewright ask --engine generative --state "a = 1, b = 2" --question "Is b greater than a?"
109
+ # -> {"kind": "noul", "answer": "yes", "confidence": 1.0, "reason": "...", "calibrated": false}
110
+ uv run eyewright ask --engine auto --state "..." --question "..."
111
+ ```
112
+
113
+ Where a deterministic check exists (`a > b`), call the check — eyewright is for
114
+ judgment, not arithmetic.
115
+
116
+ Gate confidence on any von map must read the pre-band posterior (`noul_raw`);
117
+ von is retired from the consult roster (2026-10-06). See `docs/THREE-EYE.md`.
118
+
119
+ ## Layout
120
+
121
+ - `eyewright/gate.py` — three-eye policy (`decide_gate`) + systemone clients.
122
+ - `eyewright/decide.py` — `ask()`: one call, any typed decision.
123
+ - `eyewright/server.py` — `eyewright serve` HTTP service.
124
+ - `eyewright/mcp_server.py` — `eyewright mcp` stdio MCP tools.
125
+ - `eyewright/cli.py` — `eyewright gate|ask|serve|mcp`.
126
+ - `bench/` — validation bench + saved responses; `docs/` — THREE-EYE,
127
+ GATE-ANALYSIS, RESULTS, ENSEMBLE.
128
+
129
+ ## Services it talks to
130
+
131
+ | service | address | notes |
132
+ |---|---|---|
133
+ | laya-serve | `127.0.0.1:8000` | key file in `model-lab/decision-bench/`; `"model":"multilingual"` forces that checkpoint |
134
+ | Ollama | `127.0.0.1:11434` | `tev1:4b` via `/v1/systemone` |
135
+ | von | `127.0.0.1:8001` | `von-sdk`; retired as a consult |
136
+
137
+ ## Roadmap
138
+
139
+ 1. ✅ `eyewright ask` (package + CLI) and `eyewright serve` (HTTP).
140
+ 2. ✅ `eyewright mcp` (stdio MCP: `eyewright_gate`, `eyewright_ask`) — wired into the
141
+ OpenCode and Cursor configs.
142
+ 3. ✅ Generative engine behind `ask` (`--engine generative` / `auto`).
143
+ 4. Next: post-hoc calibration of the generative engine (ticket path D); wire as
144
+ a verifier provider into ctxd's Audit plane / control-plane's verifier
145
+ family (advisory, never the truth).
@@ -0,0 +1,34 @@
1
+ # bench
2
+
3
+ Validation bench for the Oracle's gate policy. Scripts run **from this
4
+ directory** (they read the saved response files by name).
5
+
6
+ - `three_eye_analysis.py` — consult-pair selection on the 600-noul gate proxy
7
+ (writes nothing; prints the THREE-EYE table).
8
+ - `gate_analysis.py` — committee vs laya-TD-alone at matched coverage
9
+ (2,000-decision replay; GATE-ANALYSIS.md).
10
+ - `ensemble.py` / `ensemble_bins.py` — main-vs-trio committee analysis.
11
+ - `von_calibration_check.py` — von shipped vs calibrated on the gate proxy
12
+ (reads the calibrated run from `C:\Users\kcpat\ai-cache\von-calibrated\` when
13
+ present; gates on `noul_raw`).
14
+ - `run_von_noul.py` — re-asks von only the noul questions (writes to the
15
+ ai-cache dir; unique temp + atomic replace).
16
+ - `build_von_labels.py` — rebuild `von-labels-train.jsonl` from the
17
+ typed-decisions train parquet (`D:\datasets\typed-decisions`).
18
+ - `score.py`, `run_laya.py`, `run_systemone.py`, `export_cases.py` — the
19
+ original decision-model bench harness.
20
+
21
+ Data: `cases.jsonl` (400 test cases), `responses-*.jsonl` (saved per model),
22
+ `von-labels-train.jsonl`, `von-calibration.json`.
23
+
24
+ Reproduce (from `bench/`):
25
+
26
+ ```powershell
27
+ uv run python three_eye_analysis.py
28
+ uv run python gate_analysis.py
29
+ uv run python von_calibration_check.py
30
+ ```
31
+
32
+ Keys live outside the repo (secrets): `model-lab/decision-bench/.laya-api-key`,
33
+ `.von-api-key`. The calibrated von run is written to
34
+ `C:\Users\kcpat\ai-cache\von-calibrated\` (outside git).
@@ -0,0 +1,72 @@
1
+ # Rigor notes: where the 600-noul bench stands (2026-10-09)
2
+
3
+ **Verdict: this is a screening bench, not a confirmatory benchmark.** It is good
4
+ for ranking candidates cheaply under a fixed protocol; it should not carry
5
+ adoption claims ("X is better") without the upgrades below. Keep using it for
6
+ fast comparisons — label results *exploratory*.
7
+
8
+ ## What is already right
9
+
10
+ - Decision models are called through their **native decision APIs** (`/v1/systemone`),
11
+ not chat-prompted.
12
+ - **Raw responses are committed** per candidate (`responses-*.jsonl`) — re-scorable.
13
+ - Calibration discipline: refits are fitted on the **train** split, scored on test.
14
+ - Local models are versioned by **digest** (model-lab snapshots); protocol written
15
+ down (THREE-EYE.md).
16
+ - The bench targets the gate's exact shape (noul confidence), with cost/latency
17
+ considered in GATE-ANALYSIS.
18
+
19
+ ## Known weaknesses
20
+
21
+ 1. **Test-set reuse (selection leakage).** The 600 rows selected the consult pair,
22
+ the thresholds, the von verdict, and now the Jev/Gemma roster. Every round turns
23
+ test into dev. No frozen holdout remains.
24
+ 2. **No uncertainty reported.** Example bootstrap on the incumbent row
25
+ (`stats_ci.py`): acted coverage 23.7% [20.3, 27.2]; acted accuracy 90.8%
26
+ [85.8, 95.3]; consensus 87 firings [70, 104] at 85.1% [77.2, 92.2]. The
27
+ "+0.5 pt committee vs single gate" (GATE-ANALYSIS) is ~7 decisions out of
28
+ 1,478 — noise.
29
+ 3. **Proxy labels, unaudited.** Gold is teacher-generated (3 samples); no human
30
+ recheck, no inter-rater agreement. The ticket's own "loop-artifact holdout"
31
+ is the missing external-validity piece.
32
+ 4. **Single operating point.** No coverage-risk curves, no reliability diagrams;
33
+ Brier/ECE reported ad hoc.
34
+ 5. **Mixed confidence kinds.** Decision models emit posteriors; generative
35
+ candidates (gemma, Jev-over-API, Jevify) emit **self-reported** confidence.
36
+ Thresholding both at 0.75/0.9 compares different objects until the generative
37
+ side is calibrated (tagged `"calibrated": false` in `eyewright.ask`).
38
+ 6. **Prompt / API drift.** LLM candidates are prompt-sensitive; OpenRouter models
39
+ change under the same id (`typesafe/jev-router` even routes to varying upstream
40
+ models, dynamic pricing). No paraphrase ablation, seeds, or version pinning.
41
+ 7. **Field/scope traps already hit once.**
42
+ - Gate on `noul_raw` (pre-band), not band `noul` (clamped to [0.80,0.85] /
43
+ [0.15,0.20] — cannot pass 0.9).
44
+ - Scope rows explicitly: the "0.413" von row is the full 2,000-decision average;
45
+ the gate table is noul-only (0.613).
46
+
47
+ ## Upgrade path (ordered by ROI)
48
+
49
+ 1. **Protocol freeze** — `BENCH.md`: field definitions, prompts, decode params,
50
+ thresholds, primary metric, decision rule; roles (screening vs confirmatory).
51
+ 2. **Role separation** — keep the 600 as screening/dev; iterate on train-derived
52
+ rows; reserve a frozen holdout; build the **loop-artifact holdout** (real
53
+ artifacts, done/not-done + defects).
54
+ 3. **Stats as standard** — bootstrap CIs; paired McNemar vs incumbent; Holm
55
+ correction across candidates; report n and intervals everywhere.
56
+ 4. **Label audit** — human re-adjudicate ~100 rows; report agreement; bound
57
+ conclusions by label noise.
58
+ 5. **Curves, not cells** — coverage-risk (accuracy vs acted %) as the headline;
59
+ reliability diagrams; cost/latency columns; adopt on dominance, not one cell.
60
+ 6. **Robustness** — 3 prompt paraphrases x 2 seeds for generative candidates;
61
+ option-order shuffle; pin API model versions + dates per run.
62
+
63
+ ## Candidate-specific notes (2026-10-09)
64
+
65
+ - `clef-flash` is blocked twice: "non-finite logit" on Ollama 0.35.1 **and** 0.40.0,
66
+ and the `9b-mxfp8` tag requires an MLX runtime (unavailable on Windows). Excluded.
67
+ - `typesafe/jev-router` (OpenRouter) is a **router**, not a fixed model: it picks
68
+ an upstream model per request (a ping returned `openai/gpt-6-luna`) with dynamic
69
+ pricing. Results measure the router product; upstream hits are tallied in the
70
+ screening output.
71
+ - `jevify-gemma4-26b-a4b` is a local Gemma-4 MoE (4B active) fine-tune; via chat,
72
+ JSON-contract prompt, `think=false`.
@@ -0,0 +1,52 @@
1
+ # Screening run: Jev / Gemma / Jevify vs the Oracle on the 600-noul gate proxy (2026-10-09)
2
+
3
+ Same rows and policy as the original Oracle-vs-laya comparison (THREE-EYE.md):
4
+ main accept >= 0.75, reject <= 0.25, uncertain band consults; consensus = >= 2
5
+ confident consults (>= 0.9) agreeing (majority for triples); conflict = one confident
6
+ + another consult within 0.05 of main.
7
+
8
+ **Screening, not confirmatory** — see `RIGOR.md`. Generative candidates' confidences
9
+ are self-reported (uncalibrated); `jev-router` answers are routed to upstream models;
10
+ `jev-1.13` is called through the native OpenRouter decisions API (`--zdr`).
11
+
12
+ ## Solo stats (600 noul decisions)
13
+
14
+ | model | n | acc | >=0.9 | Brier | ECE |
15
+ |---|---|---|---|---|---|
16
+ | laya-typed-decisions | 600 | 0.8567 | 0.0 | 0.1436 | 0.1921 |
17
+ | tev1:4b | 600 | 0.7883 | 0.45 | 0.1528 | 0.0495 |
18
+ | laya-multilingual | 600 | 0.4967 | 0.598 | 0.4117 | 0.3757 |
19
+ | von (retired) | 600 | 0.6133 | 0.0 | 0.2473 | 0.1246 |
20
+ | jev-1.13 * | 600 | 0.7833 | 0.122 | 0.139 | 0.0805 |
21
+ | jev-router * | 598 | 0.8194 | 0.554 | 0.1353 | 0.0703 |
22
+ | jevify-gemma * | 598 | 0.8177 | 0.908 | 0.1687 | 0.1506 |
23
+ | gemma4:12b * | 600 | 0.8233 | 0.858 | 0.1615 | 0.132 |
24
+
25
+ \* generative candidate (or native decisions API): confidence is the model's own, not a calibrated posterior.
26
+
27
+ ## Gate table (same policy)
28
+
29
+ | config | acted n | acted % (95% CI) | acted acc (95% CI) | consensus n/acc | conflict | not-acted\* |
30
+ |---|---|---|---|---|---|---|
31
+ | incumbent: laya / tev1 + laya-ml | 142 | 23.7% [20.3–27.2] | 90.8% [85.7–95.1] | 87/85.1% | 24 | 434 |
32
+ | laya / tev1 + jev-1.13 | 60 | 10.0% [7.7–12.5] | 100.0% [100.0–100.0] | 5/100.0% | 47 | 493 |
33
+ | laya / tev1 + jev-router | 142 | 23.7% [20.2–27.2] | 99.3% [97.7–100.0] | 87/98.9% | 4 | 454 |
34
+ | laya / tev1 + jevify-gemma | 211 | 35.2% [31.3–39.0] | 93.8% [90.5–97.0] | 156/91.7% | 23 | 366 |
35
+ | laya / tev1 + gemma4:12b | 206 | 34.3% [30.3–38.2] | 94.7% [91.4–97.7] | 151/92.7% | 20 | 374 |
36
+ | TRIPLE laya / tev1 + jev-1.13 + gemma4:12b | 207 | 34.5% [30.5–38.3] | 94.7% [91.4–97.7] | 152/92.8% | 55 | 338 |
37
+ | TRIPLE laya / tev1 + jev-router + gemma4:12b | 245 | 40.8% [36.8–44.8] | 93.5% [90.5–96.4] | 190/91.6% | 45 | 310 |
38
+ | laya / jev-1.13 + gemma4:12b (no tev1) | 61 | 10.2% [7.8–12.7] | 100.0% [100.0–100.0] | 6/100.0% | 59 | 480 |
39
+ | jev-1.13 main / tev1 + laya-ml | 195 | 32.5% [28.7–36.2] | 91.3% [87.1–94.9] | 61/78.7% | 15 | 390 |
40
+ | jev-router main / tev1 + laya-ml | 364 | 60.9% [56.9–64.7] | 78.0% [73.8–82.3] | 6/50.0% | 2 | 232 |
41
+ | jevify-gemma main / tev1 + laya-ml | 326 | 54.5% [50.5–58.4] | 80.1% [75.8–84.3] | 0/- | 0 | 272 |
42
+ | gemma4:12b main / tev1 + laya-ml | 323 | 53.8% [49.7–57.8] | 80.8% [76.4–85.0] | 0/- | 0 | 277 |
43
+ | reference: laya / tev1 + von | 55 | 9.2% [7.0–11.7] | 100.0% [100.0–100.0] | 0/- | 19 | 526 |
44
+
45
+ \* not-acted = main-rejects (p <= 0.25) + review paths (conflicts counted separately).
46
+
47
+ ## jev-router upstream models (what it routed to)
48
+
49
+ - deepseek/deepseek-v4.1-flash: 520 calls
50
+ - google/gemini-3.8-flash: 100 calls
51
+ - openai/gpt-6.1-sol: 2 calls
52
+
@@ -0,0 +1,45 @@
1
+ """Build cases-train.jsonl (1,800 noul) from von-labels-train.jsonl and a
2
+ deterministic fit/holdout split (seed 0).
3
+
4
+ cases-train.jsonl all 1,800
5
+ cases-train-fit.jsonl 1,200 (calibration fitting only)
6
+ cases-train-holdout.jsonl 600 (frozen for evaluation; never fit on)
7
+ """
8
+
9
+ import json
10
+ import random
11
+ from pathlib import Path
12
+
13
+ SRC = Path("von-labels-train.jsonl")
14
+ HOLDOUT_N = 600
15
+
16
+ rows = []
17
+ for i, line in enumerate(open(SRC, encoding="utf-8"), 1):
18
+ r = json.loads(line)
19
+ q = r["question"]
20
+ gold = str(r["gold"]).strip().lower()
21
+ label = "true" if gold in ("yes", "true", "1") else "false"
22
+ rows.append({
23
+ "id": f"train_{i:06d}",
24
+ "workflow": "train",
25
+ "state": r["state"],
26
+ "questions": {"q": {"type": "noul",
27
+ "instructions": q.get("instructions", ""),
28
+ "criteria": q.get("criteria") or {}}},
29
+ "gold": {"q": {"type": "noul", "label": label}},
30
+ })
31
+
32
+ idx = list(range(len(rows)))
33
+ random.Random(0).shuffle(idx)
34
+ hold = set(idx[:HOLDOUT_N])
35
+
36
+
37
+ def dump(path, items):
38
+ Path(path).write_text(
39
+ "\n".join(json.dumps(r, ensure_ascii=False) for r in items) + "\n", encoding="utf-8")
40
+
41
+
42
+ dump("cases-train.jsonl", rows)
43
+ dump("cases-train-fit.jsonl", [r for i, r in enumerate(rows) if i not in hold])
44
+ dump("cases-train-holdout.jsonl", [r for i, r in enumerate(rows) if i in hold])
45
+ print(f"train={len(rows)} fit={len(rows) - len(hold)} holdout={len(hold)}")
@@ -0,0 +1,43 @@
1
+ """Build von calibration labels from the typed-decisions TRAIN split (noul questions only).
2
+
3
+ Output: von-labels-train.jsonl, one wire question per line:
4
+ {"state": <state>, "instructions": <noul instructions>, "choices": {"true":..., "false":...}, "gold": "true"|"false"}
5
+ """
6
+ import json
7
+ from pathlib import Path
8
+
9
+ import pandas as pd
10
+
11
+ SRC = Path(r"D:\datasets\typed-decisions\all\train-00000-of-00001.parquet")
12
+ OUT = Path(__file__).parent / "von-labels-train.jsonl"
13
+
14
+ df = pd.read_parquet(SRC)
15
+ n = 0
16
+ with OUT.open("w", encoding="utf-8") as f:
17
+ for _, row in df.iterrows():
18
+ state = json.loads(row["state"])
19
+ questions = json.loads(row["questions"])
20
+ gold = json.loads(row["gold"])
21
+ for qname, q in questions.items():
22
+ if q.get("type") != "noul":
23
+ continue
24
+ g = gold.get(qname) or {}
25
+ label = g.get("label")
26
+ if label is None:
27
+ continue
28
+ choices = q.get("criteria") or {
29
+ "true": "The statement holds.",
30
+ "false": "The statement is false.",
31
+ }
32
+ f.write(json.dumps({
33
+ "state": state,
34
+ "question": {
35
+ "type": "noul",
36
+ "instructions": q.get("instructions", ""),
37
+ "criteria": choices,
38
+ },
39
+ "gold": "yes" if str(label).lower() == "true" else "no",
40
+ }, ensure_ascii=False) + "\n")
41
+ n += 1
42
+
43
+ print(f"wrote {n} noul calibration rows -> {OUT}")
@@ -0,0 +1,162 @@
1
+ """Fit a confidence calibration for a consult on the train split, apply to the
2
+ frozen holdout + test, and report ECE/Brier before/after.
3
+
4
+ Fit split: cases-train-fit.jsonl (1,200 rows). Evaluation: cases-train-holdout.jsonl
5
+ (600, never fit on) and cases.jsonl (the original test 600). Two maps are fit and
6
+ compared: isotonic (PAV) and a Platt-style scalar on the logit.
7
+
8
+ uv run python calibrate_consults.py --name gemma4:12b \
9
+ --raw-train responses-train-gemma12b.jsonl --raw-test responses-gemma12b.jsonl \
10
+ --out-train-cal responses-train-gemma12b-cal.jsonl \
11
+ --out-test-cal responses-gemma12b-cal.jsonl
12
+ """
13
+
14
+ import argparse
15
+ import json
16
+ import math
17
+
18
+ ap = argparse.ArgumentParser()
19
+ ap.add_argument("--name", required=True)
20
+ ap.add_argument("--raw-train", required=True)
21
+ ap.add_argument("--raw-test", required=True)
22
+ ap.add_argument("--out-train-cal", required=True)
23
+ ap.add_argument("--out-test-cal", required=True)
24
+ args = ap.parse_args()
25
+
26
+
27
+ def load_gold(path):
28
+ g = {}
29
+ for line in open(path, encoding="utf-8"):
30
+ c = json.loads(line)
31
+ for q, gold in c["gold"].items():
32
+ if gold.get("type") == "noul":
33
+ g[(c["id"], q)] = 1.0 if str(gold.get("label")).lower() == "true" else 0.0
34
+ return g
35
+
36
+
37
+ def load_p(path):
38
+ out = {}
39
+ for line in open(path, encoding="utf-8"):
40
+ r = json.loads(line)
41
+ if not r.get("ok"):
42
+ continue
43
+ for q, a in (r.get("resp") or {}).get("answers", {}).items():
44
+ p = a.get("noul_raw", a.get("noul"))
45
+ if p is not None:
46
+ out[(r["id"], q)] = float(p)
47
+ return out
48
+
49
+
50
+ def fit_isotonic(points):
51
+ blocks = []
52
+ for x, y in sorted(points):
53
+ blocks.append([1.0, y, x])
54
+ while len(blocks) >= 2 and blocks[-2][1] / blocks[-2][0] > blocks[-1][1] / blocks[-1][0]:
55
+ w2, wy2, _ = blocks.pop()
56
+ w1, wy1, _ = blocks.pop()
57
+ blocks.append([w1 + w2, wy1 + wy2, x])
58
+ xs = [b[2] for b in blocks]
59
+ ys = [b[1] / b[0] for b in blocks]
60
+ return xs, ys
61
+
62
+
63
+ def apply_isotonic(x, xs, ys):
64
+ if x <= xs[0]:
65
+ return ys[0]
66
+ if x >= xs[-1]:
67
+ return ys[-1]
68
+ for i in range(1, len(xs)):
69
+ if x <= xs[i]:
70
+ x0, x1, y0, y1 = xs[i - 1], xs[i], ys[i - 1], ys[i]
71
+ return y1 if x1 == x0 else y0 + (y1 - y0) * (x - x0) / (x1 - x0)
72
+ return ys[-1]
73
+
74
+
75
+ def _logit(p, eps=1e-6):
76
+ p = min(1 - eps, max(eps, p))
77
+ return math.log(p / (1 - p))
78
+
79
+
80
+ def fit_platt(points):
81
+ a, b = 1.0, 0.0
82
+ for _ in range(500):
83
+ ga = gb = 0.0
84
+ for x, y in points:
85
+ z = a * _logit(x) + b
86
+ q = 1 / (1 + math.exp(-z))
87
+ ga += (q - y) * _logit(x)
88
+ gb += (q - y)
89
+ a -= 0.1 * ga / len(points)
90
+ b -= 0.1 * gb / len(points)
91
+ return a, b
92
+
93
+
94
+ def apply_platt(x, a, b):
95
+ return 1 / (1 + math.exp(-(a * _logit(x) + b)))
96
+
97
+
98
+ def ece_brier(points, fn):
99
+ xs = [(fn(x), y) for x, y in points]
100
+ brier = sum((q - y) ** 2 for q, y in xs) / len(xs)
101
+ buckets = {}
102
+ for q, y in xs:
103
+ conf = q if q >= 0.5 else 1 - q
104
+ hit = (q >= 0.5) == (y == 1.0)
105
+ b = min(int(conf * 10), 9)
106
+ d = buckets.setdefault(b, [0.0, 0, 0])
107
+ d[0] += conf
108
+ d[1] += hit
109
+ d[2] += 1
110
+ ece = sum(v[2] / len(xs) * abs(v[0] / v[2] - v[1] / v[2]) for v in buckets.values())
111
+ return round(ece, 4), round(brier, 4)
112
+
113
+
114
+ gold_train = load_gold("cases-train.jsonl")
115
+ gold_test = load_gold("cases.jsonl")
116
+ fit_ids = {json.loads(l)["id"] for l in open("cases-train-fit.jsonl", encoding="utf-8")}
117
+ hold_ids = {json.loads(l)["id"] for l in open("cases-train-holdout.jsonl", encoding="utf-8")}
118
+ raw_train = load_p(args.raw_train)
119
+ raw_test = load_p(args.raw_test)
120
+
121
+ fit_pts = [(p, gold_train[k]) for k, p in raw_train.items() if k in gold_train and k[0] in fit_ids]
122
+ hold_pts = [(p, gold_train[k]) for k, p in raw_train.items() if k in gold_train and k[0] in hold_ids]
123
+ test_pts = [(p, gold_test[k]) for k, p in raw_test.items() if k in gold_test]
124
+
125
+ xs, ys = fit_isotonic(fit_pts)
126
+ a, b = fit_platt(fit_pts)
127
+
128
+
129
+ def m_iso(p):
130
+ return apply_isotonic(p, xs, ys)
131
+
132
+
133
+ def m_platt(p):
134
+ return apply_platt(p, a, b)
135
+
136
+
137
+ print(f"# Calibration for {args.name}")
138
+ print(f"fit={len(fit_pts)} holdout={len(hold_pts)} test={len(test_pts)} platt a={a:.3f} b={b:.3f}")
139
+ print(f"{'split':<10} {'raw ECE/Brier':<20} {'isotonic':<20} {'platt':<20}")
140
+ for label, pts in (("fit", fit_pts), ("holdout", hold_pts), ("test", test_pts)):
141
+ print(f"{label:<10} {str(ece_brier(pts, lambda x: x)):<20} "
142
+ f"{str(ece_brier(pts, m_iso)):<20} {str(ece_brier(pts, m_platt)):<20}")
143
+
144
+
145
+ def write_cal(src, dst, fn):
146
+ with open(dst, "w", encoding="utf-8") as fo:
147
+ for line in open(src, encoding="utf-8"):
148
+ r = json.loads(line)
149
+ for q, a in (r.get("resp") or {}).get("answers", {}).items():
150
+ p = a.get("noul_raw", a.get("noul"))
151
+ if p is not None:
152
+ c = round(fn(float(p)), 4)
153
+ a["noul"] = c
154
+ a["noul_raw"] = c
155
+ a["calibrated_from"] = round(float(p), 4)
156
+ a["calibration"] = "isotonic"
157
+ fo.write(json.dumps(r, ensure_ascii=False) + "\n")
158
+
159
+
160
+ write_cal(args.raw_train, args.out_train_cal, m_iso)
161
+ write_cal(args.raw_test, args.out_test_cal, m_iso)
162
+ print(f"\nwrote {args.out_train_cal} and {args.out_test_cal}")