judgetap 0.2.1.dev42001__tar.gz → 0.2.1.dev44001__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/PKG-INFO +5 -2
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/README.md +2 -1
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/docs/SPEC.md +2 -1
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/pyproject.toml +2 -1
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/__init__.py +1 -1
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/cascade.py +10 -5
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/__init__.py +6 -1
- judgetap-0.2.1.dev44001/src/judgetap/engines/gliner.py +132 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/evaluate.py +25 -2
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/hook.py +25 -5
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/install.py +20 -1
- judgetap-0.2.1.dev44001/src/judgetap/suites.py +177 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_cascade.py +24 -0
- judgetap-0.2.1.dev44001/tests/test_engine_gliner.py +213 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_agents.py +1 -1
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_hook.py +8 -3
- judgetap-0.2.1.dev44001/tests/test_suites.py +169 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/uv.lock +1594 -1589
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.github/workflows/ci.yml +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.github/workflows/demo.yml +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.github/workflows/release.yml +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.gitignore +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.python-version +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.release-please-manifest.json +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/CHANGELOG.md +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/CONTRIBUTING.md +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/LICENSE +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/docs/demo.tape +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/release-please-config.json +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/_compat.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/api.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/cli.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/dashboard/__init__.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/dashboard/data.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/dashboard/page.html +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/dashboard/server.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/decision_log.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engine.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/agentjev.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/jev.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/julia.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/laya.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/llm.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/errors.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/__init__.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/core.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/loop.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/rules.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/stop.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/py.typed +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/secrets.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/testing.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/types.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_api.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_call_accounting.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_calls.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_dashboard.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_decision_log.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_engine_jev.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_engine_julia.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_engine_llm.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_engine_local.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_evaluate.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_core.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_loop.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_polish_76.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_rules.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_stop.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_trust.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_llm_logprobs.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_questions.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_robustness_68.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_secrets.py +0 -0
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: judgetap
|
|
3
|
-
Version: 0.2.1.
|
|
3
|
+
Version: 0.2.1.dev44001
|
|
4
4
|
Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
|
|
5
5
|
Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
|
|
6
6
|
Author-email: Omer Bar-Ness <omer@zsquared.io>
|
|
7
7
|
License-Expression: Apache-2.0
|
|
8
8
|
License-File: LICENSE
|
|
9
9
|
Requires-Python: >=3.12
|
|
10
|
+
Provides-Extra: gliner
|
|
11
|
+
Requires-Dist: gliner2; extra == 'gliner'
|
|
10
12
|
Provides-Extra: keychain
|
|
11
13
|
Requires-Dist: keyring>=24; extra == 'keychain'
|
|
12
14
|
Provides-Extra: laya
|
|
@@ -44,7 +46,7 @@ judgetap guard test "git push --force origin main"
|
|
|
44
46
|
|
|
45
47
|
- **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
|
|
46
48
|
- **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
|
|
47
|
-
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
|
|
49
|
+
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost. `judgetap eval --suite banking77 --limit 200 --engines jev` runs a public suite instead (`ag_news`, `banking77`): it is downloaded from Hugging Face on first use, converted to judgetap's case format and cached in `~/.judgetap/suites`; nothing is bundled. `--limit N` takes the same fixed-seed sample every run. Each suite's licence and source URL are listed in `src/judgetap/suites.py`.
|
|
48
50
|
- **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
|
|
49
51
|
|
|
50
52
|
## Library
|
|
@@ -74,6 +76,7 @@ sj.configure(sj.Cascade([jev, flash]))
|
|
|
74
76
|
| `jev@<url>` / `typesafe:<url>` | Any TypeSafe-compatible server | none on localhost; remote needs the key and https |
|
|
75
77
|
| `laya` | Laya, open weights, runs in process (`pip install "judgetap[laya]"`) | none |
|
|
76
78
|
| `julia` / `julia:<path>` | [Julia-1](https://huggingface.co/SupersonicLabs/Julia-1), 144M open weights, runs in process on CPU (download into `~/.judgetap/models/Julia-1` and `pip install -e` it; relative paths never come from the working directory). Loads per process, so for the guard prefer a server engine | none |
|
|
79
|
+
| `gliner[:<hf model>]` | Fastino GLiNER2.5-Decide, 340M encoder, CPU or GPU (`pip install "judgetap[gliner]"`); only a full probability map over every option counts as calibrated; a winner-only score (the others share the rest evenly) or a bare label is uncalibrated. Loads in-process, so the guard refuses it: use a server engine there | none |
|
|
77
80
|
| `agentjev` / `agentjev:<url>` | A local AgentJev server | none |
|
|
78
81
|
| `llm:<model>` | Any LiteLLM model (`pip install "judgetap[llm]"`); probabilities self-reported | the provider's |
|
|
79
82
|
|
|
@@ -27,7 +27,7 @@ judgetap guard test "git push --force origin main"
|
|
|
27
27
|
|
|
28
28
|
- **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
|
|
29
29
|
- **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
|
|
30
|
-
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
|
|
30
|
+
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost. `judgetap eval --suite banking77 --limit 200 --engines jev` runs a public suite instead (`ag_news`, `banking77`): it is downloaded from Hugging Face on first use, converted to judgetap's case format and cached in `~/.judgetap/suites`; nothing is bundled. `--limit N` takes the same fixed-seed sample every run. Each suite's licence and source URL are listed in `src/judgetap/suites.py`.
|
|
31
31
|
- **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
|
|
32
32
|
|
|
33
33
|
## Library
|
|
@@ -57,6 +57,7 @@ sj.configure(sj.Cascade([jev, flash]))
|
|
|
57
57
|
| `jev@<url>` / `typesafe:<url>` | Any TypeSafe-compatible server | none on localhost; remote needs the key and https |
|
|
58
58
|
| `laya` | Laya, open weights, runs in process (`pip install "judgetap[laya]"`) | none |
|
|
59
59
|
| `julia` / `julia:<path>` | [Julia-1](https://huggingface.co/SupersonicLabs/Julia-1), 144M open weights, runs in process on CPU (download into `~/.judgetap/models/Julia-1` and `pip install -e` it; relative paths never come from the working directory). Loads per process, so for the guard prefer a server engine | none |
|
|
60
|
+
| `gliner[:<hf model>]` | Fastino GLiNER2.5-Decide, 340M encoder, CPU or GPU (`pip install "judgetap[gliner]"`); only a full probability map over every option counts as calibrated; a winner-only score (the others share the rest evenly) or a bare label is uncalibrated. Loads in-process, so the guard refuses it: use a server engine there | none |
|
|
60
61
|
| `agentjev` / `agentjev:<url>` | A local AgentJev server | none |
|
|
61
62
|
| `llm:<model>` | Any LiteLLM model (`pip install "judgetap[llm]"`); probabilities self-reported | the provider's |
|
|
62
63
|
|
|
@@ -37,6 +37,7 @@ sj.batch([...questions], context=...) -> list[Decision] # one pa
|
|
|
37
37
|
| `jev` | hosted (TypeSafe console, Vercel AI Gateway) | native batch; the reference shape |
|
|
38
38
|
| `laya` | local, open weights (Apache-2.0) | via transformers; GPU optional |
|
|
39
39
|
| `julia` | local, open weights (Apache-2.0), 144M | Julia-1 runtime from its model repo; CPU by default; Jev-shaped API; 2-20 options per question |
|
|
40
|
+
| `gliner` | local, open weights (Apache-2.0), Fastino GLiNER2.5-Decide via `gliner2` | all questions as heads in one pass; full probabilities when gliner2 returns them, else the winner's probability with the rest split evenly, else the label alone (`calibrated=False`) |
|
|
40
41
|
| `agentjev` | local, open weights | ~50 ms per pass |
|
|
41
42
|
| `llm` | any structured-output LLM via LiteLLM | OpenAI, Gemini, Anthropic, Ollama; probabilities are the model's own JSON estimate, flagged `calibrated=False` (see `?logprobs` below) |
|
|
42
43
|
| `llm:<model>?logprobs` | same, OpenAI-compatible servers exposing `logprobs` | options listed as letters, one single-token call per question; the distribution is the letters' `top_logprobs` renormalised (missing letters get 0, none present is an error); token probabilities, not calibrated (`calibrated=False`; check with `judgetap eval`); per-question calls run concurrently (max 8) and each is reported as a Call; more than 26 options is a local error raised before any request; falls back to JSON mode for good if the provider rejects logprobs |
|
|
@@ -52,7 +53,7 @@ escalate_below = 0.8 # p under this goes to the next engine
|
|
|
52
53
|
on_exhausted = "raise" # or "return_last", or a callback (e.g. ask a human)
|
|
53
54
|
```
|
|
54
55
|
|
|
55
|
-
Each hop is recorded on the `Decision`. Engine errors and timeouts fall through the same way.
|
|
56
|
+
Each hop is recorded on the `Decision`. Engine errors and timeouts fall through the same way, and so does an uncalibrated answer (`calibrated=False`) whatever its p, while an engine is left to ask; the last engine's answer is judged on p alone.
|
|
56
57
|
|
|
57
58
|
### 4. Calibration check
|
|
58
59
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "judgetap"
|
|
3
|
-
version = "0.2.1.
|
|
3
|
+
version = "0.2.1.dev44001"
|
|
4
4
|
description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -17,6 +17,7 @@ snapjudge = "judgetap.cli:main" # deprecated alias, removed after one release
|
|
|
17
17
|
[project.optional-dependencies]
|
|
18
18
|
llm = ["litellm>=1.40"]
|
|
19
19
|
laya = ["laya"]
|
|
20
|
+
gliner = ["gliner2"]
|
|
20
21
|
keychain = ["keyring>=24"]
|
|
21
22
|
|
|
22
23
|
[build-system]
|
|
@@ -175,11 +175,16 @@ class Cascade:
|
|
|
175
175
|
def _pending(self, questions, attempts) -> list[int]:
|
|
176
176
|
return [i for i in range(len(questions)) if not self._confident(attempts[i])]
|
|
177
177
|
|
|
178
|
-
def _confident(self, attempts: Sequence[Attempt]) -> bool:
|
|
178
|
+
def _confident(self, attempts: Sequence[Attempt], *, final: bool = False) -> bool:
|
|
179
|
+
"""Whether the latest answer stops the cascade. An uncalibrated p
|
|
180
|
+
(calibrated=False) never stops it while another engine is left to
|
|
181
|
+
ask; once none is (final), it is judged on p like any other."""
|
|
182
|
+
last = attempts[-1] if attempts else None
|
|
179
183
|
return (
|
|
180
|
-
|
|
181
|
-
and
|
|
182
|
-
and
|
|
184
|
+
last is not None
|
|
185
|
+
and last.p is not None
|
|
186
|
+
and last.p >= self.escalate_below
|
|
187
|
+
and (final or last.answer is None or last.answer.calibrated)
|
|
183
188
|
)
|
|
184
189
|
|
|
185
190
|
def _record(self, engine, questions, pending, answers, attempts) -> None:
|
|
@@ -220,7 +225,7 @@ class Cascade:
|
|
|
220
225
|
hops = _hops(attempts)
|
|
221
226
|
answered = [a for a in attempts if a.answer is not None]
|
|
222
227
|
spent = _spent(attempts)
|
|
223
|
-
if self._confident(attempts):
|
|
228
|
+
if self._confident(attempts, final=True):
|
|
224
229
|
return _as_result(attempts[-1], hops, spent)
|
|
225
230
|
if callable(self.on_exhausted):
|
|
226
231
|
raw = RawAnswer(
|
|
@@ -58,8 +58,13 @@ def load(spec: str | None = None) -> Engine:
|
|
|
58
58
|
from judgetap.engines.agentjev import DEFAULT_URL, AgentJevEngine
|
|
59
59
|
|
|
60
60
|
return AgentJevEngine(url=arg or DEFAULT_URL)
|
|
61
|
+
if name == "gliner":
|
|
62
|
+
from judgetap.engines.gliner import DEFAULT_MODEL as GLINER_MODEL
|
|
63
|
+
from judgetap.engines.gliner import GlinerEngine
|
|
64
|
+
|
|
65
|
+
return GlinerEngine(model=arg or GLINER_MODEL)
|
|
61
66
|
raise JudgetapError(
|
|
62
|
-
f"unknown engine {name!r} in spec {spec!r}; known: jev, jev@<url>, typesafe:<url>, llm, laya, julia, agentjev"
|
|
67
|
+
f"unknown engine {name!r} in spec {spec!r}; known: jev, jev@<url>, typesafe:<url>, llm, laya, julia, agentjev, gliner"
|
|
63
68
|
)
|
|
64
69
|
|
|
65
70
|
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""GLiNER2.5-Decide (Fastino, Apache-2.0) in process, through `gliner2`.
|
|
2
|
+
|
|
3
|
+
A 340M DeBERTa encoder that picks labels for several questions ("heads") in
|
|
4
|
+
one forward pass, on CPU or GPU (huggingface.co/fastino/GLiNER2.5-Decide).
|
|
5
|
+
Needs `pip install 'judgetap[gliner]'`; the checkpoint downloads from
|
|
6
|
+
Hugging Face on first use.
|
|
7
|
+
|
|
8
|
+
What comes back depends on the gliner2 code path. Only a full `probabilities`
|
|
9
|
+
map covering every option is marked `calibrated`; anything less is
|
|
10
|
+
`calibrated=False`, so a cascade escalates it to its next engine whatever
|
|
11
|
+
its p:
|
|
12
|
+
- a full `probabilities` map per head (the classifier path): renormalised
|
|
13
|
+
over the options, calibrated;
|
|
14
|
+
- a map missing some options: renormalised over the ones present, uncalibrated;
|
|
15
|
+
- `{"label", "confidence"}` (the extractor path): the winner gets its score
|
|
16
|
+
and the other labels split the rest evenly, uncalibrated (gliner2 doesn't
|
|
17
|
+
report them, and the score isn't a calibrated probability);
|
|
18
|
+
- a bare label (older versions): p=1.0, uncalibrated.
|
|
19
|
+
|
|
20
|
+
Loaded models are cached per process (by model id, at most
|
|
21
|
+
MAX_CACHED_MODELS). The guard refuses this engine: its hook is a new process
|
|
22
|
+
per action, so the model would load on every guarded action.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import asyncio
|
|
28
|
+
import json
|
|
29
|
+
import threading
|
|
30
|
+
from collections.abc import Mapping, Sequence
|
|
31
|
+
from typing import Any
|
|
32
|
+
|
|
33
|
+
from judgetap.engine import Context, RawAnswer, plain_context
|
|
34
|
+
from judgetap.errors import JudgetapError
|
|
35
|
+
from judgetap.types import NO, YES, Question
|
|
36
|
+
|
|
37
|
+
DEFAULT_MODEL = "fastino/GLiNER2.5-Decide"
|
|
38
|
+
MAX_CACHED_MODELS = 2
|
|
39
|
+
_models: dict[str, Any] = {}
|
|
40
|
+
_models_lock = threading.Lock()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class GlinerError(JudgetapError):
|
|
44
|
+
"""gliner2 is missing, failed to load or classify, or answered in an
|
|
45
|
+
unexpected shape. The underlying exception is chained."""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _task(q: Question) -> dict[str, Any]:
|
|
49
|
+
"""One classification head: its labels, with the question as the prompt."""
|
|
50
|
+
labels = [YES, NO] if q.kind == "yesno" else list(q.options)
|
|
51
|
+
return {"labels": labels, "prompt": q.text}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _text(context: Context) -> str:
|
|
55
|
+
state = plain_context(context)
|
|
56
|
+
return state if isinstance(state, str) else json.dumps(state)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _distribution(q: Question, answer: Any) -> tuple[dict[str, float], bool]:
|
|
60
|
+
"""(distribution over the question's options, calibrated)."""
|
|
61
|
+
options = list(q.options)
|
|
62
|
+
if isinstance(answer, Mapping) and isinstance(answer.get("probabilities"), Mapping):
|
|
63
|
+
probs = {str(k): float(v) for k, v in answer["probabilities"].items()}
|
|
64
|
+
total = sum(probs.get(o, 0.0) for o in options)
|
|
65
|
+
if total > 0:
|
|
66
|
+
complete = all(o in probs for o in options)
|
|
67
|
+
return {o: probs.get(o, 0.0) / total for o in options}, complete
|
|
68
|
+
if isinstance(answer, Mapping) and "label" in answer:
|
|
69
|
+
label, p = str(answer["label"]), float(answer.get("confidence", 1.0))
|
|
70
|
+
rest = (1.0 - p) / (len(options) - 1) if len(options) > 1 else 0.0
|
|
71
|
+
return {o: (p if o == label else rest) for o in options}, False
|
|
72
|
+
if isinstance(answer, Mapping) and "value" in answer:
|
|
73
|
+
answer = answer["value"]
|
|
74
|
+
label = str(answer)
|
|
75
|
+
return {o: (1.0 if o == label else 0.0) for o in options}, False
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class GlinerEngine:
|
|
79
|
+
def __init__(self, model: str = DEFAULT_MODEL, *, extractor: Any = None) -> None:
|
|
80
|
+
self.name = "gliner"
|
|
81
|
+
self.model = model
|
|
82
|
+
self._extractor = extractor
|
|
83
|
+
|
|
84
|
+
def _get(self) -> Any:
|
|
85
|
+
if self._extractor is not None:
|
|
86
|
+
return self._extractor
|
|
87
|
+
# One load per process and model, even with concurrent first calls.
|
|
88
|
+
with _models_lock:
|
|
89
|
+
if self.model not in _models:
|
|
90
|
+
try:
|
|
91
|
+
from gliner2 import AutoExtractor
|
|
92
|
+
except ImportError as err:
|
|
93
|
+
raise GlinerError(
|
|
94
|
+
"the gliner engine needs gliner2: pip install 'judgetap[gliner]'"
|
|
95
|
+
) from err
|
|
96
|
+
try:
|
|
97
|
+
_models[self.model] = AutoExtractor.from_pretrained(self.model)
|
|
98
|
+
except Exception as err:
|
|
99
|
+
raise GlinerError(f"could not load {self.model}: {err}") from err
|
|
100
|
+
while len(_models) > MAX_CACHED_MODELS:
|
|
101
|
+
del _models[next(iter(_models))] # oldest load first
|
|
102
|
+
else:
|
|
103
|
+
_models[self.model] = _models.pop(self.model) # most recently used
|
|
104
|
+
self._extractor = _models[self.model]
|
|
105
|
+
return self._extractor
|
|
106
|
+
|
|
107
|
+
def decide(
|
|
108
|
+
self, questions: Sequence[Question], context: Context
|
|
109
|
+
) -> Sequence[RawAnswer]:
|
|
110
|
+
ids = [f"q{i}" for i in range(len(questions))]
|
|
111
|
+
tasks = {i: _task(q) for i, q in zip(ids, questions, strict=True)}
|
|
112
|
+
extractor, text = self._get(), _text(context)
|
|
113
|
+
try:
|
|
114
|
+
try:
|
|
115
|
+
result = extractor.classify_text(text, tasks, include_confidence=True)
|
|
116
|
+
except TypeError:
|
|
117
|
+
result = extractor.classify_text(text, tasks) # no confidence support
|
|
118
|
+
except Exception as err:
|
|
119
|
+
raise GlinerError(f"gliner2 classification failed: {err}") from err
|
|
120
|
+
try:
|
|
121
|
+
answers = []
|
|
122
|
+
for i, q in zip(ids, questions, strict=True):
|
|
123
|
+
dist, calibrated = _distribution(q, result[i])
|
|
124
|
+
answers.append(RawAnswer(dist, cost_usd=0.0, calibrated=calibrated))
|
|
125
|
+
return answers
|
|
126
|
+
except (KeyError, TypeError, ValueError) as err:
|
|
127
|
+
raise GlinerError(f"unexpected gliner2 result shape: {err!r}") from err
|
|
128
|
+
|
|
129
|
+
async def adecide(
|
|
130
|
+
self, questions: Sequence[Question], context: Context
|
|
131
|
+
) -> Sequence[RawAnswer]:
|
|
132
|
+
return await asyncio.to_thread(self.decide, questions, context)
|
|
@@ -13,6 +13,7 @@ from __future__ import annotations
|
|
|
13
13
|
|
|
14
14
|
import json
|
|
15
15
|
import math
|
|
16
|
+
import sys
|
|
16
17
|
from collections.abc import Iterable, Sequence
|
|
17
18
|
from dataclasses import asdict, dataclass, field
|
|
18
19
|
from pathlib import Path
|
|
@@ -194,13 +195,35 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|
|
194
195
|
parser = argparse.ArgumentParser(
|
|
195
196
|
prog="judgetap eval", description=__doc__.split("\n")[0]
|
|
196
197
|
)
|
|
197
|
-
parser.add_argument("cases", help="JSONL file of labelled cases")
|
|
198
|
+
parser.add_argument("cases", nargs="?", help="JSONL file of labelled cases")
|
|
199
|
+
parser.add_argument(
|
|
200
|
+
"--suite", help="public suite instead of a file (ag_news, banking77)"
|
|
201
|
+
)
|
|
202
|
+
parser.add_argument(
|
|
203
|
+
"--limit", type=int, help="deterministic sample of N cases (fixed seed)"
|
|
204
|
+
)
|
|
198
205
|
parser.add_argument(
|
|
199
206
|
"--engines", required=True, help="comma-separated specs, e.g. jev,laya"
|
|
200
207
|
)
|
|
201
208
|
parser.add_argument("--json", action="store_true", help="JSON instead of Markdown")
|
|
202
209
|
args = parser.parse_args(argv)
|
|
203
|
-
cases
|
|
210
|
+
if (args.cases is None) == (args.suite is None):
|
|
211
|
+
parser.error("give exactly one of a cases file or --suite")
|
|
212
|
+
if args.suite is not None:
|
|
213
|
+
from judgetap.suites import suite_path
|
|
214
|
+
|
|
215
|
+
if not args.suite.strip():
|
|
216
|
+
parser.error("--suite needs a suite name (ag_news, banking77)")
|
|
217
|
+
try:
|
|
218
|
+
path: str | Path = suite_path(args.suite)
|
|
219
|
+
except JudgetapError as err:
|
|
220
|
+
print(f"judgetap eval: {err}", file=sys.stderr)
|
|
221
|
+
return 1
|
|
222
|
+
else:
|
|
223
|
+
path = args.cases
|
|
224
|
+
from judgetap.suites import sample
|
|
225
|
+
|
|
226
|
+
cases = sample(load_cases(path), args.limit)
|
|
204
227
|
reports = [evaluate(cases, load(spec)) for spec in args.engines.split(",")]
|
|
205
228
|
print(to_json(reports) if args.json else to_markdown(reports), end="")
|
|
206
229
|
return 0
|
|
@@ -167,7 +167,11 @@ def _engine():
|
|
|
167
167
|
if not spec:
|
|
168
168
|
return None
|
|
169
169
|
from judgetap.engines import load
|
|
170
|
+
from judgetap.guard.install import in_process_error
|
|
170
171
|
|
|
172
|
+
if error := in_process_error(spec):
|
|
173
|
+
# Caught by the caller: rules only, failing closed.
|
|
174
|
+
raise RuntimeError(error)
|
|
171
175
|
engine = load(spec)
|
|
172
176
|
# The engine looked the key up once when built; reuse that, don't hit the
|
|
173
177
|
# keychain a second time on every guarded action.
|
|
@@ -292,14 +296,30 @@ def run(
|
|
|
292
296
|
record: bool = True,
|
|
293
297
|
agent: str = "claude-code",
|
|
294
298
|
) -> int:
|
|
295
|
-
"""Hook entry point; returns the exit code.
|
|
296
|
-
|
|
299
|
+
"""Hook entry point; returns the exit code. Input the guard can't read
|
|
300
|
+
asks the user (it never saw the action, so it can't vouch for it); any
|
|
301
|
+
later failure allows the action and says so."""
|
|
297
302
|
start = time.perf_counter()
|
|
298
303
|
code, err_text = 0, ""
|
|
299
304
|
try:
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
305
|
+
try:
|
|
306
|
+
payload = json.load(stdin)
|
|
307
|
+
if not isinstance(payload, dict):
|
|
308
|
+
raise TypeError(f"expected a JSON object, got {type(payload).__name__}")
|
|
309
|
+
payload = normalise(payload, agent)
|
|
310
|
+
action = action_from_hook(payload)
|
|
311
|
+
except Exception as err: # noqa: BLE001 -- unreadable input fails closed
|
|
312
|
+
verdict = Verdict(
|
|
313
|
+
"ask",
|
|
314
|
+
"none",
|
|
315
|
+
"couldn't read the hook input",
|
|
316
|
+
error=f"{type(err).__name__}: {err}",
|
|
317
|
+
)
|
|
318
|
+
out, code, err_text = respond(verdict, agent)
|
|
319
|
+
action = payload = None
|
|
320
|
+
if payload is None:
|
|
321
|
+
pass
|
|
322
|
+
elif action is None:
|
|
303
323
|
out = {"permission": "allow"} if agent == "cursor" else None
|
|
304
324
|
else:
|
|
305
325
|
verdict = _decide(action, start, payload.get("session_id"))
|
|
@@ -205,7 +205,24 @@ SPEC_PATTERN = re.compile(
|
|
|
205
205
|
) # [ ] for IPv6 hosts, ? for llm options
|
|
206
206
|
|
|
207
207
|
|
|
208
|
-
KNOWN_ENGINES = frozenset(
|
|
208
|
+
KNOWN_ENGINES = frozenset(
|
|
209
|
+
{"jev", "llm", "laya", "julia", "agentjev", "typesafe", "gliner"}
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
# Engines that load a model into the calling process. The guard hook is a new
|
|
214
|
+
# process per action, so these would reload the model on every guarded action.
|
|
215
|
+
IN_PROCESS_ENGINES = frozenset({"gliner"})
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def in_process_error(spec: str) -> str | None:
|
|
219
|
+
name = spec.partition(":")[0].partition("@")[0]
|
|
220
|
+
if name in IN_PROCESS_ENGINES:
|
|
221
|
+
return (
|
|
222
|
+
f"{name} loads its model in-process, and the guard hook is a new process "
|
|
223
|
+
"per action; use a server engine for the guard"
|
|
224
|
+
)
|
|
225
|
+
return None
|
|
209
226
|
|
|
210
227
|
|
|
211
228
|
def validate_engine(spec: str):
|
|
@@ -217,6 +234,8 @@ def validate_engine(spec: str):
|
|
|
217
234
|
raise ValueError(
|
|
218
235
|
f"unknown engine {spec!r}; known: {', '.join(sorted(KNOWN_ENGINES))}"
|
|
219
236
|
)
|
|
237
|
+
if error := in_process_error(spec):
|
|
238
|
+
raise ValueError(error)
|
|
220
239
|
from judgetap.engines import load
|
|
221
240
|
|
|
222
241
|
try:
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""Public eval suites, converted to judgetap's case format on demand.
|
|
2
|
+
|
|
3
|
+
Nothing here is vendored: the first ``judgetap eval --suite <name>`` pages the
|
|
4
|
+
split out of the Hugging Face datasets-server JSON API (plain HTTPS, stdlib
|
|
5
|
+
only), converts each row to a ``choice`` case and caches the JSONL under
|
|
6
|
+
``~/.judgetap/suites`` (override with ``JUDGETAP_SUITES_DIR``). Later runs read
|
|
7
|
+
the cache. ``--limit N`` takes a deterministic sample: a fixed seed, so the same
|
|
8
|
+
N cases every run and on every machine.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import random
|
|
16
|
+
import urllib.parse
|
|
17
|
+
import urllib.request
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from judgetap.errors import JudgetapError
|
|
23
|
+
|
|
24
|
+
ROWS_API = "https://datasets-server.huggingface.co/rows"
|
|
25
|
+
PAGE = 100 # the rows API's maximum page size
|
|
26
|
+
SEED = 88
|
|
27
|
+
TIMEOUT = 30
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class Suite:
|
|
32
|
+
name: str
|
|
33
|
+
dataset: str # Hugging Face dataset id
|
|
34
|
+
split: str
|
|
35
|
+
question: str
|
|
36
|
+
licence: str
|
|
37
|
+
source: str # human-readable page for the dataset
|
|
38
|
+
config: str = "default"
|
|
39
|
+
text_field: str = "text"
|
|
40
|
+
label_field: str = "label"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
REGISTRY: dict[str, Suite] = {
|
|
44
|
+
s.name: s
|
|
45
|
+
for s in (
|
|
46
|
+
Suite(
|
|
47
|
+
name="ag_news",
|
|
48
|
+
dataset="fancyzhx/ag_news",
|
|
49
|
+
split="test",
|
|
50
|
+
question="Which topic is this news article about?",
|
|
51
|
+
licence="unknown; AG's corpus is provided for non-commercial research use",
|
|
52
|
+
source="https://huggingface.co/datasets/fancyzhx/ag_news",
|
|
53
|
+
),
|
|
54
|
+
Suite(
|
|
55
|
+
name="banking77",
|
|
56
|
+
dataset="legacy-datasets/banking77",
|
|
57
|
+
split="test",
|
|
58
|
+
question="Which intent does this banking customer message express?",
|
|
59
|
+
licence="CC-BY-4.0",
|
|
60
|
+
source="https://huggingface.co/datasets/legacy-datasets/banking77",
|
|
61
|
+
),
|
|
62
|
+
)
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def cache_dir() -> Path:
|
|
67
|
+
override = os.environ.get("JUDGETAP_SUITES_DIR")
|
|
68
|
+
return Path(override) if override else Path.home() / ".judgetap" / "suites"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _get_json(url: str) -> dict[str, Any]:
|
|
72
|
+
req = urllib.request.Request(url, headers={"User-Agent": "judgetap-eval"})
|
|
73
|
+
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
|
|
74
|
+
return json.load(resp)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _page_url(suite: Suite, offset: int) -> str:
|
|
78
|
+
query = urllib.parse.urlencode(
|
|
79
|
+
{
|
|
80
|
+
"dataset": suite.dataset,
|
|
81
|
+
"config": suite.config,
|
|
82
|
+
"split": suite.split,
|
|
83
|
+
"offset": offset,
|
|
84
|
+
"length": PAGE,
|
|
85
|
+
}
|
|
86
|
+
)
|
|
87
|
+
return f"{ROWS_API}?{query}"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _label_names(features: list[dict[str, Any]], field: str) -> list[str]:
|
|
91
|
+
for feature in features:
|
|
92
|
+
if feature.get("name") == field:
|
|
93
|
+
# The live rows API puts the ClassLabel under "type" (checked
|
|
94
|
+
# 2026-09-27); "feature" is accepted too, the key older docs name.
|
|
95
|
+
for key in ("type", "feature"):
|
|
96
|
+
spec = feature.get(key)
|
|
97
|
+
names = spec.get("names") if isinstance(spec, dict) else None
|
|
98
|
+
if isinstance(names, list) and names:
|
|
99
|
+
return [str(n) for n in names]
|
|
100
|
+
raise JudgetapError(f"no class-label names for {field!r} in the dataset")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _to_case(suite: Suite, names: list[str], item: Any, index: int) -> dict[str, Any]:
|
|
104
|
+
try:
|
|
105
|
+
row = item["row"]
|
|
106
|
+
context = row[suite.text_field]
|
|
107
|
+
label = int(row[suite.label_field])
|
|
108
|
+
if not isinstance(context, str) or not 0 <= label < len(names):
|
|
109
|
+
raise ValueError(f"label {label} or text is out of range")
|
|
110
|
+
except (KeyError, TypeError, ValueError) as err:
|
|
111
|
+
raise JudgetapError(
|
|
112
|
+
f"suite {suite.name!r}: malformed row {index}: {err!r}"
|
|
113
|
+
) from err
|
|
114
|
+
return {
|
|
115
|
+
"kind": "choice",
|
|
116
|
+
"question": suite.question,
|
|
117
|
+
"options": names,
|
|
118
|
+
"context": context,
|
|
119
|
+
"label": names[label],
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def download(suite: Suite) -> list[dict[str, Any]]:
|
|
124
|
+
"""Every row of the split, as judgetap case dicts."""
|
|
125
|
+
cases: list[dict[str, Any]] = []
|
|
126
|
+
names: list[str] | None = None
|
|
127
|
+
offset, total = 0, None
|
|
128
|
+
while total is None or offset < total:
|
|
129
|
+
try:
|
|
130
|
+
page = _get_json(_page_url(suite, offset))
|
|
131
|
+
except (OSError, ValueError) as err:
|
|
132
|
+
raise JudgetapError(f"downloading suite {suite.name!r}: {err}") from err
|
|
133
|
+
if names is None:
|
|
134
|
+
names = _label_names(page.get("features", []), suite.label_field)
|
|
135
|
+
total = int(page.get("num_rows_total", 0))
|
|
136
|
+
rows = page.get("rows", [])
|
|
137
|
+
if not rows:
|
|
138
|
+
if offset < total:
|
|
139
|
+
raise JudgetapError(
|
|
140
|
+
f"suite {suite.name!r}: empty page at offset {offset} "
|
|
141
|
+
f"of {total} rows; refusing to cache a partial suite"
|
|
142
|
+
)
|
|
143
|
+
break
|
|
144
|
+
for item in rows:
|
|
145
|
+
cases.append(_to_case(suite, names, item, offset + len(cases)))
|
|
146
|
+
offset += len(rows)
|
|
147
|
+
if not cases:
|
|
148
|
+
raise JudgetapError(f"suite {suite.name!r} downloaded no rows")
|
|
149
|
+
return cases
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def suite_path(name: str) -> Path:
|
|
153
|
+
"""The cached JSONL for ``name``, downloading it the first time."""
|
|
154
|
+
suite = REGISTRY.get(name)
|
|
155
|
+
if suite is None:
|
|
156
|
+
raise JudgetapError(
|
|
157
|
+
f"unknown suite {name!r}; available: {', '.join(sorted(REGISTRY))}"
|
|
158
|
+
)
|
|
159
|
+
path = cache_dir() / f"{suite.name}-{suite.split}.jsonl"
|
|
160
|
+
if path.exists():
|
|
161
|
+
return path
|
|
162
|
+
cases = download(suite)
|
|
163
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
164
|
+
tmp = path.with_suffix(".jsonl.part")
|
|
165
|
+
tmp.write_text("".join(json.dumps(c) + "\n" for c in cases))
|
|
166
|
+
tmp.replace(path) # a half-written download never looks cached
|
|
167
|
+
return path
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def sample(items: list[Any], limit: int | None) -> list[Any]:
|
|
171
|
+
"""A fixed-seed sample of ``limit`` items, kept in their original order."""
|
|
172
|
+
if limit is None or limit >= len(items):
|
|
173
|
+
return items
|
|
174
|
+
if limit < 1:
|
|
175
|
+
raise JudgetapError("--limit must be at least 1")
|
|
176
|
+
picked = sorted(random.Random(SEED).sample(range(len(items)), limit))
|
|
177
|
+
return [items[i] for i in picked]
|
|
@@ -229,3 +229,27 @@ def test_from_config_scalar_cascade_is_a_config_error(tmp_path):
|
|
|
229
229
|
cfg.write_text('cascade = "jev"\n')
|
|
230
230
|
with pytest.raises(sj.JudgetapError):
|
|
231
231
|
from_config(cfg)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
class Uncalibrated(StaticEngine):
|
|
235
|
+
def decide(self, questions, context):
|
|
236
|
+
self.calls.append((tuple(questions), context))
|
|
237
|
+
return [
|
|
238
|
+
sj.RawAnswer({"yes": 0.0, "no": 1.0}, calibrated=False) for _ in questions
|
|
239
|
+
]
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def test_uncalibrated_answer_escalates_whatever_its_p():
|
|
243
|
+
bare, strong = Uncalibrated(lambda q, c: {}, name="bare"), eng("strong", 0.9)
|
|
244
|
+
d = sj.yesno("q", engine=sj.Cascade([bare, strong]))
|
|
245
|
+
assert (d.engine, d.meta["hops"]) == ("strong", ["bare", "strong"])
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def test_last_engine_uncalibrated_answer_is_judged_on_p():
|
|
249
|
+
d = sj.yesno(
|
|
250
|
+
"q",
|
|
251
|
+
engine=sj.Cascade(
|
|
252
|
+
[eng("cheap", 0.6), Uncalibrated(lambda q, c: {}, name="bare")]
|
|
253
|
+
),
|
|
254
|
+
)
|
|
255
|
+
assert d.engine == "bare" and d.calibrated is False
|