judgetap 0.2.1.dev42001__tar.gz → 0.2.1.dev44001__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/PKG-INFO +5 -2
  2. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/README.md +2 -1
  3. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/docs/SPEC.md +2 -1
  4. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/pyproject.toml +2 -1
  5. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/__init__.py +1 -1
  6. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/cascade.py +10 -5
  7. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/__init__.py +6 -1
  8. judgetap-0.2.1.dev44001/src/judgetap/engines/gliner.py +132 -0
  9. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/evaluate.py +25 -2
  10. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/hook.py +25 -5
  11. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/install.py +20 -1
  12. judgetap-0.2.1.dev44001/src/judgetap/suites.py +177 -0
  13. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_cascade.py +24 -0
  14. judgetap-0.2.1.dev44001/tests/test_engine_gliner.py +213 -0
  15. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_agents.py +1 -1
  16. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_hook.py +8 -3
  17. judgetap-0.2.1.dev44001/tests/test_suites.py +169 -0
  18. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/uv.lock +1594 -1589
  19. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.github/workflows/ci.yml +0 -0
  20. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.github/workflows/demo.yml +0 -0
  21. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.github/workflows/release.yml +0 -0
  22. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.gitignore +0 -0
  23. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.python-version +0 -0
  24. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/.release-please-manifest.json +0 -0
  25. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/CHANGELOG.md +0 -0
  26. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/CONTRIBUTING.md +0 -0
  27. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/LICENSE +0 -0
  28. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/docs/demo.tape +0 -0
  29. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/release-please-config.json +0 -0
  30. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/_compat.py +0 -0
  31. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/api.py +0 -0
  32. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/cli.py +0 -0
  33. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/dashboard/__init__.py +0 -0
  34. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/dashboard/data.py +0 -0
  35. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/dashboard/page.html +0 -0
  36. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/dashboard/server.py +0 -0
  37. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/decision_log.py +0 -0
  38. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engine.py +0 -0
  39. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/agentjev.py +0 -0
  40. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/jev.py +0 -0
  41. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/julia.py +0 -0
  42. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/laya.py +0 -0
  43. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/engines/llm.py +0 -0
  44. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/errors.py +0 -0
  45. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/__init__.py +0 -0
  46. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/core.py +0 -0
  47. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/loop.py +0 -0
  48. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/rules.py +0 -0
  49. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/guard/stop.py +0 -0
  50. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/py.typed +0 -0
  51. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/secrets.py +0 -0
  52. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/testing.py +0 -0
  53. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/src/judgetap/types.py +0 -0
  54. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_api.py +0 -0
  55. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_call_accounting.py +0 -0
  56. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_calls.py +0 -0
  57. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_dashboard.py +0 -0
  58. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_decision_log.py +0 -0
  59. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_engine_jev.py +0 -0
  60. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_engine_julia.py +0 -0
  61. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_engine_llm.py +0 -0
  62. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_engine_local.py +0 -0
  63. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_evaluate.py +0 -0
  64. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_core.py +0 -0
  65. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_loop.py +0 -0
  66. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_polish_76.py +0 -0
  67. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_rules.py +0 -0
  68. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_stop.py +0 -0
  69. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_guard_trust.py +0 -0
  70. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_llm_logprobs.py +0 -0
  71. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_questions.py +0 -0
  72. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_robustness_68.py +0 -0
  73. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev44001}/tests/test_secrets.py +0 -0
@@ -1,12 +1,14 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: judgetap
3
- Version: 0.2.1.dev42001
3
+ Version: 0.2.1.dev44001
4
4
  Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
5
5
  Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
6
6
  Author-email: Omer Bar-Ness <omer@zsquared.io>
7
7
  License-Expression: Apache-2.0
8
8
  License-File: LICENSE
9
9
  Requires-Python: >=3.12
10
+ Provides-Extra: gliner
11
+ Requires-Dist: gliner2; extra == 'gliner'
10
12
  Provides-Extra: keychain
11
13
  Requires-Dist: keyring>=24; extra == 'keychain'
12
14
  Provides-Extra: laya
@@ -44,7 +46,7 @@ judgetap guard test "git push --force origin main"
44
46
 
45
47
  - **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
46
48
  - **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
47
- - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
49
+ - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost. `judgetap eval --suite banking77 --limit 200 --engines jev` runs a public suite instead (`ag_news`, `banking77`): it is downloaded from Hugging Face on first use, converted to judgetap's case format and cached in `~/.judgetap/suites`; nothing is bundled. `--limit N` takes the same fixed-seed sample every run. Each suite's licence and source URL are listed in `src/judgetap/suites.py`.
48
50
  - **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
49
51
 
50
52
  ## Library
@@ -74,6 +76,7 @@ sj.configure(sj.Cascade([jev, flash]))
74
76
  | `jev@<url>` / `typesafe:<url>` | Any TypeSafe-compatible server | none on localhost; remote needs the key and https |
75
77
  | `laya` | Laya, open weights, runs in process (`pip install "judgetap[laya]"`) | none |
76
78
  | `julia` / `julia:<path>` | [Julia-1](https://huggingface.co/SupersonicLabs/Julia-1), 144M open weights, runs in process on CPU (download into `~/.judgetap/models/Julia-1` and `pip install -e` it; relative paths never come from the working directory). Loads per process, so for the guard prefer a server engine | none |
79
+ | `gliner[:<hf model>]` | Fastino GLiNER2.5-Decide, 340M encoder, CPU or GPU (`pip install "judgetap[gliner]"`); only a full probability map over every option counts as calibrated; a winner-only score (the others share the rest evenly) or a bare label is uncalibrated. Loads in-process, so the guard refuses it: use a server engine there | none |
77
80
  | `agentjev` / `agentjev:<url>` | A local AgentJev server | none |
78
81
  | `llm:<model>` | Any LiteLLM model (`pip install "judgetap[llm]"`); probabilities self-reported | the provider's |
79
82
 
@@ -27,7 +27,7 @@ judgetap guard test "git push --force origin main"
27
27
 
28
28
  - **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
29
29
  - **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
30
- - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
30
+ - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost. `judgetap eval --suite banking77 --limit 200 --engines jev` runs a public suite instead (`ag_news`, `banking77`): it is downloaded from Hugging Face on first use, converted to judgetap's case format and cached in `~/.judgetap/suites`; nothing is bundled. `--limit N` takes the same fixed-seed sample every run. Each suite's licence and source URL are listed in `src/judgetap/suites.py`.
31
31
  - **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
32
32
 
33
33
  ## Library
@@ -57,6 +57,7 @@ sj.configure(sj.Cascade([jev, flash]))
57
57
  | `jev@<url>` / `typesafe:<url>` | Any TypeSafe-compatible server | none on localhost; remote needs the key and https |
58
58
  | `laya` | Laya, open weights, runs in process (`pip install "judgetap[laya]"`) | none |
59
59
  | `julia` / `julia:<path>` | [Julia-1](https://huggingface.co/SupersonicLabs/Julia-1), 144M open weights, runs in process on CPU (download into `~/.judgetap/models/Julia-1` and `pip install -e` it; relative paths never come from the working directory). Loads per process, so for the guard prefer a server engine | none |
60
+ | `gliner[:<hf model>]` | Fastino GLiNER2.5-Decide, 340M encoder, CPU or GPU (`pip install "judgetap[gliner]"`); only a full probability map over every option counts as calibrated; a winner-only score (the others share the rest evenly) or a bare label is uncalibrated. Loads in-process, so the guard refuses it: use a server engine there | none |
60
61
  | `agentjev` / `agentjev:<url>` | A local AgentJev server | none |
61
62
  | `llm:<model>` | Any LiteLLM model (`pip install "judgetap[llm]"`); probabilities self-reported | the provider's |
62
63
 
@@ -37,6 +37,7 @@ sj.batch([...questions], context=...) -> list[Decision] # one pa
37
37
  | `jev` | hosted (TypeSafe console, Vercel AI Gateway) | native batch; the reference shape |
38
38
  | `laya` | local, open weights (Apache-2.0) | via transformers; GPU optional |
39
39
  | `julia` | local, open weights (Apache-2.0), 144M | Julia-1 runtime from its model repo; CPU by default; Jev-shaped API; 2-20 options per question |
40
+ | `gliner` | local, open weights (Apache-2.0), Fastino GLiNER2.5-Decide via `gliner2` | all questions as heads in one pass; full probabilities when gliner2 returns them, else the winner's probability with the rest split evenly, else the label alone (`calibrated=False`) |
40
41
  | `agentjev` | local, open weights | ~50 ms per pass |
41
42
  | `llm` | any structured-output LLM via LiteLLM | OpenAI, Gemini, Anthropic, Ollama; probabilities are the model's own JSON estimate, flagged `calibrated=False` (see `?logprobs` below) |
42
43
  | `llm:<model>?logprobs` | same, OpenAI-compatible servers exposing `logprobs` | options listed as letters, one single-token call per question; the distribution is the letters' `top_logprobs` renormalised (missing letters get 0, none present is an error); token probabilities, not calibrated (`calibrated=False`; check with `judgetap eval`); per-question calls run concurrently (max 8) and each is reported as a Call; more than 26 options is a local error raised before any request; falls back to JSON mode for good if the provider rejects logprobs |
@@ -52,7 +53,7 @@ escalate_below = 0.8 # p under this goes to the next engine
52
53
  on_exhausted = "raise" # or "return_last", or a callback (e.g. ask a human)
53
54
  ```
54
55
 
55
- Each hop is recorded on the `Decision`. Engine errors and timeouts fall through the same way.
56
+ Each hop is recorded on the `Decision`. Engine errors and timeouts fall through the same way, and so does an uncalibrated answer (`calibrated=False`) whatever its p, while an engine is left to ask; the last engine's answer is judged on p alone.
56
57
 
57
58
  ### 4. Calibration check
58
59
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "judgetap"
3
- version = "0.2.1.dev42001"
3
+ version = "0.2.1.dev44001"
4
4
  description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -17,6 +17,7 @@ snapjudge = "judgetap.cli:main" # deprecated alias, removed after one release
17
17
  [project.optional-dependencies]
18
18
  llm = ["litellm>=1.40"]
19
19
  laya = ["laya"]
20
+ gliner = ["gliner2"]
20
21
  keychain = ["keyring>=24"]
21
22
 
22
23
  [build-system]
@@ -23,7 +23,7 @@ from judgetap.errors import (
23
23
  )
24
24
  from judgetap.types import Decision, Question
25
25
 
26
- __version__ = "0.2.1.dev42001" # x-release-please-version
26
+ __version__ = "0.2.1.dev44001" # x-release-please-version
27
27
 
28
28
  __all__ = [
29
29
  "Cascade",
@@ -175,11 +175,16 @@ class Cascade:
175
175
  def _pending(self, questions, attempts) -> list[int]:
176
176
  return [i for i in range(len(questions)) if not self._confident(attempts[i])]
177
177
 
178
- def _confident(self, attempts: Sequence[Attempt]) -> bool:
178
+ def _confident(self, attempts: Sequence[Attempt], *, final: bool = False) -> bool:
179
+ """Whether the latest answer stops the cascade. An uncalibrated p
180
+ (calibrated=False) never stops it while another engine is left to
181
+ ask; once none is (final), it is judged on p like any other."""
182
+ last = attempts[-1] if attempts else None
179
183
  return (
180
- bool(attempts)
181
- and attempts[-1].p is not None
182
- and (attempts[-1].p >= self.escalate_below)
184
+ last is not None
185
+ and last.p is not None
186
+ and last.p >= self.escalate_below
187
+ and (final or last.answer is None or last.answer.calibrated)
183
188
  )
184
189
 
185
190
  def _record(self, engine, questions, pending, answers, attempts) -> None:
@@ -220,7 +225,7 @@ class Cascade:
220
225
  hops = _hops(attempts)
221
226
  answered = [a for a in attempts if a.answer is not None]
222
227
  spent = _spent(attempts)
223
- if self._confident(attempts):
228
+ if self._confident(attempts, final=True):
224
229
  return _as_result(attempts[-1], hops, spent)
225
230
  if callable(self.on_exhausted):
226
231
  raw = RawAnswer(
@@ -58,8 +58,13 @@ def load(spec: str | None = None) -> Engine:
58
58
  from judgetap.engines.agentjev import DEFAULT_URL, AgentJevEngine
59
59
 
60
60
  return AgentJevEngine(url=arg or DEFAULT_URL)
61
+ if name == "gliner":
62
+ from judgetap.engines.gliner import DEFAULT_MODEL as GLINER_MODEL
63
+ from judgetap.engines.gliner import GlinerEngine
64
+
65
+ return GlinerEngine(model=arg or GLINER_MODEL)
61
66
  raise JudgetapError(
62
- f"unknown engine {name!r} in spec {spec!r}; known: jev, jev@<url>, typesafe:<url>, llm, laya, julia, agentjev"
67
+ f"unknown engine {name!r} in spec {spec!r}; known: jev, jev@<url>, typesafe:<url>, llm, laya, julia, agentjev, gliner"
63
68
  )
64
69
 
65
70
 
@@ -0,0 +1,132 @@
1
+ """GLiNER2.5-Decide (Fastino, Apache-2.0) in process, through `gliner2`.
2
+
3
+ A 340M DeBERTa encoder that picks labels for several questions ("heads") in
4
+ one forward pass, on CPU or GPU (huggingface.co/fastino/GLiNER2.5-Decide).
5
+ Needs `pip install 'judgetap[gliner]'`; the checkpoint downloads from
6
+ Hugging Face on first use.
7
+
8
+ What comes back depends on the gliner2 code path. Only a full `probabilities`
9
+ map covering every option is marked `calibrated`; anything less is
10
+ `calibrated=False`, so a cascade escalates it to its next engine whatever
11
+ its p:
12
+ - a full `probabilities` map per head (the classifier path): renormalised
13
+ over the options, calibrated;
14
+ - a map missing some options: renormalised over the ones present, uncalibrated;
15
+ - `{"label", "confidence"}` (the extractor path): the winner gets its score
16
+ and the other labels split the rest evenly, uncalibrated (gliner2 doesn't
17
+ report them, and the score isn't a calibrated probability);
18
+ - a bare label (older versions): p=1.0, uncalibrated.
19
+
20
+ Loaded models are cached per process (by model id, at most
21
+ MAX_CACHED_MODELS). The guard refuses this engine: its hook is a new process
22
+ per action, so the model would load on every guarded action.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import asyncio
28
+ import json
29
+ import threading
30
+ from collections.abc import Mapping, Sequence
31
+ from typing import Any
32
+
33
+ from judgetap.engine import Context, RawAnswer, plain_context
34
+ from judgetap.errors import JudgetapError
35
+ from judgetap.types import NO, YES, Question
36
+
37
+ DEFAULT_MODEL = "fastino/GLiNER2.5-Decide"
38
+ MAX_CACHED_MODELS = 2
39
+ _models: dict[str, Any] = {}
40
+ _models_lock = threading.Lock()
41
+
42
+
43
+ class GlinerError(JudgetapError):
44
+ """gliner2 is missing, failed to load or classify, or answered in an
45
+ unexpected shape. The underlying exception is chained."""
46
+
47
+
48
+ def _task(q: Question) -> dict[str, Any]:
49
+ """One classification head: its labels, with the question as the prompt."""
50
+ labels = [YES, NO] if q.kind == "yesno" else list(q.options)
51
+ return {"labels": labels, "prompt": q.text}
52
+
53
+
54
+ def _text(context: Context) -> str:
55
+ state = plain_context(context)
56
+ return state if isinstance(state, str) else json.dumps(state)
57
+
58
+
59
+ def _distribution(q: Question, answer: Any) -> tuple[dict[str, float], bool]:
60
+ """(distribution over the question's options, calibrated)."""
61
+ options = list(q.options)
62
+ if isinstance(answer, Mapping) and isinstance(answer.get("probabilities"), Mapping):
63
+ probs = {str(k): float(v) for k, v in answer["probabilities"].items()}
64
+ total = sum(probs.get(o, 0.0) for o in options)
65
+ if total > 0:
66
+ complete = all(o in probs for o in options)
67
+ return {o: probs.get(o, 0.0) / total for o in options}, complete
68
+ if isinstance(answer, Mapping) and "label" in answer:
69
+ label, p = str(answer["label"]), float(answer.get("confidence", 1.0))
70
+ rest = (1.0 - p) / (len(options) - 1) if len(options) > 1 else 0.0
71
+ return {o: (p if o == label else rest) for o in options}, False
72
+ if isinstance(answer, Mapping) and "value" in answer:
73
+ answer = answer["value"]
74
+ label = str(answer)
75
+ return {o: (1.0 if o == label else 0.0) for o in options}, False
76
+
77
+
78
+ class GlinerEngine:
79
+ def __init__(self, model: str = DEFAULT_MODEL, *, extractor: Any = None) -> None:
80
+ self.name = "gliner"
81
+ self.model = model
82
+ self._extractor = extractor
83
+
84
+ def _get(self) -> Any:
85
+ if self._extractor is not None:
86
+ return self._extractor
87
+ # One load per process and model, even with concurrent first calls.
88
+ with _models_lock:
89
+ if self.model not in _models:
90
+ try:
91
+ from gliner2 import AutoExtractor
92
+ except ImportError as err:
93
+ raise GlinerError(
94
+ "the gliner engine needs gliner2: pip install 'judgetap[gliner]'"
95
+ ) from err
96
+ try:
97
+ _models[self.model] = AutoExtractor.from_pretrained(self.model)
98
+ except Exception as err:
99
+ raise GlinerError(f"could not load {self.model}: {err}") from err
100
+ while len(_models) > MAX_CACHED_MODELS:
101
+ del _models[next(iter(_models))] # oldest load first
102
+ else:
103
+ _models[self.model] = _models.pop(self.model) # most recently used
104
+ self._extractor = _models[self.model]
105
+ return self._extractor
106
+
107
+ def decide(
108
+ self, questions: Sequence[Question], context: Context
109
+ ) -> Sequence[RawAnswer]:
110
+ ids = [f"q{i}" for i in range(len(questions))]
111
+ tasks = {i: _task(q) for i, q in zip(ids, questions, strict=True)}
112
+ extractor, text = self._get(), _text(context)
113
+ try:
114
+ try:
115
+ result = extractor.classify_text(text, tasks, include_confidence=True)
116
+ except TypeError:
117
+ result = extractor.classify_text(text, tasks) # no confidence support
118
+ except Exception as err:
119
+ raise GlinerError(f"gliner2 classification failed: {err}") from err
120
+ try:
121
+ answers = []
122
+ for i, q in zip(ids, questions, strict=True):
123
+ dist, calibrated = _distribution(q, result[i])
124
+ answers.append(RawAnswer(dist, cost_usd=0.0, calibrated=calibrated))
125
+ return answers
126
+ except (KeyError, TypeError, ValueError) as err:
127
+ raise GlinerError(f"unexpected gliner2 result shape: {err!r}") from err
128
+
129
+ async def adecide(
130
+ self, questions: Sequence[Question], context: Context
131
+ ) -> Sequence[RawAnswer]:
132
+ return await asyncio.to_thread(self.decide, questions, context)
@@ -13,6 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  import json
15
15
  import math
16
+ import sys
16
17
  from collections.abc import Iterable, Sequence
17
18
  from dataclasses import asdict, dataclass, field
18
19
  from pathlib import Path
@@ -194,13 +195,35 @@ def main(argv: Sequence[str] | None = None) -> int:
194
195
  parser = argparse.ArgumentParser(
195
196
  prog="judgetap eval", description=__doc__.split("\n")[0]
196
197
  )
197
- parser.add_argument("cases", help="JSONL file of labelled cases")
198
+ parser.add_argument("cases", nargs="?", help="JSONL file of labelled cases")
199
+ parser.add_argument(
200
+ "--suite", help="public suite instead of a file (ag_news, banking77)"
201
+ )
202
+ parser.add_argument(
203
+ "--limit", type=int, help="deterministic sample of N cases (fixed seed)"
204
+ )
198
205
  parser.add_argument(
199
206
  "--engines", required=True, help="comma-separated specs, e.g. jev,laya"
200
207
  )
201
208
  parser.add_argument("--json", action="store_true", help="JSON instead of Markdown")
202
209
  args = parser.parse_args(argv)
203
- cases = load_cases(args.cases)
210
+ if (args.cases is None) == (args.suite is None):
211
+ parser.error("give exactly one of a cases file or --suite")
212
+ if args.suite is not None:
213
+ from judgetap.suites import suite_path
214
+
215
+ if not args.suite.strip():
216
+ parser.error("--suite needs a suite name (ag_news, banking77)")
217
+ try:
218
+ path: str | Path = suite_path(args.suite)
219
+ except JudgetapError as err:
220
+ print(f"judgetap eval: {err}", file=sys.stderr)
221
+ return 1
222
+ else:
223
+ path = args.cases
224
+ from judgetap.suites import sample
225
+
226
+ cases = sample(load_cases(path), args.limit)
204
227
  reports = [evaluate(cases, load(spec)) for spec in args.engines.split(",")]
205
228
  print(to_json(reports) if args.json else to_markdown(reports), end="")
206
229
  return 0
@@ -167,7 +167,11 @@ def _engine():
167
167
  if not spec:
168
168
  return None
169
169
  from judgetap.engines import load
170
+ from judgetap.guard.install import in_process_error
170
171
 
172
+ if error := in_process_error(spec):
173
+ # Caught by the caller: rules only, failing closed.
174
+ raise RuntimeError(error)
171
175
  engine = load(spec)
172
176
  # The engine looked the key up once when built; reuse that, don't hit the
173
177
  # keychain a second time on every guarded action.
@@ -292,14 +296,30 @@ def run(
292
296
  record: bool = True,
293
297
  agent: str = "claude-code",
294
298
  ) -> int:
295
- """Hook entry point; returns the exit code. Failures never block: they
296
- allow the action and say so, since a crashing hook is ignored anyway."""
299
+ """Hook entry point; returns the exit code. Input the guard can't read
300
+ asks the user (it never saw the action, so it can't vouch for it); any
301
+ later failure allows the action and says so."""
297
302
  start = time.perf_counter()
298
303
  code, err_text = 0, ""
299
304
  try:
300
- payload = normalise(json.load(stdin), agent)
301
- action = action_from_hook(payload)
302
- if action is None:
305
+ try:
306
+ payload = json.load(stdin)
307
+ if not isinstance(payload, dict):
308
+ raise TypeError(f"expected a JSON object, got {type(payload).__name__}")
309
+ payload = normalise(payload, agent)
310
+ action = action_from_hook(payload)
311
+ except Exception as err: # noqa: BLE001 -- unreadable input fails closed
312
+ verdict = Verdict(
313
+ "ask",
314
+ "none",
315
+ "couldn't read the hook input",
316
+ error=f"{type(err).__name__}: {err}",
317
+ )
318
+ out, code, err_text = respond(verdict, agent)
319
+ action = payload = None
320
+ if payload is None:
321
+ pass
322
+ elif action is None:
303
323
  out = {"permission": "allow"} if agent == "cursor" else None
304
324
  else:
305
325
  verdict = _decide(action, start, payload.get("session_id"))
@@ -205,7 +205,24 @@ SPEC_PATTERN = re.compile(
205
205
  ) # [ ] for IPv6 hosts, ? for llm options
206
206
 
207
207
 
208
- KNOWN_ENGINES = frozenset({"jev", "llm", "laya", "julia", "agentjev", "typesafe"})
208
+ KNOWN_ENGINES = frozenset(
209
+ {"jev", "llm", "laya", "julia", "agentjev", "typesafe", "gliner"}
210
+ )
211
+
212
+
213
+ # Engines that load a model into the calling process. The guard hook is a new
214
+ # process per action, so these would reload the model on every guarded action.
215
+ IN_PROCESS_ENGINES = frozenset({"gliner"})
216
+
217
+
218
+ def in_process_error(spec: str) -> str | None:
219
+ name = spec.partition(":")[0].partition("@")[0]
220
+ if name in IN_PROCESS_ENGINES:
221
+ return (
222
+ f"{name} loads its model in-process, and the guard hook is a new process "
223
+ "per action; use a server engine for the guard"
224
+ )
225
+ return None
209
226
 
210
227
 
211
228
  def validate_engine(spec: str):
@@ -217,6 +234,8 @@ def validate_engine(spec: str):
217
234
  raise ValueError(
218
235
  f"unknown engine {spec!r}; known: {', '.join(sorted(KNOWN_ENGINES))}"
219
236
  )
237
+ if error := in_process_error(spec):
238
+ raise ValueError(error)
220
239
  from judgetap.engines import load
221
240
 
222
241
  try:
@@ -0,0 +1,177 @@
1
+ """Public eval suites, converted to judgetap's case format on demand.
2
+
3
+ Nothing here is vendored: the first ``judgetap eval --suite <name>`` pages the
4
+ split out of the Hugging Face datasets-server JSON API (plain HTTPS, stdlib
5
+ only), converts each row to a ``choice`` case and caches the JSONL under
6
+ ``~/.judgetap/suites`` (override with ``JUDGETAP_SUITES_DIR``). Later runs read
7
+ the cache. ``--limit N`` takes a deterministic sample: a fixed seed, so the same
8
+ N cases every run and on every machine.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import os
15
+ import random
16
+ import urllib.parse
17
+ import urllib.request
18
+ from dataclasses import dataclass
19
+ from pathlib import Path
20
+ from typing import Any
21
+
22
+ from judgetap.errors import JudgetapError
23
+
24
+ ROWS_API = "https://datasets-server.huggingface.co/rows"
25
+ PAGE = 100 # the rows API's maximum page size
26
+ SEED = 88
27
+ TIMEOUT = 30
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class Suite:
32
+ name: str
33
+ dataset: str # Hugging Face dataset id
34
+ split: str
35
+ question: str
36
+ licence: str
37
+ source: str # human-readable page for the dataset
38
+ config: str = "default"
39
+ text_field: str = "text"
40
+ label_field: str = "label"
41
+
42
+
43
+ REGISTRY: dict[str, Suite] = {
44
+ s.name: s
45
+ for s in (
46
+ Suite(
47
+ name="ag_news",
48
+ dataset="fancyzhx/ag_news",
49
+ split="test",
50
+ question="Which topic is this news article about?",
51
+ licence="unknown; AG's corpus is provided for non-commercial research use",
52
+ source="https://huggingface.co/datasets/fancyzhx/ag_news",
53
+ ),
54
+ Suite(
55
+ name="banking77",
56
+ dataset="legacy-datasets/banking77",
57
+ split="test",
58
+ question="Which intent does this banking customer message express?",
59
+ licence="CC-BY-4.0",
60
+ source="https://huggingface.co/datasets/legacy-datasets/banking77",
61
+ ),
62
+ )
63
+ }
64
+
65
+
66
+ def cache_dir() -> Path:
67
+ override = os.environ.get("JUDGETAP_SUITES_DIR")
68
+ return Path(override) if override else Path.home() / ".judgetap" / "suites"
69
+
70
+
71
+ def _get_json(url: str) -> dict[str, Any]:
72
+ req = urllib.request.Request(url, headers={"User-Agent": "judgetap-eval"})
73
+ with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
74
+ return json.load(resp)
75
+
76
+
77
+ def _page_url(suite: Suite, offset: int) -> str:
78
+ query = urllib.parse.urlencode(
79
+ {
80
+ "dataset": suite.dataset,
81
+ "config": suite.config,
82
+ "split": suite.split,
83
+ "offset": offset,
84
+ "length": PAGE,
85
+ }
86
+ )
87
+ return f"{ROWS_API}?{query}"
88
+
89
+
90
+ def _label_names(features: list[dict[str, Any]], field: str) -> list[str]:
91
+ for feature in features:
92
+ if feature.get("name") == field:
93
+ # The live rows API puts the ClassLabel under "type" (checked
94
+ # 2026-09-27); "feature" is accepted too, the key older docs name.
95
+ for key in ("type", "feature"):
96
+ spec = feature.get(key)
97
+ names = spec.get("names") if isinstance(spec, dict) else None
98
+ if isinstance(names, list) and names:
99
+ return [str(n) for n in names]
100
+ raise JudgetapError(f"no class-label names for {field!r} in the dataset")
101
+
102
+
103
+ def _to_case(suite: Suite, names: list[str], item: Any, index: int) -> dict[str, Any]:
104
+ try:
105
+ row = item["row"]
106
+ context = row[suite.text_field]
107
+ label = int(row[suite.label_field])
108
+ if not isinstance(context, str) or not 0 <= label < len(names):
109
+ raise ValueError(f"label {label} or text is out of range")
110
+ except (KeyError, TypeError, ValueError) as err:
111
+ raise JudgetapError(
112
+ f"suite {suite.name!r}: malformed row {index}: {err!r}"
113
+ ) from err
114
+ return {
115
+ "kind": "choice",
116
+ "question": suite.question,
117
+ "options": names,
118
+ "context": context,
119
+ "label": names[label],
120
+ }
121
+
122
+
123
+ def download(suite: Suite) -> list[dict[str, Any]]:
124
+ """Every row of the split, as judgetap case dicts."""
125
+ cases: list[dict[str, Any]] = []
126
+ names: list[str] | None = None
127
+ offset, total = 0, None
128
+ while total is None or offset < total:
129
+ try:
130
+ page = _get_json(_page_url(suite, offset))
131
+ except (OSError, ValueError) as err:
132
+ raise JudgetapError(f"downloading suite {suite.name!r}: {err}") from err
133
+ if names is None:
134
+ names = _label_names(page.get("features", []), suite.label_field)
135
+ total = int(page.get("num_rows_total", 0))
136
+ rows = page.get("rows", [])
137
+ if not rows:
138
+ if offset < total:
139
+ raise JudgetapError(
140
+ f"suite {suite.name!r}: empty page at offset {offset} "
141
+ f"of {total} rows; refusing to cache a partial suite"
142
+ )
143
+ break
144
+ for item in rows:
145
+ cases.append(_to_case(suite, names, item, offset + len(cases)))
146
+ offset += len(rows)
147
+ if not cases:
148
+ raise JudgetapError(f"suite {suite.name!r} downloaded no rows")
149
+ return cases
150
+
151
+
152
+ def suite_path(name: str) -> Path:
153
+ """The cached JSONL for ``name``, downloading it the first time."""
154
+ suite = REGISTRY.get(name)
155
+ if suite is None:
156
+ raise JudgetapError(
157
+ f"unknown suite {name!r}; available: {', '.join(sorted(REGISTRY))}"
158
+ )
159
+ path = cache_dir() / f"{suite.name}-{suite.split}.jsonl"
160
+ if path.exists():
161
+ return path
162
+ cases = download(suite)
163
+ path.parent.mkdir(parents=True, exist_ok=True)
164
+ tmp = path.with_suffix(".jsonl.part")
165
+ tmp.write_text("".join(json.dumps(c) + "\n" for c in cases))
166
+ tmp.replace(path) # a half-written download never looks cached
167
+ return path
168
+
169
+
170
+ def sample(items: list[Any], limit: int | None) -> list[Any]:
171
+ """A fixed-seed sample of ``limit`` items, kept in their original order."""
172
+ if limit is None or limit >= len(items):
173
+ return items
174
+ if limit < 1:
175
+ raise JudgetapError("--limit must be at least 1")
176
+ picked = sorted(random.Random(SEED).sample(range(len(items)), limit))
177
+ return [items[i] for i in picked]
@@ -229,3 +229,27 @@ def test_from_config_scalar_cascade_is_a_config_error(tmp_path):
229
229
  cfg.write_text('cascade = "jev"\n')
230
230
  with pytest.raises(sj.JudgetapError):
231
231
  from_config(cfg)
232
+
233
+
234
+ class Uncalibrated(StaticEngine):
235
+ def decide(self, questions, context):
236
+ self.calls.append((tuple(questions), context))
237
+ return [
238
+ sj.RawAnswer({"yes": 0.0, "no": 1.0}, calibrated=False) for _ in questions
239
+ ]
240
+
241
+
242
+ def test_uncalibrated_answer_escalates_whatever_its_p():
243
+ bare, strong = Uncalibrated(lambda q, c: {}, name="bare"), eng("strong", 0.9)
244
+ d = sj.yesno("q", engine=sj.Cascade([bare, strong]))
245
+ assert (d.engine, d.meta["hops"]) == ("strong", ["bare", "strong"])
246
+
247
+
248
+ def test_last_engine_uncalibrated_answer_is_judged_on_p():
249
+ d = sj.yesno(
250
+ "q",
251
+ engine=sj.Cascade(
252
+ [eng("cheap", 0.6), Uncalibrated(lambda q, c: {}, name="bare")]
253
+ ),
254
+ )
255
+ assert d.engine == "bare" and d.calibrated is False