judgetap 0.2.1.dev42001__tar.gz → 0.2.1.dev43001__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/PKG-INFO +2 -2
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/README.md +1 -1
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/pyproject.toml +1 -1
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/__init__.py +1 -1
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/evaluate.py +25 -2
- judgetap-0.2.1.dev43001/src/judgetap/suites.py +177 -0
- judgetap-0.2.1.dev43001/tests/test_suites.py +169 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.github/workflows/ci.yml +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.github/workflows/demo.yml +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.github/workflows/release.yml +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.gitignore +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.python-version +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.release-please-manifest.json +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/CHANGELOG.md +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/CONTRIBUTING.md +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/LICENSE +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/docs/SPEC.md +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/docs/demo.tape +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/release-please-config.json +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/_compat.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/api.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/cascade.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/cli.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/dashboard/__init__.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/dashboard/data.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/dashboard/page.html +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/dashboard/server.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/decision_log.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engine.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/__init__.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/agentjev.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/jev.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/julia.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/laya.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/llm.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/errors.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/__init__.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/core.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/hook.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/install.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/loop.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/rules.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/stop.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/py.typed +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/secrets.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/testing.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/types.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_api.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_call_accounting.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_calls.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_cascade.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_dashboard.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_decision_log.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_engine_jev.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_engine_julia.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_engine_llm.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_engine_local.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_evaluate.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_agents.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_core.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_hook.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_loop.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_polish_76.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_rules.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_stop.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_trust.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_llm_logprobs.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_questions.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_robustness_68.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_secrets.py +0 -0
- {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/uv.lock +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: judgetap
|
|
3
|
-
Version: 0.2.1.
|
|
3
|
+
Version: 0.2.1.dev43001
|
|
4
4
|
Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
|
|
5
5
|
Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
|
|
6
6
|
Author-email: Omer Bar-Ness <omer@zsquared.io>
|
|
@@ -44,7 +44,7 @@ judgetap guard test "git push --force origin main"
|
|
|
44
44
|
|
|
45
45
|
- **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
|
|
46
46
|
- **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
|
|
47
|
-
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
|
|
47
|
+
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost. `judgetap eval --suite banking77 --limit 200 --engines jev` runs a public suite instead (`ag_news`, `banking77`): it is downloaded from Hugging Face on first use, converted to judgetap's case format and cached in `~/.judgetap/suites`; nothing is bundled. `--limit N` takes the same fixed-seed sample every run. Each suite's licence and source URL are listed in `src/judgetap/suites.py`.
|
|
48
48
|
- **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
|
|
49
49
|
|
|
50
50
|
## Library
|
|
@@ -27,7 +27,7 @@ judgetap guard test "git push --force origin main"
|
|
|
27
27
|
|
|
28
28
|
- **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
|
|
29
29
|
- **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
|
|
30
|
-
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
|
|
30
|
+
- **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost. `judgetap eval --suite banking77 --limit 200 --engines jev` runs a public suite instead (`ag_news`, `banking77`): it is downloaded from Hugging Face on first use, converted to judgetap's case format and cached in `~/.judgetap/suites`; nothing is bundled. `--limit N` takes the same fixed-seed sample every run. Each suite's licence and source URL are listed in `src/judgetap/suites.py`.
|
|
31
31
|
- **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
|
|
32
32
|
|
|
33
33
|
## Library
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "judgetap"
|
|
3
|
-
version = "0.2.1.
|
|
3
|
+
version = "0.2.1.dev43001"
|
|
4
4
|
description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -13,6 +13,7 @@ from __future__ import annotations
|
|
|
13
13
|
|
|
14
14
|
import json
|
|
15
15
|
import math
|
|
16
|
+
import sys
|
|
16
17
|
from collections.abc import Iterable, Sequence
|
|
17
18
|
from dataclasses import asdict, dataclass, field
|
|
18
19
|
from pathlib import Path
|
|
@@ -194,13 +195,35 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|
|
194
195
|
parser = argparse.ArgumentParser(
|
|
195
196
|
prog="judgetap eval", description=__doc__.split("\n")[0]
|
|
196
197
|
)
|
|
197
|
-
parser.add_argument("cases", help="JSONL file of labelled cases")
|
|
198
|
+
parser.add_argument("cases", nargs="?", help="JSONL file of labelled cases")
|
|
199
|
+
parser.add_argument(
|
|
200
|
+
"--suite", help="public suite instead of a file (ag_news, banking77)"
|
|
201
|
+
)
|
|
202
|
+
parser.add_argument(
|
|
203
|
+
"--limit", type=int, help="deterministic sample of N cases (fixed seed)"
|
|
204
|
+
)
|
|
198
205
|
parser.add_argument(
|
|
199
206
|
"--engines", required=True, help="comma-separated specs, e.g. jev,laya"
|
|
200
207
|
)
|
|
201
208
|
parser.add_argument("--json", action="store_true", help="JSON instead of Markdown")
|
|
202
209
|
args = parser.parse_args(argv)
|
|
203
|
-
cases
|
|
210
|
+
if (args.cases is None) == (args.suite is None):
|
|
211
|
+
parser.error("give exactly one of a cases file or --suite")
|
|
212
|
+
if args.suite is not None:
|
|
213
|
+
from judgetap.suites import suite_path
|
|
214
|
+
|
|
215
|
+
if not args.suite.strip():
|
|
216
|
+
parser.error("--suite needs a suite name (ag_news, banking77)")
|
|
217
|
+
try:
|
|
218
|
+
path: str | Path = suite_path(args.suite)
|
|
219
|
+
except JudgetapError as err:
|
|
220
|
+
print(f"judgetap eval: {err}", file=sys.stderr)
|
|
221
|
+
return 1
|
|
222
|
+
else:
|
|
223
|
+
path = args.cases
|
|
224
|
+
from judgetap.suites import sample
|
|
225
|
+
|
|
226
|
+
cases = sample(load_cases(path), args.limit)
|
|
204
227
|
reports = [evaluate(cases, load(spec)) for spec in args.engines.split(",")]
|
|
205
228
|
print(to_json(reports) if args.json else to_markdown(reports), end="")
|
|
206
229
|
return 0
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""Public eval suites, converted to judgetap's case format on demand.
|
|
2
|
+
|
|
3
|
+
Nothing here is vendored: the first ``judgetap eval --suite <name>`` pages the
|
|
4
|
+
split out of the Hugging Face datasets-server JSON API (plain HTTPS, stdlib
|
|
5
|
+
only), converts each row to a ``choice`` case and caches the JSONL under
|
|
6
|
+
``~/.judgetap/suites`` (override with ``JUDGETAP_SUITES_DIR``). Later runs read
|
|
7
|
+
the cache. ``--limit N`` takes a deterministic sample: a fixed seed, so the same
|
|
8
|
+
N cases every run and on every machine.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import random
|
|
16
|
+
import urllib.parse
|
|
17
|
+
import urllib.request
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from judgetap.errors import JudgetapError
|
|
23
|
+
|
|
24
|
+
ROWS_API = "https://datasets-server.huggingface.co/rows"
|
|
25
|
+
PAGE = 100 # the rows API's maximum page size
|
|
26
|
+
SEED = 88
|
|
27
|
+
TIMEOUT = 30
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class Suite:
|
|
32
|
+
name: str
|
|
33
|
+
dataset: str # Hugging Face dataset id
|
|
34
|
+
split: str
|
|
35
|
+
question: str
|
|
36
|
+
licence: str
|
|
37
|
+
source: str # human-readable page for the dataset
|
|
38
|
+
config: str = "default"
|
|
39
|
+
text_field: str = "text"
|
|
40
|
+
label_field: str = "label"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
REGISTRY: dict[str, Suite] = {
|
|
44
|
+
s.name: s
|
|
45
|
+
for s in (
|
|
46
|
+
Suite(
|
|
47
|
+
name="ag_news",
|
|
48
|
+
dataset="fancyzhx/ag_news",
|
|
49
|
+
split="test",
|
|
50
|
+
question="Which topic is this news article about?",
|
|
51
|
+
licence="unknown; AG's corpus is provided for non-commercial research use",
|
|
52
|
+
source="https://huggingface.co/datasets/fancyzhx/ag_news",
|
|
53
|
+
),
|
|
54
|
+
Suite(
|
|
55
|
+
name="banking77",
|
|
56
|
+
dataset="legacy-datasets/banking77",
|
|
57
|
+
split="test",
|
|
58
|
+
question="Which intent does this banking customer message express?",
|
|
59
|
+
licence="CC-BY-4.0",
|
|
60
|
+
source="https://huggingface.co/datasets/legacy-datasets/banking77",
|
|
61
|
+
),
|
|
62
|
+
)
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def cache_dir() -> Path:
|
|
67
|
+
override = os.environ.get("JUDGETAP_SUITES_DIR")
|
|
68
|
+
return Path(override) if override else Path.home() / ".judgetap" / "suites"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _get_json(url: str) -> dict[str, Any]:
|
|
72
|
+
req = urllib.request.Request(url, headers={"User-Agent": "judgetap-eval"})
|
|
73
|
+
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
|
|
74
|
+
return json.load(resp)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _page_url(suite: Suite, offset: int) -> str:
|
|
78
|
+
query = urllib.parse.urlencode(
|
|
79
|
+
{
|
|
80
|
+
"dataset": suite.dataset,
|
|
81
|
+
"config": suite.config,
|
|
82
|
+
"split": suite.split,
|
|
83
|
+
"offset": offset,
|
|
84
|
+
"length": PAGE,
|
|
85
|
+
}
|
|
86
|
+
)
|
|
87
|
+
return f"{ROWS_API}?{query}"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _label_names(features: list[dict[str, Any]], field: str) -> list[str]:
|
|
91
|
+
for feature in features:
|
|
92
|
+
if feature.get("name") == field:
|
|
93
|
+
# The live rows API puts the ClassLabel under "type" (checked
|
|
94
|
+
# 2026-09-27); "feature" is accepted too, the key older docs name.
|
|
95
|
+
for key in ("type", "feature"):
|
|
96
|
+
spec = feature.get(key)
|
|
97
|
+
names = spec.get("names") if isinstance(spec, dict) else None
|
|
98
|
+
if isinstance(names, list) and names:
|
|
99
|
+
return [str(n) for n in names]
|
|
100
|
+
raise JudgetapError(f"no class-label names for {field!r} in the dataset")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _to_case(suite: Suite, names: list[str], item: Any, index: int) -> dict[str, Any]:
|
|
104
|
+
try:
|
|
105
|
+
row = item["row"]
|
|
106
|
+
context = row[suite.text_field]
|
|
107
|
+
label = int(row[suite.label_field])
|
|
108
|
+
if not isinstance(context, str) or not 0 <= label < len(names):
|
|
109
|
+
raise ValueError(f"label {label} or text is out of range")
|
|
110
|
+
except (KeyError, TypeError, ValueError) as err:
|
|
111
|
+
raise JudgetapError(
|
|
112
|
+
f"suite {suite.name!r}: malformed row {index}: {err!r}"
|
|
113
|
+
) from err
|
|
114
|
+
return {
|
|
115
|
+
"kind": "choice",
|
|
116
|
+
"question": suite.question,
|
|
117
|
+
"options": names,
|
|
118
|
+
"context": context,
|
|
119
|
+
"label": names[label],
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def download(suite: Suite) -> list[dict[str, Any]]:
|
|
124
|
+
"""Every row of the split, as judgetap case dicts."""
|
|
125
|
+
cases: list[dict[str, Any]] = []
|
|
126
|
+
names: list[str] | None = None
|
|
127
|
+
offset, total = 0, None
|
|
128
|
+
while total is None or offset < total:
|
|
129
|
+
try:
|
|
130
|
+
page = _get_json(_page_url(suite, offset))
|
|
131
|
+
except (OSError, ValueError) as err:
|
|
132
|
+
raise JudgetapError(f"downloading suite {suite.name!r}: {err}") from err
|
|
133
|
+
if names is None:
|
|
134
|
+
names = _label_names(page.get("features", []), suite.label_field)
|
|
135
|
+
total = int(page.get("num_rows_total", 0))
|
|
136
|
+
rows = page.get("rows", [])
|
|
137
|
+
if not rows:
|
|
138
|
+
if offset < total:
|
|
139
|
+
raise JudgetapError(
|
|
140
|
+
f"suite {suite.name!r}: empty page at offset {offset} "
|
|
141
|
+
f"of {total} rows; refusing to cache a partial suite"
|
|
142
|
+
)
|
|
143
|
+
break
|
|
144
|
+
for item in rows:
|
|
145
|
+
cases.append(_to_case(suite, names, item, offset + len(cases)))
|
|
146
|
+
offset += len(rows)
|
|
147
|
+
if not cases:
|
|
148
|
+
raise JudgetapError(f"suite {suite.name!r} downloaded no rows")
|
|
149
|
+
return cases
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def suite_path(name: str) -> Path:
|
|
153
|
+
"""The cached JSONL for ``name``, downloading it the first time."""
|
|
154
|
+
suite = REGISTRY.get(name)
|
|
155
|
+
if suite is None:
|
|
156
|
+
raise JudgetapError(
|
|
157
|
+
f"unknown suite {name!r}; available: {', '.join(sorted(REGISTRY))}"
|
|
158
|
+
)
|
|
159
|
+
path = cache_dir() / f"{suite.name}-{suite.split}.jsonl"
|
|
160
|
+
if path.exists():
|
|
161
|
+
return path
|
|
162
|
+
cases = download(suite)
|
|
163
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
164
|
+
tmp = path.with_suffix(".jsonl.part")
|
|
165
|
+
tmp.write_text("".join(json.dumps(c) + "\n" for c in cases))
|
|
166
|
+
tmp.replace(path) # a half-written download never looks cached
|
|
167
|
+
return path
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def sample(items: list[Any], limit: int | None) -> list[Any]:
|
|
171
|
+
"""A fixed-seed sample of ``limit`` items, kept in their original order."""
|
|
172
|
+
if limit is None or limit >= len(items):
|
|
173
|
+
return items
|
|
174
|
+
if limit < 1:
|
|
175
|
+
raise JudgetapError("--limit must be at least 1")
|
|
176
|
+
picked = sorted(random.Random(SEED).sample(range(len(items)), limit))
|
|
177
|
+
return [items[i] for i in picked]
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
import judgetap as sj
|
|
6
|
+
from judgetap import suites
|
|
7
|
+
from judgetap.evaluate import load_cases, main
|
|
8
|
+
|
|
9
|
+
NAMES = ["World", "Sports", "Business", "Sci/Tech"]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def fake_api(total, calls):
|
|
13
|
+
def get(url):
|
|
14
|
+
calls.append(url)
|
|
15
|
+
offset = int(url.split("offset=")[1].split("&")[0])
|
|
16
|
+
rows = [
|
|
17
|
+
{"row_idx": i, "row": {"text": f"article {i}", "label": i % 4}}
|
|
18
|
+
for i in range(offset, min(offset + suites.PAGE, total))
|
|
19
|
+
]
|
|
20
|
+
return {
|
|
21
|
+
"features": [
|
|
22
|
+
{"name": "text", "type": {"dtype": "string"}},
|
|
23
|
+
{"name": "label", "type": {"names": NAMES, "_type": "ClassLabel"}},
|
|
24
|
+
],
|
|
25
|
+
"rows": rows,
|
|
26
|
+
"num_rows_total": total,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
return get
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@pytest.fixture
|
|
33
|
+
def cache(tmp_path, monkeypatch):
|
|
34
|
+
monkeypatch.setenv("JUDGETAP_SUITES_DIR", str(tmp_path))
|
|
35
|
+
return tmp_path
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def test_registry_records_licence_and_source():
|
|
39
|
+
assert set(suites.REGISTRY) >= {"ag_news", "banking77"}
|
|
40
|
+
for s in suites.REGISTRY.values():
|
|
41
|
+
assert s.licence and s.source.startswith("https://huggingface.co/")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_download_pages_converts_and_caches(cache, monkeypatch):
|
|
45
|
+
calls = []
|
|
46
|
+
monkeypatch.setattr(suites, "_get_json", fake_api(250, calls))
|
|
47
|
+
path = suites.suite_path("ag_news")
|
|
48
|
+
assert len(calls) == 3 and "dataset=fancyzhx%2Fag_news" in calls[0]
|
|
49
|
+
assert path.parent == cache
|
|
50
|
+
cases = load_cases(path)
|
|
51
|
+
assert len(cases) == 250
|
|
52
|
+
assert cases[1].label == "Sports" and cases[1].context == "article 1"
|
|
53
|
+
assert list(cases[0].question.options) == NAMES
|
|
54
|
+
suites.suite_path("ag_news") # cached: no further requests
|
|
55
|
+
assert len(calls) == 3
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_failed_download_leaves_no_cache(cache, monkeypatch):
|
|
59
|
+
def boom(url):
|
|
60
|
+
raise OSError("offline")
|
|
61
|
+
|
|
62
|
+
monkeypatch.setattr(suites, "_get_json", boom)
|
|
63
|
+
with pytest.raises(sj.JudgetapError, match="offline"):
|
|
64
|
+
suites.suite_path("banking77")
|
|
65
|
+
assert list(cache.iterdir()) == []
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_unknown_suite_lists_available():
|
|
69
|
+
with pytest.raises(sj.JudgetapError, match="ag_news, banking77"):
|
|
70
|
+
suites.suite_path("nope")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def test_sample_is_deterministic_and_ordered():
|
|
74
|
+
items = list(range(1000))
|
|
75
|
+
a, b = suites.sample(items, 10), suites.sample(items, 10)
|
|
76
|
+
assert a == b == sorted(a) and len(set(a)) == 10
|
|
77
|
+
assert suites.sample(items, None) is items
|
|
78
|
+
assert suites.sample(items, 5000) is items
|
|
79
|
+
with pytest.raises(sj.JudgetapError):
|
|
80
|
+
suites.sample(items, 0)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_cli_suite_with_limit(cache, monkeypatch, capsys):
|
|
84
|
+
from judgetap import engines
|
|
85
|
+
from judgetap.testing import StaticEngine
|
|
86
|
+
|
|
87
|
+
monkeypatch.setattr(suites, "_get_json", fake_api(40, []))
|
|
88
|
+
monkeypatch.setattr(
|
|
89
|
+
engines,
|
|
90
|
+
"load",
|
|
91
|
+
lambda spec: StaticEngine(lambda q, c: {"World": 1.0}, name=spec),
|
|
92
|
+
)
|
|
93
|
+
assert main(["--suite", "ag_news", "--limit", "7", "--engines", "e", "--json"]) == 0
|
|
94
|
+
report = json.loads(capsys.readouterr().out)[0]
|
|
95
|
+
assert report["cases"] == 7
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def test_cli_needs_exactly_one_source(tmp_path):
|
|
99
|
+
with pytest.raises(SystemExit):
|
|
100
|
+
main(["--engines", "x"])
|
|
101
|
+
with pytest.raises(SystemExit):
|
|
102
|
+
main([str(tmp_path / "c.jsonl"), "--suite", "ag_news", "--engines", "x"])
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _page(features, rows, total):
|
|
106
|
+
return {"features": features, "rows": rows, "num_rows_total": total}
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
LABEL_ONLY = [{"name": "label", "type": {"names": NAMES, "_type": "ClassLabel"}}]
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def test_label_names_live_type_key_and_feature_key():
|
|
113
|
+
# The live rows API shape, captured 2026-09-27, uses "type".
|
|
114
|
+
assert suites._label_names(LABEL_ONLY, "label") == NAMES
|
|
115
|
+
alt = [{"name": "label", "feature": {"names": NAMES}}]
|
|
116
|
+
assert suites._label_names(alt, "label") == NAMES
|
|
117
|
+
with pytest.raises(sj.JudgetapError, match="class-label"):
|
|
118
|
+
suites._label_names([{"name": "label", "type": {"dtype": "int64"}}], "label")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def test_empty_intermediate_page_raises_and_caches_nothing(cache, monkeypatch):
|
|
122
|
+
real = fake_api(250, [])
|
|
123
|
+
|
|
124
|
+
def get(url):
|
|
125
|
+
page = real(url)
|
|
126
|
+
if "offset=100" in url:
|
|
127
|
+
page["rows"] = []
|
|
128
|
+
return page
|
|
129
|
+
|
|
130
|
+
monkeypatch.setattr(suites, "_get_json", get)
|
|
131
|
+
with pytest.raises(sj.JudgetapError, match="empty page at offset 100 of 250"):
|
|
132
|
+
suites.suite_path("ag_news")
|
|
133
|
+
assert list(cache.iterdir()) == []
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
@pytest.mark.parametrize(
|
|
137
|
+
"row",
|
|
138
|
+
[
|
|
139
|
+
{"text": "x"},
|
|
140
|
+
{"label": 1},
|
|
141
|
+
{"text": "x", "label": 9},
|
|
142
|
+
{"text": "x", "label": -1},
|
|
143
|
+
{"text": "x", "label": "nan"},
|
|
144
|
+
{"text": None, "label": 1},
|
|
145
|
+
],
|
|
146
|
+
)
|
|
147
|
+
def test_malformed_row_raises_judgetap_error(cache, monkeypatch, row):
|
|
148
|
+
page = _page(LABEL_ONLY, [{"row_idx": 0, "row": row}], 1)
|
|
149
|
+
monkeypatch.setattr(suites, "_get_json", lambda url: page)
|
|
150
|
+
with pytest.raises(sj.JudgetapError, match="malformed row 0"):
|
|
151
|
+
suites.suite_path("ag_news")
|
|
152
|
+
assert list(cache.iterdir()) == []
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def test_cli_empty_suite_is_a_usage_error(capsys):
|
|
156
|
+
with pytest.raises(SystemExit) as exc:
|
|
157
|
+
main(["--suite", "", "--engines", "x"])
|
|
158
|
+
assert exc.value.code == 2
|
|
159
|
+
assert "needs a suite name" in capsys.readouterr().err
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def test_cli_suite_failure_is_a_clean_error(cache, monkeypatch, capsys):
|
|
163
|
+
def boom(url):
|
|
164
|
+
raise OSError("offline")
|
|
165
|
+
|
|
166
|
+
monkeypatch.setattr(suites, "_get_json", boom)
|
|
167
|
+
assert main(["--suite", "ag_news", "--engines", "x"]) == 1
|
|
168
|
+
err = capsys.readouterr().err
|
|
169
|
+
assert "offline" in err and "Traceback" not in err
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|