judgetap 0.2.1.dev42001__tar.gz → 0.2.1.dev43001__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/PKG-INFO +2 -2
  2. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/README.md +1 -1
  3. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/pyproject.toml +1 -1
  4. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/__init__.py +1 -1
  5. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/evaluate.py +25 -2
  6. judgetap-0.2.1.dev43001/src/judgetap/suites.py +177 -0
  7. judgetap-0.2.1.dev43001/tests/test_suites.py +169 -0
  8. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.github/workflows/ci.yml +0 -0
  9. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.github/workflows/demo.yml +0 -0
  10. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.github/workflows/release.yml +0 -0
  11. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.gitignore +0 -0
  12. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.python-version +0 -0
  13. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/.release-please-manifest.json +0 -0
  14. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/CHANGELOG.md +0 -0
  15. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/CONTRIBUTING.md +0 -0
  16. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/LICENSE +0 -0
  17. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/docs/SPEC.md +0 -0
  18. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/docs/demo.tape +0 -0
  19. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/release-please-config.json +0 -0
  20. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/_compat.py +0 -0
  21. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/api.py +0 -0
  22. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/cascade.py +0 -0
  23. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/cli.py +0 -0
  24. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/dashboard/__init__.py +0 -0
  25. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/dashboard/data.py +0 -0
  26. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/dashboard/page.html +0 -0
  27. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/dashboard/server.py +0 -0
  28. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/decision_log.py +0 -0
  29. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engine.py +0 -0
  30. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/__init__.py +0 -0
  31. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/agentjev.py +0 -0
  32. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/jev.py +0 -0
  33. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/julia.py +0 -0
  34. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/laya.py +0 -0
  35. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/engines/llm.py +0 -0
  36. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/errors.py +0 -0
  37. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/__init__.py +0 -0
  38. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/core.py +0 -0
  39. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/hook.py +0 -0
  40. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/install.py +0 -0
  41. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/loop.py +0 -0
  42. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/rules.py +0 -0
  43. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/guard/stop.py +0 -0
  44. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/py.typed +0 -0
  45. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/secrets.py +0 -0
  46. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/testing.py +0 -0
  47. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/src/judgetap/types.py +0 -0
  48. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_api.py +0 -0
  49. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_call_accounting.py +0 -0
  50. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_calls.py +0 -0
  51. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_cascade.py +0 -0
  52. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_dashboard.py +0 -0
  53. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_decision_log.py +0 -0
  54. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_engine_jev.py +0 -0
  55. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_engine_julia.py +0 -0
  56. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_engine_llm.py +0 -0
  57. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_engine_local.py +0 -0
  58. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_evaluate.py +0 -0
  59. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_agents.py +0 -0
  60. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_core.py +0 -0
  61. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_hook.py +0 -0
  62. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_loop.py +0 -0
  63. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_polish_76.py +0 -0
  64. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_rules.py +0 -0
  65. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_stop.py +0 -0
  66. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_guard_trust.py +0 -0
  67. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_llm_logprobs.py +0 -0
  68. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_questions.py +0 -0
  69. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_robustness_68.py +0 -0
  70. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/tests/test_secrets.py +0 -0
  71. {judgetap-0.2.1.dev42001 → judgetap-0.2.1.dev43001}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: judgetap
3
- Version: 0.2.1.dev42001
3
+ Version: 0.2.1.dev43001
4
4
  Summary: Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development.
5
5
  Project-URL: Homepage, https://github.com/mergesafe-ai/judgetap
6
6
  Author-email: Omer Bar-Ness <omer@zsquared.io>
@@ -44,7 +44,7 @@ judgetap guard test "git push --force origin main"
44
44
 
45
45
  - **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
46
46
  - **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
47
- - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
47
+ - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost. `judgetap eval --suite banking77 --limit 200 --engines jev` runs a public suite instead (`ag_news`, `banking77`): it is downloaded from Hugging Face on first use, converted to judgetap's case format and cached in `~/.judgetap/suites`; nothing is bundled. `--limit N` takes the same fixed-seed sample every run. Each suite's licence and source URL are listed in `src/judgetap/suites.py`.
48
48
  - **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
49
49
 
50
50
  ## Library
@@ -27,7 +27,7 @@ judgetap guard test "git push --force origin main"
27
27
 
28
28
  - **Guard.** A hook that checks every shell command (and, in Claude Code, every file write and edit) before it runs. Rules catch common destructive forms: recursive deletes outside the workspace, force-pushes and pushes to protected branches, `DROP`/`DELETE` without `WHERE`, `terraform destroy`, and secrets written to files. With an engine configured, a model judges the rest: is it irreversible? off-task? against a rule in `AGENTS.md`? Without an engine, the rules fail closed for shell commands (anything they can't vouch for asks you); file writes and edits are checked for secrets, your own rules, and writes to the guard's own configuration (which ask).
29
29
  - **Library.** One API (`choice`, `score`, `yesno`, `batch`) over every Jev-style decision engine, with a cascade that escalates low-confidence answers to a stronger engine.
30
- - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost.
30
+ - **Eval.** `judgetap eval cases.jsonl --engines jev,laya` compares engines on your labelled cases: accuracy, calibration (ECE), latency and cost. `judgetap eval --suite banking77 --limit 200 --engines jev` runs a public suite instead (`ag_news`, `banking77`): it is downloaded from Hugging Face on first use, converted to judgetap's case format and cached in `~/.judgetap/suites`; nothing is bundled. `--limit N` takes the same fixed-seed sample every run. Each suite's licence and source URL are listed in `src/judgetap/suites.py`.
31
31
  - **Dashboard.** `judgetap dashboard` is a local page with recent decisions, holds, asks, latency and cost per engine. You can mark a hold as a false alarm.
32
32
 
33
33
  ## Library
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "judgetap"
3
- version = "0.2.1.dev42001"
3
+ version = "0.2.1.dev43001"
4
4
  description = "Fast typed decisions (choice, score, yes/no) across Jev-style engines, plus a guard for coding agents. Early development."
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -23,7 +23,7 @@ from judgetap.errors import (
23
23
  )
24
24
  from judgetap.types import Decision, Question
25
25
 
26
- __version__ = "0.2.1.dev42001" # x-release-please-version
26
+ __version__ = "0.2.1.dev43001" # x-release-please-version
27
27
 
28
28
  __all__ = [
29
29
  "Cascade",
@@ -13,6 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  import json
15
15
  import math
16
+ import sys
16
17
  from collections.abc import Iterable, Sequence
17
18
  from dataclasses import asdict, dataclass, field
18
19
  from pathlib import Path
@@ -194,13 +195,35 @@ def main(argv: Sequence[str] | None = None) -> int:
194
195
  parser = argparse.ArgumentParser(
195
196
  prog="judgetap eval", description=__doc__.split("\n")[0]
196
197
  )
197
- parser.add_argument("cases", help="JSONL file of labelled cases")
198
+ parser.add_argument("cases", nargs="?", help="JSONL file of labelled cases")
199
+ parser.add_argument(
200
+ "--suite", help="public suite instead of a file (ag_news, banking77)"
201
+ )
202
+ parser.add_argument(
203
+ "--limit", type=int, help="deterministic sample of N cases (fixed seed)"
204
+ )
198
205
  parser.add_argument(
199
206
  "--engines", required=True, help="comma-separated specs, e.g. jev,laya"
200
207
  )
201
208
  parser.add_argument("--json", action="store_true", help="JSON instead of Markdown")
202
209
  args = parser.parse_args(argv)
203
- cases = load_cases(args.cases)
210
+ if (args.cases is None) == (args.suite is None):
211
+ parser.error("give exactly one of a cases file or --suite")
212
+ if args.suite is not None:
213
+ from judgetap.suites import suite_path
214
+
215
+ if not args.suite.strip():
216
+ parser.error("--suite needs a suite name (ag_news, banking77)")
217
+ try:
218
+ path: str | Path = suite_path(args.suite)
219
+ except JudgetapError as err:
220
+ print(f"judgetap eval: {err}", file=sys.stderr)
221
+ return 1
222
+ else:
223
+ path = args.cases
224
+ from judgetap.suites import sample
225
+
226
+ cases = sample(load_cases(path), args.limit)
204
227
  reports = [evaluate(cases, load(spec)) for spec in args.engines.split(",")]
205
228
  print(to_json(reports) if args.json else to_markdown(reports), end="")
206
229
  return 0
@@ -0,0 +1,177 @@
1
+ """Public eval suites, converted to judgetap's case format on demand.
2
+
3
+ Nothing here is vendored: the first ``judgetap eval --suite <name>`` pages the
4
+ split out of the Hugging Face datasets-server JSON API (plain HTTPS, stdlib
5
+ only), converts each row to a ``choice`` case and caches the JSONL under
6
+ ``~/.judgetap/suites`` (override with ``JUDGETAP_SUITES_DIR``). Later runs read
7
+ the cache. ``--limit N`` takes a deterministic sample: a fixed seed, so the same
8
+ N cases every run and on every machine.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import os
15
+ import random
16
+ import urllib.parse
17
+ import urllib.request
18
+ from dataclasses import dataclass
19
+ from pathlib import Path
20
+ from typing import Any
21
+
22
+ from judgetap.errors import JudgetapError
23
+
24
+ ROWS_API = "https://datasets-server.huggingface.co/rows"
25
+ PAGE = 100 # the rows API's maximum page size
26
+ SEED = 88
27
+ TIMEOUT = 30
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class Suite:
32
+ name: str
33
+ dataset: str # Hugging Face dataset id
34
+ split: str
35
+ question: str
36
+ licence: str
37
+ source: str # human-readable page for the dataset
38
+ config: str = "default"
39
+ text_field: str = "text"
40
+ label_field: str = "label"
41
+
42
+
43
+ REGISTRY: dict[str, Suite] = {
44
+ s.name: s
45
+ for s in (
46
+ Suite(
47
+ name="ag_news",
48
+ dataset="fancyzhx/ag_news",
49
+ split="test",
50
+ question="Which topic is this news article about?",
51
+ licence="unknown; AG's corpus is provided for non-commercial research use",
52
+ source="https://huggingface.co/datasets/fancyzhx/ag_news",
53
+ ),
54
+ Suite(
55
+ name="banking77",
56
+ dataset="legacy-datasets/banking77",
57
+ split="test",
58
+ question="Which intent does this banking customer message express?",
59
+ licence="CC-BY-4.0",
60
+ source="https://huggingface.co/datasets/legacy-datasets/banking77",
61
+ ),
62
+ )
63
+ }
64
+
65
+
66
+ def cache_dir() -> Path:
67
+ override = os.environ.get("JUDGETAP_SUITES_DIR")
68
+ return Path(override) if override else Path.home() / ".judgetap" / "suites"
69
+
70
+
71
+ def _get_json(url: str) -> dict[str, Any]:
72
+ req = urllib.request.Request(url, headers={"User-Agent": "judgetap-eval"})
73
+ with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
74
+ return json.load(resp)
75
+
76
+
77
+ def _page_url(suite: Suite, offset: int) -> str:
78
+ query = urllib.parse.urlencode(
79
+ {
80
+ "dataset": suite.dataset,
81
+ "config": suite.config,
82
+ "split": suite.split,
83
+ "offset": offset,
84
+ "length": PAGE,
85
+ }
86
+ )
87
+ return f"{ROWS_API}?{query}"
88
+
89
+
90
+ def _label_names(features: list[dict[str, Any]], field: str) -> list[str]:
91
+ for feature in features:
92
+ if feature.get("name") == field:
93
+ # The live rows API puts the ClassLabel under "type" (checked
94
+ # 2026-09-27); "feature" is accepted too, the key older docs name.
95
+ for key in ("type", "feature"):
96
+ spec = feature.get(key)
97
+ names = spec.get("names") if isinstance(spec, dict) else None
98
+ if isinstance(names, list) and names:
99
+ return [str(n) for n in names]
100
+ raise JudgetapError(f"no class-label names for {field!r} in the dataset")
101
+
102
+
103
+ def _to_case(suite: Suite, names: list[str], item: Any, index: int) -> dict[str, Any]:
104
+ try:
105
+ row = item["row"]
106
+ context = row[suite.text_field]
107
+ label = int(row[suite.label_field])
108
+ if not isinstance(context, str) or not 0 <= label < len(names):
109
+ raise ValueError(f"label {label} or text is out of range")
110
+ except (KeyError, TypeError, ValueError) as err:
111
+ raise JudgetapError(
112
+ f"suite {suite.name!r}: malformed row {index}: {err!r}"
113
+ ) from err
114
+ return {
115
+ "kind": "choice",
116
+ "question": suite.question,
117
+ "options": names,
118
+ "context": context,
119
+ "label": names[label],
120
+ }
121
+
122
+
123
+ def download(suite: Suite) -> list[dict[str, Any]]:
124
+ """Every row of the split, as judgetap case dicts."""
125
+ cases: list[dict[str, Any]] = []
126
+ names: list[str] | None = None
127
+ offset, total = 0, None
128
+ while total is None or offset < total:
129
+ try:
130
+ page = _get_json(_page_url(suite, offset))
131
+ except (OSError, ValueError) as err:
132
+ raise JudgetapError(f"downloading suite {suite.name!r}: {err}") from err
133
+ if names is None:
134
+ names = _label_names(page.get("features", []), suite.label_field)
135
+ total = int(page.get("num_rows_total", 0))
136
+ rows = page.get("rows", [])
137
+ if not rows:
138
+ if offset < total:
139
+ raise JudgetapError(
140
+ f"suite {suite.name!r}: empty page at offset {offset} "
141
+ f"of {total} rows; refusing to cache a partial suite"
142
+ )
143
+ break
144
+ for item in rows:
145
+ cases.append(_to_case(suite, names, item, offset + len(cases)))
146
+ offset += len(rows)
147
+ if not cases:
148
+ raise JudgetapError(f"suite {suite.name!r} downloaded no rows")
149
+ return cases
150
+
151
+
152
+ def suite_path(name: str) -> Path:
153
+ """The cached JSONL for ``name``, downloading it the first time."""
154
+ suite = REGISTRY.get(name)
155
+ if suite is None:
156
+ raise JudgetapError(
157
+ f"unknown suite {name!r}; available: {', '.join(sorted(REGISTRY))}"
158
+ )
159
+ path = cache_dir() / f"{suite.name}-{suite.split}.jsonl"
160
+ if path.exists():
161
+ return path
162
+ cases = download(suite)
163
+ path.parent.mkdir(parents=True, exist_ok=True)
164
+ tmp = path.with_suffix(".jsonl.part")
165
+ tmp.write_text("".join(json.dumps(c) + "\n" for c in cases))
166
+ tmp.replace(path) # a half-written download never looks cached
167
+ return path
168
+
169
+
170
+ def sample(items: list[Any], limit: int | None) -> list[Any]:
171
+ """A fixed-seed sample of ``limit`` items, kept in their original order."""
172
+ if limit is None or limit >= len(items):
173
+ return items
174
+ if limit < 1:
175
+ raise JudgetapError("--limit must be at least 1")
176
+ picked = sorted(random.Random(SEED).sample(range(len(items)), limit))
177
+ return [items[i] for i in picked]
@@ -0,0 +1,169 @@
1
+ import json
2
+
3
+ import pytest
4
+
5
+ import judgetap as sj
6
+ from judgetap import suites
7
+ from judgetap.evaluate import load_cases, main
8
+
9
+ NAMES = ["World", "Sports", "Business", "Sci/Tech"]
10
+
11
+
12
+ def fake_api(total, calls):
13
+ def get(url):
14
+ calls.append(url)
15
+ offset = int(url.split("offset=")[1].split("&")[0])
16
+ rows = [
17
+ {"row_idx": i, "row": {"text": f"article {i}", "label": i % 4}}
18
+ for i in range(offset, min(offset + suites.PAGE, total))
19
+ ]
20
+ return {
21
+ "features": [
22
+ {"name": "text", "type": {"dtype": "string"}},
23
+ {"name": "label", "type": {"names": NAMES, "_type": "ClassLabel"}},
24
+ ],
25
+ "rows": rows,
26
+ "num_rows_total": total,
27
+ }
28
+
29
+ return get
30
+
31
+
32
+ @pytest.fixture
33
+ def cache(tmp_path, monkeypatch):
34
+ monkeypatch.setenv("JUDGETAP_SUITES_DIR", str(tmp_path))
35
+ return tmp_path
36
+
37
+
38
+ def test_registry_records_licence_and_source():
39
+ assert set(suites.REGISTRY) >= {"ag_news", "banking77"}
40
+ for s in suites.REGISTRY.values():
41
+ assert s.licence and s.source.startswith("https://huggingface.co/")
42
+
43
+
44
+ def test_download_pages_converts_and_caches(cache, monkeypatch):
45
+ calls = []
46
+ monkeypatch.setattr(suites, "_get_json", fake_api(250, calls))
47
+ path = suites.suite_path("ag_news")
48
+ assert len(calls) == 3 and "dataset=fancyzhx%2Fag_news" in calls[0]
49
+ assert path.parent == cache
50
+ cases = load_cases(path)
51
+ assert len(cases) == 250
52
+ assert cases[1].label == "Sports" and cases[1].context == "article 1"
53
+ assert list(cases[0].question.options) == NAMES
54
+ suites.suite_path("ag_news") # cached: no further requests
55
+ assert len(calls) == 3
56
+
57
+
58
+ def test_failed_download_leaves_no_cache(cache, monkeypatch):
59
+ def boom(url):
60
+ raise OSError("offline")
61
+
62
+ monkeypatch.setattr(suites, "_get_json", boom)
63
+ with pytest.raises(sj.JudgetapError, match="offline"):
64
+ suites.suite_path("banking77")
65
+ assert list(cache.iterdir()) == []
66
+
67
+
68
+ def test_unknown_suite_lists_available():
69
+ with pytest.raises(sj.JudgetapError, match="ag_news, banking77"):
70
+ suites.suite_path("nope")
71
+
72
+
73
+ def test_sample_is_deterministic_and_ordered():
74
+ items = list(range(1000))
75
+ a, b = suites.sample(items, 10), suites.sample(items, 10)
76
+ assert a == b == sorted(a) and len(set(a)) == 10
77
+ assert suites.sample(items, None) is items
78
+ assert suites.sample(items, 5000) is items
79
+ with pytest.raises(sj.JudgetapError):
80
+ suites.sample(items, 0)
81
+
82
+
83
+ def test_cli_suite_with_limit(cache, monkeypatch, capsys):
84
+ from judgetap import engines
85
+ from judgetap.testing import StaticEngine
86
+
87
+ monkeypatch.setattr(suites, "_get_json", fake_api(40, []))
88
+ monkeypatch.setattr(
89
+ engines,
90
+ "load",
91
+ lambda spec: StaticEngine(lambda q, c: {"World": 1.0}, name=spec),
92
+ )
93
+ assert main(["--suite", "ag_news", "--limit", "7", "--engines", "e", "--json"]) == 0
94
+ report = json.loads(capsys.readouterr().out)[0]
95
+ assert report["cases"] == 7
96
+
97
+
98
+ def test_cli_needs_exactly_one_source(tmp_path):
99
+ with pytest.raises(SystemExit):
100
+ main(["--engines", "x"])
101
+ with pytest.raises(SystemExit):
102
+ main([str(tmp_path / "c.jsonl"), "--suite", "ag_news", "--engines", "x"])
103
+
104
+
105
+ def _page(features, rows, total):
106
+ return {"features": features, "rows": rows, "num_rows_total": total}
107
+
108
+
109
+ LABEL_ONLY = [{"name": "label", "type": {"names": NAMES, "_type": "ClassLabel"}}]
110
+
111
+
112
+ def test_label_names_live_type_key_and_feature_key():
113
+ # The live rows API shape, captured 2026-09-27, uses "type".
114
+ assert suites._label_names(LABEL_ONLY, "label") == NAMES
115
+ alt = [{"name": "label", "feature": {"names": NAMES}}]
116
+ assert suites._label_names(alt, "label") == NAMES
117
+ with pytest.raises(sj.JudgetapError, match="class-label"):
118
+ suites._label_names([{"name": "label", "type": {"dtype": "int64"}}], "label")
119
+
120
+
121
+ def test_empty_intermediate_page_raises_and_caches_nothing(cache, monkeypatch):
122
+ real = fake_api(250, [])
123
+
124
+ def get(url):
125
+ page = real(url)
126
+ if "offset=100" in url:
127
+ page["rows"] = []
128
+ return page
129
+
130
+ monkeypatch.setattr(suites, "_get_json", get)
131
+ with pytest.raises(sj.JudgetapError, match="empty page at offset 100 of 250"):
132
+ suites.suite_path("ag_news")
133
+ assert list(cache.iterdir()) == []
134
+
135
+
136
+ @pytest.mark.parametrize(
137
+ "row",
138
+ [
139
+ {"text": "x"},
140
+ {"label": 1},
141
+ {"text": "x", "label": 9},
142
+ {"text": "x", "label": -1},
143
+ {"text": "x", "label": "nan"},
144
+ {"text": None, "label": 1},
145
+ ],
146
+ )
147
+ def test_malformed_row_raises_judgetap_error(cache, monkeypatch, row):
148
+ page = _page(LABEL_ONLY, [{"row_idx": 0, "row": row}], 1)
149
+ monkeypatch.setattr(suites, "_get_json", lambda url: page)
150
+ with pytest.raises(sj.JudgetapError, match="malformed row 0"):
151
+ suites.suite_path("ag_news")
152
+ assert list(cache.iterdir()) == []
153
+
154
+
155
+ def test_cli_empty_suite_is_a_usage_error(capsys):
156
+ with pytest.raises(SystemExit) as exc:
157
+ main(["--suite", "", "--engines", "x"])
158
+ assert exc.value.code == 2
159
+ assert "needs a suite name" in capsys.readouterr().err
160
+
161
+
162
+ def test_cli_suite_failure_is_a_clean_error(cache, monkeypatch, capsys):
163
+ def boom(url):
164
+ raise OSError("offline")
165
+
166
+ monkeypatch.setattr(suites, "_get_json", boom)
167
+ assert main(["--suite", "ag_news", "--engines", "x"]) == 1
168
+ err = capsys.readouterr().err
169
+ assert "offline" in err and "Traceback" not in err