spancheck 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. spancheck-0.1.0/LICENSE +21 -0
  2. spancheck-0.1.0/PKG-INFO +180 -0
  3. spancheck-0.1.0/README.md +158 -0
  4. spancheck-0.1.0/pyproject.toml +43 -0
  5. spancheck-0.1.0/setup.cfg +4 -0
  6. spancheck-0.1.0/src/spancheck/__init__.py +47 -0
  7. spancheck-0.1.0/src/spancheck/_env.py +25 -0
  8. spancheck-0.1.0/src/spancheck/_text.py +29 -0
  9. spancheck-0.1.0/src/spancheck/adapter.py +90 -0
  10. spancheck-0.1.0/src/spancheck/audit.py +135 -0
  11. spancheck-0.1.0/src/spancheck/calibrate.py +98 -0
  12. spancheck-0.1.0/src/spancheck/cli.py +148 -0
  13. spancheck-0.1.0/src/spancheck/core.py +267 -0
  14. spancheck-0.1.0/src/spancheck/demo.py +43 -0
  15. spancheck-0.1.0/src/spancheck/graders.py +127 -0
  16. spancheck-0.1.0/src/spancheck/judge.py +102 -0
  17. spancheck-0.1.0/src/spancheck/prompts/entailment_v1.txt +15 -0
  18. spancheck-0.1.0/src/spancheck/provider.py +32 -0
  19. spancheck-0.1.0/src/spancheck/span.py +126 -0
  20. spancheck-0.1.0/src/spancheck.egg-info/PKG-INFO +180 -0
  21. spancheck-0.1.0/src/spancheck.egg-info/SOURCES.txt +30 -0
  22. spancheck-0.1.0/src/spancheck.egg-info/dependency_links.txt +1 -0
  23. spancheck-0.1.0/src/spancheck.egg-info/entry_points.txt +2 -0
  24. spancheck-0.1.0/src/spancheck.egg-info/requires.txt +3 -0
  25. spancheck-0.1.0/src/spancheck.egg-info/top_level.txt +1 -0
  26. spancheck-0.1.0/tests/test_audit.py +134 -0
  27. spancheck-0.1.0/tests/test_cli.py +75 -0
  28. spancheck-0.1.0/tests/test_core.py +174 -0
  29. spancheck-0.1.0/tests/test_env.py +20 -0
  30. spancheck-0.1.0/tests/test_judge.py +109 -0
  31. spancheck-0.1.0/tests/test_scaffold.py +31 -0
  32. spancheck-0.1.0/tests/test_span.py +171 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Trevor J. Romack
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,180 @@
1
+ Metadata-Version: 2.4
2
+ Name: spancheck
3
+ Version: 0.1.0
4
+ Summary: Score a grounded-answer system on citation accuracy, abstention correctness, hallucination rate, cost and latency — with an audit log a compliance reviewer can read.
5
+ Author-email: "Trevor J. Romack" <tjromack@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/tjromack/spancheck
8
+ Project-URL: Source, https://github.com/tjromack/spancheck
9
+ Keywords: rag,evaluation,llm,hallucination,citation,abstention,guardrails
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Topic :: Software Development :: Quality Assurance
16
+ Requires-Python: >=3.11
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE
19
+ Provides-Extra: dev
20
+ Requires-Dist: pytest>=8; extra == "dev"
21
+ Dynamic: license-file
22
+
23
+ # spancheck
24
+
25
+ > © 2026 Trevor J. Romack — **MIT-licensed** ([LICENSE](LICENSE)) · `pip install spancheck` · tjromack@gmail.com
26
+
27
+ **Score a grounded-answer system on citation accuracy, abstention correctness, hallucination rate, cost and latency —
28
+ and get an audit log a compliance reviewer can read.**
29
+
30
+ `spancheck` is an installable, corpus-agnostic evaluation library for RAG and retrieval-then-answer pipelines. Point it
31
+ at any system through a thin adapter, give it a test set, and it produces a scorecard on the four metrics that decide
32
+ whether the system's answers can be trusted — plus a versioned audit log that reconstructs *why* each answer passed or
33
+ failed.
34
+
35
+ Library first, CLI second, GitHub Action third: the Python API is the product; the command line and the Action are thin
36
+ wrappers over it.
37
+
38
+ ---
39
+
40
+ ## The four metrics
41
+
42
+ | Metric | Question it answers | How it's scored |
43
+ |---|---|---|
44
+ | **Citation accuracy** | Does the cited span actually support the claim? | Deterministic span check *(the capability built first and tested deepest)* |
45
+ | **Abstention correctness** | Does it decline when the answer isn't in the corpus, and answer when it is? | Deterministic |
46
+ | **Hallucination rate** | What share of answers assert something the retrieved context doesn't support? | Deterministic groundedness + calibrated LLM-judge for the qualitative residue |
47
+ | **Cost & latency** | Tokens / dollars / wall-clock per answer, against a budget | Deterministic, from a cached run — no network |
48
+
49
+ ## Status
50
+
51
+ **Complete and dogfooded.** The measurement core (`Case` / `evaluate` / `Run` / `Scorecard` / `diff` / `gate`), the thin
52
+ `adapter` contract, the deterministic graders, **citation-span verification** — the capability this build exists to
53
+ add — the **cost/latency + versioned audit log** layer, the **calibrated LLM-judge**, and the **CLI + GitHub Action**
54
+ are implemented, dependency-free (the judge reaches a model only through a caller-supplied callable), and tested
55
+ (68 tests, green on a clean `pip install` and in CI). Citation verification splits into **provenance** (is the cited
56
+ span really in the retrieved context, verbatim?) and **support** (does the span cover the claim?); both must hold, so a
57
+ fabricated quote or a right-answer-wrong-citation cannot pass. A captured run re-scores **offline** (`Run.rescore`), and
58
+ `run.audit_log()` emits a versioned, self-describing record with per-citation verdicts (schema in
59
+ [`docs/AUDIT-LOG.md`](docs/AUDIT-LOG.md)). The LLM-judge is **opt-in** — the default grader set stays deterministic and
60
+ offline — and is **calibrated before it is trusted**.
61
+
62
+ ## Results (measured)
63
+
64
+ `spancheck` was run end-to-end against a real product pipeline as a black box (a grounded answer-a-document tool on
65
+ `claude-sonnet-5`, over its own sample contract), and the judge was calibrated against a human-labelled gold set. Full
66
+ write-up in [`docs/CASE-STUDY.md`](docs/CASE-STUDY.md); the audit log is committed at
67
+ [`dogfood/suver_audit.json`](dogfood/suver_audit.json).
68
+
69
+ | Metric | Result |
70
+ |---|---|
71
+ | Citation accuracy (real pipeline) | **1.00** — every cited span real and supporting |
72
+ | Hallucination (groundedness) | **0 hallucinations** (pass-rate 1.00) |
73
+ | Abstention correctness | 11/12 — spancheck **caught one false-abstention** on the real system |
74
+ | Overall (12 cases) | **0.967** |
75
+ | Judge vs human gold set — stub / `claude-sonnet-5` | **0.75 / 1.00** |
76
+ | Tests | **68**, clean `pip install` + CI (3.11 & 3.12) |
77
+
78
+ The one miss is the point: the eval found a real recall gap in a shipping system — a measured behaviour, not a spancheck
79
+ bug. Reproduce with `python dogfood/run_dogfood.py` and `python -m spancheck.calibrate --real`.
80
+
81
+ ## Command line & CI
82
+
83
+ The CLI is a thin wrapper over the library. It works out of the box against a shipped demo target — no key, no user
84
+ code:
85
+
86
+ ```bash
87
+ pip install spancheck # or `pip install -e .` from a clone
88
+ spancheck run examples/cases.jsonl --target spancheck.demo:system --out audit.json
89
+ spancheck score audit.json # recompute metrics from the cache, offline
90
+ spancheck gate audit.json citation_accuracy=1.0 groundedness=0.9 # exit 1 on failure (for CI)
91
+ ```
92
+
93
+ Point `--target` at your own system with `module:callable` (any `input -> answer` function; the answer may be a string
94
+ or a dict with `contexts`/`citations`/`usage`).
95
+
96
+ The deterministic path needs no key. The **opt-in LLM-judge** (`--real` calibration, or `judge_support(provider=...)`)
97
+ reads `ANTHROPIC_API_KEY` — set it in the environment, or drop it in a gitignored `.env` at the repo root
98
+ (`ANTHROPIC_API_KEY=sk-ant-...`), which the CLI and `spancheck.calibrate` load automatically.
99
+
100
+ As a **GitHub Action** (fails a PR on a regression):
101
+
102
+ ```yaml
103
+ - uses: tjromack/spancheck@main
104
+ with:
105
+ cases: eval/cases.jsonl
106
+ target: myapp.rag:answer
107
+ thresholds: "citation_accuracy=1.0 groundedness=0.9"
108
+ baseline: baseline-audit.json # optional: also fail on a pass-rate drop
109
+ ```
110
+
111
+ ## Lineage
112
+
113
+ `spancheck` is the spinoff of [`llm-eval-guardrails-harness`](https://github.com/tjromack/llm-eval-guardrails-harness) —
114
+ the version that proved the method against a single real target (a regulatory RAG copilot), with LLM-judge agreement
115
+ 1.00 against a human gold set. This package is the same method, extracted from an in-repo lab, generalised to any
116
+ corpus, and packaged so a stranger can install it and point it at their own system. What carried over unchanged, what
117
+ was rewritten and why, and what the harness got wrong that this fixes are recorded in [`LINEAGE.md`](LINEAGE.md).
118
+
119
+ ## Quickstart
120
+
121
+ > The Phase-1 core below runs today (`pip install -e .`). Lines marked *(Phase 3)* / *(Phase 2)* are the intended shape
122
+ > of what those phases add; see `Status` and `TODO.md`.
123
+
124
+ ```python
125
+ from spancheck import Case, evaluate, adapter, citation_accuracy
126
+
127
+ # 1. Wrap your system in a thin adapter: input -> {answer, contexts, citations, usage, latency_ms}
128
+ # (optional — evaluate() also accepts a plain input->answer callable and times it for you)
129
+ my_system = adapter(lambda q: my_rag_pipeline(q))
130
+
131
+ # 2. Describe your test set — answerable, unanswerable, and adversarial cases
132
+ cases = [
133
+ Case(id="q1", input="What is the notice period?", category="answerable",
134
+ expected="30 days", meta={"answerable": True}),
135
+ Case(id="q2", input="What is the company's revenue?", category="unanswerable",
136
+ meta={"answerable": False}), # should abstain
137
+ ]
138
+
139
+ # 3. Score it — a run is captured once, then the metrics compute offline
140
+ run = evaluate(cases, my_system)
141
+ print(run.scorecard().by_grader()) # citation accuracy, abstention correctness, groundedness, PII-leak
142
+ run.save("run.json") # persist a run for baselines / regression gating
143
+ run.audit_log("audit.json") # the versioned, compliance-readable record (per-citation verdicts + cost/latency)
144
+ run.rescore([citation_accuracy(support_threshold=0.8)]) # re-score the SAME run offline — no system call
145
+ ```
146
+
147
+ ```bash
148
+ # CLI (thin wrapper over the same API) — command surface exists; wired to the library in Phase 5
149
+ spancheck run cases.jsonl --target my_module:my_system --out audit.json
150
+ spancheck score audit.json # recompute metrics from a cached run, no network
151
+ ```
152
+
153
+ ## Limits — what a passing score does *not* claim
154
+
155
+ - **Deterministic support is a lexical proxy, not entailment.** By default, citation *support* is scored by how much of
156
+ the claim's wording the cited span covers — fast, offline, and unable to see a span that shares the claim's words but
157
+ **contradicts** it ("the notice period is *not* 30 days"). The **opt-in LLM-judge** upgrades this to true entailment
158
+ (`citation_accuracy(support_fn=judge_support(provider=...))`), and is **calibrated against a human gold set before it
159
+ is trusted** (`python -m spancheck.calibrate`). The deterministic core flags the proxy's boundary rather than hiding
160
+ it (a test pins the false-positive; the judge catches it).
161
+ - **Provenance requires a verbatim quote.** A paraphrased citation fails provenance by design — a deliberate incentive
162
+ for a system to quote its sources exactly. `spancheck` does not (yet) match a citation by meaning.
163
+ - **A citation isn't scoped to its named source.** A cited span found in *any* retrieved context passes; tying a
164
+ citation to the specific `source_id` it names is a planned refinement.
165
+ - **Groundedness is a word-overlap proxy too**, and the metrics are only as good as the test set you bring. `spancheck`
166
+ measures a system against cases you author; it does not generate them.
167
+
168
+ ## Design pins
169
+
170
+ - **Library-first, CLI-second** — the API is the product.
171
+ - **No provider lock-in** — a thin adapter from day one; stub by default, any model behind a callable.
172
+ - **Every metric computable without network given a cached run** — capture once, re-score offline.
173
+ - **The audit-log schema is versioned** — a breaking change is a major version bump; it's a contract with a reviewer.
174
+ - **Judge prompts live in version-controlled files** — a rubric change is a reviewable, version-stamped diff.
175
+ - **No dependency on any other project's internals** — standalone; dogfooded as a black box, never coupled.
176
+
177
+ ## License
178
+
179
+ **MIT** — see [LICENSE](LICENSE). Free to use, modify, and distribute; the library is a tool meant to be run against
180
+ your own systems.
@@ -0,0 +1,158 @@
1
+ # spancheck
2
+
3
+ > © 2026 Trevor J. Romack — **MIT-licensed** ([LICENSE](LICENSE)) · `pip install spancheck` · tjromack@gmail.com
4
+
5
+ **Score a grounded-answer system on citation accuracy, abstention correctness, hallucination rate, cost and latency —
6
+ and get an audit log a compliance reviewer can read.**
7
+
8
+ `spancheck` is an installable, corpus-agnostic evaluation library for RAG and retrieval-then-answer pipelines. Point it
9
+ at any system through a thin adapter, give it a test set, and it produces a scorecard on the four metrics that decide
10
+ whether the system's answers can be trusted — plus a versioned audit log that reconstructs *why* each answer passed or
11
+ failed.
12
+
13
+ Library first, CLI second, GitHub Action third: the Python API is the product; the command line and the Action are thin
14
+ wrappers over it.
15
+
16
+ ---
17
+
18
+ ## The four metrics
19
+
20
+ | Metric | Question it answers | How it's scored |
21
+ |---|---|---|
22
+ | **Citation accuracy** | Does the cited span actually support the claim? | Deterministic span check *(the capability built first and tested deepest)* |
23
+ | **Abstention correctness** | Does it decline when the answer isn't in the corpus, and answer when it is? | Deterministic |
24
+ | **Hallucination rate** | What share of answers assert something the retrieved context doesn't support? | Deterministic groundedness + calibrated LLM-judge for the qualitative residue |
25
+ | **Cost & latency** | Tokens / dollars / wall-clock per answer, against a budget | Deterministic, from a cached run — no network |
26
+
27
+ ## Status
28
+
29
+ **Complete and dogfooded.** The measurement core (`Case` / `evaluate` / `Run` / `Scorecard` / `diff` / `gate`), the thin
30
+ `adapter` contract, the deterministic graders, **citation-span verification** — the capability this build exists to
31
+ add — the **cost/latency + versioned audit log** layer, the **calibrated LLM-judge**, and the **CLI + GitHub Action**
32
+ are implemented, dependency-free (the judge reaches a model only through a caller-supplied callable), and tested
33
+ (68 tests, green on a clean `pip install` and in CI). Citation verification splits into **provenance** (is the cited
34
+ span really in the retrieved context, verbatim?) and **support** (does the span cover the claim?); both must hold, so a
35
+ fabricated quote or a right-answer-wrong-citation cannot pass. A captured run re-scores **offline** (`Run.rescore`), and
36
+ `run.audit_log()` emits a versioned, self-describing record with per-citation verdicts (schema in
37
+ [`docs/AUDIT-LOG.md`](docs/AUDIT-LOG.md)). The LLM-judge is **opt-in** — the default grader set stays deterministic and
38
+ offline — and is **calibrated before it is trusted**.
39
+
40
+ ## Results (measured)
41
+
42
+ `spancheck` was run end-to-end against a real product pipeline as a black box (a grounded answer-a-document tool on
43
+ `claude-sonnet-5`, over its own sample contract), and the judge was calibrated against a human-labelled gold set. Full
44
+ write-up in [`docs/CASE-STUDY.md`](docs/CASE-STUDY.md); the audit log is committed at
45
+ [`dogfood/suver_audit.json`](dogfood/suver_audit.json).
46
+
47
+ | Metric | Result |
48
+ |---|---|
49
+ | Citation accuracy (real pipeline) | **1.00** — every cited span real and supporting |
50
+ | Hallucination (groundedness) | **0 hallucinations** (pass-rate 1.00) |
51
+ | Abstention correctness | 11/12 — spancheck **caught one false-abstention** on the real system |
52
+ | Overall (12 cases) | **0.967** |
53
+ | Judge vs human gold set — stub / `claude-sonnet-5` | **0.75 / 1.00** |
54
+ | Tests | **68**, clean `pip install` + CI (3.11 & 3.12) |
55
+
56
+ The one miss is the point: the eval found a real recall gap in a shipping system — a measured behaviour, not a spancheck
57
+ bug. Reproduce with `python dogfood/run_dogfood.py` and `python -m spancheck.calibrate --real`.
58
+
59
+ ## Command line & CI
60
+
61
+ The CLI is a thin wrapper over the library. It works out of the box against a shipped demo target — no key, no user
62
+ code:
63
+
64
+ ```bash
65
+ pip install spancheck # or `pip install -e .` from a clone
66
+ spancheck run examples/cases.jsonl --target spancheck.demo:system --out audit.json
67
+ spancheck score audit.json # recompute metrics from the cache, offline
68
+ spancheck gate audit.json citation_accuracy=1.0 groundedness=0.9 # exit 1 on failure (for CI)
69
+ ```
70
+
71
+ Point `--target` at your own system with `module:callable` (any `input -> answer` function; the answer may be a string
72
+ or a dict with `contexts`/`citations`/`usage`).
73
+
74
+ The deterministic path needs no key. The **opt-in LLM-judge** (`--real` calibration, or `judge_support(provider=...)`)
75
+ reads `ANTHROPIC_API_KEY` — set it in the environment, or drop it in a gitignored `.env` at the repo root
76
+ (`ANTHROPIC_API_KEY=sk-ant-...`), which the CLI and `spancheck.calibrate` load automatically.
77
+
78
+ As a **GitHub Action** (fails a PR on a regression):
79
+
80
+ ```yaml
81
+ - uses: tjromack/spancheck@main
82
+ with:
83
+ cases: eval/cases.jsonl
84
+ target: myapp.rag:answer
85
+ thresholds: "citation_accuracy=1.0 groundedness=0.9"
86
+ baseline: baseline-audit.json # optional: also fail on a pass-rate drop
87
+ ```
88
+
89
+ ## Lineage
90
+
91
+ `spancheck` is the spinoff of [`llm-eval-guardrails-harness`](https://github.com/tjromack/llm-eval-guardrails-harness) —
92
+ the version that proved the method against a single real target (a regulatory RAG copilot), with LLM-judge agreement
93
+ 1.00 against a human gold set. This package is the same method, extracted from an in-repo lab, generalised to any
94
+ corpus, and packaged so a stranger can install it and point it at their own system. What carried over unchanged, what
95
+ was rewritten and why, and what the harness got wrong that this fixes are recorded in [`LINEAGE.md`](LINEAGE.md).
96
+
97
+ ## Quickstart
98
+
99
+ > The Phase-1 core below runs today (`pip install -e .`). Lines marked *(Phase 3)* / *(Phase 2)* are the intended shape
100
+ > of what those phases add; see `Status` and `TODO.md`.
101
+
102
+ ```python
103
+ from spancheck import Case, evaluate, adapter, citation_accuracy
104
+
105
+ # 1. Wrap your system in a thin adapter: input -> {answer, contexts, citations, usage, latency_ms}
106
+ # (optional — evaluate() also accepts a plain input->answer callable and times it for you)
107
+ my_system = adapter(lambda q: my_rag_pipeline(q))
108
+
109
+ # 2. Describe your test set — answerable, unanswerable, and adversarial cases
110
+ cases = [
111
+ Case(id="q1", input="What is the notice period?", category="answerable",
112
+ expected="30 days", meta={"answerable": True}),
113
+ Case(id="q2", input="What is the company's revenue?", category="unanswerable",
114
+ meta={"answerable": False}), # should abstain
115
+ ]
116
+
117
+ # 3. Score it — a run is captured once, then the metrics compute offline
118
+ run = evaluate(cases, my_system)
119
+ print(run.scorecard().by_grader()) # citation accuracy, abstention correctness, groundedness, PII-leak
120
+ run.save("run.json") # persist a run for baselines / regression gating
121
+ run.audit_log("audit.json") # the versioned, compliance-readable record (per-citation verdicts + cost/latency)
122
+ run.rescore([citation_accuracy(support_threshold=0.8)]) # re-score the SAME run offline — no system call
123
+ ```
124
+
125
+ ```bash
126
+ # CLI (thin wrapper over the same API) — command surface exists; wired to the library in Phase 5
127
+ spancheck run cases.jsonl --target my_module:my_system --out audit.json
128
+ spancheck score audit.json # recompute metrics from a cached run, no network
129
+ ```
130
+
131
+ ## Limits — what a passing score does *not* claim
132
+
133
+ - **Deterministic support is a lexical proxy, not entailment.** By default, citation *support* is scored by how much of
134
+ the claim's wording the cited span covers — fast, offline, and unable to see a span that shares the claim's words but
135
+ **contradicts** it ("the notice period is *not* 30 days"). The **opt-in LLM-judge** upgrades this to true entailment
136
+ (`citation_accuracy(support_fn=judge_support(provider=...))`), and is **calibrated against a human gold set before it
137
+ is trusted** (`python -m spancheck.calibrate`). The deterministic core flags the proxy's boundary rather than hiding
138
+ it (a test pins the false-positive; the judge catches it).
139
+ - **Provenance requires a verbatim quote.** A paraphrased citation fails provenance by design — a deliberate incentive
140
+ for a system to quote its sources exactly. `spancheck` does not (yet) match a citation by meaning.
141
+ - **A citation isn't scoped to its named source.** A cited span found in *any* retrieved context passes; tying a
142
+ citation to the specific `source_id` it names is a planned refinement.
143
+ - **Groundedness is a word-overlap proxy too**, and the metrics are only as good as the test set you bring. `spancheck`
144
+ measures a system against cases you author; it does not generate them.
145
+
146
+ ## Design pins
147
+
148
+ - **Library-first, CLI-second** — the API is the product.
149
+ - **No provider lock-in** — a thin adapter from day one; stub by default, any model behind a callable.
150
+ - **Every metric computable without network given a cached run** — capture once, re-score offline.
151
+ - **The audit-log schema is versioned** — a breaking change is a major version bump; it's a contract with a reviewer.
152
+ - **Judge prompts live in version-controlled files** — a rubric change is a reviewable, version-stamped diff.
153
+ - **No dependency on any other project's internals** — standalone; dogfooded as a black box, never coupled.
154
+
155
+ ## License
156
+
157
+ **MIT** — see [LICENSE](LICENSE). Free to use, modify, and distribute; the library is a tool meant to be run against
158
+ your own systems.
@@ -0,0 +1,43 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "spancheck"
7
+ version = "0.1.0"
8
+ description = "Score a grounded-answer system on citation accuracy, abstention correctness, hallucination rate, cost and latency — with an audit log a compliance reviewer can read."
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Trevor J. Romack", email = "tjromack@gmail.com" }]
14
+ keywords = ["rag", "evaluation", "llm", "hallucination", "citation", "abstention", "guardrails"]
15
+ classifiers = [
16
+ "Programming Language :: Python :: 3",
17
+ "Programming Language :: Python :: 3.11",
18
+ "Programming Language :: Python :: 3.12",
19
+ "Operating System :: OS Independent",
20
+ "Intended Audience :: Developers",
21
+ "Topic :: Software Development :: Quality Assurance",
22
+ ]
23
+ # Standard-library first (design pin: a hard dependency is added only with a DECISIONS entry). None yet.
24
+ dependencies = []
25
+
26
+ [project.optional-dependencies]
27
+ dev = ["pytest>=8"]
28
+
29
+ [project.urls]
30
+ Homepage = "https://github.com/tjromack/spancheck"
31
+ Source = "https://github.com/tjromack/spancheck"
32
+
33
+ [project.scripts]
34
+ spancheck = "spancheck.cli:main"
35
+
36
+ [tool.setuptools.packages.find]
37
+ where = ["src"]
38
+
39
+ [tool.setuptools.package-data]
40
+ spancheck = ["prompts/*.txt"]
41
+
42
+ [tool.pytest.ini_options]
43
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,47 @@
1
+ """spancheck — score a grounded-answer system on citation accuracy, abstention correctness,
2
+ hallucination rate, cost and latency, and emit an audit log a compliance reviewer can read.
3
+
4
+ Public API (Phase 1 — the ported measurement core + the thin adapter contract):
5
+
6
+ from spancheck import Case, evaluate, adapter, Run, Scorecard, gate, diff
7
+
8
+ Deterministic graders live in `spancheck.graders`. Citation-span verification (the capability this build exists to
9
+ add) lands in Phase 2; the versioned audit log in Phase 3; the calibrated LLM-judge in Phase 4. See TODO.md.
10
+
11
+ Nothing here hard-codes a model provider or depends on any other project's internals (CLAUDE.md design pins).
12
+ """
13
+ from __future__ import annotations
14
+
15
+ from .adapter import Output, adapter, normalize
16
+ from .core import (
17
+ Case,
18
+ GradeResult,
19
+ CaseResult,
20
+ Run,
21
+ Scorecard,
22
+ evaluate,
23
+ run_eval,
24
+ diff,
25
+ gate,
26
+ default_graders,
27
+ )
28
+ from .span import citation_accuracy, verify_citation, CitationVerdict
29
+ from .audit import build_audit_log, cost_latency, SCHEMA_VERSION
30
+ from .judge import judge_entailment, judge_support, llm_judge, load_prompt
31
+ from .calibrate import calibrate_judge
32
+ from . import graders
33
+
34
+ __version__ = "0.0.1"
35
+
36
+ __all__ = [
37
+ # the declared public surface
38
+ "Case", "evaluate", "adapter", "Run", "Scorecard", "gate", "diff",
39
+ # citation-span verification (the capability spancheck is named for)
40
+ "citation_accuracy", "verify_citation", "CitationVerdict",
41
+ # cost/latency + the versioned audit log
42
+ "build_audit_log", "cost_latency", "SCHEMA_VERSION",
43
+ # the calibrated LLM-judge (opt-in; stub by default, no provider lock-in)
44
+ "judge_entailment", "judge_support", "llm_judge", "load_prompt", "calibrate_judge",
45
+ # supporting types + the grader library
46
+ "Output", "normalize", "GradeResult", "CaseResult", "run_eval", "default_graders", "graders",
47
+ ]
@@ -0,0 +1,25 @@
1
+ """Tiny, dependency-free .env loader.
2
+
3
+ Reads KEY=VALUE lines from a .env file into os.environ **without overwriting** anything already set (a real
4
+ environment variable always wins). Called at CLI/calibration startup so a key placed in `.env` "just works" — no
5
+ python-dotenv dependency, no vendor lock-in. `.env` is gitignored; a key never enters the repo.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import os
10
+
11
+
12
+ def load_dotenv(path: str = ".env") -> None:
13
+ try:
14
+ with open(path, encoding="utf-8") as f:
15
+ for line in f:
16
+ line = line.strip()
17
+ if not line or line.startswith("#") or "=" not in line:
18
+ continue
19
+ key, _, value = line.partition("=")
20
+ key = key.strip()
21
+ value = value.strip().strip('"').strip("'")
22
+ if key and key not in os.environ:
23
+ os.environ[key] = value
24
+ except FileNotFoundError:
25
+ pass
@@ -0,0 +1,29 @@
1
+ """Shared text helpers used by graders and the citation-span verifier.
2
+
3
+ Kept in one place so the deterministic checks agree on what a "word" is and how text is normalised.
4
+ Stdlib only.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import re
9
+
10
+ WORD = re.compile(r"[a-z0-9]+")
11
+
12
+ STOP = set("the a an and or of to in on for is are was were be been it its this that with as at by from "
13
+ "your you our we i they he she them his her their not no do does did have has had will would can "
14
+ "could should may might must shall into over under about than then so if but".split())
15
+
16
+
17
+ def norm(s) -> str:
18
+ """Lowercase and collapse all whitespace to single spaces — the normalisation used for span provenance."""
19
+ return re.sub(r"\s+", " ", (s or "")).strip().lower()
20
+
21
+
22
+ def words(s) -> set:
23
+ """All word tokens (lowercased) — used as the haystack when checking coverage."""
24
+ return set(WORD.findall((s or "").lower()))
25
+
26
+
27
+ def content_words(s):
28
+ """Meaningful tokens: lowercased words that aren't stopwords and are longer than two characters."""
29
+ return [w for w in WORD.findall((s or "").lower()) if w not in STOP and len(w) > 2]
@@ -0,0 +1,90 @@
1
+ """The thin adapter contract — the seam between spancheck and any system under test.
2
+
3
+ Design pin #2 (no provider lock-in) and #3 (every metric computable offline from a captured run) both live here:
4
+ a system's answer is normalised **once** into a typed `Output` that carries everything the metrics need
5
+ (the answer, the retrieved contexts, the citations, token/cost usage, and latency), so scoring never has to
6
+ call the system again.
7
+
8
+ `spancheck` never hard-codes a vendor. A caller supplies a plain `input -> answer` callable; `adapter()` wraps it,
9
+ times it, and normalises whatever it returns. The return may be:
10
+
11
+ - a **str** — just the answer text;
12
+ - a **dict** — with any of `answer`/`text`, `contexts`/`context`, `citations`, `usage`, `latency_ms`, `abstained`;
13
+ - an **Output** — already normalised (idempotent).
14
+
15
+ Rewritten from evallab's loose "str-or-dict" handling into a pinned dataclass so cost/latency and citations are
16
+ first-class and always present (see LINEAGE.md, DECISIONS.md AB-DEC 006).
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import time
21
+ from dataclasses import dataclass, field, asdict
22
+
23
+
24
+ @dataclass
25
+ class Output:
26
+ """The normalised result of one system call — the single shape every grader reads.
27
+
28
+ `citations` is carried through as-is in Phase 1; its shape is firmed up by the citation-span verifier
29
+ (Phase 2). `abstained` is an explicit signal when the system provides one; when it is None the graders
30
+ infer abstention from the answer text.
31
+ """
32
+ answer: str = ""
33
+ contexts: list = field(default_factory=list) # retrieved context passages the answer should be grounded in
34
+ citations: list = field(default_factory=list) # citations backing the answer (span shape fixed in Phase 2)
35
+ usage: dict = field(default_factory=dict) # e.g. {"input_tokens", "output_tokens", "cost_usd"}
36
+ latency_ms: float | None = None # wall-clock for the call; filled by adapter/evaluate if absent
37
+ abstained: bool | None = None # explicit abstention signal, if the system emits one
38
+ raw: object = None # the untouched original return, for the audit log
39
+
40
+ def to_dict(self) -> dict:
41
+ d = asdict(self)
42
+ # `raw` may not be JSON-serialisable; keep a string fallback for the audit log rather than crash.
43
+ try:
44
+ import json
45
+ json.dumps(d["raw"])
46
+ except (TypeError, ValueError):
47
+ d["raw"] = repr(self.raw)
48
+ return d
49
+
50
+
51
+ def normalize(value) -> Output:
52
+ """Coerce a system's return (str | dict | Output) into an Output. Idempotent on Output."""
53
+ if isinstance(value, Output):
54
+ return value
55
+ if isinstance(value, dict):
56
+ answer = value.get("answer", value.get("text", ""))
57
+ ctx = value.get("contexts", value.get("context", []))
58
+ if not isinstance(ctx, list):
59
+ ctx = [ctx] if ctx else []
60
+ citations = value.get("citations", [])
61
+ if not isinstance(citations, list):
62
+ citations = [citations]
63
+ return Output(
64
+ answer="" if answer is None else str(answer),
65
+ contexts=[str(c) for c in ctx],
66
+ citations=citations,
67
+ usage=dict(value.get("usage", {}) or {}),
68
+ latency_ms=value.get("latency_ms"),
69
+ abstained=value.get("abstained"),
70
+ raw=value,
71
+ )
72
+ # a bare string (or anything else) is just the answer text
73
+ return Output(answer="" if value is None else str(value), raw=value)
74
+
75
+
76
+ def adapter(fn, *, capture_time: bool = True):
77
+ """Wrap a caller's `input -> result` callable so it returns a normalised, timed `Output`.
78
+
79
+ Using `adapter()` is optional — `evaluate()` normalises and times its system either way — but it lets a caller
80
+ attach usage/citations at the source and hand `evaluate` a system that already speaks the contract.
81
+ """
82
+ def run(inp) -> Output:
83
+ start = time.perf_counter()
84
+ result = fn(inp)
85
+ elapsed_ms = (time.perf_counter() - start) * 1000.0
86
+ out = normalize(result)
87
+ if capture_time and out.latency_ms is None:
88
+ out.latency_ms = round(elapsed_ms, 3)
89
+ return out
90
+ return run