spancheck 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spancheck-0.1.0/LICENSE +21 -0
- spancheck-0.1.0/PKG-INFO +180 -0
- spancheck-0.1.0/README.md +158 -0
- spancheck-0.1.0/pyproject.toml +43 -0
- spancheck-0.1.0/setup.cfg +4 -0
- spancheck-0.1.0/src/spancheck/__init__.py +47 -0
- spancheck-0.1.0/src/spancheck/_env.py +25 -0
- spancheck-0.1.0/src/spancheck/_text.py +29 -0
- spancheck-0.1.0/src/spancheck/adapter.py +90 -0
- spancheck-0.1.0/src/spancheck/audit.py +135 -0
- spancheck-0.1.0/src/spancheck/calibrate.py +98 -0
- spancheck-0.1.0/src/spancheck/cli.py +148 -0
- spancheck-0.1.0/src/spancheck/core.py +267 -0
- spancheck-0.1.0/src/spancheck/demo.py +43 -0
- spancheck-0.1.0/src/spancheck/graders.py +127 -0
- spancheck-0.1.0/src/spancheck/judge.py +102 -0
- spancheck-0.1.0/src/spancheck/prompts/entailment_v1.txt +15 -0
- spancheck-0.1.0/src/spancheck/provider.py +32 -0
- spancheck-0.1.0/src/spancheck/span.py +126 -0
- spancheck-0.1.0/src/spancheck.egg-info/PKG-INFO +180 -0
- spancheck-0.1.0/src/spancheck.egg-info/SOURCES.txt +30 -0
- spancheck-0.1.0/src/spancheck.egg-info/dependency_links.txt +1 -0
- spancheck-0.1.0/src/spancheck.egg-info/entry_points.txt +2 -0
- spancheck-0.1.0/src/spancheck.egg-info/requires.txt +3 -0
- spancheck-0.1.0/src/spancheck.egg-info/top_level.txt +1 -0
- spancheck-0.1.0/tests/test_audit.py +134 -0
- spancheck-0.1.0/tests/test_cli.py +75 -0
- spancheck-0.1.0/tests/test_core.py +174 -0
- spancheck-0.1.0/tests/test_env.py +20 -0
- spancheck-0.1.0/tests/test_judge.py +109 -0
- spancheck-0.1.0/tests/test_scaffold.py +31 -0
- spancheck-0.1.0/tests/test_span.py +171 -0
spancheck-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Trevor J. Romack
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
spancheck-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: spancheck
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Score a grounded-answer system on citation accuracy, abstention correctness, hallucination rate, cost and latency — with an audit log a compliance reviewer can read.
|
|
5
|
+
Author-email: "Trevor J. Romack" <tjromack@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/tjromack/spancheck
|
|
8
|
+
Project-URL: Source, https://github.com/tjromack/spancheck
|
|
9
|
+
Keywords: rag,evaluation,llm,hallucination,citation,abstention,guardrails
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
16
|
+
Requires-Python: >=3.11
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
21
|
+
Dynamic: license-file
|
|
22
|
+
|
|
23
|
+
# spancheck
|
|
24
|
+
|
|
25
|
+
> © 2026 Trevor J. Romack — **MIT-licensed** ([LICENSE](LICENSE)) · `pip install spancheck` · tjromack@gmail.com
|
|
26
|
+
|
|
27
|
+
**Score a grounded-answer system on citation accuracy, abstention correctness, hallucination rate, cost and latency —
|
|
28
|
+
and get an audit log a compliance reviewer can read.**
|
|
29
|
+
|
|
30
|
+
`spancheck` is an installable, corpus-agnostic evaluation library for RAG and retrieval-then-answer pipelines. Point it
|
|
31
|
+
at any system through a thin adapter, give it a test set, and it produces a scorecard on the four metrics that decide
|
|
32
|
+
whether the system's answers can be trusted — plus a versioned audit log that reconstructs *why* each answer passed or
|
|
33
|
+
failed.
|
|
34
|
+
|
|
35
|
+
Library first, CLI second, GitHub Action third: the Python API is the product; the command line and the Action are thin
|
|
36
|
+
wrappers over it.
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
## The four metrics
|
|
41
|
+
|
|
42
|
+
| Metric | Question it answers | How it's scored |
|
|
43
|
+
|---|---|---|
|
|
44
|
+
| **Citation accuracy** | Does the cited span actually support the claim? | Deterministic span check *(the capability built first and tested deepest)* |
|
|
45
|
+
| **Abstention correctness** | Does it decline when the answer isn't in the corpus, and answer when it is? | Deterministic |
|
|
46
|
+
| **Hallucination rate** | What share of answers assert something the retrieved context doesn't support? | Deterministic groundedness + calibrated LLM-judge for the qualitative residue |
|
|
47
|
+
| **Cost & latency** | Tokens / dollars / wall-clock per answer, against a budget | Deterministic, from a cached run — no network |
|
|
48
|
+
|
|
49
|
+
## Status
|
|
50
|
+
|
|
51
|
+
**Complete and dogfooded.** The measurement core (`Case` / `evaluate` / `Run` / `Scorecard` / `diff` / `gate`), the thin
|
|
52
|
+
`adapter` contract, the deterministic graders, **citation-span verification** — the capability this build exists to
|
|
53
|
+
add — the **cost/latency + versioned audit log** layer, the **calibrated LLM-judge**, and the **CLI + GitHub Action**
|
|
54
|
+
are implemented, dependency-free (the judge reaches a model only through a caller-supplied callable), and tested
|
|
55
|
+
(68 tests, green on a clean `pip install` and in CI). Citation verification splits into **provenance** (is the cited
|
|
56
|
+
span really in the retrieved context, verbatim?) and **support** (does the span cover the claim?); both must hold, so a
|
|
57
|
+
fabricated quote or a right-answer-wrong-citation cannot pass. A captured run re-scores **offline** (`Run.rescore`), and
|
|
58
|
+
`run.audit_log()` emits a versioned, self-describing record with per-citation verdicts (schema in
|
|
59
|
+
[`docs/AUDIT-LOG.md`](docs/AUDIT-LOG.md)). The LLM-judge is **opt-in** — the default grader set stays deterministic and
|
|
60
|
+
offline — and is **calibrated before it is trusted**.
|
|
61
|
+
|
|
62
|
+
## Results (measured)
|
|
63
|
+
|
|
64
|
+
`spancheck` was run end-to-end against a real product pipeline as a black box (a grounded answer-a-document tool on
|
|
65
|
+
`claude-sonnet-5`, over its own sample contract), and the judge was calibrated against a human-labelled gold set. Full
|
|
66
|
+
write-up in [`docs/CASE-STUDY.md`](docs/CASE-STUDY.md); the audit log is committed at
|
|
67
|
+
[`dogfood/suver_audit.json`](dogfood/suver_audit.json).
|
|
68
|
+
|
|
69
|
+
| Metric | Result |
|
|
70
|
+
|---|---|
|
|
71
|
+
| Citation accuracy (real pipeline) | **1.00** — every cited span real and supporting |
|
|
72
|
+
| Hallucination (groundedness) | **0 hallucinations** (pass-rate 1.00) |
|
|
73
|
+
| Abstention correctness | 11/12 — spancheck **caught one false-abstention** on the real system |
|
|
74
|
+
| Overall (12 cases) | **0.967** |
|
|
75
|
+
| Judge vs human gold set — stub / `claude-sonnet-5` | **0.75 / 1.00** |
|
|
76
|
+
| Tests | **68**, clean `pip install` + CI (3.11 & 3.12) |
|
|
77
|
+
|
|
78
|
+
The one miss is the point: the eval found a real recall gap in a shipping system — a measured behaviour, not a spancheck
|
|
79
|
+
bug. Reproduce with `python dogfood/run_dogfood.py` and `python -m spancheck.calibrate --real`.
|
|
80
|
+
|
|
81
|
+
## Command line & CI
|
|
82
|
+
|
|
83
|
+
The CLI is a thin wrapper over the library. It works out of the box against a shipped demo target — no key, no user
|
|
84
|
+
code:
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
pip install spancheck # or `pip install -e .` from a clone
|
|
88
|
+
spancheck run examples/cases.jsonl --target spancheck.demo:system --out audit.json
|
|
89
|
+
spancheck score audit.json # recompute metrics from the cache, offline
|
|
90
|
+
spancheck gate audit.json citation_accuracy=1.0 groundedness=0.9 # exit 1 on failure (for CI)
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Point `--target` at your own system with `module:callable` (any `input -> answer` function; the answer may be a string
|
|
94
|
+
or a dict with `contexts`/`citations`/`usage`).
|
|
95
|
+
|
|
96
|
+
The deterministic path needs no key. The **opt-in LLM-judge** (`--real` calibration, or `judge_support(provider=...)`)
|
|
97
|
+
reads `ANTHROPIC_API_KEY` — set it in the environment, or drop it in a gitignored `.env` at the repo root
|
|
98
|
+
(`ANTHROPIC_API_KEY=sk-ant-...`), which the CLI and `spancheck.calibrate` load automatically.
|
|
99
|
+
|
|
100
|
+
As a **GitHub Action** (fails a PR on a regression):
|
|
101
|
+
|
|
102
|
+
```yaml
|
|
103
|
+
- uses: tjromack/spancheck@main
|
|
104
|
+
with:
|
|
105
|
+
cases: eval/cases.jsonl
|
|
106
|
+
target: myapp.rag:answer
|
|
107
|
+
thresholds: "citation_accuracy=1.0 groundedness=0.9"
|
|
108
|
+
baseline: baseline-audit.json # optional: also fail on a pass-rate drop
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
## Lineage
|
|
112
|
+
|
|
113
|
+
`spancheck` is the spinoff of [`llm-eval-guardrails-harness`](https://github.com/tjromack/llm-eval-guardrails-harness) —
|
|
114
|
+
the version that proved the method against a single real target (a regulatory RAG copilot), with LLM-judge agreement
|
|
115
|
+
1.00 against a human gold set. This package is the same method, extracted from an in-repo lab, generalised to any
|
|
116
|
+
corpus, and packaged so a stranger can install it and point it at their own system. What carried over unchanged, what
|
|
117
|
+
was rewritten and why, and what the harness got wrong that this fixes are recorded in [`LINEAGE.md`](LINEAGE.md).
|
|
118
|
+
|
|
119
|
+
## Quickstart
|
|
120
|
+
|
|
121
|
+
> The Phase-1 core below runs today (`pip install -e .`). Lines marked *(Phase 3)* / *(Phase 2)* are the intended shape
|
|
122
|
+
> of what those phases add; see `Status` and `TODO.md`.
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
from spancheck import Case, evaluate, adapter, citation_accuracy
|
|
126
|
+
|
|
127
|
+
# 1. Wrap your system in a thin adapter: input -> {answer, contexts, citations, usage, latency_ms}
|
|
128
|
+
# (optional — evaluate() also accepts a plain input->answer callable and times it for you)
|
|
129
|
+
my_system = adapter(lambda q: my_rag_pipeline(q))
|
|
130
|
+
|
|
131
|
+
# 2. Describe your test set — answerable, unanswerable, and adversarial cases
|
|
132
|
+
cases = [
|
|
133
|
+
Case(id="q1", input="What is the notice period?", category="answerable",
|
|
134
|
+
expected="30 days", meta={"answerable": True}),
|
|
135
|
+
Case(id="q2", input="What is the company's revenue?", category="unanswerable",
|
|
136
|
+
meta={"answerable": False}), # should abstain
|
|
137
|
+
]
|
|
138
|
+
|
|
139
|
+
# 3. Score it — a run is captured once, then the metrics compute offline
|
|
140
|
+
run = evaluate(cases, my_system)
|
|
141
|
+
print(run.scorecard().by_grader()) # citation accuracy, abstention correctness, groundedness, PII-leak
|
|
142
|
+
run.save("run.json") # persist a run for baselines / regression gating
|
|
143
|
+
run.audit_log("audit.json") # the versioned, compliance-readable record (per-citation verdicts + cost/latency)
|
|
144
|
+
run.rescore([citation_accuracy(support_threshold=0.8)]) # re-score the SAME run offline — no system call
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
```bash
|
|
148
|
+
# CLI (thin wrapper over the same API) — command surface exists; wired to the library in Phase 5
|
|
149
|
+
spancheck run cases.jsonl --target my_module:my_system --out audit.json
|
|
150
|
+
spancheck score audit.json # recompute metrics from a cached run, no network
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
## Limits — what a passing score does *not* claim
|
|
154
|
+
|
|
155
|
+
- **Deterministic support is a lexical proxy, not entailment.** By default, citation *support* is scored by how much of
|
|
156
|
+
the claim's wording the cited span covers — fast, offline, and unable to see a span that shares the claim's words but
|
|
157
|
+
**contradicts** it ("the notice period is *not* 30 days"). The **opt-in LLM-judge** upgrades this to true entailment
|
|
158
|
+
(`citation_accuracy(support_fn=judge_support(provider=...))`), and is **calibrated against a human gold set before it
|
|
159
|
+
is trusted** (`python -m spancheck.calibrate`). The deterministic core flags the proxy's boundary rather than hiding
|
|
160
|
+
it (a test pins the false-positive; the judge catches it).
|
|
161
|
+
- **Provenance requires a verbatim quote.** A paraphrased citation fails provenance by design — a deliberate incentive
|
|
162
|
+
for a system to quote its sources exactly. `spancheck` does not (yet) match a citation by meaning.
|
|
163
|
+
- **A citation isn't scoped to its named source.** A cited span found in *any* retrieved context passes; tying a
|
|
164
|
+
citation to the specific `source_id` it names is a planned refinement.
|
|
165
|
+
- **Groundedness is a word-overlap proxy too**, and the metrics are only as good as the test set you bring. `spancheck`
|
|
166
|
+
measures a system against cases you author; it does not generate them.
|
|
167
|
+
|
|
168
|
+
## Design pins
|
|
169
|
+
|
|
170
|
+
- **Library-first, CLI-second** — the API is the product.
|
|
171
|
+
- **No provider lock-in** — a thin adapter from day one; stub by default, any model behind a callable.
|
|
172
|
+
- **Every metric computable without network given a cached run** — capture once, re-score offline.
|
|
173
|
+
- **The audit-log schema is versioned** — a breaking change is a major version bump; it's a contract with a reviewer.
|
|
174
|
+
- **Judge prompts live in version-controlled files** — a rubric change is a reviewable, version-stamped diff.
|
|
175
|
+
- **No dependency on any other project's internals** — standalone; dogfooded as a black box, never coupled.
|
|
176
|
+
|
|
177
|
+
## License
|
|
178
|
+
|
|
179
|
+
**MIT** — see [LICENSE](LICENSE). Free to use, modify, and distribute; the library is a tool meant to be run against
|
|
180
|
+
your own systems.
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
# spancheck
|
|
2
|
+
|
|
3
|
+
> © 2026 Trevor J. Romack — **MIT-licensed** ([LICENSE](LICENSE)) · `pip install spancheck` · tjromack@gmail.com
|
|
4
|
+
|
|
5
|
+
**Score a grounded-answer system on citation accuracy, abstention correctness, hallucination rate, cost and latency —
|
|
6
|
+
and get an audit log a compliance reviewer can read.**
|
|
7
|
+
|
|
8
|
+
`spancheck` is an installable, corpus-agnostic evaluation library for RAG and retrieval-then-answer pipelines. Point it
|
|
9
|
+
at any system through a thin adapter, give it a test set, and it produces a scorecard on the four metrics that decide
|
|
10
|
+
whether the system's answers can be trusted — plus a versioned audit log that reconstructs *why* each answer passed or
|
|
11
|
+
failed.
|
|
12
|
+
|
|
13
|
+
Library first, CLI second, GitHub Action third: the Python API is the product; the command line and the Action are thin
|
|
14
|
+
wrappers over it.
|
|
15
|
+
|
|
16
|
+
---
|
|
17
|
+
|
|
18
|
+
## The four metrics
|
|
19
|
+
|
|
20
|
+
| Metric | Question it answers | How it's scored |
|
|
21
|
+
|---|---|---|
|
|
22
|
+
| **Citation accuracy** | Does the cited span actually support the claim? | Deterministic span check *(the capability built first and tested deepest)* |
|
|
23
|
+
| **Abstention correctness** | Does it decline when the answer isn't in the corpus, and answer when it is? | Deterministic |
|
|
24
|
+
| **Hallucination rate** | What share of answers assert something the retrieved context doesn't support? | Deterministic groundedness + calibrated LLM-judge for the qualitative residue |
|
|
25
|
+
| **Cost & latency** | Tokens / dollars / wall-clock per answer, against a budget | Deterministic, from a cached run — no network |
|
|
26
|
+
|
|
27
|
+
## Status
|
|
28
|
+
|
|
29
|
+
**Complete and dogfooded.** The measurement core (`Case` / `evaluate` / `Run` / `Scorecard` / `diff` / `gate`), the thin
|
|
30
|
+
`adapter` contract, the deterministic graders, **citation-span verification** — the capability this build exists to
|
|
31
|
+
add — the **cost/latency + versioned audit log** layer, the **calibrated LLM-judge**, and the **CLI + GitHub Action**
|
|
32
|
+
are implemented, dependency-free (the judge reaches a model only through a caller-supplied callable), and tested
|
|
33
|
+
(68 tests, green on a clean `pip install` and in CI). Citation verification splits into **provenance** (is the cited
|
|
34
|
+
span really in the retrieved context, verbatim?) and **support** (does the span cover the claim?); both must hold, so a
|
|
35
|
+
fabricated quote or a right-answer-wrong-citation cannot pass. A captured run re-scores **offline** (`Run.rescore`), and
|
|
36
|
+
`run.audit_log()` emits a versioned, self-describing record with per-citation verdicts (schema in
|
|
37
|
+
[`docs/AUDIT-LOG.md`](docs/AUDIT-LOG.md)). The LLM-judge is **opt-in** — the default grader set stays deterministic and
|
|
38
|
+
offline — and is **calibrated before it is trusted**.
|
|
39
|
+
|
|
40
|
+
## Results (measured)
|
|
41
|
+
|
|
42
|
+
`spancheck` was run end-to-end against a real product pipeline as a black box (a grounded answer-a-document tool on
|
|
43
|
+
`claude-sonnet-5`, over its own sample contract), and the judge was calibrated against a human-labelled gold set. Full
|
|
44
|
+
write-up in [`docs/CASE-STUDY.md`](docs/CASE-STUDY.md); the audit log is committed at
|
|
45
|
+
[`dogfood/suver_audit.json`](dogfood/suver_audit.json).
|
|
46
|
+
|
|
47
|
+
| Metric | Result |
|
|
48
|
+
|---|---|
|
|
49
|
+
| Citation accuracy (real pipeline) | **1.00** — every cited span real and supporting |
|
|
50
|
+
| Hallucination (groundedness) | **0 hallucinations** (pass-rate 1.00) |
|
|
51
|
+
| Abstention correctness | 11/12 — spancheck **caught one false-abstention** on the real system |
|
|
52
|
+
| Overall (12 cases) | **0.967** |
|
|
53
|
+
| Judge vs human gold set — stub / `claude-sonnet-5` | **0.75 / 1.00** |
|
|
54
|
+
| Tests | **68**, clean `pip install` + CI (3.11 & 3.12) |
|
|
55
|
+
|
|
56
|
+
The one miss is the point: the eval found a real recall gap in a shipping system — a measured behaviour, not a spancheck
|
|
57
|
+
bug. Reproduce with `python dogfood/run_dogfood.py` and `python -m spancheck.calibrate --real`.
|
|
58
|
+
|
|
59
|
+
## Command line & CI
|
|
60
|
+
|
|
61
|
+
The CLI is a thin wrapper over the library. It works out of the box against a shipped demo target — no key, no user
|
|
62
|
+
code:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
pip install spancheck # or `pip install -e .` from a clone
|
|
66
|
+
spancheck run examples/cases.jsonl --target spancheck.demo:system --out audit.json
|
|
67
|
+
spancheck score audit.json # recompute metrics from the cache, offline
|
|
68
|
+
spancheck gate audit.json citation_accuracy=1.0 groundedness=0.9 # exit 1 on failure (for CI)
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Point `--target` at your own system with `module:callable` (any `input -> answer` function; the answer may be a string
|
|
72
|
+
or a dict with `contexts`/`citations`/`usage`).
|
|
73
|
+
|
|
74
|
+
The deterministic path needs no key. The **opt-in LLM-judge** (`--real` calibration, or `judge_support(provider=...)`)
|
|
75
|
+
reads `ANTHROPIC_API_KEY` — set it in the environment, or drop it in a gitignored `.env` at the repo root
|
|
76
|
+
(`ANTHROPIC_API_KEY=sk-ant-...`), which the CLI and `spancheck.calibrate` load automatically.
|
|
77
|
+
|
|
78
|
+
As a **GitHub Action** (fails a PR on a regression):
|
|
79
|
+
|
|
80
|
+
```yaml
|
|
81
|
+
- uses: tjromack/spancheck@main
|
|
82
|
+
with:
|
|
83
|
+
cases: eval/cases.jsonl
|
|
84
|
+
target: myapp.rag:answer
|
|
85
|
+
thresholds: "citation_accuracy=1.0 groundedness=0.9"
|
|
86
|
+
baseline: baseline-audit.json # optional: also fail on a pass-rate drop
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## Lineage
|
|
90
|
+
|
|
91
|
+
`spancheck` is the spinoff of [`llm-eval-guardrails-harness`](https://github.com/tjromack/llm-eval-guardrails-harness) —
|
|
92
|
+
the version that proved the method against a single real target (a regulatory RAG copilot), with LLM-judge agreement
|
|
93
|
+
1.00 against a human gold set. This package is the same method, extracted from an in-repo lab, generalised to any
|
|
94
|
+
corpus, and packaged so a stranger can install it and point it at their own system. What carried over unchanged, what
|
|
95
|
+
was rewritten and why, and what the harness got wrong that this fixes are recorded in [`LINEAGE.md`](LINEAGE.md).
|
|
96
|
+
|
|
97
|
+
## Quickstart
|
|
98
|
+
|
|
99
|
+
> The Phase-1 core below runs today (`pip install -e .`). Lines marked *(Phase 3)* / *(Phase 2)* are the intended shape
|
|
100
|
+
> of what those phases add; see `Status` and `TODO.md`.
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from spancheck import Case, evaluate, adapter, citation_accuracy
|
|
104
|
+
|
|
105
|
+
# 1. Wrap your system in a thin adapter: input -> {answer, contexts, citations, usage, latency_ms}
|
|
106
|
+
# (optional — evaluate() also accepts a plain input->answer callable and times it for you)
|
|
107
|
+
my_system = adapter(lambda q: my_rag_pipeline(q))
|
|
108
|
+
|
|
109
|
+
# 2. Describe your test set — answerable, unanswerable, and adversarial cases
|
|
110
|
+
cases = [
|
|
111
|
+
Case(id="q1", input="What is the notice period?", category="answerable",
|
|
112
|
+
expected="30 days", meta={"answerable": True}),
|
|
113
|
+
Case(id="q2", input="What is the company's revenue?", category="unanswerable",
|
|
114
|
+
meta={"answerable": False}), # should abstain
|
|
115
|
+
]
|
|
116
|
+
|
|
117
|
+
# 3. Score it — a run is captured once, then the metrics compute offline
|
|
118
|
+
run = evaluate(cases, my_system)
|
|
119
|
+
print(run.scorecard().by_grader()) # citation accuracy, abstention correctness, groundedness, PII-leak
|
|
120
|
+
run.save("run.json") # persist a run for baselines / regression gating
|
|
121
|
+
run.audit_log("audit.json") # the versioned, compliance-readable record (per-citation verdicts + cost/latency)
|
|
122
|
+
run.rescore([citation_accuracy(support_threshold=0.8)]) # re-score the SAME run offline — no system call
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
# CLI (thin wrapper over the same API) — command surface exists; wired to the library in Phase 5
|
|
127
|
+
spancheck run cases.jsonl --target my_module:my_system --out audit.json
|
|
128
|
+
spancheck score audit.json # recompute metrics from a cached run, no network
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
## Limits — what a passing score does *not* claim
|
|
132
|
+
|
|
133
|
+
- **Deterministic support is a lexical proxy, not entailment.** By default, citation *support* is scored by how much of
|
|
134
|
+
the claim's wording the cited span covers — fast, offline, and unable to see a span that shares the claim's words but
|
|
135
|
+
**contradicts** it ("the notice period is *not* 30 days"). The **opt-in LLM-judge** upgrades this to true entailment
|
|
136
|
+
(`citation_accuracy(support_fn=judge_support(provider=...))`), and is **calibrated against a human gold set before it
|
|
137
|
+
is trusted** (`python -m spancheck.calibrate`). The deterministic core flags the proxy's boundary rather than hiding
|
|
138
|
+
it (a test pins the false-positive; the judge catches it).
|
|
139
|
+
- **Provenance requires a verbatim quote.** A paraphrased citation fails provenance by design — a deliberate incentive
|
|
140
|
+
for a system to quote its sources exactly. `spancheck` does not (yet) match a citation by meaning.
|
|
141
|
+
- **A citation isn't scoped to its named source.** A cited span found in *any* retrieved context passes; tying a
|
|
142
|
+
citation to the specific `source_id` it names is a planned refinement.
|
|
143
|
+
- **Groundedness is a word-overlap proxy too**, and the metrics are only as good as the test set you bring. `spancheck`
|
|
144
|
+
measures a system against cases you author; it does not generate them.
|
|
145
|
+
|
|
146
|
+
## Design pins
|
|
147
|
+
|
|
148
|
+
- **Library-first, CLI-second** — the API is the product.
|
|
149
|
+
- **No provider lock-in** — a thin adapter from day one; stub by default, any model behind a callable.
|
|
150
|
+
- **Every metric computable without network given a cached run** — capture once, re-score offline.
|
|
151
|
+
- **The audit-log schema is versioned** — a breaking change is a major version bump; it's a contract with a reviewer.
|
|
152
|
+
- **Judge prompts live in version-controlled files** — a rubric change is a reviewable, version-stamped diff.
|
|
153
|
+
- **No dependency on any other project's internals** — standalone; dogfooded as a black box, never coupled.
|
|
154
|
+
|
|
155
|
+
## License
|
|
156
|
+
|
|
157
|
+
**MIT** — see [LICENSE](LICENSE). Free to use, modify, and distribute; the library is a tool meant to be run against
|
|
158
|
+
your own systems.
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "spancheck"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Score a grounded-answer system on citation accuracy, abstention correctness, hallucination rate, cost and latency — with an audit log a compliance reviewer can read."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Trevor J. Romack", email = "tjromack@gmail.com" }]
|
|
14
|
+
keywords = ["rag", "evaluation", "llm", "hallucination", "citation", "abstention", "guardrails"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Programming Language :: Python :: 3.11",
|
|
18
|
+
"Programming Language :: Python :: 3.12",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Intended Audience :: Developers",
|
|
21
|
+
"Topic :: Software Development :: Quality Assurance",
|
|
22
|
+
]
|
|
23
|
+
# Standard-library first (design pin: a hard dependency is added only with a DECISIONS entry). None yet.
|
|
24
|
+
dependencies = []
|
|
25
|
+
|
|
26
|
+
[project.optional-dependencies]
|
|
27
|
+
dev = ["pytest>=8"]
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/tjromack/spancheck"
|
|
31
|
+
Source = "https://github.com/tjromack/spancheck"
|
|
32
|
+
|
|
33
|
+
[project.scripts]
|
|
34
|
+
spancheck = "spancheck.cli:main"
|
|
35
|
+
|
|
36
|
+
[tool.setuptools.packages.find]
|
|
37
|
+
where = ["src"]
|
|
38
|
+
|
|
39
|
+
[tool.setuptools.package-data]
|
|
40
|
+
spancheck = ["prompts/*.txt"]
|
|
41
|
+
|
|
42
|
+
[tool.pytest.ini_options]
|
|
43
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""spancheck — score a grounded-answer system on citation accuracy, abstention correctness,
|
|
2
|
+
hallucination rate, cost and latency, and emit an audit log a compliance reviewer can read.
|
|
3
|
+
|
|
4
|
+
Public API (Phase 1 — the ported measurement core + the thin adapter contract):
|
|
5
|
+
|
|
6
|
+
from spancheck import Case, evaluate, adapter, Run, Scorecard, gate, diff
|
|
7
|
+
|
|
8
|
+
Deterministic graders live in `spancheck.graders`. Citation-span verification (the capability this build exists to
|
|
9
|
+
add) lands in Phase 2; the versioned audit log in Phase 3; the calibrated LLM-judge in Phase 4. See TODO.md.
|
|
10
|
+
|
|
11
|
+
Nothing here hard-codes a model provider or depends on any other project's internals (CLAUDE.md design pins).
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from .adapter import Output, adapter, normalize
|
|
16
|
+
from .core import (
|
|
17
|
+
Case,
|
|
18
|
+
GradeResult,
|
|
19
|
+
CaseResult,
|
|
20
|
+
Run,
|
|
21
|
+
Scorecard,
|
|
22
|
+
evaluate,
|
|
23
|
+
run_eval,
|
|
24
|
+
diff,
|
|
25
|
+
gate,
|
|
26
|
+
default_graders,
|
|
27
|
+
)
|
|
28
|
+
from .span import citation_accuracy, verify_citation, CitationVerdict
|
|
29
|
+
from .audit import build_audit_log, cost_latency, SCHEMA_VERSION
|
|
30
|
+
from .judge import judge_entailment, judge_support, llm_judge, load_prompt
|
|
31
|
+
from .calibrate import calibrate_judge
|
|
32
|
+
from . import graders
|
|
33
|
+
|
|
34
|
+
__version__ = "0.0.1"
|
|
35
|
+
|
|
36
|
+
__all__ = [
|
|
37
|
+
# the declared public surface
|
|
38
|
+
"Case", "evaluate", "adapter", "Run", "Scorecard", "gate", "diff",
|
|
39
|
+
# citation-span verification (the capability spancheck is named for)
|
|
40
|
+
"citation_accuracy", "verify_citation", "CitationVerdict",
|
|
41
|
+
# cost/latency + the versioned audit log
|
|
42
|
+
"build_audit_log", "cost_latency", "SCHEMA_VERSION",
|
|
43
|
+
# the calibrated LLM-judge (opt-in; stub by default, no provider lock-in)
|
|
44
|
+
"judge_entailment", "judge_support", "llm_judge", "load_prompt", "calibrate_judge",
|
|
45
|
+
# supporting types + the grader library
|
|
46
|
+
"Output", "normalize", "GradeResult", "CaseResult", "run_eval", "default_graders", "graders",
|
|
47
|
+
]
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Tiny, dependency-free .env loader.
|
|
2
|
+
|
|
3
|
+
Reads KEY=VALUE lines from a .env file into os.environ **without overwriting** anything already set (a real
|
|
4
|
+
environment variable always wins). Called at CLI/calibration startup so a key placed in `.env` "just works" — no
|
|
5
|
+
python-dotenv dependency, no vendor lock-in. `.env` is gitignored; a key never enters the repo.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def load_dotenv(path: str = ".env") -> None:
|
|
13
|
+
try:
|
|
14
|
+
with open(path, encoding="utf-8") as f:
|
|
15
|
+
for line in f:
|
|
16
|
+
line = line.strip()
|
|
17
|
+
if not line or line.startswith("#") or "=" not in line:
|
|
18
|
+
continue
|
|
19
|
+
key, _, value = line.partition("=")
|
|
20
|
+
key = key.strip()
|
|
21
|
+
value = value.strip().strip('"').strip("'")
|
|
22
|
+
if key and key not in os.environ:
|
|
23
|
+
os.environ[key] = value
|
|
24
|
+
except FileNotFoundError:
|
|
25
|
+
pass
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Shared text helpers used by graders and the citation-span verifier.
|
|
2
|
+
|
|
3
|
+
Kept in one place so the deterministic checks agree on what a "word" is and how text is normalised.
|
|
4
|
+
Stdlib only.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
|
|
10
|
+
WORD = re.compile(r"[a-z0-9]+")
|
|
11
|
+
|
|
12
|
+
STOP = set("the a an and or of to in on for is are was were be been it its this that with as at by from "
|
|
13
|
+
"your you our we i they he she them his her their not no do does did have has had will would can "
|
|
14
|
+
"could should may might must shall into over under about than then so if but".split())
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def norm(s) -> str:
|
|
18
|
+
"""Lowercase and collapse all whitespace to single spaces — the normalisation used for span provenance."""
|
|
19
|
+
return re.sub(r"\s+", " ", (s or "")).strip().lower()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def words(s) -> set:
|
|
23
|
+
"""All word tokens (lowercased) — used as the haystack when checking coverage."""
|
|
24
|
+
return set(WORD.findall((s or "").lower()))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def content_words(s):
|
|
28
|
+
"""Meaningful tokens: lowercased words that aren't stopwords and are longer than two characters."""
|
|
29
|
+
return [w for w in WORD.findall((s or "").lower()) if w not in STOP and len(w) > 2]
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""The thin adapter contract — the seam between spancheck and any system under test.
|
|
2
|
+
|
|
3
|
+
Design pin #2 (no provider lock-in) and #3 (every metric computable offline from a captured run) both live here:
|
|
4
|
+
a system's answer is normalised **once** into a typed `Output` that carries everything the metrics need
|
|
5
|
+
(the answer, the retrieved contexts, the citations, token/cost usage, and latency), so scoring never has to
|
|
6
|
+
call the system again.
|
|
7
|
+
|
|
8
|
+
`spancheck` never hard-codes a vendor. A caller supplies a plain `input -> answer` callable; `adapter()` wraps it,
|
|
9
|
+
times it, and normalises whatever it returns. The return may be:
|
|
10
|
+
|
|
11
|
+
- a **str** — just the answer text;
|
|
12
|
+
- a **dict** — with any of `answer`/`text`, `contexts`/`context`, `citations`, `usage`, `latency_ms`, `abstained`;
|
|
13
|
+
- an **Output** — already normalised (idempotent).
|
|
14
|
+
|
|
15
|
+
Rewritten from evallab's loose "str-or-dict" handling into a pinned dataclass so cost/latency and citations are
|
|
16
|
+
first-class and always present (see LINEAGE.md, DECISIONS.md AB-DEC 006).
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import time
|
|
21
|
+
from dataclasses import dataclass, field, asdict
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class Output:
|
|
26
|
+
"""The normalised result of one system call — the single shape every grader reads.
|
|
27
|
+
|
|
28
|
+
`citations` is carried through as-is in Phase 1; its shape is firmed up by the citation-span verifier
|
|
29
|
+
(Phase 2). `abstained` is an explicit signal when the system provides one; when it is None the graders
|
|
30
|
+
infer abstention from the answer text.
|
|
31
|
+
"""
|
|
32
|
+
answer: str = ""
|
|
33
|
+
contexts: list = field(default_factory=list) # retrieved context passages the answer should be grounded in
|
|
34
|
+
citations: list = field(default_factory=list) # citations backing the answer (span shape fixed in Phase 2)
|
|
35
|
+
usage: dict = field(default_factory=dict) # e.g. {"input_tokens", "output_tokens", "cost_usd"}
|
|
36
|
+
latency_ms: float | None = None # wall-clock for the call; filled by adapter/evaluate if absent
|
|
37
|
+
abstained: bool | None = None # explicit abstention signal, if the system emits one
|
|
38
|
+
raw: object = None # the untouched original return, for the audit log
|
|
39
|
+
|
|
40
|
+
def to_dict(self) -> dict:
|
|
41
|
+
d = asdict(self)
|
|
42
|
+
# `raw` may not be JSON-serialisable; keep a string fallback for the audit log rather than crash.
|
|
43
|
+
try:
|
|
44
|
+
import json
|
|
45
|
+
json.dumps(d["raw"])
|
|
46
|
+
except (TypeError, ValueError):
|
|
47
|
+
d["raw"] = repr(self.raw)
|
|
48
|
+
return d
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def normalize(value) -> Output:
|
|
52
|
+
"""Coerce a system's return (str | dict | Output) into an Output. Idempotent on Output."""
|
|
53
|
+
if isinstance(value, Output):
|
|
54
|
+
return value
|
|
55
|
+
if isinstance(value, dict):
|
|
56
|
+
answer = value.get("answer", value.get("text", ""))
|
|
57
|
+
ctx = value.get("contexts", value.get("context", []))
|
|
58
|
+
if not isinstance(ctx, list):
|
|
59
|
+
ctx = [ctx] if ctx else []
|
|
60
|
+
citations = value.get("citations", [])
|
|
61
|
+
if not isinstance(citations, list):
|
|
62
|
+
citations = [citations]
|
|
63
|
+
return Output(
|
|
64
|
+
answer="" if answer is None else str(answer),
|
|
65
|
+
contexts=[str(c) for c in ctx],
|
|
66
|
+
citations=citations,
|
|
67
|
+
usage=dict(value.get("usage", {}) or {}),
|
|
68
|
+
latency_ms=value.get("latency_ms"),
|
|
69
|
+
abstained=value.get("abstained"),
|
|
70
|
+
raw=value,
|
|
71
|
+
)
|
|
72
|
+
# a bare string (or anything else) is just the answer text
|
|
73
|
+
return Output(answer="" if value is None else str(value), raw=value)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def adapter(fn, *, capture_time: bool = True):
|
|
77
|
+
"""Wrap a caller's `input -> result` callable so it returns a normalised, timed `Output`.
|
|
78
|
+
|
|
79
|
+
Using `adapter()` is optional — `evaluate()` normalises and times its system either way — but it lets a caller
|
|
80
|
+
attach usage/citations at the source and hand `evaluate` a system that already speaks the contract.
|
|
81
|
+
"""
|
|
82
|
+
def run(inp) -> Output:
|
|
83
|
+
start = time.perf_counter()
|
|
84
|
+
result = fn(inp)
|
|
85
|
+
elapsed_ms = (time.perf_counter() - start) * 1000.0
|
|
86
|
+
out = normalize(result)
|
|
87
|
+
if capture_time and out.latency_ms is None:
|
|
88
|
+
out.latency_ms = round(elapsed_ms, 3)
|
|
89
|
+
return out
|
|
90
|
+
return run
|