figured 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- figured/__init__.py +31 -0
- figured/__main__.py +41 -0
- figured/core.py +66 -0
- figured/derive.py +359 -0
- figured/evidence.py +202 -0
- figured/extract.py +149 -0
- figured/judge.py +69 -0
- figured/policy.py +60 -0
- figured/py.typed +0 -0
- figured/report.py +105 -0
- figured-0.1.0.dist-info/METADATA +211 -0
- figured-0.1.0.dist-info/RECORD +15 -0
- figured-0.1.0.dist-info/WHEEL +4 -0
- figured-0.1.0.dist-info/entry_points.txt +2 -0
- figured-0.1.0.dist-info/licenses/LICENSE +21 -0
figured/__init__.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""figured: show your work.
|
|
2
|
+
|
|
3
|
+
Verify that every number in an LLM-generated answer traces to the rows it was derived from,
|
|
4
|
+
directly or as a sum, difference, ratio, or percentage of them.
|
|
5
|
+
|
|
6
|
+
>>> from figured import trace
|
|
7
|
+
>>> report = trace("California has 39.3 million people.", [{"state": "CA", "pop": 39346023}])
|
|
8
|
+
>>> report.ok
|
|
9
|
+
True
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from figured.core import trace
|
|
13
|
+
from figured.evidence import build_evidence
|
|
14
|
+
from figured.extract import Figure, extract_numbers
|
|
15
|
+
from figured.policy import DERIVATIONS, LENIENT, STRICT, Policy
|
|
16
|
+
from figured.report import Report, Result
|
|
17
|
+
|
|
18
|
+
__version__ = "0.1.0"
|
|
19
|
+
__all__ = [
|
|
20
|
+
"DERIVATIONS",
|
|
21
|
+
"LENIENT",
|
|
22
|
+
"STRICT",
|
|
23
|
+
"Figure",
|
|
24
|
+
"Policy",
|
|
25
|
+
"Report",
|
|
26
|
+
"Result",
|
|
27
|
+
"__version__",
|
|
28
|
+
"build_evidence",
|
|
29
|
+
"extract_numbers",
|
|
30
|
+
"trace",
|
|
31
|
+
]
|
figured/__main__.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Command line: figured "answer text" --rows rows.json [--tolerance 0.015] [--json]
|
|
2
|
+
|
|
3
|
+
Exits 1 when any figure is untraceable, so it can gate a CI step or a pipeline.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import argparse
|
|
9
|
+
import json
|
|
10
|
+
import sys
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from figured.core import trace
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def main(argv: list[str] | None = None) -> int:
|
|
18
|
+
p = argparse.ArgumentParser(prog="figured", description="Check that the numbers in a text trace to rows.")
|
|
19
|
+
p.add_argument("text", help="the generated text, or '-' to read it from stdin")
|
|
20
|
+
p.add_argument(
|
|
21
|
+
"--rows", required=True, help="JSON file: a list of objects, a list of arrays, or {columns, rows}"
|
|
22
|
+
)
|
|
23
|
+
p.add_argument("--tolerance", type=float, default=None, help="relative tolerance, default 0.015")
|
|
24
|
+
p.add_argument("--strict-percent", action="store_true", help="flag percentages that match nothing")
|
|
25
|
+
p.add_argument("--json", action="store_true", help="print the full report as JSON")
|
|
26
|
+
args = p.parse_args(argv)
|
|
27
|
+
|
|
28
|
+
text = sys.stdin.read() if args.text == "-" else args.text
|
|
29
|
+
rows = json.loads(Path(args.rows).read_text())
|
|
30
|
+
overrides: dict[str, Any] = {}
|
|
31
|
+
if args.tolerance is not None:
|
|
32
|
+
overrides["rel_tolerance"] = args.tolerance
|
|
33
|
+
if args.strict_percent:
|
|
34
|
+
overrides["unmatched_percent"] = "flag"
|
|
35
|
+
report = trace(text, rows, **overrides)
|
|
36
|
+
print(json.dumps(report.to_dict(), indent=2) if args.json else report.explain())
|
|
37
|
+
return 0 if report.ok else 1
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
if __name__ == "__main__":
|
|
41
|
+
raise SystemExit(main())
|
figured/core.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""The one function: trace(text, rows) -> Report."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from figured.derive import Index
|
|
9
|
+
from figured.evidence import build_evidence
|
|
10
|
+
from figured.extract import Figure, extract_numbers
|
|
11
|
+
from figured.policy import Policy
|
|
12
|
+
from figured.report import Report, Result
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def trace(
|
|
16
|
+
text: str,
|
|
17
|
+
rows: Any = None,
|
|
18
|
+
*,
|
|
19
|
+
results: Iterable[Any] | None = None,
|
|
20
|
+
policy: Policy | None = None,
|
|
21
|
+
**overrides: Any,
|
|
22
|
+
) -> Report:
|
|
23
|
+
"""Check that every substantive number in `text` traces to `rows`.
|
|
24
|
+
|
|
25
|
+
`rows` is one result set in any common shape: a list of dicts, a list of sequences, a
|
|
26
|
+
pandas DataFrame, a DB-API cursor, or a {"columns": [...], "rows": [...]} mapping.
|
|
27
|
+
`results` is several of those, for answers written from more than one query.
|
|
28
|
+
Policy options can be passed as keywords: trace(text, rows, rel_tolerance=0.01).
|
|
29
|
+
"""
|
|
30
|
+
pol = (policy or Policy()).with_overrides(**overrides)
|
|
31
|
+
ev = build_evidence(rows, results, parse_strings=pol.parse_strings)
|
|
32
|
+
figures = extract_numbers(text)
|
|
33
|
+
if ev.empty:
|
|
34
|
+
return Report(text, [_without_evidence(f, pol) for f in figures], 0)
|
|
35
|
+
index = Index(ev, pol)
|
|
36
|
+
return Report(text, [_check(f, index, pol) for f in figures], ev.size)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _without_evidence(fig: Figure, pol: Policy) -> Result:
|
|
40
|
+
if _is_year(fig, pol):
|
|
41
|
+
return Result(fig, "ignored", reason="year")
|
|
42
|
+
if fig.is_percent:
|
|
43
|
+
return Result(fig, "ignored", reason="percent without evidence")
|
|
44
|
+
if abs(fig.value) > pol.flag_without_evidence_above:
|
|
45
|
+
return Result(fig, "ungrounded", reason="no evidence")
|
|
46
|
+
return Result(fig, "ignored", reason="small")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _check(fig: Figure, index: Index, pol: Policy) -> Result:
|
|
50
|
+
if _is_year(fig, pol):
|
|
51
|
+
return Result(fig, "ignored", reason="year")
|
|
52
|
+
if not fig.is_percent and abs(fig.value) <= pol.ignore_below:
|
|
53
|
+
return Result(fig, "ignored", reason="small")
|
|
54
|
+
match = index.lookup(fig.value, pol, is_percent=fig.is_percent)
|
|
55
|
+
if match is None and fig.is_range:
|
|
56
|
+
match = index.lookup_range(*fig.bounds, pol, is_percent=fig.is_percent)
|
|
57
|
+
if match is not None:
|
|
58
|
+
return Result(fig, "grounded", match)
|
|
59
|
+
if fig.is_percent and pol.unmatched_percent == "pass":
|
|
60
|
+
return Result(fig, "ignored", reason="unmatched percent allowed by policy")
|
|
61
|
+
return Result(fig, "ungrounded")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _is_year(fig: Figure, pol: Policy) -> bool:
|
|
65
|
+
lo, hi = pol.year_range
|
|
66
|
+
return pol.ignore_years and fig.looks_like_year and lo <= fig.value <= hi
|
figured/derive.py
ADDED
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
"""Everything the rows could legitimately produce, searched on demand rather than enumerated.
|
|
2
|
+
|
|
3
|
+
Cells, column sums, and adjacent-cell sums are indexed once. Differences, ratios, percentages,
|
|
4
|
+
and percent changes are found per figure by solving for the partner cell and bisecting for it,
|
|
5
|
+
so the cost is O(cells · log cells) per figure instead of O(cells²) up front.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from bisect import bisect_left, bisect_right
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
|
|
14
|
+
from figured.evidence import Cell, Evidence
|
|
15
|
+
from figured.policy import Policy
|
|
16
|
+
|
|
17
|
+
RANK = {
|
|
18
|
+
"cell": 0,
|
|
19
|
+
"column_sum": 1,
|
|
20
|
+
"row_sum": 2,
|
|
21
|
+
"difference": 3,
|
|
22
|
+
"ratio": 4,
|
|
23
|
+
"percent": 4,
|
|
24
|
+
"percent_change": 5,
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True, slots=True)
|
|
29
|
+
class Candidate:
|
|
30
|
+
value: float
|
|
31
|
+
kind: str
|
|
32
|
+
explanation: str
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def rank(self) -> int:
|
|
36
|
+
return RANK[self.kind]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True, slots=True)
|
|
40
|
+
class Match:
|
|
41
|
+
candidate: Candidate
|
|
42
|
+
error: float
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def kind(self) -> str:
|
|
46
|
+
return self.candidate.kind
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def explanation(self) -> str:
|
|
50
|
+
return self.candidate.explanation
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def value(self) -> float:
|
|
54
|
+
return self.candidate.value
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def fmt(v: float) -> str:
|
|
58
|
+
if abs(v) >= 1e15:
|
|
59
|
+
return f"{v:.3g}"
|
|
60
|
+
if float(v).is_integer():
|
|
61
|
+
return f"{int(v):,}"
|
|
62
|
+
if abs(v) >= 100:
|
|
63
|
+
return f"{v:,.1f}"
|
|
64
|
+
return f"{v:,.4g}"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
Scorer = Callable[[float], float | None]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class Index:
|
|
71
|
+
"""Sorted views over the evidence plus on-demand derivation search."""
|
|
72
|
+
|
|
73
|
+
def __init__(self, ev: Evidence, policy: Policy) -> None:
|
|
74
|
+
self.ev = ev
|
|
75
|
+
self.policy = policy
|
|
76
|
+
allowed = policy.derivations
|
|
77
|
+
self.allowed = allowed
|
|
78
|
+
|
|
79
|
+
sets = ev.results
|
|
80
|
+
flat_abs: list[float] = []
|
|
81
|
+
self.set_start: list[int] = []
|
|
82
|
+
for rs in sets:
|
|
83
|
+
self.set_start.append(len(flat_abs))
|
|
84
|
+
flat_abs.extend(map(abs, rs.values))
|
|
85
|
+
self.flat_abs = flat_abs
|
|
86
|
+
if "cell" in allowed and flat_abs:
|
|
87
|
+
order = sorted(range(len(flat_abs)), key=flat_abs.__getitem__)
|
|
88
|
+
self.cell_keys = [flat_abs[i] for i in order]
|
|
89
|
+
self.cell_order = order
|
|
90
|
+
else:
|
|
91
|
+
self.cell_keys = []
|
|
92
|
+
self.cell_order = []
|
|
93
|
+
|
|
94
|
+
agg: list[tuple[float, str, str]] = []
|
|
95
|
+
if "column_sum" in allowed:
|
|
96
|
+
agg.extend(self._column_sums())
|
|
97
|
+
agg.sort(key=lambda t: t[0])
|
|
98
|
+
self.agg_keys = [a for a, _, _ in agg]
|
|
99
|
+
self.agg_items = agg
|
|
100
|
+
self._row_sums: list[tuple[float, int, int, int, int]] | None = None
|
|
101
|
+
self._row_sum_keys: list[float] = []
|
|
102
|
+
|
|
103
|
+
pair_idx: list[int] = []
|
|
104
|
+
for rs, base in zip(sets, self.set_start, strict=True):
|
|
105
|
+
limit = rs.row_start[min(policy.max_rows, len(rs.row_start) - 1)]
|
|
106
|
+
pair_idx.extend(range(base, base + limit))
|
|
107
|
+
pair_idx = pair_idx[: policy.max_cells]
|
|
108
|
+
signed = self._signed
|
|
109
|
+
pair_idx.sort(key=signed)
|
|
110
|
+
self.pair_vals = [signed(i) for i in pair_idx]
|
|
111
|
+
self.pair_refs = pair_idx
|
|
112
|
+
by_abs = sorted(pair_idx, key=flat_abs.__getitem__)
|
|
113
|
+
self.pair_abs = [flat_abs[i] for i in by_abs]
|
|
114
|
+
self.pair_abs_refs = by_abs
|
|
115
|
+
|
|
116
|
+
def _signed(self, g: int) -> float:
|
|
117
|
+
rs_i = bisect_right(self.set_start, g) - 1
|
|
118
|
+
return self.ev.results[rs_i].values[g - self.set_start[rs_i]]
|
|
119
|
+
|
|
120
|
+
def cell(self, g: int) -> Cell:
|
|
121
|
+
rs_i = bisect_right(self.set_start, g) - 1
|
|
122
|
+
return self.ev.results[rs_i].cell(g - self.set_start[rs_i])
|
|
123
|
+
|
|
124
|
+
def lookup(self, value: float, policy: Policy, *, is_percent: bool = False) -> Match | None:
|
|
125
|
+
v = abs(value)
|
|
126
|
+
tol = max(v * policy.rel_tolerance, policy.abs_tolerance)
|
|
127
|
+
|
|
128
|
+
def score(d: float) -> float | None:
|
|
129
|
+
if not policy.close(v, d):
|
|
130
|
+
return None
|
|
131
|
+
return abs(v - d) / d if d else abs(v - d)
|
|
132
|
+
|
|
133
|
+
return self._search(v - tol, v + tol, v, score, is_percent)
|
|
134
|
+
|
|
135
|
+
def lookup_range(self, lo: float, hi: float, policy: Policy, *, is_percent: bool = False) -> Match | None:
|
|
136
|
+
lo_t = min(lo * (1 - policy.rel_tolerance), lo - policy.abs_tolerance)
|
|
137
|
+
hi_t = max(hi * (1 + policy.rel_tolerance), hi + policy.abs_tolerance)
|
|
138
|
+
|
|
139
|
+
def score(d: float) -> float | None:
|
|
140
|
+
if d < lo_t or d > hi_t:
|
|
141
|
+
return None
|
|
142
|
+
return 0.0 if lo <= d <= hi else min(abs(d - lo), abs(d - hi)) / max(d, 1e-12)
|
|
143
|
+
|
|
144
|
+
return self._search(lo_t, hi_t, (lo + hi) / 2, score, is_percent)
|
|
145
|
+
|
|
146
|
+
def _search(self, lo: float, hi: float, v: float, score: Scorer, is_percent: bool) -> Match | None:
|
|
147
|
+
allowed = self.allowed
|
|
148
|
+
if self.cell_keys:
|
|
149
|
+
m = self._best_sorted(self.cell_keys, lo, hi, score, self._cell_candidate)
|
|
150
|
+
if m:
|
|
151
|
+
return m
|
|
152
|
+
if self.agg_keys:
|
|
153
|
+
m = self._best_sorted(self.agg_keys, lo, hi, score, self._agg_candidate)
|
|
154
|
+
if m:
|
|
155
|
+
return m
|
|
156
|
+
if "row_sum" in allowed:
|
|
157
|
+
sums = self._row_sum_index()
|
|
158
|
+
if sums:
|
|
159
|
+
m = self._best_sorted(self._row_sum_keys, lo, hi, score, self._row_candidate)
|
|
160
|
+
if m:
|
|
161
|
+
return m
|
|
162
|
+
if not self.pair_vals:
|
|
163
|
+
return None
|
|
164
|
+
if "difference" in allowed:
|
|
165
|
+
m = self._differences(lo, hi, score)
|
|
166
|
+
if m:
|
|
167
|
+
return m
|
|
168
|
+
kind = "percent" if is_percent else "ratio"
|
|
169
|
+
if kind in allowed:
|
|
170
|
+
m = self._ratios(lo, hi, kind, score)
|
|
171
|
+
if m:
|
|
172
|
+
return m
|
|
173
|
+
if "percent_change" in allowed:
|
|
174
|
+
m = self._percent_changes(lo, hi, score)
|
|
175
|
+
if m:
|
|
176
|
+
return m
|
|
177
|
+
return None
|
|
178
|
+
|
|
179
|
+
@staticmethod
|
|
180
|
+
def _best_sorted(
|
|
181
|
+
keys: list[float], lo: float, hi: float, score: Scorer, make: Callable[[int], Candidate]
|
|
182
|
+
) -> Match | None:
|
|
183
|
+
best: tuple[float, int] | None = None
|
|
184
|
+
for i in range(bisect_left(keys, lo), bisect_right(keys, hi)):
|
|
185
|
+
err = score(keys[i])
|
|
186
|
+
if err is not None and (best is None or err < best[0]):
|
|
187
|
+
best = (err, i)
|
|
188
|
+
return Match(make(best[1]), best[0]) if best else None
|
|
189
|
+
|
|
190
|
+
def _cell_candidate(self, i: int) -> Candidate:
|
|
191
|
+
c = self.cell(self.cell_order[i])
|
|
192
|
+
return Candidate(c.value, "cell", f"{c.ref()} = {fmt(c.value)}")
|
|
193
|
+
|
|
194
|
+
def _agg_candidate(self, i: int) -> Candidate:
|
|
195
|
+
a, kind, text = self.agg_items[i]
|
|
196
|
+
return Candidate(a, kind, text)
|
|
197
|
+
|
|
198
|
+
def _row_candidate(self, i: int) -> Candidate:
|
|
199
|
+
assert self._row_sums is not None
|
|
200
|
+
_, rs_index, r, a, b = self._row_sums[i]
|
|
201
|
+
rs = self.ev.results[rs_index]
|
|
202
|
+
s = rs.row_start[r]
|
|
203
|
+
signed = sum(rs.values[s + a : s + b + 1])
|
|
204
|
+
head, tail = rs.columns[rs.col_of[s + a]], rs.columns[rs.col_of[s + b]]
|
|
205
|
+
return Candidate(signed, "row_sum", f"{head}..{tail}[{rs.labels[r]}] summed = {fmt(signed)}")
|
|
206
|
+
|
|
207
|
+
def _column_sums(self) -> list[tuple[float, str, str]]:
|
|
208
|
+
out: list[tuple[float, str, str]] = []
|
|
209
|
+
for rs in self.ev.results:
|
|
210
|
+
for c, (total, n) in enumerate(zip(rs.col_totals, rs.col_counts, strict=True)):
|
|
211
|
+
if n >= 2:
|
|
212
|
+
out.append(
|
|
213
|
+
(abs(total), "column_sum", f"sum of {rs.columns[c]} over {n} rows = {fmt(total)}")
|
|
214
|
+
)
|
|
215
|
+
return out
|
|
216
|
+
|
|
217
|
+
def _row_sum_index(self) -> list[tuple[float, int, int, int, int]]:
|
|
218
|
+
"""Adjacent-cell sums (2 to 6 cells) over the first max_rows rows, formatted only on a match."""
|
|
219
|
+
if self._row_sums is None:
|
|
220
|
+
out: list[tuple[float, int, int, int, int]] = []
|
|
221
|
+
for rs in self.ev.results:
|
|
222
|
+
vals, starts, idx = rs.values, rs.row_start, rs.index
|
|
223
|
+
for r in range(min(self.policy.max_rows, len(starts) - 1)):
|
|
224
|
+
s, e = starts[r], starts[r + 1]
|
|
225
|
+
n = e - s
|
|
226
|
+
for i in range(n - 1):
|
|
227
|
+
total = vals[s + i]
|
|
228
|
+
for j in range(i + 1, min(n, i + 6)):
|
|
229
|
+
total += vals[s + j]
|
|
230
|
+
out.append((abs(total), idx, r, i, j))
|
|
231
|
+
out.sort(key=lambda t: t[0])
|
|
232
|
+
self._row_sums = out
|
|
233
|
+
self._row_sum_keys = [t[0] for t in out]
|
|
234
|
+
return self._row_sums
|
|
235
|
+
|
|
236
|
+
def _differences(self, lo: float, hi: float, score: Scorer) -> Match | None:
|
|
237
|
+
"""Pairs with |a − b| in [lo, hi]. Both windows slide right as b grows, so two pointers suffice."""
|
|
238
|
+
vals, n = self.pair_vals, len(self.pair_vals)
|
|
239
|
+
best: tuple[float, int, int] | None = None
|
|
240
|
+
s1 = e1 = s2 = e2 = 0
|
|
241
|
+
for j in range(n):
|
|
242
|
+
b = vals[j]
|
|
243
|
+
a_lo, a_hi = b + lo, b + hi
|
|
244
|
+
while s1 < n and vals[s1] < a_lo:
|
|
245
|
+
s1 += 1
|
|
246
|
+
while e1 < n and vals[e1] <= a_hi:
|
|
247
|
+
e1 += 1
|
|
248
|
+
for i in range(s1, e1):
|
|
249
|
+
if i != j:
|
|
250
|
+
err = score(vals[i] - b)
|
|
251
|
+
if err is not None and (best is None or err < best[0]):
|
|
252
|
+
best = (err, i, j)
|
|
253
|
+
a_lo, a_hi = b - hi, b - lo
|
|
254
|
+
while s2 < n and vals[s2] < a_lo:
|
|
255
|
+
s2 += 1
|
|
256
|
+
while e2 < n and vals[e2] <= a_hi:
|
|
257
|
+
e2 += 1
|
|
258
|
+
for i in range(s2, e2):
|
|
259
|
+
if i != j:
|
|
260
|
+
err = score(b - vals[i])
|
|
261
|
+
if err is not None and (best is None or err < best[0]):
|
|
262
|
+
best = (err, i, j)
|
|
263
|
+
if best is None:
|
|
264
|
+
return None
|
|
265
|
+
_, i, j = best
|
|
266
|
+
ca, cb = self.cell(self.pair_refs[i]), self.cell(self.pair_refs[j])
|
|
267
|
+
d = ca.value - cb.value
|
|
268
|
+
return Match(
|
|
269
|
+
Candidate(
|
|
270
|
+
d, "difference", f"{ca.ref()} − {cb.ref()} = {fmt(ca.value)} − {fmt(cb.value)} = {fmt(d)}"
|
|
271
|
+
),
|
|
272
|
+
best[0],
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
def _ratios(self, lo: float, hi: float, kind: str, score: Scorer) -> Match | None:
|
|
276
|
+
"""A percent figure is searched as a ÷ b × 100, a plain figure as a ÷ b."""
|
|
277
|
+
scale = 100.0 if kind == "percent" else 1.0
|
|
278
|
+
vals, n = self.pair_abs, len(self.pair_abs)
|
|
279
|
+
best: tuple[float, int, int, str, float] | None = None
|
|
280
|
+
r_lo, r_hi = lo / scale, hi / scale
|
|
281
|
+
if r_hi <= 0:
|
|
282
|
+
return None
|
|
283
|
+
r_lo = max(r_lo, 1e-300)
|
|
284
|
+
s1 = e1 = s2 = e2 = 0
|
|
285
|
+
for j in range(n):
|
|
286
|
+
b = vals[j]
|
|
287
|
+
if b == 0:
|
|
288
|
+
continue
|
|
289
|
+
a_lo, a_hi = b * r_lo, b * r_hi
|
|
290
|
+
while s1 < n and vals[s1] < a_lo:
|
|
291
|
+
s1 += 1
|
|
292
|
+
while e1 < n and vals[e1] <= a_hi:
|
|
293
|
+
e1 += 1
|
|
294
|
+
for i in range(s1, e1):
|
|
295
|
+
if i != j:
|
|
296
|
+
err = score(vals[i] / b * scale)
|
|
297
|
+
if err is not None and (best is None or err < best[0]):
|
|
298
|
+
best = (err, i, j, kind, scale)
|
|
299
|
+
a_lo, a_hi = b / r_hi, b / r_lo
|
|
300
|
+
while s2 < n and vals[s2] < a_lo:
|
|
301
|
+
s2 += 1
|
|
302
|
+
while e2 < n and vals[e2] <= a_hi:
|
|
303
|
+
e2 += 1
|
|
304
|
+
for i in range(s2, e2):
|
|
305
|
+
if i != j and vals[i]:
|
|
306
|
+
err = score(b / vals[i] * scale)
|
|
307
|
+
if err is not None and (best is None or err < best[0]):
|
|
308
|
+
best = (err, j, i, kind, scale)
|
|
309
|
+
if best is None:
|
|
310
|
+
return None
|
|
311
|
+
err, num_i, den_i, kind, scale = best
|
|
312
|
+
ca, cb = self.cell(self.pair_abs_refs[num_i]), self.cell(self.pair_abs_refs[den_i])
|
|
313
|
+
value = abs(ca.value / cb.value) * scale
|
|
314
|
+
text = f"{ca.ref()} ÷ {cb.ref()} = {fmt(value)}" + ("%" if scale == 100.0 else "")
|
|
315
|
+
return Match(Candidate(value, kind, text), err)
|
|
316
|
+
|
|
317
|
+
def _percent_changes(self, lo: float, hi: float, score: Scorer) -> Match | None:
|
|
318
|
+
"""(a − b) ÷ b × 100 in ±[lo, hi]. Positive b slides monotonically; negatives fall back to bisect."""
|
|
319
|
+
vals, n = self.pair_vals, len(self.pair_vals)
|
|
320
|
+
f_lo, f_hi = lo / 100, hi / 100
|
|
321
|
+
best: tuple[float, int, int] | None = None
|
|
322
|
+
s1 = e1 = s2 = e2 = 0
|
|
323
|
+
for j in range(n):
|
|
324
|
+
b = vals[j]
|
|
325
|
+
if b == 0:
|
|
326
|
+
continue
|
|
327
|
+
if b > 0:
|
|
328
|
+
a_lo, a_hi = b * (1 + f_lo), b * (1 + f_hi)
|
|
329
|
+
while s1 < n and vals[s1] < a_lo:
|
|
330
|
+
s1 += 1
|
|
331
|
+
while e1 < n and vals[e1] <= a_hi:
|
|
332
|
+
e1 += 1
|
|
333
|
+
r1 = range(s1, e1)
|
|
334
|
+
a_lo, a_hi = b * (1 - f_hi), b * (1 - f_lo)
|
|
335
|
+
while s2 < n and vals[s2] < a_lo:
|
|
336
|
+
s2 += 1
|
|
337
|
+
while e2 < n and vals[e2] <= a_hi:
|
|
338
|
+
e2 += 1
|
|
339
|
+
r2 = range(s2, e2)
|
|
340
|
+
else:
|
|
341
|
+
w1 = sorted((b * (1 + f_lo), b * (1 + f_hi)))
|
|
342
|
+
w2 = sorted((b * (1 - f_hi), b * (1 - f_lo)))
|
|
343
|
+
r1 = range(bisect_left(vals, w1[0]), bisect_right(vals, w1[1]))
|
|
344
|
+
r2 = range(bisect_left(vals, w2[0]), bisect_right(vals, w2[1]))
|
|
345
|
+
for rng in (r1, r2):
|
|
346
|
+
for i in rng:
|
|
347
|
+
if i != j:
|
|
348
|
+
err = score(abs((vals[i] - b) / b * 100))
|
|
349
|
+
if err is not None and (best is None or err < best[0]):
|
|
350
|
+
best = (err, i, j)
|
|
351
|
+
if best is None:
|
|
352
|
+
return None
|
|
353
|
+
_, i, j = best
|
|
354
|
+
ca, cb = self.cell(self.pair_refs[i]), self.cell(self.pair_refs[j])
|
|
355
|
+
pc = (ca.value - cb.value) / cb.value * 100
|
|
356
|
+
return Match(
|
|
357
|
+
Candidate(pc, "percent_change", f"({ca.ref()} − {cb.ref()}) ÷ {cb.ref()} = {fmt(pc)}%"),
|
|
358
|
+
best[0],
|
|
359
|
+
)
|
figured/evidence.py
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""Normalize whatever the caller has (rows, dicts, a DataFrame, a cursor) into numeric cells.
|
|
2
|
+
|
|
3
|
+
Storage is columnar and flat: one list of values across all result sets, with offsets. Cell
|
|
4
|
+
objects are created only for the handful of cells that end up in an explanation.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import math
|
|
10
|
+
import re
|
|
11
|
+
from bisect import bisect_right
|
|
12
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from decimal import Decimal
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
_NUMERIC_STRING = re.compile(r"^[-+]?[$€£¥]?\s?(?:\d{1,3}(?:,\d{3})+|\d+)(?:\.\d+)?(?:[eE][-+]?\d+)?%?$")
|
|
18
|
+
_MONEY = str.maketrans("", "", ",$€£¥")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True, slots=True)
|
|
22
|
+
class Cell:
|
|
23
|
+
value: float
|
|
24
|
+
column: str
|
|
25
|
+
row: int
|
|
26
|
+
result: int
|
|
27
|
+
label: str
|
|
28
|
+
|
|
29
|
+
def ref(self) -> str:
|
|
30
|
+
return f"{self.column}[{self.label}]"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(slots=True)
|
|
34
|
+
class ResultSet:
|
|
35
|
+
columns: list[str]
|
|
36
|
+
index: int
|
|
37
|
+
values: list[float] = field(default_factory=list)
|
|
38
|
+
col_of: list[int] = field(default_factory=list)
|
|
39
|
+
row_start: list[int] = field(default_factory=list)
|
|
40
|
+
labels: list[str] = field(default_factory=list)
|
|
41
|
+
col_totals: list[float] = field(default_factory=list)
|
|
42
|
+
col_counts: list[int] = field(default_factory=list)
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def rows(self) -> list[list[Cell]]:
|
|
46
|
+
return [
|
|
47
|
+
[self.cell(k) for k in range(self.row_start[r], self.row_start[r + 1])]
|
|
48
|
+
for r in range(len(self.labels))
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
def row_of(self, k: int) -> int:
|
|
52
|
+
return bisect_right(self.row_start, k) - 1
|
|
53
|
+
|
|
54
|
+
def cell(self, k: int) -> Cell:
|
|
55
|
+
r = self.row_of(k)
|
|
56
|
+
return Cell(self.values[k], self.columns[self.col_of[k]], r, self.index, self.labels[r])
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(slots=True)
|
|
60
|
+
class Evidence:
|
|
61
|
+
results: list[ResultSet] = field(default_factory=list)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def cells(self) -> list[Cell]:
|
|
65
|
+
return [rs.cell(k) for rs in self.results for k in range(len(rs.values))]
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def size(self) -> int:
|
|
69
|
+
return sum(len(rs.values) for rs in self.results)
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def empty(self) -> bool:
|
|
73
|
+
return self.size == 0
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def build_evidence(
|
|
77
|
+
rows: Any = None,
|
|
78
|
+
results: Iterable[Any] | None = None,
|
|
79
|
+
*,
|
|
80
|
+
parse_strings: bool = True,
|
|
81
|
+
) -> Evidence:
|
|
82
|
+
"""Accept one result set (`rows`) and/or several (`results`) in any common shape."""
|
|
83
|
+
sets: list[Any] = []
|
|
84
|
+
if rows is not None:
|
|
85
|
+
sets.append(rows)
|
|
86
|
+
if results is not None:
|
|
87
|
+
sets.extend(results)
|
|
88
|
+
ev = Evidence()
|
|
89
|
+
for i, obj in enumerate(sets):
|
|
90
|
+
columns, raw_rows = _to_table(obj)
|
|
91
|
+
ev.results.append(_result_set(columns, raw_rows, i, parse_strings))
|
|
92
|
+
return ev
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _to_table(obj: Any) -> tuple[list[str], list[Sequence[Any]]]:
|
|
96
|
+
if obj is None:
|
|
97
|
+
return [], []
|
|
98
|
+
if isinstance(obj, Mapping) and "rows" in obj:
|
|
99
|
+
cols = [str(c) for c in obj.get("columns") or []]
|
|
100
|
+
return cols, [_dict_row(r, cols) if isinstance(r, Mapping) else r for r in obj["rows"]]
|
|
101
|
+
if hasattr(obj, "to_dict") and hasattr(obj, "columns"):
|
|
102
|
+
cols = [str(c) for c in obj.columns]
|
|
103
|
+
return cols, [[rec.get(c) for c in cols] for rec in obj.to_dict("records")]
|
|
104
|
+
if hasattr(obj, "fetchall") and hasattr(obj, "description"):
|
|
105
|
+
return [str(d[0]) for d in obj.description or []], list(obj.fetchall())
|
|
106
|
+
rows = obj if isinstance(obj, list) else list(obj)
|
|
107
|
+
if not rows:
|
|
108
|
+
return [], []
|
|
109
|
+
first = rows[0]
|
|
110
|
+
if isinstance(first, Mapping):
|
|
111
|
+
keys: list[Any] = list(first.keys())
|
|
112
|
+
seen = set(keys)
|
|
113
|
+
for r in rows:
|
|
114
|
+
if len(r) != len(keys) or r.keys() != seen:
|
|
115
|
+
for k in r:
|
|
116
|
+
if k not in seen:
|
|
117
|
+
seen.add(k)
|
|
118
|
+
keys.append(k)
|
|
119
|
+
return [str(k) for k in keys], [[r.get(c) for c in keys] for r in rows]
|
|
120
|
+
if isinstance(first, str | bytes | int | float | Decimal):
|
|
121
|
+
return [], [[r] for r in rows]
|
|
122
|
+
return [], rows
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _dict_row(r: Mapping[str, Any], cols: list[str]) -> list[Any]:
|
|
126
|
+
return [r.get(c) for c in cols] if cols else list(r.values())
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _result_set(
|
|
130
|
+
columns: list[str], raw_rows: list[Sequence[Any]], index: int, parse_strings: bool
|
|
131
|
+
) -> ResultSet:
|
|
132
|
+
width = max((len(r) for r in raw_rows), default=0)
|
|
133
|
+
if len(columns) < width:
|
|
134
|
+
columns = columns + [f"c{i}" for i in range(len(columns), width)]
|
|
135
|
+
rs = ResultSet(columns, index)
|
|
136
|
+
values, col_of, row_start, labels = rs.values, rs.col_of, rs.row_start, rs.labels
|
|
137
|
+
totals, counts = [0.0] * width, [0] * width
|
|
138
|
+
push_v, push_c, push_label, push_start = values.append, col_of.append, labels.append, row_start.append
|
|
139
|
+
isfinite = math.isfinite
|
|
140
|
+
for raw in raw_rows:
|
|
141
|
+
push_start(len(values))
|
|
142
|
+
label = ""
|
|
143
|
+
for ci, v in enumerate(raw):
|
|
144
|
+
t = type(v)
|
|
145
|
+
if t is float:
|
|
146
|
+
if not isfinite(v):
|
|
147
|
+
continue
|
|
148
|
+
elif t is int:
|
|
149
|
+
v = float(v)
|
|
150
|
+
elif t is str:
|
|
151
|
+
s = v.strip()
|
|
152
|
+
if _NUMERIC_STRING.match(s):
|
|
153
|
+
if not parse_strings:
|
|
154
|
+
continue
|
|
155
|
+
num = _parse_numeric_string(s)
|
|
156
|
+
if num is None:
|
|
157
|
+
continue
|
|
158
|
+
v = num
|
|
159
|
+
else:
|
|
160
|
+
if not label and s:
|
|
161
|
+
label = s[:40]
|
|
162
|
+
continue
|
|
163
|
+
else:
|
|
164
|
+
num = to_number(v, parse_strings)
|
|
165
|
+
if num is None:
|
|
166
|
+
continue
|
|
167
|
+
v = num
|
|
168
|
+
push_v(v)
|
|
169
|
+
push_c(ci)
|
|
170
|
+
totals[ci] += v
|
|
171
|
+
counts[ci] += 1
|
|
172
|
+
push_label(label or f"row {len(labels)}")
|
|
173
|
+
push_start(len(values))
|
|
174
|
+
rs.col_totals, rs.col_counts = totals, counts
|
|
175
|
+
return rs
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _parse_numeric_string(s: str) -> float | None:
|
|
179
|
+
try:
|
|
180
|
+
return float(s.rstrip("%").translate(_MONEY))
|
|
181
|
+
except ValueError:
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def to_number(v: Any, parse_strings: bool = True) -> float | None:
|
|
186
|
+
if v is None or isinstance(v, bool):
|
|
187
|
+
return None
|
|
188
|
+
if isinstance(v, int | float):
|
|
189
|
+
f = float(v)
|
|
190
|
+
return f if math.isfinite(f) else None
|
|
191
|
+
if isinstance(v, Decimal):
|
|
192
|
+
return float(v) if v.is_finite() else None
|
|
193
|
+
if isinstance(v, str):
|
|
194
|
+
s = v.strip()
|
|
195
|
+
return _parse_numeric_string(s) if parse_strings and _NUMERIC_STRING.match(s) else None
|
|
196
|
+
if hasattr(v, "__float__"):
|
|
197
|
+
try:
|
|
198
|
+
f = float(v)
|
|
199
|
+
except (TypeError, ValueError):
|
|
200
|
+
return None
|
|
201
|
+
return f if math.isfinite(f) else None
|
|
202
|
+
return None
|
figured/extract.py
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Find the numbers in a piece of text, with their positions and meaning."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
SCALES: dict[str, float] = {
|
|
9
|
+
"thousand": 1e3,
|
|
10
|
+
"k": 1e3,
|
|
11
|
+
"million": 1e6,
|
|
12
|
+
"mn": 1e6,
|
|
13
|
+
"m": 1e6,
|
|
14
|
+
"mm": 1e6,
|
|
15
|
+
"billion": 1e9,
|
|
16
|
+
"bn": 1e9,
|
|
17
|
+
"b": 1e9,
|
|
18
|
+
"trillion": 1e12,
|
|
19
|
+
"tn": 1e12,
|
|
20
|
+
"t": 1e12,
|
|
21
|
+
}
|
|
22
|
+
PERCENT_WORDS = frozenset({"%", "percent", "pct", "percentage points", "pp"})
|
|
23
|
+
|
|
24
|
+
_NUMBER = re.compile(
|
|
25
|
+
r"""
|
|
26
|
+
(?<![\w.])
|
|
27
|
+
(?P<sign>[-−]\s?)?
|
|
28
|
+
(?P<currency>[$€£¥])?
|
|
29
|
+
(?P<body>
|
|
30
|
+
\d+(?:\.\d+)?[eE][+-]?\d+
|
|
31
|
+
| \d{1,3}(?:,\d{3})+(?:\.\d+)?
|
|
32
|
+
| \d+(?:\.\d+)?
|
|
33
|
+
)
|
|
34
|
+
(?!\d)
|
|
35
|
+
(?!(?:st|nd|rd|th)\b)
|
|
36
|
+
(?:
|
|
37
|
+
\s?(?P<pct>%)
|
|
38
|
+
| \s?(?P<word>percentage\ points|percent|pct|pp|thousand|million|billion|trillion|mn|mm|bn|tn|k|m|b|t)\b
|
|
39
|
+
)?
|
|
40
|
+
(?![A-Za-z_])
|
|
41
|
+
""",
|
|
42
|
+
re.VERBOSE | re.IGNORECASE,
|
|
43
|
+
)
|
|
44
|
+
_START = re.compile(r"(?:[-−]\s?)?[$€£¥]?\d")
|
|
45
|
+
_RANGE_JOIN = re.compile(r"^\s*(?:to|and|or|-|–|—)\s*$", re.IGNORECASE)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True)
|
|
49
|
+
class Figure:
|
|
50
|
+
"""One number found in the text."""
|
|
51
|
+
|
|
52
|
+
literal: str
|
|
53
|
+
value: float
|
|
54
|
+
start: int
|
|
55
|
+
end: int
|
|
56
|
+
is_percent: bool = False
|
|
57
|
+
is_currency: bool = False
|
|
58
|
+
has_scale: bool = False
|
|
59
|
+
range_partner: float | None = None
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def is_range(self) -> bool:
|
|
63
|
+
return self.range_partner is not None
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def bounds(self) -> tuple[float, float]:
|
|
67
|
+
other = self.range_partner if self.range_partner is not None else self.value
|
|
68
|
+
lo, hi = sorted((abs(self.value), abs(other)))
|
|
69
|
+
return lo, hi
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def looks_like_year(self) -> bool:
|
|
73
|
+
body = self.literal.lstrip("-−$€£¥ ").strip()
|
|
74
|
+
return body.isdigit() and len(body) == 4 and not self.is_percent and not self.has_scale
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def extract_numbers(text: str) -> list[Figure]:
|
|
78
|
+
"""Return every number-like token in reading order.
|
|
79
|
+
|
|
80
|
+
Handles thousands separators, decimals, scientific notation, currency symbols, scale
|
|
81
|
+
words (thousand, million, k, bn, ...), percent markers, and ranges such as
|
|
82
|
+
"40 to 50 million", where the scale of the second number is applied to the first.
|
|
83
|
+
"""
|
|
84
|
+
out: list[Figure] = []
|
|
85
|
+
pos = 0
|
|
86
|
+
match = _NUMBER.match
|
|
87
|
+
for start in _START.finditer(text):
|
|
88
|
+
at = start.start()
|
|
89
|
+
if at < pos:
|
|
90
|
+
continue
|
|
91
|
+
m = match(text, at) or (match(text, start.end() - 1) if start.end() - 1 > at else None)
|
|
92
|
+
if m is None:
|
|
93
|
+
continue
|
|
94
|
+
pos = m.end()
|
|
95
|
+
body = m.group("body").replace(",", "")
|
|
96
|
+
value = float(body)
|
|
97
|
+
word = (m.group("word") or "").lower()
|
|
98
|
+
pct = bool(m.group("pct")) or word in PERCENT_WORDS
|
|
99
|
+
scale = SCALES.get(word)
|
|
100
|
+
if scale:
|
|
101
|
+
value *= scale
|
|
102
|
+
if m.group("sign"):
|
|
103
|
+
value = -value
|
|
104
|
+
out.append(
|
|
105
|
+
Figure(
|
|
106
|
+
literal=m.group(0).strip(),
|
|
107
|
+
value=value,
|
|
108
|
+
start=m.start(),
|
|
109
|
+
end=m.end(),
|
|
110
|
+
is_percent=pct,
|
|
111
|
+
is_currency=bool(m.group("currency")),
|
|
112
|
+
has_scale=scale is not None,
|
|
113
|
+
)
|
|
114
|
+
)
|
|
115
|
+
return _apply_ranges(text, out)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _apply_ranges(text: str, figures: list[Figure]) -> list[Figure]:
|
|
119
|
+
if len(figures) < 2:
|
|
120
|
+
return figures
|
|
121
|
+
fixed = list(figures)
|
|
122
|
+
for i in range(len(fixed) - 1):
|
|
123
|
+
a, b = fixed[i], fixed[i + 1]
|
|
124
|
+
if not _RANGE_JOIN.match(text[a.end : b.start]):
|
|
125
|
+
continue
|
|
126
|
+
if b.has_scale and not a.has_scale and not a.is_percent:
|
|
127
|
+
a = Figure(a.literal, a.value * _scale_of(b), a.start, a.end, False, a.is_currency, True)
|
|
128
|
+
elif b.is_percent and not a.is_percent and not a.has_scale:
|
|
129
|
+
a = Figure(a.literal, a.value, a.start, a.end, True, a.is_currency, False)
|
|
130
|
+
if a.is_percent == b.is_percent and a.looks_like_year == b.looks_like_year and not a.looks_like_year:
|
|
131
|
+
fixed[i] = Figure(
|
|
132
|
+
a.literal, a.value, a.start, a.end, a.is_percent, a.is_currency, a.has_scale, b.value
|
|
133
|
+
)
|
|
134
|
+
fixed[i + 1] = Figure(
|
|
135
|
+
b.literal, b.value, b.start, b.end, b.is_percent, b.is_currency, b.has_scale, a.value
|
|
136
|
+
)
|
|
137
|
+
else:
|
|
138
|
+
fixed[i] = a
|
|
139
|
+
return fixed
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _scale_of(fig: Figure) -> float:
|
|
143
|
+
word = fig.literal.split()[-1].lower() if " " in fig.literal else _trailing_word(fig.literal)
|
|
144
|
+
return SCALES.get(word, 1.0)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _trailing_word(literal: str) -> str:
|
|
148
|
+
m = re.search(r"[a-z]+$", literal, re.IGNORECASE)
|
|
149
|
+
return m.group(0).lower() if m else ""
|
figured/judge.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Optional second opinion from a model, for what arithmetic cannot see: wrong words around right numbers.
|
|
2
|
+
|
|
3
|
+
Requires the `judge` extra: pip install "figured[judge]". The judge receives the question, the
|
|
4
|
+
rows, and the answer, nothing else, and returns a strict verdict.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
from typing import Any, Literal
|
|
11
|
+
|
|
12
|
+
JUDGE_SYSTEM = """You audit answers produced by a data assistant. You are given the user's question, the rows the assistant retrieved, and the assistant's answer. Judge strictly and only from what is shown.
|
|
13
|
+
|
|
14
|
+
faithful: every number in the answer is supported by the rows, allowing rounding and values derived from them (sums, differences, ratios, percentages), and every comparative word (higher, lower, most, fewer) agrees with the rows.
|
|
15
|
+
responsive: the answer addresses the question that was asked, or clearly explains what part cannot be answered.
|
|
16
|
+
caveats_ok: if the answer relies on an approximation or omits part of the question, it says so. True when no caveat was needed.
|
|
17
|
+
issues: short, specific strings; empty when none.
|
|
18
|
+
verdict: "pass" only if faithful and responsive are both true."""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def judge(
|
|
22
|
+
question: str,
|
|
23
|
+
answer: str,
|
|
24
|
+
rows: Any,
|
|
25
|
+
*,
|
|
26
|
+
model: str = "claude-opus-5",
|
|
27
|
+
client: Any = None,
|
|
28
|
+
max_rows: int = 30,
|
|
29
|
+
) -> dict[str, Any]:
|
|
30
|
+
"""Return {"verdict", "faithful", "responsive", "caveats_ok", "issues", "model"}."""
|
|
31
|
+
try:
|
|
32
|
+
import anthropic
|
|
33
|
+
from pydantic import BaseModel
|
|
34
|
+
except ImportError as exc: # pragma: no cover
|
|
35
|
+
raise ImportError('the judge needs the extra: pip install "figured[judge]"') from exc
|
|
36
|
+
|
|
37
|
+
class Verdict(BaseModel): # type: ignore[misc]
|
|
38
|
+
faithful: bool
|
|
39
|
+
responsive: bool
|
|
40
|
+
caveats_ok: bool
|
|
41
|
+
issues: list[str]
|
|
42
|
+
verdict: Literal["pass", "fail"]
|
|
43
|
+
|
|
44
|
+
from figured.evidence import build_evidence
|
|
45
|
+
|
|
46
|
+
ev = build_evidence(rows)
|
|
47
|
+
evidence = [
|
|
48
|
+
{
|
|
49
|
+
"columns": rs.columns,
|
|
50
|
+
"rows": [[c.value for c in row] for row in rs.rows[:max_rows]],
|
|
51
|
+
}
|
|
52
|
+
for rs in ev.results
|
|
53
|
+
]
|
|
54
|
+
prompt = (
|
|
55
|
+
f"Question:\n{question}\n\nRetrieved rows (JSON):\n{json.dumps(evidence)[:12000]}"
|
|
56
|
+
f"\n\nAssistant answer:\n{answer}"
|
|
57
|
+
)
|
|
58
|
+
client = client or anthropic.Anthropic()
|
|
59
|
+
resp = client.messages.parse(
|
|
60
|
+
model=model,
|
|
61
|
+
max_tokens=600,
|
|
62
|
+
system=JUDGE_SYSTEM,
|
|
63
|
+
messages=[{"role": "user", "content": prompt}],
|
|
64
|
+
output_format=Verdict,
|
|
65
|
+
)
|
|
66
|
+
parsed = resp.parsed_output
|
|
67
|
+
if parsed is None:
|
|
68
|
+
return {"verdict": "error", "issues": ["no parsed output"], "model": model}
|
|
69
|
+
return {**parsed.model_dump(), "model": model}
|
figured/policy.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Tunable rules for what counts as grounded."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, replace
|
|
6
|
+
from typing import Any, Literal
|
|
7
|
+
|
|
8
|
+
DERIVATIONS = ("cell", "column_sum", "row_sum", "difference", "ratio", "percent", "percent_change")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True)
|
|
12
|
+
class Policy:
|
|
13
|
+
"""How strictly to match, what to ignore, and which derivations to allow.
|
|
14
|
+
|
|
15
|
+
rel_tolerance: relative error allowed between a figure and a candidate (0.015 is 1.5%).
|
|
16
|
+
abs_tolerance: absolute error allowed in addition to the relative one.
|
|
17
|
+
ignore_below: figures at or below this absolute value are not checked (counts of items, rankings).
|
|
18
|
+
ignore_years: treat bare four-digit integers between year_range as years and skip them.
|
|
19
|
+
unmatched_percent: "pass" lets a percentage through when nothing matches, since shares of a
|
|
20
|
+
total outside the rows are common; "flag" treats it like any other figure.
|
|
21
|
+
max_rows / max_cells: how many rows and flat cells feed the pairwise derivations.
|
|
22
|
+
derivations: which candidate kinds are generated.
|
|
23
|
+
parse_strings: coerce numeric strings in the rows ("39,346,023", "$1,200", "12%").
|
|
24
|
+
flag_without_evidence_above: with no rows at all, figures above this are flagged.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
rel_tolerance: float = 0.015
|
|
28
|
+
abs_tolerance: float = 0.0
|
|
29
|
+
ignore_below: float = 100.0
|
|
30
|
+
ignore_years: bool = True
|
|
31
|
+
year_range: tuple[int, int] = (1900, 2100)
|
|
32
|
+
unmatched_percent: Literal["pass", "flag"] = "pass"
|
|
33
|
+
max_rows: int = 12
|
|
34
|
+
max_cells: int = 40
|
|
35
|
+
derivations: frozenset[str] = frozenset(DERIVATIONS)
|
|
36
|
+
parse_strings: bool = True
|
|
37
|
+
flag_without_evidence_above: float = 1000.0
|
|
38
|
+
|
|
39
|
+
def with_overrides(self, **overrides: Any) -> Policy:
|
|
40
|
+
if not overrides:
|
|
41
|
+
return self
|
|
42
|
+
if "derivations" in overrides and not isinstance(overrides["derivations"], frozenset):
|
|
43
|
+
overrides["derivations"] = frozenset(overrides["derivations"])
|
|
44
|
+
unknown = set(overrides) - set(self.__dataclass_fields__)
|
|
45
|
+
if unknown:
|
|
46
|
+
raise TypeError(f"unknown policy option(s): {', '.join(sorted(unknown))}")
|
|
47
|
+
return replace(self, **overrides)
|
|
48
|
+
|
|
49
|
+
def close(self, a: float, b: float) -> bool:
|
|
50
|
+
if a == b:
|
|
51
|
+
return True
|
|
52
|
+
if abs(a - b) <= self.abs_tolerance:
|
|
53
|
+
return True
|
|
54
|
+
if b == 0:
|
|
55
|
+
return abs(a) <= self.abs_tolerance
|
|
56
|
+
return abs(a - b) / abs(b) <= self.rel_tolerance
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
STRICT = Policy(rel_tolerance=0.005, unmatched_percent="flag", ignore_below=10.0)
|
|
60
|
+
LENIENT = Policy(rel_tolerance=0.05)
|
figured/py.typed
ADDED
|
File without changes
|
figured/report.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""What came back: every figure, whether it traced, and how."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any, Literal
|
|
7
|
+
|
|
8
|
+
from figured.derive import Match, fmt
|
|
9
|
+
from figured.extract import Figure
|
|
10
|
+
|
|
11
|
+
Status = Literal["grounded", "ungrounded", "ignored"]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class Result:
|
|
16
|
+
figure: Figure
|
|
17
|
+
status: Status
|
|
18
|
+
match: Match | None = None
|
|
19
|
+
reason: str = ""
|
|
20
|
+
|
|
21
|
+
@property
|
|
22
|
+
def literal(self) -> str:
|
|
23
|
+
return self.figure.literal
|
|
24
|
+
|
|
25
|
+
def to_dict(self) -> dict[str, Any]:
|
|
26
|
+
d: dict[str, Any] = {
|
|
27
|
+
"literal": self.figure.literal,
|
|
28
|
+
"value": self.figure.value,
|
|
29
|
+
"span": [self.figure.start, self.figure.end],
|
|
30
|
+
"status": self.status,
|
|
31
|
+
}
|
|
32
|
+
if self.match:
|
|
33
|
+
d["match"] = {
|
|
34
|
+
"kind": self.match.kind,
|
|
35
|
+
"value": self.match.value,
|
|
36
|
+
"error": self.match.error,
|
|
37
|
+
"explanation": self.match.explanation,
|
|
38
|
+
}
|
|
39
|
+
if self.reason:
|
|
40
|
+
d["reason"] = self.reason
|
|
41
|
+
return d
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass
|
|
45
|
+
class Report:
|
|
46
|
+
text: str
|
|
47
|
+
results: list[Result]
|
|
48
|
+
evidence_cells: int
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def ok(self) -> bool:
|
|
52
|
+
return not self.ungrounded
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def grounded(self) -> list[Result]:
|
|
56
|
+
return [r for r in self.results if r.status == "grounded"]
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def ungrounded(self) -> list[str]:
|
|
60
|
+
return [r.figure.literal for r in self.results if r.status == "ungrounded"]
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def ignored(self) -> list[Result]:
|
|
64
|
+
return [r for r in self.results if r.status == "ignored"]
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def checked(self) -> int:
|
|
68
|
+
return sum(1 for r in self.results if r.status != "ignored")
|
|
69
|
+
|
|
70
|
+
def caveat(self, limit: int = 4) -> str:
|
|
71
|
+
if self.ok:
|
|
72
|
+
return ""
|
|
73
|
+
shown = ", ".join(self.ungrounded[:limit])
|
|
74
|
+
more = len(self.ungrounded) - limit
|
|
75
|
+
tail = f" and {more} more" if more > 0 else ""
|
|
76
|
+
return (
|
|
77
|
+
f"Note: these figures could not be traced to the data: {shown}{tail}. Treat them as approximate."
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
def explain(self) -> str:
|
|
81
|
+
head = "OK" if self.ok else "UNGROUNDED"
|
|
82
|
+
lines = [f"{head} · {self.checked} checked · {len(self.ungrounded)} untraceable"]
|
|
83
|
+
for r in self.results:
|
|
84
|
+
if r.status == "grounded" and r.match:
|
|
85
|
+
lines.append(f" ✓ {r.literal:<16} {r.match.kind:<14} {r.match.explanation}")
|
|
86
|
+
elif r.status == "ungrounded":
|
|
87
|
+
lines.append(f" ✗ {r.literal:<16} no cell, sum, difference, or ratio within tolerance")
|
|
88
|
+
else:
|
|
89
|
+
lines.append(f" · {r.literal:<16} ignored ({r.reason})")
|
|
90
|
+
return "\n".join(lines)
|
|
91
|
+
|
|
92
|
+
def to_dict(self) -> dict[str, Any]:
|
|
93
|
+
return {
|
|
94
|
+
"ok": self.ok,
|
|
95
|
+
"checked": self.checked,
|
|
96
|
+
"ungrounded": self.ungrounded,
|
|
97
|
+
"evidence_cells": self.evidence_cells,
|
|
98
|
+
"figures": [r.to_dict() for r in self.results],
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
def __repr__(self) -> str:
|
|
102
|
+
return f"Report(ok={self.ok}, checked={self.checked}, ungrounded={self.ungrounded})"
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
__all__ = ["Report", "Result", "Status", "fmt"]
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: figured
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Show your work: verify that every number in an LLM-generated answer traces to the rows it was derived from.
|
|
5
|
+
Project-URL: Homepage, https://github.com/nisheshshukla/figured
|
|
6
|
+
Project-URL: Repository, https://github.com/nisheshshukla/figured
|
|
7
|
+
Project-URL: Issues, https://github.com/nisheshshukla/figured/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/nisheshshukla/figured/blob/main/CHANGELOG.md
|
|
9
|
+
Author: Nishesh Shukla
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: evaluation,faithfulness,grounding,guardrails,hallucination,llm,text-to-sql
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
23
|
+
Classifier: Typing :: Typed
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: hypothesis>=6.100; extra == 'dev'
|
|
27
|
+
Requires-Dist: mypy>=1.11; extra == 'dev'
|
|
28
|
+
Requires-Dist: pytest-cov>=5; extra == 'dev'
|
|
29
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
30
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
31
|
+
Provides-Extra: judge
|
|
32
|
+
Requires-Dist: anthropic>=1.0; extra == 'judge'
|
|
33
|
+
Requires-Dist: pydantic>=2.0; extra == 'judge'
|
|
34
|
+
Description-Content-Type: text/markdown
|
|
35
|
+
|
|
36
|
+
# figured
|
|
37
|
+
|
|
38
|
+
**Show your work.** Verify that every number in an LLM-generated answer traces to the rows it was written from.
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
from figured import trace
|
|
42
|
+
|
|
43
|
+
rows = [{"state": "California", "pop": 39_346_023}, {"state": "Texas", "pop": 28_635_442}]
|
|
44
|
+
answer = "California has 39.3 million people, about 10.7 million more than Texas, and 4.1 million of them moved last year."
|
|
45
|
+
|
|
46
|
+
report = trace(answer, rows)
|
|
47
|
+
report.ok # False
|
|
48
|
+
report.ungrounded # ['4.1 million']
|
|
49
|
+
print(report.explain())
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
```
|
|
53
|
+
UNGROUNDED · 3 checked · 1 untraceable
|
|
54
|
+
✓ 39.3 million cell pop[California] = 39,346,023
|
|
55
|
+
✓ 10.7 million difference pop[California] − pop[Texas] = 39,346,023 − 28,635,442 = 10,710,581
|
|
56
|
+
✗ 4.1 million no cell, sum, difference, or ratio within tolerance
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Zero dependencies. Deterministic. About 150 µs for a typical answer, 1 ms for 200 rows. Python 3.10+.
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
pip install figured
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
## Why
|
|
66
|
+
|
|
67
|
+
Text-to-SQL agents and RAG-over-tables pipelines validate the query and trust the prose. The model reads the rows and writes a paragraph, and nothing checks that the paragraph's numbers came from the rows. When it invents a figure, the SQL was fine, the rows were fine, and the user sees a confident wrong number.
|
|
68
|
+
|
|
69
|
+
The usual answer is an LLM judge, which is slow, costs money per answer, and is itself wrong sometimes: in one published test, a faithfulness metric scored a fabricated price as fully faithful five times in a row. `figured` is the deterministic check that runs on every answer before a judge is needed. It is the "grounding" step the authors of this library shipped inside a Census data agent, extracted so anyone can use it.
|
|
70
|
+
|
|
71
|
+
## What counts as grounded
|
|
72
|
+
|
|
73
|
+
Every substantive number in the text must be within a tolerance (default 1.5 percent) of something the rows could legitimately produce:
|
|
74
|
+
|
|
75
|
+
| Derivation | Example | Explanation you get back |
|
|
76
|
+
|---|---|---|
|
|
77
|
+
| cell | "39,346,023 people" | `pop[California] = 39,346,023` |
|
|
78
|
+
| column sum | "together, 1,000,000 residents" | `sum of pop over 3 rows = 1,000,000` |
|
|
79
|
+
| adjacent-cell sum | "the three youngest bands total 1,200" | `a..c[row 0] summed = 1,200` |
|
|
80
|
+
| difference | "10.7 million more than Texas" | `pop[California] − pop[Texas] = ... = 10,710,581` |
|
|
81
|
+
| ratio | "3.0 to one" | `a[row 0] ÷ b[row 0] = 3` |
|
|
82
|
+
| percent | "72.8% of California" | `pop[Texas] ÷ pop[California] = 72.8%` |
|
|
83
|
+
| percent change | "grew 2.3%" | `(y2020 − y2019) ÷ y2019 = 2.3%` |
|
|
84
|
+
|
|
85
|
+
Differences, ratios, and percentages are searched within a row and across rows. A stated range such as "between 39 and 40 million" is grounded when a candidate lies inside it. Numbers at or below 100 and bare four-digit years are ignored by default, because "top 5 counties in 2020" is not a claim about the data.
|
|
86
|
+
|
|
87
|
+
Two rules keep the search honest. A figure written as a percentage is searched as `a ÷ b × 100`, and a plain figure as `a ÷ b`, never both, so "150" cannot pass by coincidentally matching a 150% share. And the pairwise and adjacent-cell derivations cover the first `max_rows` rows (12 by default), which is the part of a result a model has usually read; cells and column sums cover every row. Raise `max_rows` if your prompt includes more.
|
|
88
|
+
|
|
89
|
+
Each grounded figure carries the derivation that matched, so a reviewer can check it by hand. Each ungrounded figure is named. Nothing blocks: you decide whether to append the caveat, change a badge, or fail a test.
|
|
90
|
+
|
|
91
|
+
## What it reads
|
|
92
|
+
|
|
93
|
+
`trace(text, rows)` accepts the rows in whatever shape you already have:
|
|
94
|
+
|
|
95
|
+
- a list of dicts, as most drivers and ORMs return
|
|
96
|
+
- a list of lists or tuples, with or without column names
|
|
97
|
+
- a `{"columns": [...], "rows": [...]}` mapping
|
|
98
|
+
- a pandas DataFrame
|
|
99
|
+
- a DB-API cursor after `execute`
|
|
100
|
+
- several result sets at once: `trace(text, results=[rows_a, rows_b])`
|
|
101
|
+
|
|
102
|
+
Numeric strings in the rows are parsed by default, so `"39,346,023"`, `"$1,200"`, and `"12%"` all count. Decimals from database drivers are handled. Booleans are not numbers.
|
|
103
|
+
|
|
104
|
+
## Text it understands
|
|
105
|
+
|
|
106
|
+
Thousands separators, decimals, scientific notation (`1.2e6`), currency symbols, scale words (`39.3 million`, `2.5bn`, `3k`), percent markers (`12%`, `12 percent`, `3 percentage points`), negatives, and ranges with a shared unit (`40 to 50 million`). Identifiers such as `B01003e1` or request ids are not mistaken for numbers, and ordinals are skipped.
|
|
107
|
+
|
|
108
|
+
## Tuning
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
from figured import trace, Policy, STRICT, LENIENT
|
|
112
|
+
|
|
113
|
+
trace(answer, rows, rel_tolerance=0.005) # tighter rounding
|
|
114
|
+
trace(answer, rows, unmatched_percent="flag") # a percentage must match something
|
|
115
|
+
trace(answer, rows, derivations={"cell", "column_sum"}) # no pairwise arithmetic
|
|
116
|
+
trace(answer, rows, policy=STRICT) # 0.5%, percentages must match, checks down to 10
|
|
117
|
+
trace(answer, rows, ignore_below=0, ignore_years=False)
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
| Option | Default | Meaning |
|
|
121
|
+
|---|---|---|
|
|
122
|
+
| `rel_tolerance` | 0.015 | relative error allowed, covers rounding to three significant figures |
|
|
123
|
+
| `abs_tolerance` | 0 | absolute error allowed in addition |
|
|
124
|
+
| `ignore_below` | 100 | figures at or below this are counts of things, not claims |
|
|
125
|
+
| `ignore_years` | True | bare four-digit integers in `year_range` are skipped |
|
|
126
|
+
| `unmatched_percent` | "pass" | shares of totals outside the rows are common, so a lone percentage passes |
|
|
127
|
+
| `max_rows`, `max_cells` | 12, 40 | how much of the result feeds the pairwise and adjacent-sum search |
|
|
128
|
+
| `derivations` | all seven | which candidate kinds are generated |
|
|
129
|
+
| `parse_strings` | True | coerce numeric strings in the rows |
|
|
130
|
+
|
|
131
|
+
## Speed
|
|
132
|
+
|
|
133
|
+
Measured with `python benchmarks/bench.py` on a laptop, one answer with nine figures:
|
|
134
|
+
|
|
135
|
+
| Result set | Time per check |
|
|
136
|
+
|---|---|
|
|
137
|
+
| 2 rows × 3 columns | 150 µs |
|
|
138
|
+
| 12 rows × 5 columns | 360 µs |
|
|
139
|
+
| 200 rows × 10 columns | 1.1 ms |
|
|
140
|
+
| 2,000 rows × 10 columns | 9 ms |
|
|
141
|
+
|
|
142
|
+
Nothing is enumerated up front. Cells and column sums are indexed once; differences, ratios, percentages, and percent changes are found per figure by solving for the partner cell and bisecting for it. Explanations are formatted only for the figure that matched. For comparison, a model-based faithfulness judge takes seconds and costs a request.
|
|
143
|
+
|
|
144
|
+
## Command line
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
figured "California has 39.3 million people." --rows rows.json
|
|
148
|
+
figured - --rows rows.json < answer.txt
|
|
149
|
+
figured "..." --rows rows.json --json --tolerance 0.01 --strict-percent
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Exit code 1 when any figure is untraceable, so it can gate a pipeline step.
|
|
153
|
+
|
|
154
|
+
## Using it in a pipeline
|
|
155
|
+
|
|
156
|
+
**After every answer**, append the caveat and flip a badge:
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
report = trace(answer, rows)
|
|
160
|
+
if not report.ok:
|
|
161
|
+
answer += "\n\n" + report.caveat()
|
|
162
|
+
badge = "check figures"
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
**In promptfoo**, as a Python assertion: see `examples/promptfoo_assert.py`.
|
|
166
|
+
|
|
167
|
+
**In DeepEval or any custom metric**, wrap `trace` and return `1 - len(report.ungrounded) / report.checked`.
|
|
168
|
+
|
|
169
|
+
**With a model judge for the rest.** Arithmetic cannot see a wrong word around a right number: "Nevada is richer than Utah" with the two correct medians reversed passes. The optional `judge` extra sends the question, the rows, and the answer to a model and returns a strict verdict on faithfulness, responsiveness, and caveats:
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
pip install "figured[judge]"
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
```python
|
|
176
|
+
from figured.judge import judge
|
|
177
|
+
|
|
178
|
+
judge("Which state is richer?", answer, rows) # {"verdict": "fail", "issues": ["comparison reversed"], ...}
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
## What it does not do
|
|
182
|
+
|
|
183
|
+
- It cannot catch a correct number attached to the wrong claim. That is what the judge extra is for.
|
|
184
|
+
- With large result sets the derived set is big, and a hallucinated figure can land within tolerance of some difference by coincidence. The defaults cap the pairwise search at 12 rows and 40 cells; tighten the tolerance or restrict `derivations` for sensitive uses. A flag on a correct figure is treated as the worse error, because people stop reading badges that cry wolf.
|
|
185
|
+
- Numbers written as words ("two million") are not extracted.
|
|
186
|
+
- It does not know what the rows mean. If the agent queried the wrong column and described it faithfully, every figure traces.
|
|
187
|
+
|
|
188
|
+
## How it compares
|
|
189
|
+
|
|
190
|
+
| | rows as evidence | derived arithmetic | deterministic | names each figure | packaged |
|
|
191
|
+
|---|---|---|---|---|---|
|
|
192
|
+
| **figured** | yes | sums, differences, ratios, percentages, ranges | yes | yes, with the derivation | pip, zero deps |
|
|
193
|
+
| llmground | no, a source string | no | yes | yes | pip |
|
|
194
|
+
| @demystify/grounding | no, cited facts | no | yes | yes | npm |
|
|
195
|
+
| pcn-core (Proof-Carrying Numbers) | claim values you supply | no | yes | yes, needs model-emitted tags | pip |
|
|
196
|
+
| NumProof | yes | yes | yes | yes | hosted API |
|
|
197
|
+
| DeepEval / Ragas faithfulness | text context | n/a | no, LLM or NLI | no | pip |
|
|
198
|
+
|
|
199
|
+
The Proof-Carrying Numbers policy vocabulary (exact, rounded, scale alias, tolerance, percent, range, year) is the clearest statement of the matching problem, and this library borrows its shape. The difference is the evidence contract: rows in, free text in, no cooperation from the model required.
|
|
200
|
+
|
|
201
|
+
## Ports
|
|
202
|
+
|
|
203
|
+
Behavior is pinned by the conformance vectors in `tests/vectors/`. A port in another language is correct when it passes them unchanged. A TypeScript port is the natural next one; open an issue if you want to take it.
|
|
204
|
+
|
|
205
|
+
## Origin
|
|
206
|
+
|
|
207
|
+
Built inside a Census data agent whose answers had to be traceable to the ACS rows behind them. The first version only derived values within a row, so a correct "about $10,900 higher" comparison across two state rows was flagged as suspect. That false flag is now a named test vector, and it is why the defaults lean toward trusting the model when the arithmetic works out.
|
|
208
|
+
|
|
209
|
+
## License
|
|
210
|
+
|
|
211
|
+
MIT.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
figured/__init__.py,sha256=KfstOh4RrhIVyLWctCyq-mqm3ViEQPTINEo5wKVagyM,801
|
|
2
|
+
figured/__main__.py,sha256=X5TxIAVDwbQeZZIKqSIRL9-o_7TyfR69olsPbLxit_U,1561
|
|
3
|
+
figured/core.py,sha256=TSQZGNvTmpltU0CihpaFU6YAl5761tI52xb0GsJvn4Q,2576
|
|
4
|
+
figured/derive.py,sha256=idVOAY57JFrkEwCHZzvZz3_e7OcNVyX8vur3zZCriks,13774
|
|
5
|
+
figured/evidence.py,sha256=b2_aqiqjAc9o2IwSnq1qQS9NIQLVzuJb6YQAANxiYFM,6652
|
|
6
|
+
figured/extract.py,sha256=WyWxPX6edy-tvMJtTAnJuvJx2VWaJmlnvuB3Rx862Us,4632
|
|
7
|
+
figured/judge.py,sha256=H36V2fsR6uNZeX7iP1vLVSQUocxQJ8LGNRLv2wSLZCs,2677
|
|
8
|
+
figured/policy.py,sha256=ScBM63WDvUZO2Q0PEfiB5QBUzSAZzlMuNnyKUW9TpFQ,2557
|
|
9
|
+
figured/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
10
|
+
figured/report.py,sha256=nrA-Uic7OmhU9cNAbPP8CW48nk9NdGlanhnKy55Fhog,3213
|
|
11
|
+
figured-0.1.0.dist-info/METADATA,sha256=c4hYUZmSJPfK48QQNlXSSBjJ0K7MjorHqyJl7xoZ-5s,11207
|
|
12
|
+
figured-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
13
|
+
figured-0.1.0.dist-info/entry_points.txt,sha256=mh91KVuIvpJy6DIGjjsmEcuJaOGMoldaFyyRXdSgW7Q,50
|
|
14
|
+
figured-0.1.0.dist-info/licenses/LICENSE,sha256=SHGKk_adXxNMC1rM63eWqTyDr6nbLVzVZcMi7OReWNA,1071
|
|
15
|
+
figured-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Nishesh Shukla
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|