figured 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
figured/__init__.py ADDED
@@ -0,0 +1,31 @@
1
+ """figured: show your work.
2
+
3
+ Verify that every number in an LLM-generated answer traces to the rows it was derived from,
4
+ directly or as a sum, difference, ratio, or percentage of them.
5
+
6
+ >>> from figured import trace
7
+ >>> report = trace("California has 39.3 million people.", [{"state": "CA", "pop": 39346023}])
8
+ >>> report.ok
9
+ True
10
+ """
11
+
12
+ from figured.core import trace
13
+ from figured.evidence import build_evidence
14
+ from figured.extract import Figure, extract_numbers
15
+ from figured.policy import DERIVATIONS, LENIENT, STRICT, Policy
16
+ from figured.report import Report, Result
17
+
18
+ __version__ = "0.1.0"
19
+ __all__ = [
20
+ "DERIVATIONS",
21
+ "LENIENT",
22
+ "STRICT",
23
+ "Figure",
24
+ "Policy",
25
+ "Report",
26
+ "Result",
27
+ "__version__",
28
+ "build_evidence",
29
+ "extract_numbers",
30
+ "trace",
31
+ ]
figured/__main__.py ADDED
@@ -0,0 +1,41 @@
1
+ """Command line: figured "answer text" --rows rows.json [--tolerance 0.015] [--json]
2
+
3
+ Exits 1 when any figure is untraceable, so it can gate a CI step or a pipeline.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import argparse
9
+ import json
10
+ import sys
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ from figured.core import trace
15
+
16
+
17
+ def main(argv: list[str] | None = None) -> int:
18
+ p = argparse.ArgumentParser(prog="figured", description="Check that the numbers in a text trace to rows.")
19
+ p.add_argument("text", help="the generated text, or '-' to read it from stdin")
20
+ p.add_argument(
21
+ "--rows", required=True, help="JSON file: a list of objects, a list of arrays, or {columns, rows}"
22
+ )
23
+ p.add_argument("--tolerance", type=float, default=None, help="relative tolerance, default 0.015")
24
+ p.add_argument("--strict-percent", action="store_true", help="flag percentages that match nothing")
25
+ p.add_argument("--json", action="store_true", help="print the full report as JSON")
26
+ args = p.parse_args(argv)
27
+
28
+ text = sys.stdin.read() if args.text == "-" else args.text
29
+ rows = json.loads(Path(args.rows).read_text())
30
+ overrides: dict[str, Any] = {}
31
+ if args.tolerance is not None:
32
+ overrides["rel_tolerance"] = args.tolerance
33
+ if args.strict_percent:
34
+ overrides["unmatched_percent"] = "flag"
35
+ report = trace(text, rows, **overrides)
36
+ print(json.dumps(report.to_dict(), indent=2) if args.json else report.explain())
37
+ return 0 if report.ok else 1
38
+
39
+
40
+ if __name__ == "__main__":
41
+ raise SystemExit(main())
figured/core.py ADDED
@@ -0,0 +1,66 @@
1
+ """The one function: trace(text, rows) -> Report."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterable
6
+ from typing import Any
7
+
8
+ from figured.derive import Index
9
+ from figured.evidence import build_evidence
10
+ from figured.extract import Figure, extract_numbers
11
+ from figured.policy import Policy
12
+ from figured.report import Report, Result
13
+
14
+
15
+ def trace(
16
+ text: str,
17
+ rows: Any = None,
18
+ *,
19
+ results: Iterable[Any] | None = None,
20
+ policy: Policy | None = None,
21
+ **overrides: Any,
22
+ ) -> Report:
23
+ """Check that every substantive number in `text` traces to `rows`.
24
+
25
+ `rows` is one result set in any common shape: a list of dicts, a list of sequences, a
26
+ pandas DataFrame, a DB-API cursor, or a {"columns": [...], "rows": [...]} mapping.
27
+ `results` is several of those, for answers written from more than one query.
28
+ Policy options can be passed as keywords: trace(text, rows, rel_tolerance=0.01).
29
+ """
30
+ pol = (policy or Policy()).with_overrides(**overrides)
31
+ ev = build_evidence(rows, results, parse_strings=pol.parse_strings)
32
+ figures = extract_numbers(text)
33
+ if ev.empty:
34
+ return Report(text, [_without_evidence(f, pol) for f in figures], 0)
35
+ index = Index(ev, pol)
36
+ return Report(text, [_check(f, index, pol) for f in figures], ev.size)
37
+
38
+
39
+ def _without_evidence(fig: Figure, pol: Policy) -> Result:
40
+ if _is_year(fig, pol):
41
+ return Result(fig, "ignored", reason="year")
42
+ if fig.is_percent:
43
+ return Result(fig, "ignored", reason="percent without evidence")
44
+ if abs(fig.value) > pol.flag_without_evidence_above:
45
+ return Result(fig, "ungrounded", reason="no evidence")
46
+ return Result(fig, "ignored", reason="small")
47
+
48
+
49
+ def _check(fig: Figure, index: Index, pol: Policy) -> Result:
50
+ if _is_year(fig, pol):
51
+ return Result(fig, "ignored", reason="year")
52
+ if not fig.is_percent and abs(fig.value) <= pol.ignore_below:
53
+ return Result(fig, "ignored", reason="small")
54
+ match = index.lookup(fig.value, pol, is_percent=fig.is_percent)
55
+ if match is None and fig.is_range:
56
+ match = index.lookup_range(*fig.bounds, pol, is_percent=fig.is_percent)
57
+ if match is not None:
58
+ return Result(fig, "grounded", match)
59
+ if fig.is_percent and pol.unmatched_percent == "pass":
60
+ return Result(fig, "ignored", reason="unmatched percent allowed by policy")
61
+ return Result(fig, "ungrounded")
62
+
63
+
64
+ def _is_year(fig: Figure, pol: Policy) -> bool:
65
+ lo, hi = pol.year_range
66
+ return pol.ignore_years and fig.looks_like_year and lo <= fig.value <= hi
figured/derive.py ADDED
@@ -0,0 +1,359 @@
1
+ """Everything the rows could legitimately produce, searched on demand rather than enumerated.
2
+
3
+ Cells, column sums, and adjacent-cell sums are indexed once. Differences, ratios, percentages,
4
+ and percent changes are found per figure by solving for the partner cell and bisecting for it,
5
+ so the cost is O(cells · log cells) per figure instead of O(cells²) up front.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from bisect import bisect_left, bisect_right
11
+ from collections.abc import Callable
12
+ from dataclasses import dataclass
13
+
14
+ from figured.evidence import Cell, Evidence
15
+ from figured.policy import Policy
16
+
17
+ RANK = {
18
+ "cell": 0,
19
+ "column_sum": 1,
20
+ "row_sum": 2,
21
+ "difference": 3,
22
+ "ratio": 4,
23
+ "percent": 4,
24
+ "percent_change": 5,
25
+ }
26
+
27
+
28
+ @dataclass(frozen=True, slots=True)
29
+ class Candidate:
30
+ value: float
31
+ kind: str
32
+ explanation: str
33
+
34
+ @property
35
+ def rank(self) -> int:
36
+ return RANK[self.kind]
37
+
38
+
39
+ @dataclass(frozen=True, slots=True)
40
+ class Match:
41
+ candidate: Candidate
42
+ error: float
43
+
44
+ @property
45
+ def kind(self) -> str:
46
+ return self.candidate.kind
47
+
48
+ @property
49
+ def explanation(self) -> str:
50
+ return self.candidate.explanation
51
+
52
+ @property
53
+ def value(self) -> float:
54
+ return self.candidate.value
55
+
56
+
57
+ def fmt(v: float) -> str:
58
+ if abs(v) >= 1e15:
59
+ return f"{v:.3g}"
60
+ if float(v).is_integer():
61
+ return f"{int(v):,}"
62
+ if abs(v) >= 100:
63
+ return f"{v:,.1f}"
64
+ return f"{v:,.4g}"
65
+
66
+
67
+ Scorer = Callable[[float], float | None]
68
+
69
+
70
+ class Index:
71
+ """Sorted views over the evidence plus on-demand derivation search."""
72
+
73
+ def __init__(self, ev: Evidence, policy: Policy) -> None:
74
+ self.ev = ev
75
+ self.policy = policy
76
+ allowed = policy.derivations
77
+ self.allowed = allowed
78
+
79
+ sets = ev.results
80
+ flat_abs: list[float] = []
81
+ self.set_start: list[int] = []
82
+ for rs in sets:
83
+ self.set_start.append(len(flat_abs))
84
+ flat_abs.extend(map(abs, rs.values))
85
+ self.flat_abs = flat_abs
86
+ if "cell" in allowed and flat_abs:
87
+ order = sorted(range(len(flat_abs)), key=flat_abs.__getitem__)
88
+ self.cell_keys = [flat_abs[i] for i in order]
89
+ self.cell_order = order
90
+ else:
91
+ self.cell_keys = []
92
+ self.cell_order = []
93
+
94
+ agg: list[tuple[float, str, str]] = []
95
+ if "column_sum" in allowed:
96
+ agg.extend(self._column_sums())
97
+ agg.sort(key=lambda t: t[0])
98
+ self.agg_keys = [a for a, _, _ in agg]
99
+ self.agg_items = agg
100
+ self._row_sums: list[tuple[float, int, int, int, int]] | None = None
101
+ self._row_sum_keys: list[float] = []
102
+
103
+ pair_idx: list[int] = []
104
+ for rs, base in zip(sets, self.set_start, strict=True):
105
+ limit = rs.row_start[min(policy.max_rows, len(rs.row_start) - 1)]
106
+ pair_idx.extend(range(base, base + limit))
107
+ pair_idx = pair_idx[: policy.max_cells]
108
+ signed = self._signed
109
+ pair_idx.sort(key=signed)
110
+ self.pair_vals = [signed(i) for i in pair_idx]
111
+ self.pair_refs = pair_idx
112
+ by_abs = sorted(pair_idx, key=flat_abs.__getitem__)
113
+ self.pair_abs = [flat_abs[i] for i in by_abs]
114
+ self.pair_abs_refs = by_abs
115
+
116
+ def _signed(self, g: int) -> float:
117
+ rs_i = bisect_right(self.set_start, g) - 1
118
+ return self.ev.results[rs_i].values[g - self.set_start[rs_i]]
119
+
120
+ def cell(self, g: int) -> Cell:
121
+ rs_i = bisect_right(self.set_start, g) - 1
122
+ return self.ev.results[rs_i].cell(g - self.set_start[rs_i])
123
+
124
+ def lookup(self, value: float, policy: Policy, *, is_percent: bool = False) -> Match | None:
125
+ v = abs(value)
126
+ tol = max(v * policy.rel_tolerance, policy.abs_tolerance)
127
+
128
+ def score(d: float) -> float | None:
129
+ if not policy.close(v, d):
130
+ return None
131
+ return abs(v - d) / d if d else abs(v - d)
132
+
133
+ return self._search(v - tol, v + tol, v, score, is_percent)
134
+
135
+ def lookup_range(self, lo: float, hi: float, policy: Policy, *, is_percent: bool = False) -> Match | None:
136
+ lo_t = min(lo * (1 - policy.rel_tolerance), lo - policy.abs_tolerance)
137
+ hi_t = max(hi * (1 + policy.rel_tolerance), hi + policy.abs_tolerance)
138
+
139
+ def score(d: float) -> float | None:
140
+ if d < lo_t or d > hi_t:
141
+ return None
142
+ return 0.0 if lo <= d <= hi else min(abs(d - lo), abs(d - hi)) / max(d, 1e-12)
143
+
144
+ return self._search(lo_t, hi_t, (lo + hi) / 2, score, is_percent)
145
+
146
+ def _search(self, lo: float, hi: float, v: float, score: Scorer, is_percent: bool) -> Match | None:
147
+ allowed = self.allowed
148
+ if self.cell_keys:
149
+ m = self._best_sorted(self.cell_keys, lo, hi, score, self._cell_candidate)
150
+ if m:
151
+ return m
152
+ if self.agg_keys:
153
+ m = self._best_sorted(self.agg_keys, lo, hi, score, self._agg_candidate)
154
+ if m:
155
+ return m
156
+ if "row_sum" in allowed:
157
+ sums = self._row_sum_index()
158
+ if sums:
159
+ m = self._best_sorted(self._row_sum_keys, lo, hi, score, self._row_candidate)
160
+ if m:
161
+ return m
162
+ if not self.pair_vals:
163
+ return None
164
+ if "difference" in allowed:
165
+ m = self._differences(lo, hi, score)
166
+ if m:
167
+ return m
168
+ kind = "percent" if is_percent else "ratio"
169
+ if kind in allowed:
170
+ m = self._ratios(lo, hi, kind, score)
171
+ if m:
172
+ return m
173
+ if "percent_change" in allowed:
174
+ m = self._percent_changes(lo, hi, score)
175
+ if m:
176
+ return m
177
+ return None
178
+
179
+ @staticmethod
180
+ def _best_sorted(
181
+ keys: list[float], lo: float, hi: float, score: Scorer, make: Callable[[int], Candidate]
182
+ ) -> Match | None:
183
+ best: tuple[float, int] | None = None
184
+ for i in range(bisect_left(keys, lo), bisect_right(keys, hi)):
185
+ err = score(keys[i])
186
+ if err is not None and (best is None or err < best[0]):
187
+ best = (err, i)
188
+ return Match(make(best[1]), best[0]) if best else None
189
+
190
+ def _cell_candidate(self, i: int) -> Candidate:
191
+ c = self.cell(self.cell_order[i])
192
+ return Candidate(c.value, "cell", f"{c.ref()} = {fmt(c.value)}")
193
+
194
+ def _agg_candidate(self, i: int) -> Candidate:
195
+ a, kind, text = self.agg_items[i]
196
+ return Candidate(a, kind, text)
197
+
198
+ def _row_candidate(self, i: int) -> Candidate:
199
+ assert self._row_sums is not None
200
+ _, rs_index, r, a, b = self._row_sums[i]
201
+ rs = self.ev.results[rs_index]
202
+ s = rs.row_start[r]
203
+ signed = sum(rs.values[s + a : s + b + 1])
204
+ head, tail = rs.columns[rs.col_of[s + a]], rs.columns[rs.col_of[s + b]]
205
+ return Candidate(signed, "row_sum", f"{head}..{tail}[{rs.labels[r]}] summed = {fmt(signed)}")
206
+
207
+ def _column_sums(self) -> list[tuple[float, str, str]]:
208
+ out: list[tuple[float, str, str]] = []
209
+ for rs in self.ev.results:
210
+ for c, (total, n) in enumerate(zip(rs.col_totals, rs.col_counts, strict=True)):
211
+ if n >= 2:
212
+ out.append(
213
+ (abs(total), "column_sum", f"sum of {rs.columns[c]} over {n} rows = {fmt(total)}")
214
+ )
215
+ return out
216
+
217
+ def _row_sum_index(self) -> list[tuple[float, int, int, int, int]]:
218
+ """Adjacent-cell sums (2 to 6 cells) over the first max_rows rows, formatted only on a match."""
219
+ if self._row_sums is None:
220
+ out: list[tuple[float, int, int, int, int]] = []
221
+ for rs in self.ev.results:
222
+ vals, starts, idx = rs.values, rs.row_start, rs.index
223
+ for r in range(min(self.policy.max_rows, len(starts) - 1)):
224
+ s, e = starts[r], starts[r + 1]
225
+ n = e - s
226
+ for i in range(n - 1):
227
+ total = vals[s + i]
228
+ for j in range(i + 1, min(n, i + 6)):
229
+ total += vals[s + j]
230
+ out.append((abs(total), idx, r, i, j))
231
+ out.sort(key=lambda t: t[0])
232
+ self._row_sums = out
233
+ self._row_sum_keys = [t[0] for t in out]
234
+ return self._row_sums
235
+
236
+ def _differences(self, lo: float, hi: float, score: Scorer) -> Match | None:
237
+ """Pairs with |a − b| in [lo, hi]. Both windows slide right as b grows, so two pointers suffice."""
238
+ vals, n = self.pair_vals, len(self.pair_vals)
239
+ best: tuple[float, int, int] | None = None
240
+ s1 = e1 = s2 = e2 = 0
241
+ for j in range(n):
242
+ b = vals[j]
243
+ a_lo, a_hi = b + lo, b + hi
244
+ while s1 < n and vals[s1] < a_lo:
245
+ s1 += 1
246
+ while e1 < n and vals[e1] <= a_hi:
247
+ e1 += 1
248
+ for i in range(s1, e1):
249
+ if i != j:
250
+ err = score(vals[i] - b)
251
+ if err is not None and (best is None or err < best[0]):
252
+ best = (err, i, j)
253
+ a_lo, a_hi = b - hi, b - lo
254
+ while s2 < n and vals[s2] < a_lo:
255
+ s2 += 1
256
+ while e2 < n and vals[e2] <= a_hi:
257
+ e2 += 1
258
+ for i in range(s2, e2):
259
+ if i != j:
260
+ err = score(b - vals[i])
261
+ if err is not None and (best is None or err < best[0]):
262
+ best = (err, i, j)
263
+ if best is None:
264
+ return None
265
+ _, i, j = best
266
+ ca, cb = self.cell(self.pair_refs[i]), self.cell(self.pair_refs[j])
267
+ d = ca.value - cb.value
268
+ return Match(
269
+ Candidate(
270
+ d, "difference", f"{ca.ref()} − {cb.ref()} = {fmt(ca.value)} − {fmt(cb.value)} = {fmt(d)}"
271
+ ),
272
+ best[0],
273
+ )
274
+
275
+ def _ratios(self, lo: float, hi: float, kind: str, score: Scorer) -> Match | None:
276
+ """A percent figure is searched as a ÷ b × 100, a plain figure as a ÷ b."""
277
+ scale = 100.0 if kind == "percent" else 1.0
278
+ vals, n = self.pair_abs, len(self.pair_abs)
279
+ best: tuple[float, int, int, str, float] | None = None
280
+ r_lo, r_hi = lo / scale, hi / scale
281
+ if r_hi <= 0:
282
+ return None
283
+ r_lo = max(r_lo, 1e-300)
284
+ s1 = e1 = s2 = e2 = 0
285
+ for j in range(n):
286
+ b = vals[j]
287
+ if b == 0:
288
+ continue
289
+ a_lo, a_hi = b * r_lo, b * r_hi
290
+ while s1 < n and vals[s1] < a_lo:
291
+ s1 += 1
292
+ while e1 < n and vals[e1] <= a_hi:
293
+ e1 += 1
294
+ for i in range(s1, e1):
295
+ if i != j:
296
+ err = score(vals[i] / b * scale)
297
+ if err is not None and (best is None or err < best[0]):
298
+ best = (err, i, j, kind, scale)
299
+ a_lo, a_hi = b / r_hi, b / r_lo
300
+ while s2 < n and vals[s2] < a_lo:
301
+ s2 += 1
302
+ while e2 < n and vals[e2] <= a_hi:
303
+ e2 += 1
304
+ for i in range(s2, e2):
305
+ if i != j and vals[i]:
306
+ err = score(b / vals[i] * scale)
307
+ if err is not None and (best is None or err < best[0]):
308
+ best = (err, j, i, kind, scale)
309
+ if best is None:
310
+ return None
311
+ err, num_i, den_i, kind, scale = best
312
+ ca, cb = self.cell(self.pair_abs_refs[num_i]), self.cell(self.pair_abs_refs[den_i])
313
+ value = abs(ca.value / cb.value) * scale
314
+ text = f"{ca.ref()} ÷ {cb.ref()} = {fmt(value)}" + ("%" if scale == 100.0 else "")
315
+ return Match(Candidate(value, kind, text), err)
316
+
317
+ def _percent_changes(self, lo: float, hi: float, score: Scorer) -> Match | None:
318
+ """(a − b) ÷ b × 100 in ±[lo, hi]. Positive b slides monotonically; negatives fall back to bisect."""
319
+ vals, n = self.pair_vals, len(self.pair_vals)
320
+ f_lo, f_hi = lo / 100, hi / 100
321
+ best: tuple[float, int, int] | None = None
322
+ s1 = e1 = s2 = e2 = 0
323
+ for j in range(n):
324
+ b = vals[j]
325
+ if b == 0:
326
+ continue
327
+ if b > 0:
328
+ a_lo, a_hi = b * (1 + f_lo), b * (1 + f_hi)
329
+ while s1 < n and vals[s1] < a_lo:
330
+ s1 += 1
331
+ while e1 < n and vals[e1] <= a_hi:
332
+ e1 += 1
333
+ r1 = range(s1, e1)
334
+ a_lo, a_hi = b * (1 - f_hi), b * (1 - f_lo)
335
+ while s2 < n and vals[s2] < a_lo:
336
+ s2 += 1
337
+ while e2 < n and vals[e2] <= a_hi:
338
+ e2 += 1
339
+ r2 = range(s2, e2)
340
+ else:
341
+ w1 = sorted((b * (1 + f_lo), b * (1 + f_hi)))
342
+ w2 = sorted((b * (1 - f_hi), b * (1 - f_lo)))
343
+ r1 = range(bisect_left(vals, w1[0]), bisect_right(vals, w1[1]))
344
+ r2 = range(bisect_left(vals, w2[0]), bisect_right(vals, w2[1]))
345
+ for rng in (r1, r2):
346
+ for i in rng:
347
+ if i != j:
348
+ err = score(abs((vals[i] - b) / b * 100))
349
+ if err is not None and (best is None or err < best[0]):
350
+ best = (err, i, j)
351
+ if best is None:
352
+ return None
353
+ _, i, j = best
354
+ ca, cb = self.cell(self.pair_refs[i]), self.cell(self.pair_refs[j])
355
+ pc = (ca.value - cb.value) / cb.value * 100
356
+ return Match(
357
+ Candidate(pc, "percent_change", f"({ca.ref()} − {cb.ref()}) ÷ {cb.ref()} = {fmt(pc)}%"),
358
+ best[0],
359
+ )
figured/evidence.py ADDED
@@ -0,0 +1,202 @@
1
+ """Normalize whatever the caller has (rows, dicts, a DataFrame, a cursor) into numeric cells.
2
+
3
+ Storage is columnar and flat: one list of values across all result sets, with offsets. Cell
4
+ objects are created only for the handful of cells that end up in an explanation.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import math
10
+ import re
11
+ from bisect import bisect_right
12
+ from collections.abc import Iterable, Mapping, Sequence
13
+ from dataclasses import dataclass, field
14
+ from decimal import Decimal
15
+ from typing import Any
16
+
17
+ _NUMERIC_STRING = re.compile(r"^[-+]?[$€£¥]?\s?(?:\d{1,3}(?:,\d{3})+|\d+)(?:\.\d+)?(?:[eE][-+]?\d+)?%?$")
18
+ _MONEY = str.maketrans("", "", ",$€£¥")
19
+
20
+
21
+ @dataclass(frozen=True, slots=True)
22
+ class Cell:
23
+ value: float
24
+ column: str
25
+ row: int
26
+ result: int
27
+ label: str
28
+
29
+ def ref(self) -> str:
30
+ return f"{self.column}[{self.label}]"
31
+
32
+
33
+ @dataclass(slots=True)
34
+ class ResultSet:
35
+ columns: list[str]
36
+ index: int
37
+ values: list[float] = field(default_factory=list)
38
+ col_of: list[int] = field(default_factory=list)
39
+ row_start: list[int] = field(default_factory=list)
40
+ labels: list[str] = field(default_factory=list)
41
+ col_totals: list[float] = field(default_factory=list)
42
+ col_counts: list[int] = field(default_factory=list)
43
+
44
+ @property
45
+ def rows(self) -> list[list[Cell]]:
46
+ return [
47
+ [self.cell(k) for k in range(self.row_start[r], self.row_start[r + 1])]
48
+ for r in range(len(self.labels))
49
+ ]
50
+
51
+ def row_of(self, k: int) -> int:
52
+ return bisect_right(self.row_start, k) - 1
53
+
54
+ def cell(self, k: int) -> Cell:
55
+ r = self.row_of(k)
56
+ return Cell(self.values[k], self.columns[self.col_of[k]], r, self.index, self.labels[r])
57
+
58
+
59
+ @dataclass(slots=True)
60
+ class Evidence:
61
+ results: list[ResultSet] = field(default_factory=list)
62
+
63
+ @property
64
+ def cells(self) -> list[Cell]:
65
+ return [rs.cell(k) for rs in self.results for k in range(len(rs.values))]
66
+
67
+ @property
68
+ def size(self) -> int:
69
+ return sum(len(rs.values) for rs in self.results)
70
+
71
+ @property
72
+ def empty(self) -> bool:
73
+ return self.size == 0
74
+
75
+
76
+ def build_evidence(
77
+ rows: Any = None,
78
+ results: Iterable[Any] | None = None,
79
+ *,
80
+ parse_strings: bool = True,
81
+ ) -> Evidence:
82
+ """Accept one result set (`rows`) and/or several (`results`) in any common shape."""
83
+ sets: list[Any] = []
84
+ if rows is not None:
85
+ sets.append(rows)
86
+ if results is not None:
87
+ sets.extend(results)
88
+ ev = Evidence()
89
+ for i, obj in enumerate(sets):
90
+ columns, raw_rows = _to_table(obj)
91
+ ev.results.append(_result_set(columns, raw_rows, i, parse_strings))
92
+ return ev
93
+
94
+
95
+ def _to_table(obj: Any) -> tuple[list[str], list[Sequence[Any]]]:
96
+ if obj is None:
97
+ return [], []
98
+ if isinstance(obj, Mapping) and "rows" in obj:
99
+ cols = [str(c) for c in obj.get("columns") or []]
100
+ return cols, [_dict_row(r, cols) if isinstance(r, Mapping) else r for r in obj["rows"]]
101
+ if hasattr(obj, "to_dict") and hasattr(obj, "columns"):
102
+ cols = [str(c) for c in obj.columns]
103
+ return cols, [[rec.get(c) for c in cols] for rec in obj.to_dict("records")]
104
+ if hasattr(obj, "fetchall") and hasattr(obj, "description"):
105
+ return [str(d[0]) for d in obj.description or []], list(obj.fetchall())
106
+ rows = obj if isinstance(obj, list) else list(obj)
107
+ if not rows:
108
+ return [], []
109
+ first = rows[0]
110
+ if isinstance(first, Mapping):
111
+ keys: list[Any] = list(first.keys())
112
+ seen = set(keys)
113
+ for r in rows:
114
+ if len(r) != len(keys) or r.keys() != seen:
115
+ for k in r:
116
+ if k not in seen:
117
+ seen.add(k)
118
+ keys.append(k)
119
+ return [str(k) for k in keys], [[r.get(c) for c in keys] for r in rows]
120
+ if isinstance(first, str | bytes | int | float | Decimal):
121
+ return [], [[r] for r in rows]
122
+ return [], rows
123
+
124
+
125
+ def _dict_row(r: Mapping[str, Any], cols: list[str]) -> list[Any]:
126
+ return [r.get(c) for c in cols] if cols else list(r.values())
127
+
128
+
129
+ def _result_set(
130
+ columns: list[str], raw_rows: list[Sequence[Any]], index: int, parse_strings: bool
131
+ ) -> ResultSet:
132
+ width = max((len(r) for r in raw_rows), default=0)
133
+ if len(columns) < width:
134
+ columns = columns + [f"c{i}" for i in range(len(columns), width)]
135
+ rs = ResultSet(columns, index)
136
+ values, col_of, row_start, labels = rs.values, rs.col_of, rs.row_start, rs.labels
137
+ totals, counts = [0.0] * width, [0] * width
138
+ push_v, push_c, push_label, push_start = values.append, col_of.append, labels.append, row_start.append
139
+ isfinite = math.isfinite
140
+ for raw in raw_rows:
141
+ push_start(len(values))
142
+ label = ""
143
+ for ci, v in enumerate(raw):
144
+ t = type(v)
145
+ if t is float:
146
+ if not isfinite(v):
147
+ continue
148
+ elif t is int:
149
+ v = float(v)
150
+ elif t is str:
151
+ s = v.strip()
152
+ if _NUMERIC_STRING.match(s):
153
+ if not parse_strings:
154
+ continue
155
+ num = _parse_numeric_string(s)
156
+ if num is None:
157
+ continue
158
+ v = num
159
+ else:
160
+ if not label and s:
161
+ label = s[:40]
162
+ continue
163
+ else:
164
+ num = to_number(v, parse_strings)
165
+ if num is None:
166
+ continue
167
+ v = num
168
+ push_v(v)
169
+ push_c(ci)
170
+ totals[ci] += v
171
+ counts[ci] += 1
172
+ push_label(label or f"row {len(labels)}")
173
+ push_start(len(values))
174
+ rs.col_totals, rs.col_counts = totals, counts
175
+ return rs
176
+
177
+
178
+ def _parse_numeric_string(s: str) -> float | None:
179
+ try:
180
+ return float(s.rstrip("%").translate(_MONEY))
181
+ except ValueError:
182
+ return None
183
+
184
+
185
+ def to_number(v: Any, parse_strings: bool = True) -> float | None:
186
+ if v is None or isinstance(v, bool):
187
+ return None
188
+ if isinstance(v, int | float):
189
+ f = float(v)
190
+ return f if math.isfinite(f) else None
191
+ if isinstance(v, Decimal):
192
+ return float(v) if v.is_finite() else None
193
+ if isinstance(v, str):
194
+ s = v.strip()
195
+ return _parse_numeric_string(s) if parse_strings and _NUMERIC_STRING.match(s) else None
196
+ if hasattr(v, "__float__"):
197
+ try:
198
+ f = float(v)
199
+ except (TypeError, ValueError):
200
+ return None
201
+ return f if math.isfinite(f) else None
202
+ return None
figured/extract.py ADDED
@@ -0,0 +1,149 @@
1
+ """Find the numbers in a piece of text, with their positions and meaning."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from dataclasses import dataclass
7
+
8
+ SCALES: dict[str, float] = {
9
+ "thousand": 1e3,
10
+ "k": 1e3,
11
+ "million": 1e6,
12
+ "mn": 1e6,
13
+ "m": 1e6,
14
+ "mm": 1e6,
15
+ "billion": 1e9,
16
+ "bn": 1e9,
17
+ "b": 1e9,
18
+ "trillion": 1e12,
19
+ "tn": 1e12,
20
+ "t": 1e12,
21
+ }
22
+ PERCENT_WORDS = frozenset({"%", "percent", "pct", "percentage points", "pp"})
23
+
24
+ _NUMBER = re.compile(
25
+ r"""
26
+ (?<![\w.])
27
+ (?P<sign>[-−]\s?)?
28
+ (?P<currency>[$€£¥])?
29
+ (?P<body>
30
+ \d+(?:\.\d+)?[eE][+-]?\d+
31
+ | \d{1,3}(?:,\d{3})+(?:\.\d+)?
32
+ | \d+(?:\.\d+)?
33
+ )
34
+ (?!\d)
35
+ (?!(?:st|nd|rd|th)\b)
36
+ (?:
37
+ \s?(?P<pct>%)
38
+ | \s?(?P<word>percentage\ points|percent|pct|pp|thousand|million|billion|trillion|mn|mm|bn|tn|k|m|b|t)\b
39
+ )?
40
+ (?![A-Za-z_])
41
+ """,
42
+ re.VERBOSE | re.IGNORECASE,
43
+ )
44
+ _START = re.compile(r"(?:[-−]\s?)?[$€£¥]?\d")
45
+ _RANGE_JOIN = re.compile(r"^\s*(?:to|and|or|-|–|—)\s*$", re.IGNORECASE)
46
+
47
+
48
+ @dataclass(frozen=True)
49
+ class Figure:
50
+ """One number found in the text."""
51
+
52
+ literal: str
53
+ value: float
54
+ start: int
55
+ end: int
56
+ is_percent: bool = False
57
+ is_currency: bool = False
58
+ has_scale: bool = False
59
+ range_partner: float | None = None
60
+
61
+ @property
62
+ def is_range(self) -> bool:
63
+ return self.range_partner is not None
64
+
65
+ @property
66
+ def bounds(self) -> tuple[float, float]:
67
+ other = self.range_partner if self.range_partner is not None else self.value
68
+ lo, hi = sorted((abs(self.value), abs(other)))
69
+ return lo, hi
70
+
71
+ @property
72
+ def looks_like_year(self) -> bool:
73
+ body = self.literal.lstrip("-−$€£¥ ").strip()
74
+ return body.isdigit() and len(body) == 4 and not self.is_percent and not self.has_scale
75
+
76
+
77
+ def extract_numbers(text: str) -> list[Figure]:
78
+ """Return every number-like token in reading order.
79
+
80
+ Handles thousands separators, decimals, scientific notation, currency symbols, scale
81
+ words (thousand, million, k, bn, ...), percent markers, and ranges such as
82
+ "40 to 50 million", where the scale of the second number is applied to the first.
83
+ """
84
+ out: list[Figure] = []
85
+ pos = 0
86
+ match = _NUMBER.match
87
+ for start in _START.finditer(text):
88
+ at = start.start()
89
+ if at < pos:
90
+ continue
91
+ m = match(text, at) or (match(text, start.end() - 1) if start.end() - 1 > at else None)
92
+ if m is None:
93
+ continue
94
+ pos = m.end()
95
+ body = m.group("body").replace(",", "")
96
+ value = float(body)
97
+ word = (m.group("word") or "").lower()
98
+ pct = bool(m.group("pct")) or word in PERCENT_WORDS
99
+ scale = SCALES.get(word)
100
+ if scale:
101
+ value *= scale
102
+ if m.group("sign"):
103
+ value = -value
104
+ out.append(
105
+ Figure(
106
+ literal=m.group(0).strip(),
107
+ value=value,
108
+ start=m.start(),
109
+ end=m.end(),
110
+ is_percent=pct,
111
+ is_currency=bool(m.group("currency")),
112
+ has_scale=scale is not None,
113
+ )
114
+ )
115
+ return _apply_ranges(text, out)
116
+
117
+
118
+ def _apply_ranges(text: str, figures: list[Figure]) -> list[Figure]:
119
+ if len(figures) < 2:
120
+ return figures
121
+ fixed = list(figures)
122
+ for i in range(len(fixed) - 1):
123
+ a, b = fixed[i], fixed[i + 1]
124
+ if not _RANGE_JOIN.match(text[a.end : b.start]):
125
+ continue
126
+ if b.has_scale and not a.has_scale and not a.is_percent:
127
+ a = Figure(a.literal, a.value * _scale_of(b), a.start, a.end, False, a.is_currency, True)
128
+ elif b.is_percent and not a.is_percent and not a.has_scale:
129
+ a = Figure(a.literal, a.value, a.start, a.end, True, a.is_currency, False)
130
+ if a.is_percent == b.is_percent and a.looks_like_year == b.looks_like_year and not a.looks_like_year:
131
+ fixed[i] = Figure(
132
+ a.literal, a.value, a.start, a.end, a.is_percent, a.is_currency, a.has_scale, b.value
133
+ )
134
+ fixed[i + 1] = Figure(
135
+ b.literal, b.value, b.start, b.end, b.is_percent, b.is_currency, b.has_scale, a.value
136
+ )
137
+ else:
138
+ fixed[i] = a
139
+ return fixed
140
+
141
+
142
+ def _scale_of(fig: Figure) -> float:
143
+ word = fig.literal.split()[-1].lower() if " " in fig.literal else _trailing_word(fig.literal)
144
+ return SCALES.get(word, 1.0)
145
+
146
+
147
+ def _trailing_word(literal: str) -> str:
148
+ m = re.search(r"[a-z]+$", literal, re.IGNORECASE)
149
+ return m.group(0).lower() if m else ""
figured/judge.py ADDED
@@ -0,0 +1,69 @@
1
+ """Optional second opinion from a model, for what arithmetic cannot see: wrong words around right numbers.
2
+
3
+ Requires the `judge` extra: pip install "figured[judge]". The judge receives the question, the
4
+ rows, and the answer, nothing else, and returns a strict verdict.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from typing import Any, Literal
11
+
12
+ JUDGE_SYSTEM = """You audit answers produced by a data assistant. You are given the user's question, the rows the assistant retrieved, and the assistant's answer. Judge strictly and only from what is shown.
13
+
14
+ faithful: every number in the answer is supported by the rows, allowing rounding and values derived from them (sums, differences, ratios, percentages), and every comparative word (higher, lower, most, fewer) agrees with the rows.
15
+ responsive: the answer addresses the question that was asked, or clearly explains what part cannot be answered.
16
+ caveats_ok: if the answer relies on an approximation or omits part of the question, it says so. True when no caveat was needed.
17
+ issues: short, specific strings; empty when none.
18
+ verdict: "pass" only if faithful and responsive are both true."""
19
+
20
+
21
+ def judge(
22
+ question: str,
23
+ answer: str,
24
+ rows: Any,
25
+ *,
26
+ model: str = "claude-opus-5",
27
+ client: Any = None,
28
+ max_rows: int = 30,
29
+ ) -> dict[str, Any]:
30
+ """Return {"verdict", "faithful", "responsive", "caveats_ok", "issues", "model"}."""
31
+ try:
32
+ import anthropic
33
+ from pydantic import BaseModel
34
+ except ImportError as exc: # pragma: no cover
35
+ raise ImportError('the judge needs the extra: pip install "figured[judge]"') from exc
36
+
37
+ class Verdict(BaseModel): # type: ignore[misc]
38
+ faithful: bool
39
+ responsive: bool
40
+ caveats_ok: bool
41
+ issues: list[str]
42
+ verdict: Literal["pass", "fail"]
43
+
44
+ from figured.evidence import build_evidence
45
+
46
+ ev = build_evidence(rows)
47
+ evidence = [
48
+ {
49
+ "columns": rs.columns,
50
+ "rows": [[c.value for c in row] for row in rs.rows[:max_rows]],
51
+ }
52
+ for rs in ev.results
53
+ ]
54
+ prompt = (
55
+ f"Question:\n{question}\n\nRetrieved rows (JSON):\n{json.dumps(evidence)[:12000]}"
56
+ f"\n\nAssistant answer:\n{answer}"
57
+ )
58
+ client = client or anthropic.Anthropic()
59
+ resp = client.messages.parse(
60
+ model=model,
61
+ max_tokens=600,
62
+ system=JUDGE_SYSTEM,
63
+ messages=[{"role": "user", "content": prompt}],
64
+ output_format=Verdict,
65
+ )
66
+ parsed = resp.parsed_output
67
+ if parsed is None:
68
+ return {"verdict": "error", "issues": ["no parsed output"], "model": model}
69
+ return {**parsed.model_dump(), "model": model}
figured/policy.py ADDED
@@ -0,0 +1,60 @@
1
+ """Tunable rules for what counts as grounded."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, replace
6
+ from typing import Any, Literal
7
+
8
+ DERIVATIONS = ("cell", "column_sum", "row_sum", "difference", "ratio", "percent", "percent_change")
9
+
10
+
11
+ @dataclass(frozen=True)
12
+ class Policy:
13
+ """How strictly to match, what to ignore, and which derivations to allow.
14
+
15
+ rel_tolerance: relative error allowed between a figure and a candidate (0.015 is 1.5%).
16
+ abs_tolerance: absolute error allowed in addition to the relative one.
17
+ ignore_below: figures at or below this absolute value are not checked (counts of items, rankings).
18
+ ignore_years: treat bare four-digit integers between year_range as years and skip them.
19
+ unmatched_percent: "pass" lets a percentage through when nothing matches, since shares of a
20
+ total outside the rows are common; "flag" treats it like any other figure.
21
+ max_rows / max_cells: how many rows and flat cells feed the pairwise derivations.
22
+ derivations: which candidate kinds are generated.
23
+ parse_strings: coerce numeric strings in the rows ("39,346,023", "$1,200", "12%").
24
+ flag_without_evidence_above: with no rows at all, figures above this are flagged.
25
+ """
26
+
27
+ rel_tolerance: float = 0.015
28
+ abs_tolerance: float = 0.0
29
+ ignore_below: float = 100.0
30
+ ignore_years: bool = True
31
+ year_range: tuple[int, int] = (1900, 2100)
32
+ unmatched_percent: Literal["pass", "flag"] = "pass"
33
+ max_rows: int = 12
34
+ max_cells: int = 40
35
+ derivations: frozenset[str] = frozenset(DERIVATIONS)
36
+ parse_strings: bool = True
37
+ flag_without_evidence_above: float = 1000.0
38
+
39
+ def with_overrides(self, **overrides: Any) -> Policy:
40
+ if not overrides:
41
+ return self
42
+ if "derivations" in overrides and not isinstance(overrides["derivations"], frozenset):
43
+ overrides["derivations"] = frozenset(overrides["derivations"])
44
+ unknown = set(overrides) - set(self.__dataclass_fields__)
45
+ if unknown:
46
+ raise TypeError(f"unknown policy option(s): {', '.join(sorted(unknown))}")
47
+ return replace(self, **overrides)
48
+
49
+ def close(self, a: float, b: float) -> bool:
50
+ if a == b:
51
+ return True
52
+ if abs(a - b) <= self.abs_tolerance:
53
+ return True
54
+ if b == 0:
55
+ return abs(a) <= self.abs_tolerance
56
+ return abs(a - b) / abs(b) <= self.rel_tolerance
57
+
58
+
59
+ STRICT = Policy(rel_tolerance=0.005, unmatched_percent="flag", ignore_below=10.0)
60
+ LENIENT = Policy(rel_tolerance=0.05)
figured/py.typed ADDED
File without changes
figured/report.py ADDED
@@ -0,0 +1,105 @@
1
+ """What came back: every figure, whether it traced, and how."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from typing import Any, Literal
7
+
8
+ from figured.derive import Match, fmt
9
+ from figured.extract import Figure
10
+
11
+ Status = Literal["grounded", "ungrounded", "ignored"]
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class Result:
16
+ figure: Figure
17
+ status: Status
18
+ match: Match | None = None
19
+ reason: str = ""
20
+
21
+ @property
22
+ def literal(self) -> str:
23
+ return self.figure.literal
24
+
25
+ def to_dict(self) -> dict[str, Any]:
26
+ d: dict[str, Any] = {
27
+ "literal": self.figure.literal,
28
+ "value": self.figure.value,
29
+ "span": [self.figure.start, self.figure.end],
30
+ "status": self.status,
31
+ }
32
+ if self.match:
33
+ d["match"] = {
34
+ "kind": self.match.kind,
35
+ "value": self.match.value,
36
+ "error": self.match.error,
37
+ "explanation": self.match.explanation,
38
+ }
39
+ if self.reason:
40
+ d["reason"] = self.reason
41
+ return d
42
+
43
+
44
+ @dataclass
45
+ class Report:
46
+ text: str
47
+ results: list[Result]
48
+ evidence_cells: int
49
+
50
+ @property
51
+ def ok(self) -> bool:
52
+ return not self.ungrounded
53
+
54
+ @property
55
+ def grounded(self) -> list[Result]:
56
+ return [r for r in self.results if r.status == "grounded"]
57
+
58
+ @property
59
+ def ungrounded(self) -> list[str]:
60
+ return [r.figure.literal for r in self.results if r.status == "ungrounded"]
61
+
62
+ @property
63
+ def ignored(self) -> list[Result]:
64
+ return [r for r in self.results if r.status == "ignored"]
65
+
66
+ @property
67
+ def checked(self) -> int:
68
+ return sum(1 for r in self.results if r.status != "ignored")
69
+
70
+ def caveat(self, limit: int = 4) -> str:
71
+ if self.ok:
72
+ return ""
73
+ shown = ", ".join(self.ungrounded[:limit])
74
+ more = len(self.ungrounded) - limit
75
+ tail = f" and {more} more" if more > 0 else ""
76
+ return (
77
+ f"Note: these figures could not be traced to the data: {shown}{tail}. Treat them as approximate."
78
+ )
79
+
80
+ def explain(self) -> str:
81
+ head = "OK" if self.ok else "UNGROUNDED"
82
+ lines = [f"{head} · {self.checked} checked · {len(self.ungrounded)} untraceable"]
83
+ for r in self.results:
84
+ if r.status == "grounded" and r.match:
85
+ lines.append(f" ✓ {r.literal:<16} {r.match.kind:<14} {r.match.explanation}")
86
+ elif r.status == "ungrounded":
87
+ lines.append(f" ✗ {r.literal:<16} no cell, sum, difference, or ratio within tolerance")
88
+ else:
89
+ lines.append(f" · {r.literal:<16} ignored ({r.reason})")
90
+ return "\n".join(lines)
91
+
92
+ def to_dict(self) -> dict[str, Any]:
93
+ return {
94
+ "ok": self.ok,
95
+ "checked": self.checked,
96
+ "ungrounded": self.ungrounded,
97
+ "evidence_cells": self.evidence_cells,
98
+ "figures": [r.to_dict() for r in self.results],
99
+ }
100
+
101
+ def __repr__(self) -> str:
102
+ return f"Report(ok={self.ok}, checked={self.checked}, ungrounded={self.ungrounded})"
103
+
104
+
105
+ __all__ = ["Report", "Result", "Status", "fmt"]
@@ -0,0 +1,211 @@
1
+ Metadata-Version: 2.5
2
+ Name: figured
3
+ Version: 0.1.0
4
+ Summary: Show your work: verify that every number in an LLM-generated answer traces to the rows it was derived from.
5
+ Project-URL: Homepage, https://github.com/nisheshshukla/figured
6
+ Project-URL: Repository, https://github.com/nisheshshukla/figured
7
+ Project-URL: Issues, https://github.com/nisheshshukla/figured/issues
8
+ Project-URL: Changelog, https://github.com/nisheshshukla/figured/blob/main/CHANGELOG.md
9
+ Author: Nishesh Shukla
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: evaluation,faithfulness,grounding,guardrails,hallucination,llm,text-to-sql
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
22
+ Classifier: Topic :: Software Development :: Quality Assurance
23
+ Classifier: Typing :: Typed
24
+ Requires-Python: >=3.10
25
+ Provides-Extra: dev
26
+ Requires-Dist: hypothesis>=6.100; extra == 'dev'
27
+ Requires-Dist: mypy>=1.11; extra == 'dev'
28
+ Requires-Dist: pytest-cov>=5; extra == 'dev'
29
+ Requires-Dist: pytest>=8; extra == 'dev'
30
+ Requires-Dist: ruff>=0.6; extra == 'dev'
31
+ Provides-Extra: judge
32
+ Requires-Dist: anthropic>=1.0; extra == 'judge'
33
+ Requires-Dist: pydantic>=2.0; extra == 'judge'
34
+ Description-Content-Type: text/markdown
35
+
36
+ # figured
37
+
38
+ **Show your work.** Verify that every number in an LLM-generated answer traces to the rows it was written from.
39
+
40
+ ```python
41
+ from figured import trace
42
+
43
+ rows = [{"state": "California", "pop": 39_346_023}, {"state": "Texas", "pop": 28_635_442}]
44
+ answer = "California has 39.3 million people, about 10.7 million more than Texas, and 4.1 million of them moved last year."
45
+
46
+ report = trace(answer, rows)
47
+ report.ok # False
48
+ report.ungrounded # ['4.1 million']
49
+ print(report.explain())
50
+ ```
51
+
52
+ ```
53
+ UNGROUNDED · 3 checked · 1 untraceable
54
+ ✓ 39.3 million cell pop[California] = 39,346,023
55
+ ✓ 10.7 million difference pop[California] − pop[Texas] = 39,346,023 − 28,635,442 = 10,710,581
56
+ ✗ 4.1 million no cell, sum, difference, or ratio within tolerance
57
+ ```
58
+
59
+ Zero dependencies. Deterministic. About 150 µs for a typical answer, 1 ms for 200 rows. Python 3.10+.
60
+
61
+ ```bash
62
+ pip install figured
63
+ ```
64
+
65
+ ## Why
66
+
67
+ Text-to-SQL agents and RAG-over-tables pipelines validate the query and trust the prose. The model reads the rows and writes a paragraph, and nothing checks that the paragraph's numbers came from the rows. When it invents a figure, the SQL was fine, the rows were fine, and the user sees a confident wrong number.
68
+
69
+ The usual answer is an LLM judge, which is slow, costs money per answer, and is itself wrong sometimes: in one published test, a faithfulness metric scored a fabricated price as fully faithful five times in a row. `figured` is the deterministic check that runs on every answer before a judge is needed. It is the "grounding" step the authors of this library shipped inside a Census data agent, extracted so anyone can use it.
70
+
71
+ ## What counts as grounded
72
+
73
+ Every substantive number in the text must be within a tolerance (default 1.5 percent) of something the rows could legitimately produce:
74
+
75
+ | Derivation | Example | Explanation you get back |
76
+ |---|---|---|
77
+ | cell | "39,346,023 people" | `pop[California] = 39,346,023` |
78
+ | column sum | "together, 1,000,000 residents" | `sum of pop over 3 rows = 1,000,000` |
79
+ | adjacent-cell sum | "the three youngest bands total 1,200" | `a..c[row 0] summed = 1,200` |
80
+ | difference | "10.7 million more than Texas" | `pop[California] − pop[Texas] = ... = 10,710,581` |
81
+ | ratio | "3.0 to one" | `a[row 0] ÷ b[row 0] = 3` |
82
+ | percent | "72.8% of California" | `pop[Texas] ÷ pop[California] = 72.8%` |
83
+ | percent change | "grew 2.3%" | `(y2020 − y2019) ÷ y2019 = 2.3%` |
84
+
85
+ Differences, ratios, and percentages are searched within a row and across rows. A stated range such as "between 39 and 40 million" is grounded when a candidate lies inside it. Numbers at or below 100 and bare four-digit years are ignored by default, because "top 5 counties in 2020" is not a claim about the data.
86
+
87
+ Two rules keep the search honest. A figure written as a percentage is searched as `a ÷ b × 100`, and a plain figure as `a ÷ b`, never both, so "150" cannot pass by coincidentally matching a 150% share. And the pairwise and adjacent-cell derivations cover the first `max_rows` rows (12 by default), which is the part of a result a model has usually read; cells and column sums cover every row. Raise `max_rows` if your prompt includes more.
88
+
89
+ Each grounded figure carries the derivation that matched, so a reviewer can check it by hand. Each ungrounded figure is named. Nothing blocks: you decide whether to append the caveat, change a badge, or fail a test.
90
+
91
+ ## What it reads
92
+
93
+ `trace(text, rows)` accepts the rows in whatever shape you already have:
94
+
95
+ - a list of dicts, as most drivers and ORMs return
96
+ - a list of lists or tuples, with or without column names
97
+ - a `{"columns": [...], "rows": [...]}` mapping
98
+ - a pandas DataFrame
99
+ - a DB-API cursor after `execute`
100
+ - several result sets at once: `trace(text, results=[rows_a, rows_b])`
101
+
102
+ Numeric strings in the rows are parsed by default, so `"39,346,023"`, `"$1,200"`, and `"12%"` all count. Decimals from database drivers are handled. Booleans are not numbers.
103
+
104
+ ## Text it understands
105
+
106
+ Thousands separators, decimals, scientific notation (`1.2e6`), currency symbols, scale words (`39.3 million`, `2.5bn`, `3k`), percent markers (`12%`, `12 percent`, `3 percentage points`), negatives, and ranges with a shared unit (`40 to 50 million`). Identifiers such as `B01003e1` or request ids are not mistaken for numbers, and ordinals are skipped.
107
+
108
+ ## Tuning
109
+
110
+ ```python
111
+ from figured import trace, Policy, STRICT, LENIENT
112
+
113
+ trace(answer, rows, rel_tolerance=0.005) # tighter rounding
114
+ trace(answer, rows, unmatched_percent="flag") # a percentage must match something
115
+ trace(answer, rows, derivations={"cell", "column_sum"}) # no pairwise arithmetic
116
+ trace(answer, rows, policy=STRICT) # 0.5%, percentages must match, checks down to 10
117
+ trace(answer, rows, ignore_below=0, ignore_years=False)
118
+ ```
119
+
120
+ | Option | Default | Meaning |
121
+ |---|---|---|
122
+ | `rel_tolerance` | 0.015 | relative error allowed, covers rounding to three significant figures |
123
+ | `abs_tolerance` | 0 | absolute error allowed in addition |
124
+ | `ignore_below` | 100 | figures at or below this are counts of things, not claims |
125
+ | `ignore_years` | True | bare four-digit integers in `year_range` are skipped |
126
+ | `unmatched_percent` | "pass" | shares of totals outside the rows are common, so a lone percentage passes |
127
+ | `max_rows`, `max_cells` | 12, 40 | how much of the result feeds the pairwise and adjacent-sum search |
128
+ | `derivations` | all seven | which candidate kinds are generated |
129
+ | `parse_strings` | True | coerce numeric strings in the rows |
130
+
131
+ ## Speed
132
+
133
+ Measured with `python benchmarks/bench.py` on a laptop, one answer with nine figures:
134
+
135
+ | Result set | Time per check |
136
+ |---|---|
137
+ | 2 rows × 3 columns | 150 µs |
138
+ | 12 rows × 5 columns | 360 µs |
139
+ | 200 rows × 10 columns | 1.1 ms |
140
+ | 2,000 rows × 10 columns | 9 ms |
141
+
142
+ Nothing is enumerated up front. Cells and column sums are indexed once; differences, ratios, percentages, and percent changes are found per figure by solving for the partner cell and bisecting for it. Explanations are formatted only for the figure that matched. For comparison, a model-based faithfulness judge takes seconds and costs a request.
143
+
144
+ ## Command line
145
+
146
+ ```bash
147
+ figured "California has 39.3 million people." --rows rows.json
148
+ figured - --rows rows.json < answer.txt
149
+ figured "..." --rows rows.json --json --tolerance 0.01 --strict-percent
150
+ ```
151
+
152
+ Exit code 1 when any figure is untraceable, so it can gate a pipeline step.
153
+
154
+ ## Using it in a pipeline
155
+
156
+ **After every answer**, append the caveat and flip a badge:
157
+
158
+ ```python
159
+ report = trace(answer, rows)
160
+ if not report.ok:
161
+ answer += "\n\n" + report.caveat()
162
+ badge = "check figures"
163
+ ```
164
+
165
+ **In promptfoo**, as a Python assertion: see `examples/promptfoo_assert.py`.
166
+
167
+ **In DeepEval or any custom metric**, wrap `trace` and return `1 - len(report.ungrounded) / report.checked`.
168
+
169
+ **With a model judge for the rest.** Arithmetic cannot see a wrong word around a right number: "Nevada is richer than Utah" with the two correct medians reversed passes. The optional `judge` extra sends the question, the rows, and the answer to a model and returns a strict verdict on faithfulness, responsiveness, and caveats:
170
+
171
+ ```bash
172
+ pip install "figured[judge]"
173
+ ```
174
+
175
+ ```python
176
+ from figured.judge import judge
177
+
178
+ judge("Which state is richer?", answer, rows) # {"verdict": "fail", "issues": ["comparison reversed"], ...}
179
+ ```
180
+
181
+ ## What it does not do
182
+
183
+ - It cannot catch a correct number attached to the wrong claim. That is what the judge extra is for.
184
+ - With large result sets the derived set is big, and a hallucinated figure can land within tolerance of some difference by coincidence. The defaults cap the pairwise search at 12 rows and 40 cells; tighten the tolerance or restrict `derivations` for sensitive uses. A flag on a correct figure is treated as the worse error, because people stop reading badges that cry wolf.
185
+ - Numbers written as words ("two million") are not extracted.
186
+ - It does not know what the rows mean. If the agent queried the wrong column and described it faithfully, every figure traces.
187
+
188
+ ## How it compares
189
+
190
+ | | rows as evidence | derived arithmetic | deterministic | names each figure | packaged |
191
+ |---|---|---|---|---|---|
192
+ | **figured** | yes | sums, differences, ratios, percentages, ranges | yes | yes, with the derivation | pip, zero deps |
193
+ | llmground | no, a source string | no | yes | yes | pip |
194
+ | @demystify/grounding | no, cited facts | no | yes | yes | npm |
195
+ | pcn-core (Proof-Carrying Numbers) | claim values you supply | no | yes | yes, needs model-emitted tags | pip |
196
+ | NumProof | yes | yes | yes | yes | hosted API |
197
+ | DeepEval / Ragas faithfulness | text context | n/a | no, LLM or NLI | no | pip |
198
+
199
+ The Proof-Carrying Numbers policy vocabulary (exact, rounded, scale alias, tolerance, percent, range, year) is the clearest statement of the matching problem, and this library borrows its shape. The difference is the evidence contract: rows in, free text in, no cooperation from the model required.
200
+
201
+ ## Ports
202
+
203
+ Behavior is pinned by the conformance vectors in `tests/vectors/`. A port in another language is correct when it passes them unchanged. A TypeScript port is the natural next one; open an issue if you want to take it.
204
+
205
+ ## Origin
206
+
207
+ Built inside a Census data agent whose answers had to be traceable to the ACS rows behind them. The first version only derived values within a row, so a correct "about $10,900 higher" comparison across two state rows was flagged as suspect. That false flag is now a named test vector, and it is why the defaults lean toward trusting the model when the arithmetic works out.
208
+
209
+ ## License
210
+
211
+ MIT.
@@ -0,0 +1,15 @@
1
+ figured/__init__.py,sha256=KfstOh4RrhIVyLWctCyq-mqm3ViEQPTINEo5wKVagyM,801
2
+ figured/__main__.py,sha256=X5TxIAVDwbQeZZIKqSIRL9-o_7TyfR69olsPbLxit_U,1561
3
+ figured/core.py,sha256=TSQZGNvTmpltU0CihpaFU6YAl5761tI52xb0GsJvn4Q,2576
4
+ figured/derive.py,sha256=idVOAY57JFrkEwCHZzvZz3_e7OcNVyX8vur3zZCriks,13774
5
+ figured/evidence.py,sha256=b2_aqiqjAc9o2IwSnq1qQS9NIQLVzuJb6YQAANxiYFM,6652
6
+ figured/extract.py,sha256=WyWxPX6edy-tvMJtTAnJuvJx2VWaJmlnvuB3Rx862Us,4632
7
+ figured/judge.py,sha256=H36V2fsR6uNZeX7iP1vLVSQUocxQJ8LGNRLv2wSLZCs,2677
8
+ figured/policy.py,sha256=ScBM63WDvUZO2Q0PEfiB5QBUzSAZzlMuNnyKUW9TpFQ,2557
9
+ figured/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
+ figured/report.py,sha256=nrA-Uic7OmhU9cNAbPP8CW48nk9NdGlanhnKy55Fhog,3213
11
+ figured-0.1.0.dist-info/METADATA,sha256=c4hYUZmSJPfK48QQNlXSSBjJ0K7MjorHqyJl7xoZ-5s,11207
12
+ figured-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
13
+ figured-0.1.0.dist-info/entry_points.txt,sha256=mh91KVuIvpJy6DIGjjsmEcuJaOGMoldaFyyRXdSgW7Q,50
14
+ figured-0.1.0.dist-info/licenses/LICENSE,sha256=SHGKk_adXxNMC1rM63eWqTyDr6nbLVzVZcMi7OReWNA,1071
15
+ figured-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ figured = figured.__main__:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Nishesh Shukla
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.