figured 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,23 @@
1
+ name: ci
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ test:
10
+ runs-on: ubuntu-latest
11
+ strategy:
12
+ matrix:
13
+ python: ["3.10", "3.11", "3.12", "3.13"]
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: actions/setup-python@v5
17
+ with:
18
+ python-version: ${{ matrix.python }}
19
+ - run: pip install -e ".[dev]"
20
+ - run: ruff check .
21
+ - run: ruff format --check .
22
+ - run: mypy
23
+ - run: pytest --cov=figured --cov-report=term-missing --cov-fail-under=95
@@ -0,0 +1,39 @@
1
+ name: release
2
+
3
+ on:
4
+ push:
5
+ tags: ["v*"]
6
+
7
+ jobs:
8
+ build:
9
+ runs-on: ubuntu-latest
10
+ steps:
11
+ - uses: actions/checkout@v4
12
+ - uses: actions/setup-python@v5
13
+ with:
14
+ python-version: "3.12"
15
+ - name: Check that the tag matches the package version
16
+ run: |
17
+ tag="${GITHUB_REF_NAME#v}"
18
+ version=$(python -c "import tomllib; print(tomllib.load(open('pyproject.toml','rb'))['project']['version'])")
19
+ test "$tag" = "$version" || { echo "tag v$tag does not match version $version"; exit 1; }
20
+ - run: pip install -e ".[dev]" build
21
+ - run: ruff check . && mypy && pytest -q
22
+ - run: python -m build
23
+ - uses: actions/upload-artifact@v4
24
+ with:
25
+ name: dist
26
+ path: dist/
27
+
28
+ publish:
29
+ needs: build
30
+ runs-on: ubuntu-latest
31
+ environment: pypi
32
+ permissions:
33
+ id-token: write
34
+ steps:
35
+ - uses: actions/download-artifact@v4
36
+ with:
37
+ name: dist
38
+ path: dist/
39
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,15 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ venv/
5
+ dist/
6
+ build/
7
+ *.egg-info/
8
+ .pytest_cache/
9
+ .mypy_cache/
10
+ .ruff_cache/
11
+ .hypothesis/
12
+ .coverage
13
+ coverage.xml
14
+ htmlcov/
15
+ .DS_Store
@@ -0,0 +1,13 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0
4
+
5
+ First release.
6
+
7
+ - `trace(text, rows)` checks every number in a text against the rows it was written from.
8
+ - Derived values: column sums, adjacent-cell sums, differences, ratios, percentages, and percent change, within and across rows.
9
+ - Number extraction with thousands separators, decimals, scientific notation, currency, scale words, percent markers, and ranges.
10
+ - Evidence from lists of dicts, lists of sequences, `{columns, rows}` mappings, pandas DataFrames, and DB-API cursors, with numeric-string parsing.
11
+ - Per-figure report with the matching derivation spelled out, a one-line caveat, JSON output, and a CLI that exits non-zero on untraceable figures.
12
+ - Optional `[judge]` extra for a model-based second opinion on comparative words.
13
+ - Language-neutral conformance vectors in `tests/vectors/`.
@@ -0,0 +1,40 @@
1
+ # Contributing
2
+
3
+ ```bash
4
+ git clone https://github.com/nisheshshukla/figured
5
+ cd figured
6
+ python -m venv .venv && source .venv/bin/activate
7
+ pip install -e ".[dev]"
8
+ pytest
9
+ ```
10
+
11
+ CI runs ruff, mypy in strict mode, and pytest with a 95 percent coverage gate on Python 3.10 through 3.13.
12
+
13
+ `python benchmarks/bench.py` prints the time per check at several result-set sizes. A change that makes the typical case slower needs a reason in the pull request.
14
+
15
+ ## Adding a case
16
+
17
+ Most behavior changes should start as a conformance vector in `tests/vectors/`. Each vector is a text, some rows, an optional policy, and what the report must say. Vectors are language-neutral so that ports in other languages can be held to the same behavior.
18
+
19
+ ## Design rules
20
+
21
+ - No runtime dependencies. Optional integrations live behind extras.
22
+ - The check must stay deterministic: no model calls, no clock, no network inside `trace`.
23
+ - A false flag is worse than a miss. When in doubt, add a derivation or a tolerance knob rather than a stricter default.
24
+ - Every grounded figure must carry an explanation a person can verify by hand.
25
+
26
+ ## Releasing
27
+
28
+ Releases are published to PyPI by the `release` workflow through trusted publishing, so no API token is stored anywhere.
29
+
30
+ One-time setup, by the repository owner:
31
+
32
+ 1. On pypi.org, under your account's Publishing settings, add a pending publisher: project `figured`, owner `nisheshshukla`, repository `figured`, workflow `release.yml`, environment `pypi`.
33
+ 2. In the GitHub repository settings, create an environment named `pypi`.
34
+
35
+ Each release:
36
+
37
+ 1. Bump `version` in `pyproject.toml` and add a section to `CHANGELOG.md`.
38
+ 2. Commit, then tag and push: `git tag v0.1.0 && git push origin main --tags`.
39
+
40
+ The workflow refuses to publish when the tag and the package version disagree.
figured-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Nishesh Shukla
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
figured-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,211 @@
1
+ Metadata-Version: 2.5
2
+ Name: figured
3
+ Version: 0.1.0
4
+ Summary: Show your work: verify that every number in an LLM-generated answer traces to the rows it was derived from.
5
+ Project-URL: Homepage, https://github.com/nisheshshukla/figured
6
+ Project-URL: Repository, https://github.com/nisheshshukla/figured
7
+ Project-URL: Issues, https://github.com/nisheshshukla/figured/issues
8
+ Project-URL: Changelog, https://github.com/nisheshshukla/figured/blob/main/CHANGELOG.md
9
+ Author: Nishesh Shukla
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: evaluation,faithfulness,grounding,guardrails,hallucination,llm,text-to-sql
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
22
+ Classifier: Topic :: Software Development :: Quality Assurance
23
+ Classifier: Typing :: Typed
24
+ Requires-Python: >=3.10
25
+ Provides-Extra: dev
26
+ Requires-Dist: hypothesis>=6.100; extra == 'dev'
27
+ Requires-Dist: mypy>=1.11; extra == 'dev'
28
+ Requires-Dist: pytest-cov>=5; extra == 'dev'
29
+ Requires-Dist: pytest>=8; extra == 'dev'
30
+ Requires-Dist: ruff>=0.6; extra == 'dev'
31
+ Provides-Extra: judge
32
+ Requires-Dist: anthropic>=1.0; extra == 'judge'
33
+ Requires-Dist: pydantic>=2.0; extra == 'judge'
34
+ Description-Content-Type: text/markdown
35
+
36
+ # figured
37
+
38
+ **Show your work.** Verify that every number in an LLM-generated answer traces to the rows it was written from.
39
+
40
+ ```python
41
+ from figured import trace
42
+
43
+ rows = [{"state": "California", "pop": 39_346_023}, {"state": "Texas", "pop": 28_635_442}]
44
+ answer = "California has 39.3 million people, about 10.7 million more than Texas, and 4.1 million of them moved last year."
45
+
46
+ report = trace(answer, rows)
47
+ report.ok # False
48
+ report.ungrounded # ['4.1 million']
49
+ print(report.explain())
50
+ ```
51
+
52
+ ```
53
+ UNGROUNDED · 3 checked · 1 untraceable
54
+ ✓ 39.3 million cell pop[California] = 39,346,023
55
+ ✓ 10.7 million difference pop[California] − pop[Texas] = 39,346,023 − 28,635,442 = 10,710,581
56
+ ✗ 4.1 million no cell, sum, difference, or ratio within tolerance
57
+ ```
58
+
59
+ Zero dependencies. Deterministic. About 150 µs for a typical answer, 1 ms for 200 rows. Python 3.10+.
60
+
61
+ ```bash
62
+ pip install figured
63
+ ```
64
+
65
+ ## Why
66
+
67
+ Text-to-SQL agents and RAG-over-tables pipelines validate the query and trust the prose. The model reads the rows and writes a paragraph, and nothing checks that the paragraph's numbers came from the rows. When it invents a figure, the SQL was fine, the rows were fine, and the user sees a confident wrong number.
68
+
69
+ The usual answer is an LLM judge, which is slow, costs money per answer, and is itself wrong sometimes: in one published test, a faithfulness metric scored a fabricated price as fully faithful five times in a row. `figured` is the deterministic check that runs on every answer before a judge is needed. It is the "grounding" step the authors of this library shipped inside a Census data agent, extracted so anyone can use it.
70
+
71
+ ## What counts as grounded
72
+
73
+ Every substantive number in the text must be within a tolerance (default 1.5 percent) of something the rows could legitimately produce:
74
+
75
+ | Derivation | Example | Explanation you get back |
76
+ |---|---|---|
77
+ | cell | "39,346,023 people" | `pop[California] = 39,346,023` |
78
+ | column sum | "together, 1,000,000 residents" | `sum of pop over 3 rows = 1,000,000` |
79
+ | adjacent-cell sum | "the three youngest bands total 1,200" | `a..c[row 0] summed = 1,200` |
80
+ | difference | "10.7 million more than Texas" | `pop[California] − pop[Texas] = ... = 10,710,581` |
81
+ | ratio | "3.0 to one" | `a[row 0] ÷ b[row 0] = 3` |
82
+ | percent | "72.8% of California" | `pop[Texas] ÷ pop[California] = 72.8%` |
83
+ | percent change | "grew 2.3%" | `(y2020 − y2019) ÷ y2019 = 2.3%` |
84
+
85
+ Differences, ratios, and percentages are searched within a row and across rows. A stated range such as "between 39 and 40 million" is grounded when a candidate lies inside it. Numbers at or below 100 and bare four-digit years are ignored by default, because "top 5 counties in 2020" is not a claim about the data.
86
+
87
+ Two rules keep the search honest. A figure written as a percentage is searched as `a ÷ b × 100`, and a plain figure as `a ÷ b`, never both, so "150" cannot pass by coincidentally matching a 150% share. And the pairwise and adjacent-cell derivations cover the first `max_rows` rows (12 by default), which is the part of a result a model has usually read; cells and column sums cover every row. Raise `max_rows` if your prompt includes more.
88
+
89
+ Each grounded figure carries the derivation that matched, so a reviewer can check it by hand. Each ungrounded figure is named. Nothing blocks: you decide whether to append the caveat, change a badge, or fail a test.
90
+
91
+ ## What it reads
92
+
93
+ `trace(text, rows)` accepts the rows in whatever shape you already have:
94
+
95
+ - a list of dicts, as most drivers and ORMs return
96
+ - a list of lists or tuples, with or without column names
97
+ - a `{"columns": [...], "rows": [...]}` mapping
98
+ - a pandas DataFrame
99
+ - a DB-API cursor after `execute`
100
+ - several result sets at once: `trace(text, results=[rows_a, rows_b])`
101
+
102
+ Numeric strings in the rows are parsed by default, so `"39,346,023"`, `"$1,200"`, and `"12%"` all count. Decimals from database drivers are handled. Booleans are not numbers.
103
+
104
+ ## Text it understands
105
+
106
+ Thousands separators, decimals, scientific notation (`1.2e6`), currency symbols, scale words (`39.3 million`, `2.5bn`, `3k`), percent markers (`12%`, `12 percent`, `3 percentage points`), negatives, and ranges with a shared unit (`40 to 50 million`). Identifiers such as `B01003e1` or request ids are not mistaken for numbers, and ordinals are skipped.
107
+
108
+ ## Tuning
109
+
110
+ ```python
111
+ from figured import trace, Policy, STRICT, LENIENT
112
+
113
+ trace(answer, rows, rel_tolerance=0.005) # tighter rounding
114
+ trace(answer, rows, unmatched_percent="flag") # a percentage must match something
115
+ trace(answer, rows, derivations={"cell", "column_sum"}) # no pairwise arithmetic
116
+ trace(answer, rows, policy=STRICT) # 0.5%, percentages must match, checks down to 10
117
+ trace(answer, rows, ignore_below=0, ignore_years=False)
118
+ ```
119
+
120
+ | Option | Default | Meaning |
121
+ |---|---|---|
122
+ | `rel_tolerance` | 0.015 | relative error allowed, covers rounding to three significant figures |
123
+ | `abs_tolerance` | 0 | absolute error allowed in addition |
124
+ | `ignore_below` | 100 | figures at or below this are counts of things, not claims |
125
+ | `ignore_years` | True | bare four-digit integers in `year_range` are skipped |
126
+ | `unmatched_percent` | "pass" | shares of totals outside the rows are common, so a lone percentage passes |
127
+ | `max_rows`, `max_cells` | 12, 40 | how much of the result feeds the pairwise and adjacent-sum search |
128
+ | `derivations` | all seven | which candidate kinds are generated |
129
+ | `parse_strings` | True | coerce numeric strings in the rows |
130
+
131
+ ## Speed
132
+
133
+ Measured with `python benchmarks/bench.py` on a laptop, one answer with nine figures:
134
+
135
+ | Result set | Time per check |
136
+ |---|---|
137
+ | 2 rows × 3 columns | 150 µs |
138
+ | 12 rows × 5 columns | 360 µs |
139
+ | 200 rows × 10 columns | 1.1 ms |
140
+ | 2,000 rows × 10 columns | 9 ms |
141
+
142
+ Nothing is enumerated up front. Cells and column sums are indexed once; differences, ratios, percentages, and percent changes are found per figure by solving for the partner cell and bisecting for it. Explanations are formatted only for the figure that matched. For comparison, a model-based faithfulness judge takes seconds and costs a request.
143
+
144
+ ## Command line
145
+
146
+ ```bash
147
+ figured "California has 39.3 million people." --rows rows.json
148
+ figured - --rows rows.json < answer.txt
149
+ figured "..." --rows rows.json --json --tolerance 0.01 --strict-percent
150
+ ```
151
+
152
+ Exit code 1 when any figure is untraceable, so it can gate a pipeline step.
153
+
154
+ ## Using it in a pipeline
155
+
156
+ **After every answer**, append the caveat and flip a badge:
157
+
158
+ ```python
159
+ report = trace(answer, rows)
160
+ if not report.ok:
161
+ answer += "\n\n" + report.caveat()
162
+ badge = "check figures"
163
+ ```
164
+
165
+ **In promptfoo**, as a Python assertion: see `examples/promptfoo_assert.py`.
166
+
167
+ **In DeepEval or any custom metric**, wrap `trace` and return `1 - len(report.ungrounded) / report.checked`.
168
+
169
+ **With a model judge for the rest.** Arithmetic cannot see a wrong word around a right number: "Nevada is richer than Utah" with the two correct medians reversed passes. The optional `judge` extra sends the question, the rows, and the answer to a model and returns a strict verdict on faithfulness, responsiveness, and caveats:
170
+
171
+ ```bash
172
+ pip install "figured[judge]"
173
+ ```
174
+
175
+ ```python
176
+ from figured.judge import judge
177
+
178
+ judge("Which state is richer?", answer, rows) # {"verdict": "fail", "issues": ["comparison reversed"], ...}
179
+ ```
180
+
181
+ ## What it does not do
182
+
183
+ - It cannot catch a correct number attached to the wrong claim. That is what the judge extra is for.
184
+ - With large result sets the derived set is big, and a hallucinated figure can land within tolerance of some difference by coincidence. The defaults cap the pairwise search at 12 rows and 40 cells; tighten the tolerance or restrict `derivations` for sensitive uses. A flag on a correct figure is treated as the worse error, because people stop reading badges that cry wolf.
185
+ - Numbers written as words ("two million") are not extracted.
186
+ - It does not know what the rows mean. If the agent queried the wrong column and described it faithfully, every figure traces.
187
+
188
+ ## How it compares
189
+
190
+ | | rows as evidence | derived arithmetic | deterministic | names each figure | packaged |
191
+ |---|---|---|---|---|---|
192
+ | **figured** | yes | sums, differences, ratios, percentages, ranges | yes | yes, with the derivation | pip, zero deps |
193
+ | llmground | no, a source string | no | yes | yes | pip |
194
+ | @demystify/grounding | no, cited facts | no | yes | yes | npm |
195
+ | pcn-core (Proof-Carrying Numbers) | claim values you supply | no | yes | yes, needs model-emitted tags | pip |
196
+ | NumProof | yes | yes | yes | yes | hosted API |
197
+ | DeepEval / Ragas faithfulness | text context | n/a | no, LLM or NLI | no | pip |
198
+
199
+ The Proof-Carrying Numbers policy vocabulary (exact, rounded, scale alias, tolerance, percent, range, year) is the clearest statement of the matching problem, and this library borrows its shape. The difference is the evidence contract: rows in, free text in, no cooperation from the model required.
200
+
201
+ ## Ports
202
+
203
+ Behavior is pinned by the conformance vectors in `tests/vectors/`. A port in another language is correct when it passes them unchanged. A TypeScript port is the natural next one; open an issue if you want to take it.
204
+
205
+ ## Origin
206
+
207
+ Built inside a Census data agent whose answers had to be traceable to the ACS rows behind them. The first version only derived values within a row, so a correct "about $10,900 higher" comparison across two state rows was flagged as suspect. That false flag is now a named test vector, and it is why the defaults lean toward trusting the model when the arithmetic works out.
208
+
209
+ ## License
210
+
211
+ MIT.
@@ -0,0 +1,176 @@
1
+ # figured
2
+
3
+ **Show your work.** Verify that every number in an LLM-generated answer traces to the rows it was written from.
4
+
5
+ ```python
6
+ from figured import trace
7
+
8
+ rows = [{"state": "California", "pop": 39_346_023}, {"state": "Texas", "pop": 28_635_442}]
9
+ answer = "California has 39.3 million people, about 10.7 million more than Texas, and 4.1 million of them moved last year."
10
+
11
+ report = trace(answer, rows)
12
+ report.ok # False
13
+ report.ungrounded # ['4.1 million']
14
+ print(report.explain())
15
+ ```
16
+
17
+ ```
18
+ UNGROUNDED · 3 checked · 1 untraceable
19
+ ✓ 39.3 million cell pop[California] = 39,346,023
20
+ ✓ 10.7 million difference pop[California] − pop[Texas] = 39,346,023 − 28,635,442 = 10,710,581
21
+ ✗ 4.1 million no cell, sum, difference, or ratio within tolerance
22
+ ```
23
+
24
+ Zero dependencies. Deterministic. About 150 µs for a typical answer, 1 ms for 200 rows. Python 3.10+.
25
+
26
+ ```bash
27
+ pip install figured
28
+ ```
29
+
30
+ ## Why
31
+
32
+ Text-to-SQL agents and RAG-over-tables pipelines validate the query and trust the prose. The model reads the rows and writes a paragraph, and nothing checks that the paragraph's numbers came from the rows. When it invents a figure, the SQL was fine, the rows were fine, and the user sees a confident wrong number.
33
+
34
+ The usual answer is an LLM judge, which is slow, costs money per answer, and is itself wrong sometimes: in one published test, a faithfulness metric scored a fabricated price as fully faithful five times in a row. `figured` is the deterministic check that runs on every answer before a judge is needed. It is the "grounding" step the authors of this library shipped inside a Census data agent, extracted so anyone can use it.
35
+
36
+ ## What counts as grounded
37
+
38
+ Every substantive number in the text must be within a tolerance (default 1.5 percent) of something the rows could legitimately produce:
39
+
40
+ | Derivation | Example | Explanation you get back |
41
+ |---|---|---|
42
+ | cell | "39,346,023 people" | `pop[California] = 39,346,023` |
43
+ | column sum | "together, 1,000,000 residents" | `sum of pop over 3 rows = 1,000,000` |
44
+ | adjacent-cell sum | "the three youngest bands total 1,200" | `a..c[row 0] summed = 1,200` |
45
+ | difference | "10.7 million more than Texas" | `pop[California] − pop[Texas] = ... = 10,710,581` |
46
+ | ratio | "3.0 to one" | `a[row 0] ÷ b[row 0] = 3` |
47
+ | percent | "72.8% of California" | `pop[Texas] ÷ pop[California] = 72.8%` |
48
+ | percent change | "grew 2.3%" | `(y2020 − y2019) ÷ y2019 = 2.3%` |
49
+
50
+ Differences, ratios, and percentages are searched within a row and across rows. A stated range such as "between 39 and 40 million" is grounded when a candidate lies inside it. Numbers at or below 100 and bare four-digit years are ignored by default, because "top 5 counties in 2020" is not a claim about the data.
51
+
52
+ Two rules keep the search honest. A figure written as a percentage is searched as `a ÷ b × 100`, and a plain figure as `a ÷ b`, never both, so "150" cannot pass by coincidentally matching a 150% share. And the pairwise and adjacent-cell derivations cover the first `max_rows` rows (12 by default), which is the part of a result a model has usually read; cells and column sums cover every row. Raise `max_rows` if your prompt includes more.
53
+
54
+ Each grounded figure carries the derivation that matched, so a reviewer can check it by hand. Each ungrounded figure is named. Nothing blocks: you decide whether to append the caveat, change a badge, or fail a test.
55
+
56
+ ## What it reads
57
+
58
+ `trace(text, rows)` accepts the rows in whatever shape you already have:
59
+
60
+ - a list of dicts, as most drivers and ORMs return
61
+ - a list of lists or tuples, with or without column names
62
+ - a `{"columns": [...], "rows": [...]}` mapping
63
+ - a pandas DataFrame
64
+ - a DB-API cursor after `execute`
65
+ - several result sets at once: `trace(text, results=[rows_a, rows_b])`
66
+
67
+ Numeric strings in the rows are parsed by default, so `"39,346,023"`, `"$1,200"`, and `"12%"` all count. Decimals from database drivers are handled. Booleans are not numbers.
68
+
69
+ ## Text it understands
70
+
71
+ Thousands separators, decimals, scientific notation (`1.2e6`), currency symbols, scale words (`39.3 million`, `2.5bn`, `3k`), percent markers (`12%`, `12 percent`, `3 percentage points`), negatives, and ranges with a shared unit (`40 to 50 million`). Identifiers such as `B01003e1` or request ids are not mistaken for numbers, and ordinals are skipped.
72
+
73
+ ## Tuning
74
+
75
+ ```python
76
+ from figured import trace, Policy, STRICT, LENIENT
77
+
78
+ trace(answer, rows, rel_tolerance=0.005) # tighter rounding
79
+ trace(answer, rows, unmatched_percent="flag") # a percentage must match something
80
+ trace(answer, rows, derivations={"cell", "column_sum"}) # no pairwise arithmetic
81
+ trace(answer, rows, policy=STRICT) # 0.5%, percentages must match, checks down to 10
82
+ trace(answer, rows, ignore_below=0, ignore_years=False)
83
+ ```
84
+
85
+ | Option | Default | Meaning |
86
+ |---|---|---|
87
+ | `rel_tolerance` | 0.015 | relative error allowed, covers rounding to three significant figures |
88
+ | `abs_tolerance` | 0 | absolute error allowed in addition |
89
+ | `ignore_below` | 100 | figures at or below this are counts of things, not claims |
90
+ | `ignore_years` | True | bare four-digit integers in `year_range` are skipped |
91
+ | `unmatched_percent` | "pass" | shares of totals outside the rows are common, so a lone percentage passes |
92
+ | `max_rows`, `max_cells` | 12, 40 | how much of the result feeds the pairwise and adjacent-sum search |
93
+ | `derivations` | all seven | which candidate kinds are generated |
94
+ | `parse_strings` | True | coerce numeric strings in the rows |
95
+
96
+ ## Speed
97
+
98
+ Measured with `python benchmarks/bench.py` on a laptop, one answer with nine figures:
99
+
100
+ | Result set | Time per check |
101
+ |---|---|
102
+ | 2 rows × 3 columns | 150 µs |
103
+ | 12 rows × 5 columns | 360 µs |
104
+ | 200 rows × 10 columns | 1.1 ms |
105
+ | 2,000 rows × 10 columns | 9 ms |
106
+
107
+ Nothing is enumerated up front. Cells and column sums are indexed once; differences, ratios, percentages, and percent changes are found per figure by solving for the partner cell and bisecting for it. Explanations are formatted only for the figure that matched. For comparison, a model-based faithfulness judge takes seconds and costs a request.
108
+
109
+ ## Command line
110
+
111
+ ```bash
112
+ figured "California has 39.3 million people." --rows rows.json
113
+ figured - --rows rows.json < answer.txt
114
+ figured "..." --rows rows.json --json --tolerance 0.01 --strict-percent
115
+ ```
116
+
117
+ Exit code 1 when any figure is untraceable, so it can gate a pipeline step.
118
+
119
+ ## Using it in a pipeline
120
+
121
+ **After every answer**, append the caveat and flip a badge:
122
+
123
+ ```python
124
+ report = trace(answer, rows)
125
+ if not report.ok:
126
+ answer += "\n\n" + report.caveat()
127
+ badge = "check figures"
128
+ ```
129
+
130
+ **In promptfoo**, as a Python assertion: see `examples/promptfoo_assert.py`.
131
+
132
+ **In DeepEval or any custom metric**, wrap `trace` and return `1 - len(report.ungrounded) / report.checked`.
133
+
134
+ **With a model judge for the rest.** Arithmetic cannot see a wrong word around a right number: "Nevada is richer than Utah" with the two correct medians reversed passes. The optional `judge` extra sends the question, the rows, and the answer to a model and returns a strict verdict on faithfulness, responsiveness, and caveats:
135
+
136
+ ```bash
137
+ pip install "figured[judge]"
138
+ ```
139
+
140
+ ```python
141
+ from figured.judge import judge
142
+
143
+ judge("Which state is richer?", answer, rows) # {"verdict": "fail", "issues": ["comparison reversed"], ...}
144
+ ```
145
+
146
+ ## What it does not do
147
+
148
+ - It cannot catch a correct number attached to the wrong claim. That is what the judge extra is for.
149
+ - With large result sets the derived set is big, and a hallucinated figure can land within tolerance of some difference by coincidence. The defaults cap the pairwise search at 12 rows and 40 cells; tighten the tolerance or restrict `derivations` for sensitive uses. A flag on a correct figure is treated as the worse error, because people stop reading badges that cry wolf.
150
+ - Numbers written as words ("two million") are not extracted.
151
+ - It does not know what the rows mean. If the agent queried the wrong column and described it faithfully, every figure traces.
152
+
153
+ ## How it compares
154
+
155
+ | | rows as evidence | derived arithmetic | deterministic | names each figure | packaged |
156
+ |---|---|---|---|---|---|
157
+ | **figured** | yes | sums, differences, ratios, percentages, ranges | yes | yes, with the derivation | pip, zero deps |
158
+ | llmground | no, a source string | no | yes | yes | pip |
159
+ | @demystify/grounding | no, cited facts | no | yes | yes | npm |
160
+ | pcn-core (Proof-Carrying Numbers) | claim values you supply | no | yes | yes, needs model-emitted tags | pip |
161
+ | NumProof | yes | yes | yes | yes | hosted API |
162
+ | DeepEval / Ragas faithfulness | text context | n/a | no, LLM or NLI | no | pip |
163
+
164
+ The Proof-Carrying Numbers policy vocabulary (exact, rounded, scale alias, tolerance, percent, range, year) is the clearest statement of the matching problem, and this library borrows its shape. The difference is the evidence contract: rows in, free text in, no cooperation from the model required.
165
+
166
+ ## Ports
167
+
168
+ Behavior is pinned by the conformance vectors in `tests/vectors/`. A port in another language is correct when it passes them unchanged. A TypeScript port is the natural next one; open an issue if you want to take it.
169
+
170
+ ## Origin
171
+
172
+ Built inside a Census data agent whose answers had to be traceable to the ACS rows behind them. The first version only derived values within a row, so a correct "about $10,900 higher" comparison across two state rows was flagged as suspect. That false flag is now a named test vector, and it is why the defaults lean toward trusting the model when the arithmetic works out.
173
+
174
+ ## License
175
+
176
+ MIT.
@@ -0,0 +1,52 @@
1
+ """Run: python benchmarks/bench.py — microseconds per trace() across result-set sizes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import random
6
+ import statistics
7
+ import time
8
+
9
+ from figured import trace
10
+
11
+ random.seed(7)
12
+
13
+ TEXT = (
14
+ "California has 39.3 million people (39,346,023), about 10.7 million more than Texas, "
15
+ "which is 72.8% of its size. Median income is $79,243, roughly $10,940 above Nevada, "
16
+ "and the region grew 2.3% while 12% of households moved. Together they hold 68 million."
17
+ )
18
+
19
+
20
+ def rows(n: int, cols: int) -> list[dict[str, float | str]]:
21
+ out = []
22
+ for i in range(n):
23
+ r: dict[str, float | str] = {"name": f"row{i}"}
24
+ for c in range(cols):
25
+ r[f"c{c}"] = float(random.randint(1_000, 50_000_000))
26
+ out.append(r)
27
+ out[0].update({"c0": 39346023.0, "c1": 28635442.0, "c2": 79243.4})
28
+ if n > 1:
29
+ out[1].update({"c0": 68303.1, "c1": 220000.0, "c2": 225060.0})
30
+ return out
31
+
32
+
33
+ def bench(label: str, data: object, reps: int) -> None:
34
+ times = []
35
+ for _ in range(reps):
36
+ t = time.perf_counter()
37
+ trace(TEXT, data)
38
+ times.append(time.perf_counter() - t)
39
+ med = statistics.median(times) * 1e6
40
+ print(f"{label:<28} {med:9.0f} µs (min {min(times) * 1e6:7.0f} µs)")
41
+
42
+
43
+ if __name__ == "__main__":
44
+ bench("2 rows × 3 cols", rows(2, 3), 300)
45
+ bench("12 rows × 5 cols", rows(12, 5), 200)
46
+ bench("200 rows × 10 cols", rows(200, 10), 50)
47
+ bench("2,000 rows × 10 cols", rows(2000, 10), 10)
48
+ bench("2,000 rows, max_cells=400", rows(2000, 10), 5) if False else None
49
+ print()
50
+ t = time.perf_counter()
51
+ r = trace(TEXT, rows(2000, 10), max_cells=400)
52
+ print(f"{'2,000 rows, max_cells=400':<28} {(time.perf_counter() - t) * 1e6:9.0f} µs ok={r.ok}")
@@ -0,0 +1,19 @@
1
+ """Run: python examples/basic.py"""
2
+
3
+ from figured import trace
4
+
5
+ rows = [
6
+ {"state": "California", "population": 39_346_023, "moe": 79_849},
7
+ {"state": "Texas", "population": 28_635_442, "moe": 65_120},
8
+ ]
9
+
10
+ answer = (
11
+ "California has about 39.3 million people (±79,849), roughly 10.7 million more than Texas. "
12
+ "Texas is 72.8% of California's size. Combined they hold 68 million people, "
13
+ "and about 4.1 million of them moved last year."
14
+ )
15
+
16
+ report = trace(answer, rows)
17
+ print(report.explain())
18
+ print()
19
+ print(report.caveat() or "Every figure traced.")
@@ -0,0 +1,39 @@
1
+ """A promptfoo Python assertion.
2
+
3
+ In promptfooconfig.yaml:
4
+
5
+ tests:
6
+ - vars:
7
+ question: "What is the population of California?"
8
+ rows: [{"state": "CA", "population": 39346023}]
9
+ assert:
10
+ - type: python
11
+ value: file://examples/promptfoo_assert.py
12
+
13
+ promptfoo calls get_assert(output, context) and uses the returned dict.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from typing import Any
19
+
20
+ from figured import trace
21
+
22
+
23
+ def get_assert(output: str, context: dict[str, Any]) -> dict[str, Any]:
24
+ rows = context.get("vars", {}).get("rows", [])
25
+ report = trace(output, rows)
26
+ return {
27
+ "pass": report.ok,
28
+ "score": 1.0 if report.ok else max(0.0, 1 - len(report.ungrounded) / max(report.checked, 1)),
29
+ "reason": report.caveat() or f"{report.checked} figures traced",
30
+ "componentResults": [
31
+ {
32
+ "pass": r.status != "ungrounded",
33
+ "score": 1.0 if r.status != "ungrounded" else 0.0,
34
+ "reason": r.literal,
35
+ }
36
+ for r in report.results
37
+ if r.status != "ignored"
38
+ ],
39
+ }