quadra-core 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quadra_core/__init__.py +41 -0
- quadra_core/cli.py +88 -0
- quadra_core/parse.py +232 -0
- quadra_core/pipeline/__init__.py +0 -0
- quadra_core/pipeline/document.py +159 -0
- quadra_core/pipeline/extract.py +556 -0
- quadra_core/pipeline/lines.py +356 -0
- quadra_core/pipeline/money.py +206 -0
- quadra_core/pipeline/ocr.py +172 -0
- quadra_core/pipeline/preprocess.py +377 -0
- quadra_core/pipeline/recheck.py +234 -0
- quadra_core/pipeline/validate.py +207 -0
- quadra_core/profiles/__init__.py +0 -0
- quadra_core/profiles/esselunga.toml +68 -0
- quadra_core/profiles/loader.py +202 -0
- quadra_core/profiles/synthetic.toml +37 -0
- quadra_core/schema/receipt-1.0.0.schema.json +155 -0
- quadra_core/testdata/__init__.py +29 -0
- quadra_core/testdata/ocr/esselunga_a.tsv +89 -0
- quadra_core/testdata/ocr/esselunga_b.tsv +123 -0
- quadra_core/testdata/ocr/esselunga_b_faded.tsv +125 -0
- quadra_core/testdata/ocr/esselunga_photo.tsv +132 -0
- quadra_core/testdata/ocr/esselunga_photo_curled.tsv +342 -0
- quadra_core/testdata/ocr/esselunga_photo_modifier.tsv +133 -0
- quadra_core/testdata/ocr/esselunga_photo_no_header.tsv +134 -0
- quadra_core/testdata/ocr/esselunga_photo_payments.tsv +113 -0
- quadra_core/testdata/ocr/esselunga_photo_table.tsv +356 -0
- quadra_core/testdata/ocr/synthetic_clean.tsv +86 -0
- quadra_core/testdata/ocr/synthetic_faded.tsv +100 -0
- quadra_core/testdata/ocr/synthetic_unbalanced.tsv +86 -0
- quadra_core-0.1.0.dist-info/METADATA +77 -0
- quadra_core-0.1.0.dist-info/RECORD +35 -0
- quadra_core-0.1.0.dist-info/WHEEL +4 -0
- quadra_core-0.1.0.dist-info/entry_points.txt +3 -0
- quadra_core-0.1.0.dist-info/licenses/LICENSE +21 -0
quadra_core/__init__.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Read a supermarket till receipt from a photo.
|
|
2
|
+
|
|
3
|
+
Tesseract does the OCR, then geometry and arithmetic do the rest. The same
|
|
4
|
+
receipt always parses the same way, and when it gets one wrong you can see why.
|
|
5
|
+
|
|
6
|
+
from quadra_core import parse_image, parse_tsv
|
|
7
|
+
|
|
8
|
+
document, status = parse_image(Path("receipt.jpg"), profile_id="esselunga")
|
|
9
|
+
|
|
10
|
+
`status` is "ok", "partial" or "failed". It never raises just because a receipt
|
|
11
|
+
was hard to read: you get a document back either way, with the problems listed
|
|
12
|
+
in `validation.warnings`.
|
|
13
|
+
|
|
14
|
+
Everything shop-specific lives in a TOML profile rather than in this code, so
|
|
15
|
+
adding a shop means writing a profile.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
from quadra_core.parse import parse_image, parse_tsv, read_input
|
|
23
|
+
from quadra_core.pipeline.document import no_review
|
|
24
|
+
from quadra_core.profiles.loader import Profile, ProfileError, available, load, select
|
|
25
|
+
|
|
26
|
+
# The JSON Schema every document validates against. Shipped with the package so
|
|
27
|
+
# callers can check output without keeping their own copy in sync.
|
|
28
|
+
SCHEMA_PATH = Path(__file__).parent / "schema" / "receipt-1.0.0.schema.json"
|
|
29
|
+
|
|
30
|
+
__all__ = [
|
|
31
|
+
"SCHEMA_PATH",
|
|
32
|
+
"Profile",
|
|
33
|
+
"ProfileError",
|
|
34
|
+
"available",
|
|
35
|
+
"load",
|
|
36
|
+
"no_review",
|
|
37
|
+
"parse_image",
|
|
38
|
+
"parse_tsv",
|
|
39
|
+
"read_input",
|
|
40
|
+
"select",
|
|
41
|
+
]
|
quadra_core/cli.py
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""`quadra-core` on the command line: read a receipt, print the document.
|
|
2
|
+
|
|
3
|
+
Small on purpose. It is here so you can try the library and check a profile
|
|
4
|
+
against a photo without writing a script. Storing, reviewing and exporting
|
|
5
|
+
receipts are jobs for whatever gets built on top.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import sys
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from quadra_core.parse import IMAGE_SUFFIXES, parse_image, parse_tsv, read_input
|
|
16
|
+
from quadra_core.profiles import loader
|
|
17
|
+
|
|
18
|
+
# Exit code carries the result, so a shell script or worker does not have to
|
|
19
|
+
# parse stdout.
|
|
20
|
+
EXIT_OK, EXIT_PARTIAL, EXIT_FAILED, EXIT_ERROR = 0, 1, 2, 3
|
|
21
|
+
|
|
22
|
+
_STATUS_EXIT = {"ok": EXIT_OK, "partial": EXIT_PARTIAL, "failed": EXIT_FAILED}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _run_parse(args: argparse.Namespace) -> int:
|
|
26
|
+
source: Path | None = args.path
|
|
27
|
+
if source is not None and source.suffix.lower() in IMAGE_SUFFIXES:
|
|
28
|
+
document, status = parse_image(
|
|
29
|
+
source, profile_id=args.profile, printed_total=args.total
|
|
30
|
+
)
|
|
31
|
+
# Only of use to something archiving the originals.
|
|
32
|
+
document.pop("_tsv", None)
|
|
33
|
+
document.pop("_prepared", None)
|
|
34
|
+
else:
|
|
35
|
+
tsv, ocr_meta = read_input(source)
|
|
36
|
+
document, status = parse_tsv(
|
|
37
|
+
tsv, profile_id=args.profile, ocr_meta=ocr_meta, printed_total=args.total
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
json.dump(document, sys.stdout, indent=2, ensure_ascii=False)
|
|
41
|
+
sys.stdout.write("\n")
|
|
42
|
+
return _STATUS_EXIT.get(status, EXIT_ERROR)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _run_profiles(_args: argparse.Namespace) -> int:
|
|
46
|
+
for profile in loader.available():
|
|
47
|
+
state = "calibrated" if profile.calibrated else "UNCALIBRATED"
|
|
48
|
+
print(f"{profile.id:<12} v{profile.version} {state}")
|
|
49
|
+
return EXIT_OK
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
53
|
+
"""Assemble the command line."""
|
|
54
|
+
ap = argparse.ArgumentParser(prog="quadra-core", description=__doc__)
|
|
55
|
+
sub = ap.add_subparsers(dest="command", required=True)
|
|
56
|
+
|
|
57
|
+
parse = sub.add_parser("parse", help="read a receipt image or Tesseract TSV")
|
|
58
|
+
parse.add_argument(
|
|
59
|
+
"path",
|
|
60
|
+
nargs="?",
|
|
61
|
+
type=Path,
|
|
62
|
+
help="image or .tsv file; omit to read TSV from stdin",
|
|
63
|
+
)
|
|
64
|
+
parse.add_argument("--profile", help="force a profile instead of fingerprinting")
|
|
65
|
+
parse.add_argument(
|
|
66
|
+
"--total",
|
|
67
|
+
type=int,
|
|
68
|
+
help="printed total in minor units, when the receipt's own is unreadable",
|
|
69
|
+
)
|
|
70
|
+
parse.set_defaults(handler=_run_parse)
|
|
71
|
+
|
|
72
|
+
profiles = sub.add_parser("profiles", help="list the shop profiles available")
|
|
73
|
+
profiles.set_defaults(handler=_run_profiles)
|
|
74
|
+
return ap
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def main(argv: list[str] | None = None) -> int:
|
|
78
|
+
"""Entry point. Returns the exit code rather than raising on a bad receipt."""
|
|
79
|
+
args = build_parser().parse_args(argv)
|
|
80
|
+
try:
|
|
81
|
+
return args.handler(args)
|
|
82
|
+
except (loader.ProfileError, OSError) as exc:
|
|
83
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
84
|
+
return EXIT_ERROR
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
if __name__ == "__main__":
|
|
88
|
+
raise SystemExit(main())
|
quadra_core/parse.py
ADDED
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
"""Photo or Tesseract TSV in, receipt document out.
|
|
2
|
+
|
|
3
|
+
The entry point for the library. `parse_tsv` takes Tesseract output directly,
|
|
4
|
+
so the parsing side runs with neither Tesseract nor Pillow installed; the image
|
|
5
|
+
path imports both lazily to keep it that way.
|
|
6
|
+
|
|
7
|
+
Bad receipt content never raises. A receipt that could not be read comes back
|
|
8
|
+
as a document saying so, so a batch of receipts does not fail on one bad photo.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import sys
|
|
14
|
+
import uuid
|
|
15
|
+
from datetime import UTC, datetime
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from quadra_core.pipeline import document as document_module
|
|
20
|
+
from quadra_core.pipeline.document import build
|
|
21
|
+
from quadra_core.pipeline.extract import extract
|
|
22
|
+
from quadra_core.pipeline.lines import (
|
|
23
|
+
drop_speckle,
|
|
24
|
+
group_lines,
|
|
25
|
+
load_tsv,
|
|
26
|
+
median_glyph_height,
|
|
27
|
+
)
|
|
28
|
+
from quadra_core.pipeline.ocr import run_on_path
|
|
29
|
+
from quadra_core.pipeline.validate import validate
|
|
30
|
+
from quadra_core.profiles import loader
|
|
31
|
+
|
|
32
|
+
IMAGE_SUFFIXES = {".jpg", ".jpeg", ".png", ".tif", ".tiff", ".bmp", ".webp"}
|
|
33
|
+
|
|
34
|
+
DEFAULT_OCR_META = {"engine": "tesseract", "lang": "ita", "psm": 4, "oem": 1}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def read_input(path: Path | None) -> tuple[str, dict]:
|
|
38
|
+
"""Return TSV plus OCR metadata, from an image, a TSV file, or stdin.
|
|
39
|
+
|
|
40
|
+
Taking TSV directly keeps the parsing core usable without Tesseract or
|
|
41
|
+
Pillow installed; the image branch imports them lazily.
|
|
42
|
+
"""
|
|
43
|
+
if path is None:
|
|
44
|
+
return sys.stdin.read(), dict(DEFAULT_OCR_META)
|
|
45
|
+
|
|
46
|
+
if path.suffix.lower() in IMAGE_SUFFIXES:
|
|
47
|
+
result, report, prepared = run_on_path(path)
|
|
48
|
+
return result.tsv, {
|
|
49
|
+
"engine": "tesseract",
|
|
50
|
+
"engine_version": result.engine_version,
|
|
51
|
+
"lang": result.lang,
|
|
52
|
+
"psm": result.psm,
|
|
53
|
+
"oem": result.oem,
|
|
54
|
+
"preprocessing": report.steps,
|
|
55
|
+
"_prepared": prepared,
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
return path.read_text(), dict(DEFAULT_OCR_META)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def parse_tsv(
|
|
62
|
+
tsv: str,
|
|
63
|
+
*,
|
|
64
|
+
profile_id: str | None,
|
|
65
|
+
receipt_id: str | None = None,
|
|
66
|
+
ocr_meta: dict | None = None,
|
|
67
|
+
prepared: object | None = None,
|
|
68
|
+
printed_total: int | None = None,
|
|
69
|
+
review: dict[str, Any] | None = None,
|
|
70
|
+
) -> tuple[dict, str]:
|
|
71
|
+
"""TSV -> receipt document. Never raises on bad receipt content.
|
|
72
|
+
|
|
73
|
+
`prepared` is the image the TSV was read from. Given one, a receipt that fails
|
|
74
|
+
to add up gets its price column read a second time; see pipeline.recheck.
|
|
75
|
+
|
|
76
|
+
`printed_total` overrides whatever total the receipt itself yielded, for the
|
|
77
|
+
case where someone typed it in because it was unreadable.
|
|
78
|
+
|
|
79
|
+
`review` says whether a person should check this receipt. That is the
|
|
80
|
+
caller's decision, not this library's, so whatever is passed in is recorded
|
|
81
|
+
as given.
|
|
82
|
+
"""
|
|
83
|
+
words = drop_speckle(load_tsv(tsv))
|
|
84
|
+
lines = group_lines(words)
|
|
85
|
+
texts = [l.text for l in lines]
|
|
86
|
+
|
|
87
|
+
confidence = 0.0
|
|
88
|
+
if profile_id:
|
|
89
|
+
matches = [p for p in loader.available() if p.id == profile_id]
|
|
90
|
+
if not matches:
|
|
91
|
+
raise loader.ProfileError(f"no such profile: {profile_id}")
|
|
92
|
+
profile = matches[0]
|
|
93
|
+
confidence = 1.0
|
|
94
|
+
else:
|
|
95
|
+
selected = loader.select(texts)
|
|
96
|
+
if selected is None:
|
|
97
|
+
raise loader.ProfileError(
|
|
98
|
+
"no profile fingerprint matched. That is a real signal: either the "
|
|
99
|
+
"receipt is from another store, or the layout changed."
|
|
100
|
+
)
|
|
101
|
+
profile, confidence = selected
|
|
102
|
+
|
|
103
|
+
extraction = extract(lines, profile)
|
|
104
|
+
if printed_total is not None:
|
|
105
|
+
# Someone read the total off the paper because the parser could not, so
|
|
106
|
+
# it beats anything OCR produced.
|
|
107
|
+
extraction.printed_total_minor = printed_total
|
|
108
|
+
extraction.printed_total_supplied = True
|
|
109
|
+
validation = validate(extraction, profile)
|
|
110
|
+
|
|
111
|
+
if prepared is not None and not validation.balanced:
|
|
112
|
+
from .pipeline import recheck # noqa: PLC0415 - only needed with an image
|
|
113
|
+
|
|
114
|
+
agreed = recheck.repair(extraction, validation, prepared, profile, lines)
|
|
115
|
+
if agreed is not None:
|
|
116
|
+
recheck.apply(extraction, agreed, lines)
|
|
117
|
+
# Re-run the whole check instead of patching the totals, so the
|
|
118
|
+
# per-line arithmetic and warnings match the new prices.
|
|
119
|
+
validation = validate(extraction, profile)
|
|
120
|
+
validation.warnings.append(
|
|
121
|
+
f"price_column_reread:{len(agreed.changes)}_of_{agreed.disputed}"
|
|
122
|
+
)
|
|
123
|
+
if agreed.recovered:
|
|
124
|
+
validation.warnings.append(f"lines_recovered:{len(agreed.recovered)}")
|
|
125
|
+
|
|
126
|
+
rid = receipt_id or uuid.uuid4().hex
|
|
127
|
+
|
|
128
|
+
doc = build(
|
|
129
|
+
receipt_id=rid,
|
|
130
|
+
lines=lines,
|
|
131
|
+
extraction=extraction,
|
|
132
|
+
validation=validation,
|
|
133
|
+
profile=profile,
|
|
134
|
+
profile_confidence=confidence,
|
|
135
|
+
ocr_meta=ocr_meta or {"engine": "tesseract", "lang": "ita", "psm": 4, "oem": 1},
|
|
136
|
+
source={"ingested_at": datetime.now(UTC).isoformat()},
|
|
137
|
+
review=(
|
|
138
|
+
review
|
|
139
|
+
if review is not None
|
|
140
|
+
else document_module.default_review(validation, extraction)
|
|
141
|
+
),
|
|
142
|
+
)
|
|
143
|
+
return doc, validation.status
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
# Below 26px Tesseract starts losing lines, so a second pass at a larger size is
|
|
147
|
+
# worth the time. A receipt at 25px was fine; one at 20px lost two lines.
|
|
148
|
+
SMALL_GLYPH = 26
|
|
149
|
+
COMFORTABLE_GLYPH = 30
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _closer(candidate: dict, current: dict) -> bool:
|
|
153
|
+
"""Whether a second reading is the better of the two.
|
|
154
|
+
|
|
155
|
+
Balancing wins. Failing that, the smaller delta wins. With no total to
|
|
156
|
+
compare against, more lines found wins, since a missed line takes its
|
|
157
|
+
money with it.
|
|
158
|
+
"""
|
|
159
|
+
a, b = candidate["totals"], current["totals"]
|
|
160
|
+
if a["balanced"] != b["balanced"]:
|
|
161
|
+
return bool(a["balanced"])
|
|
162
|
+
if a["delta_minor"] is not None and b["delta_minor"] is not None:
|
|
163
|
+
return abs(a["delta_minor"]) < abs(b["delta_minor"])
|
|
164
|
+
if (a["printed_total_minor"] is None) != (b["printed_total_minor"] is None):
|
|
165
|
+
return b["printed_total_minor"] is None
|
|
166
|
+
return len(a and candidate["line_items"]) > len(current["line_items"])
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _read_larger(prepared, tsv: str, **kwargs):
|
|
170
|
+
"""Read the photo again, enlarged, when the first read came out small.
|
|
171
|
+
|
|
172
|
+
Only runs on a receipt that did not add up, so a good one never pays for it.
|
|
173
|
+
Enlarging adds no detail, but Tesseract's line model does better at the size
|
|
174
|
+
it expects: on one real receipt it found 33 lines instead of 31.
|
|
175
|
+
"""
|
|
176
|
+
from PIL import Image # noqa: PLC0415 - the parsing core runs without Pillow
|
|
177
|
+
|
|
178
|
+
from .pipeline.ocr import run # noqa: PLC0415
|
|
179
|
+
|
|
180
|
+
glyph = median_glyph_height(drop_speckle(load_tsv(tsv)))
|
|
181
|
+
if not glyph or glyph >= SMALL_GLYPH:
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
scale = COMFORTABLE_GLYPH / glyph
|
|
185
|
+
bigger = prepared.resize(
|
|
186
|
+
(round(prepared.width * scale), round(prepared.height * scale)),
|
|
187
|
+
Image.Resampling.LANCZOS,
|
|
188
|
+
)
|
|
189
|
+
result = run(bigger, psm=4)
|
|
190
|
+
document, status = parse_tsv(result.tsv, prepared=bigger, **kwargs)
|
|
191
|
+
return document, status, result.tsv, bigger
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def parse_image(
|
|
195
|
+
path: Path,
|
|
196
|
+
*,
|
|
197
|
+
profile_id: str | None,
|
|
198
|
+
review: dict[str, Any] | None = None,
|
|
199
|
+
printed_total: int | None = None,
|
|
200
|
+
):
|
|
201
|
+
"""Parse a photo, carrying the raw TSV along for archiving.
|
|
202
|
+
|
|
203
|
+
The document comes back with two private keys, `_tsv` and `_prepared`: the
|
|
204
|
+
OCR it was built from and the image OCR actually saw. Whoever archives them
|
|
205
|
+
pops them off first. Item bounding boxes use the prepared image's
|
|
206
|
+
coordinates, so cropping by box needs that image, not the original photo.
|
|
207
|
+
"""
|
|
208
|
+
tsv, ocr_meta = read_input(path)
|
|
209
|
+
# Removed before parse_tsv: build() copies ocr_meta into the document, and a
|
|
210
|
+
# PIL image in there makes it unserialisable.
|
|
211
|
+
prepared = ocr_meta.pop("_prepared", None)
|
|
212
|
+
shared = {
|
|
213
|
+
"profile_id": profile_id,
|
|
214
|
+
"review": review,
|
|
215
|
+
"ocr_meta": ocr_meta,
|
|
216
|
+
"printed_total": printed_total,
|
|
217
|
+
}
|
|
218
|
+
document, status = parse_tsv(tsv, prepared=prepared, **shared)
|
|
219
|
+
|
|
220
|
+
if prepared is not None and not document["totals"]["balanced"]:
|
|
221
|
+
second = _read_larger(prepared, tsv, **shared)
|
|
222
|
+
if second is not None and _closer(second[0], document):
|
|
223
|
+
# The archived image must be the one the TSV describes, or crops on
|
|
224
|
+
# the review page point at the wrong pixels.
|
|
225
|
+
document, status, tsv, prepared = second
|
|
226
|
+
|
|
227
|
+
# Handed to the archiver, which pops both before the document is written.
|
|
228
|
+
document["_tsv"] = tsv
|
|
229
|
+
document["_prepared"] = prepared
|
|
230
|
+
return document, status
|
|
231
|
+
|
|
232
|
+
|
|
File without changes
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Building the JSON document for one receipt.
|
|
2
|
+
|
|
3
|
+
Money leaves here as whole cents, with a _minor suffix so the unit is obvious at
|
|
4
|
+
the call site. Unreadable fields come out null with a warning, never guessed.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import asdict
|
|
10
|
+
from decimal import Decimal
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from quadra_core.pipeline.extract import Extraction
|
|
14
|
+
from quadra_core.pipeline.lines import Line
|
|
15
|
+
from quadra_core.pipeline.validate import Validation
|
|
16
|
+
from quadra_core.profiles.loader import Profile
|
|
17
|
+
|
|
18
|
+
SCHEMA_VERSION = "1.0.0"
|
|
19
|
+
|
|
20
|
+
# Below this, the least confident word in a description is too poor to file
|
|
21
|
+
# unseen. Arithmetic cannot catch a misread product name, so this is the only
|
|
22
|
+
# thing that sends one for review.
|
|
23
|
+
LOW_CONFIDENCE_DESCRIPTION = 0.5
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def no_review(state: str = "not_required") -> dict[str, Any]:
|
|
27
|
+
"""Build the review block for a document nobody has decided anything about.
|
|
28
|
+
|
|
29
|
+
Who gets checked is up to whatever uses this library, so the default says
|
|
30
|
+
only that no decision has been made yet, rather than claiming the receipt
|
|
31
|
+
was checked or does not need checking.
|
|
32
|
+
"""
|
|
33
|
+
return {
|
|
34
|
+
"required": False,
|
|
35
|
+
"reason": None,
|
|
36
|
+
"sampled": False,
|
|
37
|
+
"mode": "none",
|
|
38
|
+
"state": state,
|
|
39
|
+
"blind_item_indexes": [],
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def default_review(
|
|
44
|
+
validation: Validation, extraction: Extraction
|
|
45
|
+
) -> dict[str, Any]:
|
|
46
|
+
"""Whether a person has to look, when the caller has not said.
|
|
47
|
+
|
|
48
|
+
The caller owns this decision and can pass its own block instead. This is
|
|
49
|
+
the default, and it errs towards asking: a receipt that does not add up, or
|
|
50
|
+
that carries a description nobody could read, is not one to file unseen.
|
|
51
|
+
"""
|
|
52
|
+
reasons = []
|
|
53
|
+
if not validation.balanced:
|
|
54
|
+
reasons.append("totals_do_not_balance")
|
|
55
|
+
if not extraction.items:
|
|
56
|
+
reasons.append("no_items")
|
|
57
|
+
worst = min((i.confidence for i in extraction.items), default=1.0)
|
|
58
|
+
if worst < LOW_CONFIDENCE_DESCRIPTION:
|
|
59
|
+
reasons.append(f"unreadable_description:{worst:.2f}")
|
|
60
|
+
|
|
61
|
+
if not reasons:
|
|
62
|
+
return no_review()
|
|
63
|
+
return {
|
|
64
|
+
"required": True,
|
|
65
|
+
"reason": "; ".join(reasons),
|
|
66
|
+
"sampled": False,
|
|
67
|
+
"mode": "full",
|
|
68
|
+
"state": "pending",
|
|
69
|
+
"blind_item_indexes": [],
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _quantity(value: Decimal | None) -> dict[str, Any] | None:
|
|
74
|
+
if value is None:
|
|
75
|
+
return None
|
|
76
|
+
# A string rather than a float, since weighed goods need fractions that
|
|
77
|
+
# binary floats cannot hold exactly.
|
|
78
|
+
return {"value": format(value.normalize(), "f"), "unit": "each"}
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def build(
|
|
82
|
+
*,
|
|
83
|
+
receipt_id: str,
|
|
84
|
+
lines: list[Line],
|
|
85
|
+
extraction: Extraction,
|
|
86
|
+
validation: Validation,
|
|
87
|
+
profile: Profile,
|
|
88
|
+
profile_confidence: float,
|
|
89
|
+
ocr_meta: dict[str, Any],
|
|
90
|
+
source: dict[str, Any],
|
|
91
|
+
review: dict[str, Any],
|
|
92
|
+
) -> dict[str, Any]:
|
|
93
|
+
"""Assemble the full receipt document."""
|
|
94
|
+
mean_conf = sum(l.conf for l in lines) / len(lines) if lines else 0.0
|
|
95
|
+
return {
|
|
96
|
+
"schema_version": SCHEMA_VERSION,
|
|
97
|
+
"receipt_id": receipt_id,
|
|
98
|
+
"source": source,
|
|
99
|
+
"profile": {
|
|
100
|
+
"id": profile.id,
|
|
101
|
+
"version": profile.version,
|
|
102
|
+
"calibrated": profile.calibrated,
|
|
103
|
+
"match_confidence": round(profile_confidence, 3),
|
|
104
|
+
},
|
|
105
|
+
"ocr": {**ocr_meta, "mean_word_confidence": round(mean_conf, 1)},
|
|
106
|
+
"merchant": {"name": None, "store_id": None, "address": None, "vat_id": None},
|
|
107
|
+
"transaction": {
|
|
108
|
+
"datetime": None,
|
|
109
|
+
"currency": profile.currency,
|
|
110
|
+
"receipt_number": None,
|
|
111
|
+
"payment_method": None,
|
|
112
|
+
},
|
|
113
|
+
"line_items": [
|
|
114
|
+
{
|
|
115
|
+
"index": i,
|
|
116
|
+
"description_raw": item.description_raw,
|
|
117
|
+
"description": item.description,
|
|
118
|
+
"quantity": _quantity(item.quantity),
|
|
119
|
+
"quantity_source": item.quantity_source,
|
|
120
|
+
"unit_price_minor": item.unit_price_minor,
|
|
121
|
+
"line_total_minor": item.line_total_minor,
|
|
122
|
+
"discounts_minor": 0,
|
|
123
|
+
"vat_code": item.vat_code,
|
|
124
|
+
"confidence": item.confidence,
|
|
125
|
+
"provenance": {"line_index": item.line_index, "bbox": list(item.bbox)},
|
|
126
|
+
"flags": item.flags,
|
|
127
|
+
}
|
|
128
|
+
for i, item in enumerate(extraction.items)
|
|
129
|
+
],
|
|
130
|
+
"adjustments": [
|
|
131
|
+
{
|
|
132
|
+
"kind": adjustment.kind,
|
|
133
|
+
"label": adjustment.label,
|
|
134
|
+
"amount_minor": adjustment.amount_minor,
|
|
135
|
+
"provenance": {"line_index": adjustment.line_index},
|
|
136
|
+
}
|
|
137
|
+
for adjustment in extraction.adjustments
|
|
138
|
+
],
|
|
139
|
+
"totals": {
|
|
140
|
+
"items_subtotal_minor": validation.items_subtotal_minor,
|
|
141
|
+
"discounts_minor": sum(a.amount_minor for a in extraction.adjustments),
|
|
142
|
+
"printed_total_minor": validation.printed_total_minor,
|
|
143
|
+
"computed_total_minor": validation.items_subtotal_minor,
|
|
144
|
+
"balanced": validation.balanced,
|
|
145
|
+
"delta_minor": validation.delta_minor,
|
|
146
|
+
},
|
|
147
|
+
"validation": {
|
|
148
|
+
"status": validation.status,
|
|
149
|
+
"checks": [asdict(c) for c in validation.checks],
|
|
150
|
+
"warnings": validation.warnings,
|
|
151
|
+
},
|
|
152
|
+
"review": {
|
|
153
|
+
**review,
|
|
154
|
+
"reviewed_at": None,
|
|
155
|
+
"duration_ms": None,
|
|
156
|
+
"rows_expanded": 0,
|
|
157
|
+
"corrections": [],
|
|
158
|
+
},
|
|
159
|
+
}
|