quadra-core 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. quadra_core/__init__.py +41 -0
  2. quadra_core/cli.py +88 -0
  3. quadra_core/parse.py +232 -0
  4. quadra_core/pipeline/__init__.py +0 -0
  5. quadra_core/pipeline/document.py +159 -0
  6. quadra_core/pipeline/extract.py +556 -0
  7. quadra_core/pipeline/lines.py +356 -0
  8. quadra_core/pipeline/money.py +206 -0
  9. quadra_core/pipeline/ocr.py +172 -0
  10. quadra_core/pipeline/preprocess.py +377 -0
  11. quadra_core/pipeline/recheck.py +234 -0
  12. quadra_core/pipeline/validate.py +207 -0
  13. quadra_core/profiles/__init__.py +0 -0
  14. quadra_core/profiles/esselunga.toml +68 -0
  15. quadra_core/profiles/loader.py +202 -0
  16. quadra_core/profiles/synthetic.toml +37 -0
  17. quadra_core/schema/receipt-1.0.0.schema.json +155 -0
  18. quadra_core/testdata/__init__.py +29 -0
  19. quadra_core/testdata/ocr/esselunga_a.tsv +89 -0
  20. quadra_core/testdata/ocr/esselunga_b.tsv +123 -0
  21. quadra_core/testdata/ocr/esselunga_b_faded.tsv +125 -0
  22. quadra_core/testdata/ocr/esselunga_photo.tsv +132 -0
  23. quadra_core/testdata/ocr/esselunga_photo_curled.tsv +342 -0
  24. quadra_core/testdata/ocr/esselunga_photo_modifier.tsv +133 -0
  25. quadra_core/testdata/ocr/esselunga_photo_no_header.tsv +134 -0
  26. quadra_core/testdata/ocr/esselunga_photo_payments.tsv +113 -0
  27. quadra_core/testdata/ocr/esselunga_photo_table.tsv +356 -0
  28. quadra_core/testdata/ocr/synthetic_clean.tsv +86 -0
  29. quadra_core/testdata/ocr/synthetic_faded.tsv +100 -0
  30. quadra_core/testdata/ocr/synthetic_unbalanced.tsv +86 -0
  31. quadra_core-0.1.0.dist-info/METADATA +77 -0
  32. quadra_core-0.1.0.dist-info/RECORD +35 -0
  33. quadra_core-0.1.0.dist-info/WHEEL +4 -0
  34. quadra_core-0.1.0.dist-info/entry_points.txt +3 -0
  35. quadra_core-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,41 @@
1
+ """Read a supermarket till receipt from a photo.
2
+
3
+ Tesseract does the OCR, then geometry and arithmetic do the rest. The same
4
+ receipt always parses the same way, and when it gets one wrong you can see why.
5
+
6
+ from quadra_core import parse_image, parse_tsv
7
+
8
+ document, status = parse_image(Path("receipt.jpg"), profile_id="esselunga")
9
+
10
+ `status` is "ok", "partial" or "failed". It never raises just because a receipt
11
+ was hard to read: you get a document back either way, with the problems listed
12
+ in `validation.warnings`.
13
+
14
+ Everything shop-specific lives in a TOML profile rather than in this code, so
15
+ adding a shop means writing a profile.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from pathlib import Path
21
+
22
+ from quadra_core.parse import parse_image, parse_tsv, read_input
23
+ from quadra_core.pipeline.document import no_review
24
+ from quadra_core.profiles.loader import Profile, ProfileError, available, load, select
25
+
26
+ # The JSON Schema every document validates against. Shipped with the package so
27
+ # callers can check output without keeping their own copy in sync.
28
+ SCHEMA_PATH = Path(__file__).parent / "schema" / "receipt-1.0.0.schema.json"
29
+
30
+ __all__ = [
31
+ "SCHEMA_PATH",
32
+ "Profile",
33
+ "ProfileError",
34
+ "available",
35
+ "load",
36
+ "no_review",
37
+ "parse_image",
38
+ "parse_tsv",
39
+ "read_input",
40
+ "select",
41
+ ]
quadra_core/cli.py ADDED
@@ -0,0 +1,88 @@
1
+ """`quadra-core` on the command line: read a receipt, print the document.
2
+
3
+ Small on purpose. It is here so you can try the library and check a profile
4
+ against a photo without writing a script. Storing, reviewing and exporting
5
+ receipts are jobs for whatever gets built on top.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ import json
12
+ import sys
13
+ from pathlib import Path
14
+
15
+ from quadra_core.parse import IMAGE_SUFFIXES, parse_image, parse_tsv, read_input
16
+ from quadra_core.profiles import loader
17
+
18
+ # Exit code carries the result, so a shell script or worker does not have to
19
+ # parse stdout.
20
+ EXIT_OK, EXIT_PARTIAL, EXIT_FAILED, EXIT_ERROR = 0, 1, 2, 3
21
+
22
+ _STATUS_EXIT = {"ok": EXIT_OK, "partial": EXIT_PARTIAL, "failed": EXIT_FAILED}
23
+
24
+
25
+ def _run_parse(args: argparse.Namespace) -> int:
26
+ source: Path | None = args.path
27
+ if source is not None and source.suffix.lower() in IMAGE_SUFFIXES:
28
+ document, status = parse_image(
29
+ source, profile_id=args.profile, printed_total=args.total
30
+ )
31
+ # Only of use to something archiving the originals.
32
+ document.pop("_tsv", None)
33
+ document.pop("_prepared", None)
34
+ else:
35
+ tsv, ocr_meta = read_input(source)
36
+ document, status = parse_tsv(
37
+ tsv, profile_id=args.profile, ocr_meta=ocr_meta, printed_total=args.total
38
+ )
39
+
40
+ json.dump(document, sys.stdout, indent=2, ensure_ascii=False)
41
+ sys.stdout.write("\n")
42
+ return _STATUS_EXIT.get(status, EXIT_ERROR)
43
+
44
+
45
+ def _run_profiles(_args: argparse.Namespace) -> int:
46
+ for profile in loader.available():
47
+ state = "calibrated" if profile.calibrated else "UNCALIBRATED"
48
+ print(f"{profile.id:<12} v{profile.version} {state}")
49
+ return EXIT_OK
50
+
51
+
52
+ def build_parser() -> argparse.ArgumentParser:
53
+ """Assemble the command line."""
54
+ ap = argparse.ArgumentParser(prog="quadra-core", description=__doc__)
55
+ sub = ap.add_subparsers(dest="command", required=True)
56
+
57
+ parse = sub.add_parser("parse", help="read a receipt image or Tesseract TSV")
58
+ parse.add_argument(
59
+ "path",
60
+ nargs="?",
61
+ type=Path,
62
+ help="image or .tsv file; omit to read TSV from stdin",
63
+ )
64
+ parse.add_argument("--profile", help="force a profile instead of fingerprinting")
65
+ parse.add_argument(
66
+ "--total",
67
+ type=int,
68
+ help="printed total in minor units, when the receipt's own is unreadable",
69
+ )
70
+ parse.set_defaults(handler=_run_parse)
71
+
72
+ profiles = sub.add_parser("profiles", help="list the shop profiles available")
73
+ profiles.set_defaults(handler=_run_profiles)
74
+ return ap
75
+
76
+
77
+ def main(argv: list[str] | None = None) -> int:
78
+ """Entry point. Returns the exit code rather than raising on a bad receipt."""
79
+ args = build_parser().parse_args(argv)
80
+ try:
81
+ return args.handler(args)
82
+ except (loader.ProfileError, OSError) as exc:
83
+ print(f"error: {exc}", file=sys.stderr)
84
+ return EXIT_ERROR
85
+
86
+
87
+ if __name__ == "__main__":
88
+ raise SystemExit(main())
quadra_core/parse.py ADDED
@@ -0,0 +1,232 @@
1
+ """Photo or Tesseract TSV in, receipt document out.
2
+
3
+ The entry point for the library. `parse_tsv` takes Tesseract output directly,
4
+ so the parsing side runs with neither Tesseract nor Pillow installed; the image
5
+ path imports both lazily to keep it that way.
6
+
7
+ Bad receipt content never raises. A receipt that could not be read comes back
8
+ as a document saying so, so a batch of receipts does not fail on one bad photo.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import sys
14
+ import uuid
15
+ from datetime import UTC, datetime
16
+ from pathlib import Path
17
+ from typing import Any
18
+
19
+ from quadra_core.pipeline import document as document_module
20
+ from quadra_core.pipeline.document import build
21
+ from quadra_core.pipeline.extract import extract
22
+ from quadra_core.pipeline.lines import (
23
+ drop_speckle,
24
+ group_lines,
25
+ load_tsv,
26
+ median_glyph_height,
27
+ )
28
+ from quadra_core.pipeline.ocr import run_on_path
29
+ from quadra_core.pipeline.validate import validate
30
+ from quadra_core.profiles import loader
31
+
32
+ IMAGE_SUFFIXES = {".jpg", ".jpeg", ".png", ".tif", ".tiff", ".bmp", ".webp"}
33
+
34
+ DEFAULT_OCR_META = {"engine": "tesseract", "lang": "ita", "psm": 4, "oem": 1}
35
+
36
+
37
+ def read_input(path: Path | None) -> tuple[str, dict]:
38
+ """Return TSV plus OCR metadata, from an image, a TSV file, or stdin.
39
+
40
+ Taking TSV directly keeps the parsing core usable without Tesseract or
41
+ Pillow installed; the image branch imports them lazily.
42
+ """
43
+ if path is None:
44
+ return sys.stdin.read(), dict(DEFAULT_OCR_META)
45
+
46
+ if path.suffix.lower() in IMAGE_SUFFIXES:
47
+ result, report, prepared = run_on_path(path)
48
+ return result.tsv, {
49
+ "engine": "tesseract",
50
+ "engine_version": result.engine_version,
51
+ "lang": result.lang,
52
+ "psm": result.psm,
53
+ "oem": result.oem,
54
+ "preprocessing": report.steps,
55
+ "_prepared": prepared,
56
+ }
57
+
58
+ return path.read_text(), dict(DEFAULT_OCR_META)
59
+
60
+
61
+ def parse_tsv(
62
+ tsv: str,
63
+ *,
64
+ profile_id: str | None,
65
+ receipt_id: str | None = None,
66
+ ocr_meta: dict | None = None,
67
+ prepared: object | None = None,
68
+ printed_total: int | None = None,
69
+ review: dict[str, Any] | None = None,
70
+ ) -> tuple[dict, str]:
71
+ """TSV -> receipt document. Never raises on bad receipt content.
72
+
73
+ `prepared` is the image the TSV was read from. Given one, a receipt that fails
74
+ to add up gets its price column read a second time; see pipeline.recheck.
75
+
76
+ `printed_total` overrides whatever total the receipt itself yielded, for the
77
+ case where someone typed it in because it was unreadable.
78
+
79
+ `review` says whether a person should check this receipt. That is the
80
+ caller's decision, not this library's, so whatever is passed in is recorded
81
+ as given.
82
+ """
83
+ words = drop_speckle(load_tsv(tsv))
84
+ lines = group_lines(words)
85
+ texts = [l.text for l in lines]
86
+
87
+ confidence = 0.0
88
+ if profile_id:
89
+ matches = [p for p in loader.available() if p.id == profile_id]
90
+ if not matches:
91
+ raise loader.ProfileError(f"no such profile: {profile_id}")
92
+ profile = matches[0]
93
+ confidence = 1.0
94
+ else:
95
+ selected = loader.select(texts)
96
+ if selected is None:
97
+ raise loader.ProfileError(
98
+ "no profile fingerprint matched. That is a real signal: either the "
99
+ "receipt is from another store, or the layout changed."
100
+ )
101
+ profile, confidence = selected
102
+
103
+ extraction = extract(lines, profile)
104
+ if printed_total is not None:
105
+ # Someone read the total off the paper because the parser could not, so
106
+ # it beats anything OCR produced.
107
+ extraction.printed_total_minor = printed_total
108
+ extraction.printed_total_supplied = True
109
+ validation = validate(extraction, profile)
110
+
111
+ if prepared is not None and not validation.balanced:
112
+ from .pipeline import recheck # noqa: PLC0415 - only needed with an image
113
+
114
+ agreed = recheck.repair(extraction, validation, prepared, profile, lines)
115
+ if agreed is not None:
116
+ recheck.apply(extraction, agreed, lines)
117
+ # Re-run the whole check instead of patching the totals, so the
118
+ # per-line arithmetic and warnings match the new prices.
119
+ validation = validate(extraction, profile)
120
+ validation.warnings.append(
121
+ f"price_column_reread:{len(agreed.changes)}_of_{agreed.disputed}"
122
+ )
123
+ if agreed.recovered:
124
+ validation.warnings.append(f"lines_recovered:{len(agreed.recovered)}")
125
+
126
+ rid = receipt_id or uuid.uuid4().hex
127
+
128
+ doc = build(
129
+ receipt_id=rid,
130
+ lines=lines,
131
+ extraction=extraction,
132
+ validation=validation,
133
+ profile=profile,
134
+ profile_confidence=confidence,
135
+ ocr_meta=ocr_meta or {"engine": "tesseract", "lang": "ita", "psm": 4, "oem": 1},
136
+ source={"ingested_at": datetime.now(UTC).isoformat()},
137
+ review=(
138
+ review
139
+ if review is not None
140
+ else document_module.default_review(validation, extraction)
141
+ ),
142
+ )
143
+ return doc, validation.status
144
+
145
+
146
+ # Below 26px Tesseract starts losing lines, so a second pass at a larger size is
147
+ # worth the time. A receipt at 25px was fine; one at 20px lost two lines.
148
+ SMALL_GLYPH = 26
149
+ COMFORTABLE_GLYPH = 30
150
+
151
+
152
+ def _closer(candidate: dict, current: dict) -> bool:
153
+ """Whether a second reading is the better of the two.
154
+
155
+ Balancing wins. Failing that, the smaller delta wins. With no total to
156
+ compare against, more lines found wins, since a missed line takes its
157
+ money with it.
158
+ """
159
+ a, b = candidate["totals"], current["totals"]
160
+ if a["balanced"] != b["balanced"]:
161
+ return bool(a["balanced"])
162
+ if a["delta_minor"] is not None and b["delta_minor"] is not None:
163
+ return abs(a["delta_minor"]) < abs(b["delta_minor"])
164
+ if (a["printed_total_minor"] is None) != (b["printed_total_minor"] is None):
165
+ return b["printed_total_minor"] is None
166
+ return len(a and candidate["line_items"]) > len(current["line_items"])
167
+
168
+
169
+ def _read_larger(prepared, tsv: str, **kwargs):
170
+ """Read the photo again, enlarged, when the first read came out small.
171
+
172
+ Only runs on a receipt that did not add up, so a good one never pays for it.
173
+ Enlarging adds no detail, but Tesseract's line model does better at the size
174
+ it expects: on one real receipt it found 33 lines instead of 31.
175
+ """
176
+ from PIL import Image # noqa: PLC0415 - the parsing core runs without Pillow
177
+
178
+ from .pipeline.ocr import run # noqa: PLC0415
179
+
180
+ glyph = median_glyph_height(drop_speckle(load_tsv(tsv)))
181
+ if not glyph or glyph >= SMALL_GLYPH:
182
+ return None
183
+
184
+ scale = COMFORTABLE_GLYPH / glyph
185
+ bigger = prepared.resize(
186
+ (round(prepared.width * scale), round(prepared.height * scale)),
187
+ Image.Resampling.LANCZOS,
188
+ )
189
+ result = run(bigger, psm=4)
190
+ document, status = parse_tsv(result.tsv, prepared=bigger, **kwargs)
191
+ return document, status, result.tsv, bigger
192
+
193
+
194
+ def parse_image(
195
+ path: Path,
196
+ *,
197
+ profile_id: str | None,
198
+ review: dict[str, Any] | None = None,
199
+ printed_total: int | None = None,
200
+ ):
201
+ """Parse a photo, carrying the raw TSV along for archiving.
202
+
203
+ The document comes back with two private keys, `_tsv` and `_prepared`: the
204
+ OCR it was built from and the image OCR actually saw. Whoever archives them
205
+ pops them off first. Item bounding boxes use the prepared image's
206
+ coordinates, so cropping by box needs that image, not the original photo.
207
+ """
208
+ tsv, ocr_meta = read_input(path)
209
+ # Removed before parse_tsv: build() copies ocr_meta into the document, and a
210
+ # PIL image in there makes it unserialisable.
211
+ prepared = ocr_meta.pop("_prepared", None)
212
+ shared = {
213
+ "profile_id": profile_id,
214
+ "review": review,
215
+ "ocr_meta": ocr_meta,
216
+ "printed_total": printed_total,
217
+ }
218
+ document, status = parse_tsv(tsv, prepared=prepared, **shared)
219
+
220
+ if prepared is not None and not document["totals"]["balanced"]:
221
+ second = _read_larger(prepared, tsv, **shared)
222
+ if second is not None and _closer(second[0], document):
223
+ # The archived image must be the one the TSV describes, or crops on
224
+ # the review page point at the wrong pixels.
225
+ document, status, tsv, prepared = second
226
+
227
+ # Handed to the archiver, which pops both before the document is written.
228
+ document["_tsv"] = tsv
229
+ document["_prepared"] = prepared
230
+ return document, status
231
+
232
+
File without changes
@@ -0,0 +1,159 @@
1
+ """Building the JSON document for one receipt.
2
+
3
+ Money leaves here as whole cents, with a _minor suffix so the unit is obvious at
4
+ the call site. Unreadable fields come out null with a warning, never guessed.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import asdict
10
+ from decimal import Decimal
11
+ from typing import Any
12
+
13
+ from quadra_core.pipeline.extract import Extraction
14
+ from quadra_core.pipeline.lines import Line
15
+ from quadra_core.pipeline.validate import Validation
16
+ from quadra_core.profiles.loader import Profile
17
+
18
+ SCHEMA_VERSION = "1.0.0"
19
+
20
+ # Below this, the least confident word in a description is too poor to file
21
+ # unseen. Arithmetic cannot catch a misread product name, so this is the only
22
+ # thing that sends one for review.
23
+ LOW_CONFIDENCE_DESCRIPTION = 0.5
24
+
25
+
26
+ def no_review(state: str = "not_required") -> dict[str, Any]:
27
+ """Build the review block for a document nobody has decided anything about.
28
+
29
+ Who gets checked is up to whatever uses this library, so the default says
30
+ only that no decision has been made yet, rather than claiming the receipt
31
+ was checked or does not need checking.
32
+ """
33
+ return {
34
+ "required": False,
35
+ "reason": None,
36
+ "sampled": False,
37
+ "mode": "none",
38
+ "state": state,
39
+ "blind_item_indexes": [],
40
+ }
41
+
42
+
43
+ def default_review(
44
+ validation: Validation, extraction: Extraction
45
+ ) -> dict[str, Any]:
46
+ """Whether a person has to look, when the caller has not said.
47
+
48
+ The caller owns this decision and can pass its own block instead. This is
49
+ the default, and it errs towards asking: a receipt that does not add up, or
50
+ that carries a description nobody could read, is not one to file unseen.
51
+ """
52
+ reasons = []
53
+ if not validation.balanced:
54
+ reasons.append("totals_do_not_balance")
55
+ if not extraction.items:
56
+ reasons.append("no_items")
57
+ worst = min((i.confidence for i in extraction.items), default=1.0)
58
+ if worst < LOW_CONFIDENCE_DESCRIPTION:
59
+ reasons.append(f"unreadable_description:{worst:.2f}")
60
+
61
+ if not reasons:
62
+ return no_review()
63
+ return {
64
+ "required": True,
65
+ "reason": "; ".join(reasons),
66
+ "sampled": False,
67
+ "mode": "full",
68
+ "state": "pending",
69
+ "blind_item_indexes": [],
70
+ }
71
+
72
+
73
+ def _quantity(value: Decimal | None) -> dict[str, Any] | None:
74
+ if value is None:
75
+ return None
76
+ # A string rather than a float, since weighed goods need fractions that
77
+ # binary floats cannot hold exactly.
78
+ return {"value": format(value.normalize(), "f"), "unit": "each"}
79
+
80
+
81
+ def build(
82
+ *,
83
+ receipt_id: str,
84
+ lines: list[Line],
85
+ extraction: Extraction,
86
+ validation: Validation,
87
+ profile: Profile,
88
+ profile_confidence: float,
89
+ ocr_meta: dict[str, Any],
90
+ source: dict[str, Any],
91
+ review: dict[str, Any],
92
+ ) -> dict[str, Any]:
93
+ """Assemble the full receipt document."""
94
+ mean_conf = sum(l.conf for l in lines) / len(lines) if lines else 0.0
95
+ return {
96
+ "schema_version": SCHEMA_VERSION,
97
+ "receipt_id": receipt_id,
98
+ "source": source,
99
+ "profile": {
100
+ "id": profile.id,
101
+ "version": profile.version,
102
+ "calibrated": profile.calibrated,
103
+ "match_confidence": round(profile_confidence, 3),
104
+ },
105
+ "ocr": {**ocr_meta, "mean_word_confidence": round(mean_conf, 1)},
106
+ "merchant": {"name": None, "store_id": None, "address": None, "vat_id": None},
107
+ "transaction": {
108
+ "datetime": None,
109
+ "currency": profile.currency,
110
+ "receipt_number": None,
111
+ "payment_method": None,
112
+ },
113
+ "line_items": [
114
+ {
115
+ "index": i,
116
+ "description_raw": item.description_raw,
117
+ "description": item.description,
118
+ "quantity": _quantity(item.quantity),
119
+ "quantity_source": item.quantity_source,
120
+ "unit_price_minor": item.unit_price_minor,
121
+ "line_total_minor": item.line_total_minor,
122
+ "discounts_minor": 0,
123
+ "vat_code": item.vat_code,
124
+ "confidence": item.confidence,
125
+ "provenance": {"line_index": item.line_index, "bbox": list(item.bbox)},
126
+ "flags": item.flags,
127
+ }
128
+ for i, item in enumerate(extraction.items)
129
+ ],
130
+ "adjustments": [
131
+ {
132
+ "kind": adjustment.kind,
133
+ "label": adjustment.label,
134
+ "amount_minor": adjustment.amount_minor,
135
+ "provenance": {"line_index": adjustment.line_index},
136
+ }
137
+ for adjustment in extraction.adjustments
138
+ ],
139
+ "totals": {
140
+ "items_subtotal_minor": validation.items_subtotal_minor,
141
+ "discounts_minor": sum(a.amount_minor for a in extraction.adjustments),
142
+ "printed_total_minor": validation.printed_total_minor,
143
+ "computed_total_minor": validation.items_subtotal_minor,
144
+ "balanced": validation.balanced,
145
+ "delta_minor": validation.delta_minor,
146
+ },
147
+ "validation": {
148
+ "status": validation.status,
149
+ "checks": [asdict(c) for c in validation.checks],
150
+ "warnings": validation.warnings,
151
+ },
152
+ "review": {
153
+ **review,
154
+ "reviewed_at": None,
155
+ "duration_ms": None,
156
+ "rows_expanded": 0,
157
+ "corrections": [],
158
+ },
159
+ }