hsa-app/OCR/receipt_group.py
Jean-Michel Tremblay 344663d87d Add local receipt OCR + grouping (OCR/)
Two local, GPU-backed tools that talk to a local ollama server (no cloud,
no API key):

- receipt_ocr.py: receipt image -> ordered list of rows of text, via
  Qwen2.5-VL 7B. Zero deps (stdlib + ollama HTTP API).
- receipt_group.py: OCR rows -> structured receipt (store, line items with
  multi-row details merged, total, metadata), via Mistral Small 24B. Detail
  rows are attached by arithmetic reconciliation, not position; a prompt
  completeness clause guards against dropped rows; a deterministic --json
  reconciliation pass audits item math, subtotal/total, and unit count.

README documents setup, an ollama operating runbook, and design rationale.
Includes the ALDI example image + its OCR rows.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-22 22:48:17 -04:00

284 lines
11 KiB
Python
Executable file

#!/usr/bin/env python3
"""Group raw receipt-OCR rows into a structured object with a local LLM.
Takes the text output of receipt_ocr.py (one row per line, optionally prefixed
with a line number) and asks a local ollama model to reconstruct the receipt:
store identity, the line items (merging multi-row items such as weighed produce
and "qty @ price" lines), the total, and any other useful metadata.
Local-only: no cloud, no API key. Uses a text model (default qwen2.5:7b).
Usage:
python3 receipt_group.py [ocr_rows.txt] # pretty summary
python3 receipt_group.py ocr_rows.txt --json # raw structured JSON
Env:
OLLAMA_HOST default http://127.0.0.1:11434
OLLAMA_MODEL default qwen2.5:7b
"""
import json
import os
import re
import sys
import urllib.request
DEFAULT_INPUT = os.path.join(os.path.dirname(__file__), "example", "data", "aldi_ocr.txt")
HOST = os.environ.get("OLLAMA_HOST", "http://127.0.0.1:11434").rstrip("/")
MODEL = os.environ.get("OLLAMA_MODEL", "mistral-small3.2:24b")
PROMPT = """You are given the OCR transcription of a single retail receipt, one \
physical row per line. Each line may start with a row number (e.g. "23 ...") — \
ignore the row number itself, it is not part of the receipt text.
Reconstruct the receipt as structured data. Rules:
- A purchased ITEM begins on a row that has a product code (a run of digits) and \
a price. EVERY row that begins with a product code starts a NEW item — never \
attach a code-bearing row to the item above it, even when the name and price are \
identical to the row above. Two identical consecutive code rows are TWO separate \
purchases, each its own entry. Do not merge or de-duplicate them.
- A DETAIL row is any row that does NOT begin with a product code, e.g.:
* quantity lines like "2 @ 1.15" (count @ unit price)
* weight lines like "1.17 lb x 1.89/lb"
* gross/tare/net lines like "(G) 3.10 lb - (T) 0.06 lb" and "(N) 3.04 lb x 1.79/lb"
Attach each detail row to the item it actually describes. Decide that by the \
MATH, not by position: count x unit price, or weight x rate, equals that item's \
charged price. (A gross/tare line that has no price of its own stays with the \
net line it accompanies.) The describing item is usually adjacent, but verify by \
reconciliation rather than assuming it is the one above or below.
- The item's "price" is the dollar amount printed on the item's own (first) row — \
the amount actually charged. Keep two decimals.
- Put the exact original row strings that belong to each item in its "rows" array, \
in order, WITHOUT the leading row number.
- Header rows (store name, branch, address, website, cashier) are NOT items.
- Payment, auth, subtotal, tax, total, item-count and marketing/footer rows are \
NOT items — pull the useful values into the metadata fields instead.
- Do not invent values. If something is not present, omit it.
COMPLETENESS — this is critical:
- Every item row AND every detail row from the input must appear in exactly ONE \
item's "rows" array, copied VERBATIM (same digits, same spacing, same price). \
You may not drop, skip, summarize, or rewrite any row.
- NEVER discard a quantity line ("2 @ 1.46"), a weight line ("1.17 lb x 1.89/lb"), \
or a gross/tare/net line. Attach each to the item its math reconciles with.
- Before you answer, walk the input top to bottom and confirm that every row is \
either placed in an item's "rows" or is a header/payment/footer row that you \
deliberately excluded. No item or detail row may be missing.
Return STRICT JSON only, matching the requested schema."""
SCHEMA = {
"type": "object",
"properties": {
"store_name": {"type": "string"},
"store_branch": {"type": "string"},
"store_address": {"type": "string"},
"items": {
"type": "array",
"items": {
"type": "object",
"properties": {
"name": {"type": "string"},
"code": {"type": "string"},
"price": {"type": "number"},
"rows": {"type": "array", "items": {"type": "string"}},
},
"required": ["name", "price", "rows"],
},
},
"total": {"type": "number"},
"metadata": {
"type": "object",
"properties": {
"cashier": {"type": "string"},
"purchase_datetime": {"type": "string"},
"payment_method": {"type": "string"},
"subtotal": {"type": "number"},
"tax": {"type": "number"},
"item_count": {"type": "integer"},
},
},
},
"required": ["store_name", "items", "total"],
}
def group(rows_text):
payload = {
"model": MODEL,
"messages": [
{"role": "user", "content": PROMPT + "\n\n--- RECEIPT OCR ---\n" + rows_text},
],
"stream": False,
"format": SCHEMA,
"options": {"temperature": 0, "num_ctx": 8192},
}
req = urllib.request.Request(
f"{HOST}/api/chat",
data=json.dumps(payload).encode("utf-8"),
headers={"Content-Type": "application/json"},
)
with urllib.request.urlopen(req, timeout=600) as resp:
body = json.load(resp)
return json.loads(body["message"]["content"])
# --- Deterministic reconciliation -------------------------------------------
# The LLM decides the grouping; arithmetic decides whether to trust it. Detail
# rows carry their own math, so we recompute each item's price from its details
# and balance the whole receipt against the printed subtotal / total / count.
QTY_RE = re.compile(r"(\d+)\s*@\s*\$?(\d+(?:\.\d+)?)") # "4 @ 0.57"
WEIGHT_RE = re.compile(r"(\d+(?:\.\d+)?)\s*lb\s*[xX]\s*\$?(\d+(?:\.\d+)?)\s*/\s*lb") # "1.17 lb x 1.89/lb"
TOL = 0.02 # cents of slack for rounding on weighed items
def check_item(it):
"""Reconcile one item from its detail rows.
Returns (status, qty, note) where status is 'ok' | 'warn' | 'none'.
'none' = no math-bearing detail row, so there's nothing to verify.
"""
price = float(it.get("price", 0))
qty = 1
expected = None
note = ""
for row in it.get("rows", [])[1:]: # skip the anchor row; details only
m = QTY_RE.search(row)
if m:
qty = int(m.group(1))
expected = round(qty * float(m.group(2)), 2)
note = f"{m.group(1)} @ {m.group(2)} = {expected:.2f}"
continue
w = WEIGHT_RE.search(row)
if w:
expected = round(float(w.group(1)) * float(w.group(2)), 2)
note = f"{w.group(1)} lb x {w.group(2)}/lb = {expected:.2f}"
if expected is None:
return ("none", qty, "")
if abs(expected - price) <= TOL:
return ("ok", qty, note)
return ("warn", qty, f"{note} != charged {price:.2f}")
def validate(obj):
items = obj.get("items", [])
meta = obj.get("metadata", {}) or {}
per_item = [check_item(it) for it in items]
items_sum = round(sum(float(it.get("price", 0)) for it in items), 2)
unit_count = sum(q for _, q, _ in per_item)
subtotal = meta.get("subtotal")
tax = meta.get("tax") or 0
total_printed = obj.get("total")
count_printed = meta.get("item_count")
report = {
"per_item": [{"status": s, "qty": q, "note": n} for s, q, n in per_item],
"items_sum": items_sum,
"subtotal_printed": subtotal,
"subtotal_ok": subtotal is not None and abs(items_sum - subtotal) <= TOL,
"total_computed": round(items_sum + tax, 2),
"total_printed": total_printed,
"total_ok": total_printed is not None and abs(items_sum + tax - total_printed) <= TOL,
"unit_count": unit_count,
"unit_count_printed": count_printed,
"unit_count_ok": count_printed is not None and unit_count == count_printed,
}
return report
# --- Rendering ----------------------------------------------------------------
MARK = {"ok": "", "warn": "", "none": " "}
def yn(ok):
return "" if ok else ""
def pretty(obj, report):
out = []
out.append(f"Store: {obj.get('store_name', '')}")
if obj.get("store_branch"):
out.append(f"Branch: {obj['store_branch']}")
if obj.get("store_address"):
out.append(f"Address: {obj['store_address']}")
out.append("")
items = obj.get("items", [])
checks = report["per_item"]
out.append(f"Items ({len(items)}):")
for it, chk in zip(items, checks):
code = it.get("code", "")
head = f" {MARK[chk['status']]} {it.get('name', '')} ${it.get('price', 0):.2f}"
if code:
head += f" [{code}]"
out.append(head)
for extra in it.get("rows", [])[1:]: # detail rows only
out.append(f" · {extra}")
if chk["status"] == "warn":
out.append(f"{chk['note']}")
out.append("")
# Receipt-level balance
out.append("Reconciliation:")
out.append(
f" Items sum: ${report['items_sum']:.2f}"
+ (
f" (printed subtotal ${report['subtotal_printed']:.2f}) {yn(report['subtotal_ok'])}"
if report["subtotal_printed"] is not None
else ""
)
)
if report["total_printed"] is not None:
out.append(
f" + tax/total: ${report['total_computed']:.2f}"
f" (printed TOTAL ${report['total_printed']:.2f}) {yn(report['total_ok'])}"
)
if report["unit_count_printed"] is not None:
out.append(
f" Unit count: {report['unit_count']}"
f" (printed {report['unit_count_printed']} ITEMS) {yn(report['unit_count_ok'])}"
)
meta = obj.get("metadata", {})
if meta:
out.append("")
out.append("Metadata:")
for k, v in meta.items():
out.append(f" {k}: {v}")
return "\n".join(out)
def main():
args = [a for a in sys.argv[1:] if not a.startswith("--")]
as_json = "--json" in sys.argv
path = args[0] if args else DEFAULT_INPUT
if not os.path.exists(path):
sys.exit(f"input not found: {path}")
with open(path, encoding="utf-8") as f:
rows_text = f.read()
obj = group(rows_text)
if as_json:
obj["validation"] = validate(obj)
print(json.dumps(obj, ensure_ascii=False, indent=2))
else:
# Just the groups: each purchased item, with the rows that belong together.
items = obj.get("items", [])
width = len(str(len(items)))
for i, it in enumerate(items, 1):
rows = it.get("rows", [])
head = rows[0] if rows else f"{it.get('name', '')} {it.get('price', 0):.2f}"
print(f"{str(i).rjust(width)} {head}")
for extra in rows[1:]:
print(f"{' ' * width} {extra}")
if __name__ == "__main__":
main()