Files
receipity/receipt-parsing-keywords.json
T
2026-08-30 12:24:29 +02:00

187 lines
5.7 KiB
JSON

{
"_comment": "Keyword/pattern config for the receipt parser. See docs/receipt-parsing.md to add a new shop. Match rules are implemented in receipt_parse_config.dart (short tokens use word boundaries; longer tokens use substring + compact OCR form).",
"known_stores": {
"_comment": "Used to detect store name from the header block (first 1-5 lines). Match against the STORE_HEADER lines.",
"list": [
"JUMBO",
"LIDL",
"KRUIDVAT",
"ALBERT HEIJN",
"AH",
"ALDI",
"PLUS",
"DIRK",
"COOP",
"SPAR",
"VOMAR",
"EKOPLAZA",
"HOOGVLIET",
"ACTION",
"ETOS",
"WORLD TOKO"
],
"fallback_rule": "If no known store name matches, take the first non-address, non-phone-number line in the header block as a best-effort store name (free text)."
},
"column_header_keywords": {
"_comment": "Lines matching these are table headers, not data — ignore entirely.",
"list": [
"OMSCHRIJVING",
"BEDRAG IN",
"ARTIKEL",
"PRIJS",
"AANTAL ARTIKEL",
"BESCHRIJVING"
],
"separator_pattern": "^[=\\-#]{3,}$"
},
"quantity_modifier_pattern": {
"_comment": "Regex to detect an 'N X price' line, optionally followed by a unit word. Group 1 = quantity, group 2 = unit price.",
"regex": "^\\s*(\\d+)\\s*[xX]\\s*(\\d+[,.]\\d{2})\\s*(PER\\s+STUK|PER\\s+KG|ST\\.?|STUKS?)?\\s*$",
"inline_regex": "(\\d+)\\s*[xX]\\s*(\\d+[,.]\\d{2})\\s*(PER\\s+STUK|PER\\s+KG)?",
"note": "Use 'regex' for standalone quantity lines (Jumbo-style). Use 'inline_regex' to detect the same pattern embedded within a product-name line (Kruidvat-style)."
},
"discount_keywords": {
"_comment": "Lines matching ANY of these AND carrying a negative amount are discounts, not products.",
"list": [
"ACTIE",
"KORTING",
"IN PRIJS VERLAAGD",
"BONUSKORTING",
"KASSAKORTING",
"ZEGELKORTING",
"REDUCTIE",
"AANBIEDING"
],
"negative_amount_pattern": "-\\d+[,.]\\d{2}",
"receipt_level_total_discount_keywords": [
"TOTAAL KORTING",
"TOTALE KORTING",
"TOTAAL BESPARING"
]
},
"total_keywords": {
"_comment": "Order matters for disambiguation — check subtotal_keywords before grand_total_keywords, since 'SUBTOTAAL' also contains 'TOTAAL'.",
"subtotal_keywords": [
"SUBTOTAAL",
"SUB TOTAAL"
],
"grand_total_keywords": [
"TOTAAL",
"TOTAL"
],
"tax_inclusive_total_keywords": [
"TOTAAL INCL. BTW",
"TOTAAL INCL BTW",
"TOTAAL EXCL. BTW",
"TOTAAL EXCL BTW"
],
"rule": "The FIRST grand_total match encountered after the last item line is the authoritative receipt total. Any match containing a discount keyword (see discount_keywords.receipt_level_total_discount_keywords) is the total discount amount, not the grand total — check discount keywords first."
},
"payment_keywords": {
"_comment": "Lines matching these mark the start of the payment block. Everything from the first match onward (until noise/footer) should be excluded from item parsing, though some fields can be stored as receipt metadata.",
"list": [
"BETAALD",
"BETALING",
"VISA",
"MASTERCARD",
"MAESTRO",
"V PAY",
"PIN",
"CONTACTLOZE BETALING",
"AKKOORD",
"KAARTNR",
"KAART:",
"TRANSACTIE",
"TERMINAL",
"MERCHANT",
"AUTH. CODE",
"AUTH CODE",
"KAARTHOUDER",
"CVM",
"CHIP",
"PAY",
"IDEAL"
],
"metadata_fields_to_extract": [
"payment_method (e.g. VISA, MASTERCARD, PIN, iDEAL)",
"transaction_id",
"auth_code",
"terminal_id"
]
},
"tax_breakdown_keywords": {
"_comment": "Lines describing BTW/tax categories, not products. Also used to detect the trailing tax-category letter appended to some item lines.",
"list": [
"BTW",
"BTW-CODE",
"BEDR.EXCL",
"BEDR.INCL",
"BEDRAG EXCL",
"BEDRAG INCL"
],
"category_row_pattern": "^[A-Z]\\s+\\d+%",
"trailing_category_letter_pattern": "\\s[A-Z]$",
"note": "trailing_category_letter_pattern strips a single trailing tax-category letter (e.g. ' B' or ' C') from the end of an item line if present, before extracting the price. Do not treat it as part of the price."
},
"noise_footer_keywords": {
"_comment": "Lines matching these (or appearing after the payment block) are noise — ignore entirely.",
"list": [
"BEWAAR",
"KASSABON",
"RETOURNEREN",
"RETOUR",
"DAGEN MET JE BON",
"WWW.",
".COM",
".NL",
"OPENINGSTIJDEN",
"MAANDAG",
"DINSDAG",
"WOENSDAG",
"DONDERDAG",
"VRIJDAG",
"ZATERDAG",
"ZONDAG",
"FACEBOOK",
"INSTAGRAM",
"BEDANKT",
"TOT ZIENS",
"MEDEWERKER",
"KASSA",
"TRANS",
"STORE",
"POS",
"REGISTREER",
"PLUS-APP",
"BESPAAR MEER",
"ZELF SERVICE",
"PRIVACY STATEMENT",
"KOOPZEGELS"
]
},
"quantity_and_price_general": {
"price_pattern": "\\d+[,.]\\d{2}",
"decimal_separator": ",",
"note": "Some receipts (e.g. World Toko) show weight-based items like '900g', '500ml', '400g' as part of the product name — these are NOT quantities/units for the parser's purposes, just part of the descriptive name. Only treat a leading integer directly adjacent to 'X' or 'x' as a quantity."
},
"line_item_count_keywords": {
"_comment": "Some receipts show an item count line near the payment block, useful as a validation cross-check (does parsed item count match this?).",
"list": [
"AANTAL ARTIKEL",
"AANTAL ARTIKELEN",
"ART.",
"ARTIKELEN"
]
}
}