# Expiry-date extraction cascade, split out of classify_ocr_server.py so it # can be imported (and offline-tested against captured OCR lines) without # loading any models. Pure regex/string logic - no torch/paddle imports. import re EXP_KEYWORD_RE = re.compile( r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)', re.IGNORECASE, ) DD_MM_YYYY_RE = re.compile( r'(? bool: if len(val) != 8 or not val.isdigit(): return False day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8]) return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099 def is_plausible_date_parts(day: str, month: str, year: str) -> bool: # Sanity gate for the lenient stage: it happily assembles junk like # "00/22/26" or "1/3/06" out of garbled digit runs. A frozen-food # expiry is always a real calendar day within a few years of today. if not (day.isdigit() and month.isdigit() and year.isdigit()): return False d, m = int(day), int(month) y = int(year) if len(year) == 4 else 2000 + int(year) return 1 <= d <= 31 and 1 <= m <= 12 and 2020 <= y <= 2039 def format_ddmmyyyy(val: str) -> str: if is_valid_ddmmyyyy_digits(val): return f"{val[0:2]}/{val[2:4]}/{val[4:8]}" return val.upper() def format_ddmmyy(val: str) -> str: if len(val) == 6 and val.isdigit(): day, month = int(val[0:2]), int(val[2:4]) if 1 <= day <= 31 and 1 <= month <= 12: return f"{val[0:2]}/{val[2:4]}/{val[4:6]}" return val.upper() def line_has_exp_keyword(line: str) -> bool: if EXP_KEYWORD_RE.search(line): return True # BB05032027 — keyword directly followed by digits return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line)) def clean_date_line(line: str) -> str: # 1) Replace "1)" with "0" cleaned = line.replace("1)", "0") # 2) Replace "()" with "0" cleaned = cleaned.replace("()", "0") # Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits) cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned) cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned) # Clean 012/112 month misrecognitions (e.g. 020122027 -> 02022027) — # but only when the line does NOT already hold a valid date: a real # "01122026" (= 01/12/2026) also matches the 112 pattern (0+112+2026) # and would be mangled into 7-digit junk. if not (DDMMYYYY_RE.search(cleaned) or DD_MM_YYYY_RE.search(cleaned)): cleaned = re.sub(r'(?02\g<2>', cleaned) cleaned = re.sub(r'(?\g<2>02\g<3>\g<4>', cleaned) cleaned = re.sub(r'(?02\g<2>', cleaned) cleaned = re.sub(r'(?\g<2>02\g<3>\g<4>', cleaned) # Run contextual replacements for _ in range(3): # letter o/O flanked by digits or boundary -> 0 cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned) # letter I/i/l/| flanked by digits -> 1 cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned) # letter S/s flanked by digits -> 5 cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned) # letter Z/z flanked by digits -> 2 cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned) # letter B flanked by digits -> 8 cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned) cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned) return cleaned def extract_expired_date(text_lines): """Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY.""" if not text_lines: return None, None, None cleaned_lines = [clean_date_line(line) for line in text_lines] def pick(match, idx, cleaned_line, formatter=None): raw = match.group(0) original_line = text_lines[idx].strip() if match.lastindex and match.lastindex >= 3: formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}" elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit(): digits = match.group(1) if len(digits) == 8: formatted = format_ddmmyyyy(digits) elif len(digits) == 6: formatted = format_ddmmyy(digits) else: formatted = digits elif formatter: formatted = formatter(raw) else: formatted = raw.strip().upper() return formatted, idx, original_line # 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027) for idx, line in enumerate(cleaned_lines): if not line_has_exp_keyword(line): continue match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line) if match: return pick(match, idx, line) # 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027) for idx, line in enumerate(cleaned_lines): if not line_has_exp_keyword(line): continue match = DD_MM_YYYY_RE.search(line) if match: return pick(match, idx, line) # 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits) for idx, line in enumerate(cleaned_lines): match = KEYWORD_DIGITS_RE.search(line) if match: digits = match.group(1) if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits): return format_ddmmyyyy(digits), idx, text_lines[idx].strip() if len(digits) == 6: return format_ddmmyy(digits), idx, text_lines[idx].strip() # 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027) for idx, line in enumerate(cleaned_lines): if not line_has_exp_keyword(line): continue match = LENIENT_DATE_RE.search(line) if match and is_plausible_date_parts(match.group(1), match.group(2), match.group(3)): return pick(match, idx, line) # 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes # detects "BB"/"Baik digunakan" as its own box, separate from the date # digits in a neighboring box, e.g. "BB" / "05032027" as two lines). for idx, line in enumerate(cleaned_lines): if not line_has_exp_keyword(line): continue for j in (idx + 1, idx - 1, idx + 2): if j < 0 or j >= len(cleaned_lines) or j == idx: continue neighbor = cleaned_lines[j] combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}" match = (BB_ATTACHED_DATE_RE.search(combined) or DDMMYYYY_RE.search(combined) or DD_MM_YYYY_RE.search(combined)) if match: report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx return pick(match, report_idx, combined) # 4) Any line — spaced DD MM YYYY (excluding store price-tag lines) for idx, line in enumerate(cleaned_lines): if PRICE_TAG_RE.search(line): continue match = DD_MM_YYYY_RE.search(line) if match: return pick(match, idx, line) # 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context) for idx, line in enumerate(cleaned_lines): if PRICE_TAG_RE.search(line): continue for match in DDMMYYYY_RE.finditer(line): digits = f"{match.group(1)}{match.group(2)}{match.group(3)}" if is_valid_ddmmyyyy_digits(digits): # Skip if this 8-digit block is the only digits and looks like SKU on label top if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line): if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I): continue return format_ddmmyyyy(digits), idx, text_lines[idx].strip() # 6) Legacy patterns (slashes, month names, etc.) date_patterns = [ r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b', r'\b\d{4}[-./]\d{2}[-./]\d{2}\b', r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b', ] for idx, line in enumerate(cleaned_lines): if not line_has_exp_keyword(line): continue for pat in date_patterns: match = re.search(pat, line, re.IGNORECASE) if match: return match.group(0).upper(), idx, text_lines[idx].strip() return None, None, None def line_contains_expired_date(line: str, expired_date: str) -> bool: if not line or not expired_date: return False digits_only = re.sub(r"\D", "", expired_date) line_digits = re.sub(r"\D", "", line) if len(digits_only) >= 6 and digits_only in line_digits: return True compact = expired_date.replace("/", "") return compact in line.replace(" ", "") or expired_date in line def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len): """Pick OCR box index for cropping; prefer the line that actually contains the date.""" if not expired_date or polys_len <= 0: return None if ( expired_idx is not None and expired_idx < polys_len and expired_idx < len(text_lines) and line_contains_expired_date(text_lines[expired_idx], expired_date) ): return expired_idx keyword_match = None for idx, line in enumerate(text_lines): if idx >= polys_len: break if not line_contains_expired_date(line, expired_date): continue if line_has_exp_keyword(line): return idx if keyword_match is None: keyword_match = idx if keyword_match is not None: return keyword_match if expired_idx is not None and expired_idx < polys_len: return expired_idx return None