Files
pfm-ocr/backend/config/date_extract.py
T
fhanyuh caf8e98378 chore: normalize line endings (CRLF -> LF)
No content changes: git diff --ignore-all-space over these files is empty.
The churn came from editing on Windows against a repo checked out with LF.
2026-08-27 10:40:49 +07:00

267 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# Expiry-date extraction cascade, split out of classify_ocr_server.py so it
# can be imported (and offline-tested against captured OCR lines) without
# loading any models. Pure regex/string logic - no torch/paddle imports.
import re
EXP_KEYWORD_RE = re.compile(
r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)',
re.IGNORECASE,
)
DD_MM_YYYY_RE = re.compile(
r'(?<!\d)(0[1-9]|[12]\d|3[01]).*?(0[1-9]|1[0-2]).*?(20\d{2})(?!\d)'
)
DDMMYYYY_RE = re.compile(
r'(?<!\d)(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)'
)
BB_ATTACHED_DATE_RE = re.compile(
r'\b(?:bb|bestbefore)\s*[:.-]?\s*(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)',
re.IGNORECASE,
)
KEYWORD_DIGITS_RE = re.compile(
r'(?:exp|expired|tgl|expiry|bbd|before|best|bb|baik|digunakan)\s*[:.-]?\s*(\d{6,8})\b',
re.IGNORECASE,
)
LENIENT_DATE_RE = re.compile(
r'(?<!\d)(\d{1,2}).*?(\d{1,2}).*?((?:20)?\d{2})(?!\d)'
)
# Store price-tag / label-printer lines ("Printed:04/05/2026 19:53",
# "Rp.6,800/PC"). The date on these is the moment the shelf label was
# printed, never the product's expiry - excluded from the keyword-less
# stages so it can't shadow the real date elsewhere on the package.
PRICE_TAG_RE = re.compile(r'(?i)printed\s*[:.]?|rp\s*[.,]?\s*\d')
def is_valid_ddmmyyyy_digits(val: str) -> bool:
if len(val) != 8 or not val.isdigit():
return False
day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8])
return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099
def is_plausible_date_parts(day: str, month: str, year: str) -> bool:
# Sanity gate for the lenient stage: it happily assembles junk like
# "00/22/26" or "1/3/06" out of garbled digit runs. A frozen-food
# expiry is always a real calendar day within a few years of today.
if not (day.isdigit() and month.isdigit() and year.isdigit()):
return False
d, m = int(day), int(month)
y = int(year) if len(year) == 4 else 2000 + int(year)
return 1 <= d <= 31 and 1 <= m <= 12 and 2020 <= y <= 2039
def format_ddmmyyyy(val: str) -> str:
if is_valid_ddmmyyyy_digits(val):
return f"{val[0:2]}/{val[2:4]}/{val[4:8]}"
return val.upper()
def format_ddmmyy(val: str) -> str:
if len(val) == 6 and val.isdigit():
day, month = int(val[0:2]), int(val[2:4])
if 1 <= day <= 31 and 1 <= month <= 12:
return f"{val[0:2]}/{val[2:4]}/{val[4:6]}"
return val.upper()
def line_has_exp_keyword(line: str) -> bool:
if EXP_KEYWORD_RE.search(line):
return True
# BB05032027 — keyword directly followed by digits
return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line))
def clean_date_line(line: str) -> str:
# 1) Replace "1)" with "0"
cleaned = line.replace("1)", "0")
# 2) Replace "()" with "0"
cleaned = cleaned.replace("()", "0")
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
# Clean 012/112 month misrecognitions (e.g. 020122027 -> 02022027) —
# but only when the line does NOT already hold a valid date: a real
# "01122026" (= 01/12/2026) also matches the 112 pattern (0+112+2026)
# and would be mangled into 7-digit junk.
if not (DDMMYYYY_RE.search(cleaned) or DD_MM_YYYY_RE.search(cleaned)):
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
# Run contextual replacements
for _ in range(3):
# letter o/O flanked by digits or boundary -> 0
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
# letter I/i/l/| flanked by digits -> 1
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
# letter S/s flanked by digits -> 5
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
# letter Z/z flanked by digits -> 2
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
# letter B flanked by digits -> 8
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
return cleaned
def extract_expired_date(text_lines):
"""Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY."""
if not text_lines:
return None, None, None
cleaned_lines = [clean_date_line(line) for line in text_lines]
def pick(match, idx, cleaned_line, formatter=None):
raw = match.group(0)
original_line = text_lines[idx].strip()
if match.lastindex and match.lastindex >= 3:
formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}"
elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit():
digits = match.group(1)
if len(digits) == 8:
formatted = format_ddmmyyyy(digits)
elif len(digits) == 6:
formatted = format_ddmmyy(digits)
else:
formatted = digits
elif formatter:
formatted = formatter(raw)
else:
formatted = raw.strip().upper()
return formatted, idx, original_line
# 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027)
for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line):
continue
match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line)
if match:
return pick(match, idx, line)
# 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027)
for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line):
continue
match = DD_MM_YYYY_RE.search(line)
if match:
return pick(match, idx, line)
# 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits)
for idx, line in enumerate(cleaned_lines):
match = KEYWORD_DIGITS_RE.search(line)
if match:
digits = match.group(1)
if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits):
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
if len(digits) == 6:
return format_ddmmyy(digits), idx, text_lines[idx].strip()
# 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027)
for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line):
continue
match = LENIENT_DATE_RE.search(line)
if match and is_plausible_date_parts(match.group(1), match.group(2), match.group(3)):
return pick(match, idx, line)
# 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes
# detects "BB"/"Baik digunakan" as its own box, separate from the date
# digits in a neighboring box, e.g. "BB" / "05032027" as two lines).
for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line):
continue
for j in (idx + 1, idx - 1, idx + 2):
if j < 0 or j >= len(cleaned_lines) or j == idx:
continue
neighbor = cleaned_lines[j]
combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}"
match = (BB_ATTACHED_DATE_RE.search(combined)
or DDMMYYYY_RE.search(combined)
or DD_MM_YYYY_RE.search(combined))
if match:
report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx
return pick(match, report_idx, combined)
# 4) Any line — spaced DD MM YYYY (excluding store price-tag lines)
for idx, line in enumerate(cleaned_lines):
if PRICE_TAG_RE.search(line):
continue
match = DD_MM_YYYY_RE.search(line)
if match:
return pick(match, idx, line)
# 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context)
for idx, line in enumerate(cleaned_lines):
if PRICE_TAG_RE.search(line):
continue
for match in DDMMYYYY_RE.finditer(line):
digits = f"{match.group(1)}{match.group(2)}{match.group(3)}"
if is_valid_ddmmyyyy_digits(digits):
# Skip if this 8-digit block is the only digits and looks like SKU on label top
if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line):
if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I):
continue
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
# 6) Legacy patterns (slashes, month names, etc.)
date_patterns = [
r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b',
r'\b\d{4}[-./]\d{2}[-./]\d{2}\b',
r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b',
]
for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line):
continue
for pat in date_patterns:
match = re.search(pat, line, re.IGNORECASE)
if match:
return match.group(0).upper(), idx, text_lines[idx].strip()
return None, None, None
def line_contains_expired_date(line: str, expired_date: str) -> bool:
if not line or not expired_date:
return False
digits_only = re.sub(r"\D", "", expired_date)
line_digits = re.sub(r"\D", "", line)
if len(digits_only) >= 6 and digits_only in line_digits:
return True
compact = expired_date.replace("/", "")
return compact in line.replace(" ", "") or expired_date in line
def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len):
"""Pick OCR box index for cropping; prefer the line that actually contains the date."""
if not expired_date or polys_len <= 0:
return None
if (
expired_idx is not None
and expired_idx < polys_len
and expired_idx < len(text_lines)
and line_contains_expired_date(text_lines[expired_idx], expired_date)
):
return expired_idx
keyword_match = None
for idx, line in enumerate(text_lines):
if idx >= polys_len:
break
if not line_contains_expired_date(line, expired_date):
continue
if line_has_exp_keyword(line):
return idx
if keyword_match is None:
keyword_match = idx
if keyword_match is not None:
return keyword_match
if expired_idx is not None and expired_idx < polys_len:
return expired_idx
return None