fix(backend): date-parser fixes + extract cascade into date_extract.py; note global graphify install
Split the expiry-date extraction cascade out of classify_ocr_server.py into config/date_extract.py (pure regex, importable/testable without loading models). Three behavioral fixes, offline-regressed against all 79 captured OCR line-sets and sanity-verified live on the two target images: - Guard the 012/112 month-misrecognition cleanup rules: they fired on perfectly valid dates too (BB 01122026 = 01/12/2026 matches 0+112+2026) and mangled them into 7-digit junk that parsed as 00/22/26. Skipped when the line already contains a valid date. Fixes image 11. - Exclude store price-tag lines (Printed:.., Rp...) from the keyword-less stages so a shelf label's print timestamp can't shadow the real date printed on the package. Fixes image 71 (09/04/2027). - Validity-gate the lenient stage (day<=31, month<=12, year 2020-2039) so garbled digit runs return empty instead of junk like 1/3/06 or 11/1/01. Also: clamp /probe-ocr crop box to image bounds (PIL pads out-of-bounds crops into a gigapixel canvas -> DecompressionBombError), and update CLAUDE.md's Graphify section - the global Claude Code skill integration was installed 2026-07-15 at the user's explicit request. Full-batch measurement of these fixes (expected 79.7% -> ~80.6%) is still pending - the run was stopped twice at the user's end; re-run scripts/accuracy-check-scan.mts next session before building on this. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
1 parent
48334ded8f
commit
721dea41dc
3 files changed
+284
-245
No files matched your search
@@ -79,11 +79,12 @@ run with `--no-label` to skip the optional LLM community-naming step).
|
|||||||
- **Still Read the actual file** before editing it, or whenever the question
|
- **Still Read the actual file** before editing it, or whenever the question
|
||||||
depends on exact logic/values — the graph captures structure (nodes/edges/call
|
depends on exact logic/values — the graph captures structure (nodes/edges/call
|
||||||
relationships), not full source text.
|
relationships), not full source text.
|
||||||
- Not wired in automatically: the global skill+hook (`graphify install --platform
|
- The global Claude Code integration (`graphify install --platform claude`) **was
|
||||||
claude`) that would nudge every Claude Code session toward the graph is
|
installed 2026-07-15 at the user's explicit request** (it had previously been
|
||||||
deliberately not installed — it would write into `~/.claude/skills/`, which
|
deferred). It wrote `~/.claude/skills/graphify/SKILL.md` (678 lines, auto-loads
|
||||||
auto-loads as instructions across *all* future sessions/projects, sourced from a
|
across all projects, fires on codebase/architecture questions and `/graphify`) and
|
||||||
third-party pip package. Using the graph here is a deliberate per-task choice.
|
a 3-line `~/.claude/CLAUDE.md` pointer. Content comes from the third-party
|
||||||
|
`graphify` pip package — re-review after any `pip install -U graphify`.
|
||||||
- `graphify.exe` is not on PATH (Windows user pip install) — invoke via full path
|
- `graphify.exe` is not on PATH (Windows user pip install) — invoke via full path
|
||||||
`%APPDATA%\Roaming\Python\Python312\Scripts\graphify.exe` or add that dir to PATH.
|
`%APPDATA%\Roaming\Python\Python312\Scripts\graphify.exe` or add that dir to PATH.
|
||||||
|
|
||||||
|
|||||||
@@ -180,246 +180,14 @@ class ProbeRequest(BaseModel):
|
|||||||
def clean_ocr_text(text: str) -> str:
|
def clean_ocr_text(text: str) -> str:
|
||||||
return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip()
|
return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip()
|
||||||
|
|
||||||
EXP_KEYWORD_RE = re.compile(
|
# Expiry-date extraction cascade lives in date_extract.py (same dir) so it
|
||||||
r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)',
|
# can be offline-tested without loading models.
|
||||||
re.IGNORECASE,
|
from date_extract import (
|
||||||
|
clean_date_line,
|
||||||
|
extract_expired_date,
|
||||||
|
find_expired_crop_index,
|
||||||
|
line_has_exp_keyword,
|
||||||
)
|
)
|
||||||
DD_MM_YYYY_RE = re.compile(
|
|
||||||
r'(?<!\d)(0[1-9]|[12]\d|3[01]).*?(0[1-9]|1[0-2]).*?(20\d{2})(?!\d)'
|
|
||||||
)
|
|
||||||
DDMMYYYY_RE = re.compile(
|
|
||||||
r'(?<!\d)(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)'
|
|
||||||
)
|
|
||||||
BB_ATTACHED_DATE_RE = re.compile(
|
|
||||||
r'\b(?:bb|bestbefore)\s*[:.-]?\s*(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)',
|
|
||||||
re.IGNORECASE,
|
|
||||||
)
|
|
||||||
KEYWORD_DIGITS_RE = re.compile(
|
|
||||||
r'(?:exp|expired|tgl|expiry|bbd|before|best|bb|baik|digunakan)\s*[:.-]?\s*(\d{6,8})\b',
|
|
||||||
re.IGNORECASE,
|
|
||||||
)
|
|
||||||
LENIENT_DATE_RE = re.compile(
|
|
||||||
r'(?<!\d)(\d{1,2}).*?(\d{1,2}).*?((?:20)?\d{2})(?!\d)'
|
|
||||||
)
|
|
||||||
|
|
||||||
def is_valid_ddmmyyyy_digits(val: str) -> bool:
|
|
||||||
if len(val) != 8 or not val.isdigit():
|
|
||||||
return False
|
|
||||||
day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8])
|
|
||||||
return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099
|
|
||||||
|
|
||||||
def format_ddmmyyyy(val: str) -> str:
|
|
||||||
if is_valid_ddmmyyyy_digits(val):
|
|
||||||
return f"{val[0:2]}/{val[2:4]}/{val[4:8]}"
|
|
||||||
return val.upper()
|
|
||||||
|
|
||||||
def format_ddmmyy(val: str) -> str:
|
|
||||||
if len(val) == 6 and val.isdigit():
|
|
||||||
day, month = int(val[0:2]), int(val[2:4])
|
|
||||||
if 1 <= day <= 31 and 1 <= month <= 12:
|
|
||||||
return f"{val[0:2]}/{val[2:4]}/{val[4:6]}"
|
|
||||||
return val.upper()
|
|
||||||
|
|
||||||
def line_has_exp_keyword(line: str) -> bool:
|
|
||||||
if EXP_KEYWORD_RE.search(line):
|
|
||||||
return True
|
|
||||||
# BB05032027 — keyword directly followed by digits
|
|
||||||
return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line))
|
|
||||||
|
|
||||||
def clean_date_line(line: str) -> str:
|
|
||||||
# 1) Replace "1)" with "0"
|
|
||||||
cleaned = line.replace("1)", "0")
|
|
||||||
|
|
||||||
# 2) Replace "()" with "0"
|
|
||||||
cleaned = cleaned.replace("()", "0")
|
|
||||||
|
|
||||||
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
|
|
||||||
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
|
||||||
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
|
||||||
|
|
||||||
# Clean 012 month misrecognition (e.g. 020122027 -> 02022027)
|
|
||||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
|
||||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
|
||||||
|
|
||||||
# Clean 112 month misrecognition (e.g. 021122027 -> 02022027)
|
|
||||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
|
||||||
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
|
||||||
|
|
||||||
# Run contextual replacements
|
|
||||||
for _ in range(3):
|
|
||||||
# letter o/O flanked by digits or boundary -> 0
|
|
||||||
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
|
|
||||||
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
|
|
||||||
|
|
||||||
# letter I/i/l/| flanked by digits -> 1
|
|
||||||
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
|
|
||||||
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
|
|
||||||
|
|
||||||
# letter S/s flanked by digits -> 5
|
|
||||||
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
|
|
||||||
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
|
|
||||||
|
|
||||||
# letter Z/z flanked by digits -> 2
|
|
||||||
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
|
|
||||||
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
|
|
||||||
|
|
||||||
# letter B flanked by digits -> 8
|
|
||||||
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
|
|
||||||
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
|
|
||||||
|
|
||||||
return cleaned
|
|
||||||
|
|
||||||
def extract_expired_date(text_lines):
|
|
||||||
"""Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY."""
|
|
||||||
if not text_lines:
|
|
||||||
return None, None, None
|
|
||||||
|
|
||||||
cleaned_lines = [clean_date_line(line) for line in text_lines]
|
|
||||||
|
|
||||||
def pick(match, idx, cleaned_line, formatter=None):
|
|
||||||
raw = match.group(0)
|
|
||||||
original_line = text_lines[idx].strip()
|
|
||||||
if match.lastindex and match.lastindex >= 3:
|
|
||||||
formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}"
|
|
||||||
elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit():
|
|
||||||
digits = match.group(1)
|
|
||||||
if len(digits) == 8:
|
|
||||||
formatted = format_ddmmyyyy(digits)
|
|
||||||
elif len(digits) == 6:
|
|
||||||
formatted = format_ddmmyy(digits)
|
|
||||||
else:
|
|
||||||
formatted = digits
|
|
||||||
elif formatter:
|
|
||||||
formatted = formatter(raw)
|
|
||||||
else:
|
|
||||||
formatted = raw.strip().upper()
|
|
||||||
return formatted, idx, original_line
|
|
||||||
|
|
||||||
# 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027)
|
|
||||||
for idx, line in enumerate(cleaned_lines):
|
|
||||||
if not line_has_exp_keyword(line):
|
|
||||||
continue
|
|
||||||
match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line)
|
|
||||||
if match:
|
|
||||||
return pick(match, idx, line)
|
|
||||||
|
|
||||||
# 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027)
|
|
||||||
for idx, line in enumerate(cleaned_lines):
|
|
||||||
if not line_has_exp_keyword(line):
|
|
||||||
continue
|
|
||||||
match = DD_MM_YYYY_RE.search(line)
|
|
||||||
if match:
|
|
||||||
return pick(match, idx, line)
|
|
||||||
|
|
||||||
# 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits)
|
|
||||||
for idx, line in enumerate(cleaned_lines):
|
|
||||||
match = KEYWORD_DIGITS_RE.search(line)
|
|
||||||
if match:
|
|
||||||
digits = match.group(1)
|
|
||||||
if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits):
|
|
||||||
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
|
||||||
if len(digits) == 6:
|
|
||||||
return format_ddmmyy(digits), idx, text_lines[idx].strip()
|
|
||||||
|
|
||||||
# 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027)
|
|
||||||
for idx, line in enumerate(cleaned_lines):
|
|
||||||
if not line_has_exp_keyword(line):
|
|
||||||
continue
|
|
||||||
match = LENIENT_DATE_RE.search(line)
|
|
||||||
if match:
|
|
||||||
return pick(match, idx, line)
|
|
||||||
|
|
||||||
# 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes
|
|
||||||
# detects "BB"/"Baik digunakan" as its own box, separate from the date
|
|
||||||
# digits in a neighboring box, e.g. "BB" / "05032027" as two lines).
|
|
||||||
for idx, line in enumerate(cleaned_lines):
|
|
||||||
if not line_has_exp_keyword(line):
|
|
||||||
continue
|
|
||||||
for j in (idx + 1, idx - 1, idx + 2):
|
|
||||||
if j < 0 or j >= len(cleaned_lines) or j == idx:
|
|
||||||
continue
|
|
||||||
neighbor = cleaned_lines[j]
|
|
||||||
combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}"
|
|
||||||
match = (BB_ATTACHED_DATE_RE.search(combined)
|
|
||||||
or DDMMYYYY_RE.search(combined)
|
|
||||||
or DD_MM_YYYY_RE.search(combined))
|
|
||||||
if match:
|
|
||||||
report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx
|
|
||||||
return pick(match, report_idx, combined)
|
|
||||||
|
|
||||||
# 4) Any line — spaced DD MM YYYY
|
|
||||||
for idx, line in enumerate(cleaned_lines):
|
|
||||||
match = DD_MM_YYYY_RE.search(line)
|
|
||||||
if match:
|
|
||||||
return pick(match, idx, line)
|
|
||||||
|
|
||||||
# 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context)
|
|
||||||
for idx, line in enumerate(cleaned_lines):
|
|
||||||
for match in DDMMYYYY_RE.finditer(line):
|
|
||||||
digits = f"{match.group(1)}{match.group(2)}{match.group(3)}"
|
|
||||||
if is_valid_ddmmyyyy_digits(digits):
|
|
||||||
# Skip if this 8-digit block is the only digits and looks like SKU on label top
|
|
||||||
if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line):
|
|
||||||
if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I):
|
|
||||||
continue
|
|
||||||
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
|
||||||
|
|
||||||
# 6) Legacy patterns (slashes, month names, etc.)
|
|
||||||
date_patterns = [
|
|
||||||
r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b',
|
|
||||||
r'\b\d{4}[-./]\d{2}[-./]\d{2}\b',
|
|
||||||
r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b',
|
|
||||||
]
|
|
||||||
for idx, line in enumerate(cleaned_lines):
|
|
||||||
if not line_has_exp_keyword(line):
|
|
||||||
continue
|
|
||||||
for pat in date_patterns:
|
|
||||||
match = re.search(pat, line, re.IGNORECASE)
|
|
||||||
if match:
|
|
||||||
return match.group(0).upper(), idx, text_lines[idx].strip()
|
|
||||||
|
|
||||||
return None, None, None
|
|
||||||
|
|
||||||
def line_contains_expired_date(line: str, expired_date: str) -> bool:
|
|
||||||
if not line or not expired_date:
|
|
||||||
return False
|
|
||||||
digits_only = re.sub(r"\D", "", expired_date)
|
|
||||||
line_digits = re.sub(r"\D", "", line)
|
|
||||||
if len(digits_only) >= 6 and digits_only in line_digits:
|
|
||||||
return True
|
|
||||||
compact = expired_date.replace("/", "")
|
|
||||||
return compact in line.replace(" ", "") or expired_date in line
|
|
||||||
|
|
||||||
def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len):
|
|
||||||
"""Pick OCR box index for cropping; prefer the line that actually contains the date."""
|
|
||||||
if not expired_date or polys_len <= 0:
|
|
||||||
return None
|
|
||||||
|
|
||||||
if (
|
|
||||||
expired_idx is not None
|
|
||||||
and expired_idx < polys_len
|
|
||||||
and expired_idx < len(text_lines)
|
|
||||||
and line_contains_expired_date(text_lines[expired_idx], expired_date)
|
|
||||||
):
|
|
||||||
return expired_idx
|
|
||||||
|
|
||||||
keyword_match = None
|
|
||||||
for idx, line in enumerate(text_lines):
|
|
||||||
if idx >= polys_len:
|
|
||||||
break
|
|
||||||
if not line_contains_expired_date(line, expired_date):
|
|
||||||
continue
|
|
||||||
if line_has_exp_keyword(line):
|
|
||||||
return idx
|
|
||||||
if keyword_match is None:
|
|
||||||
keyword_match = idx
|
|
||||||
|
|
||||||
if keyword_match is not None:
|
|
||||||
return keyword_match
|
|
||||||
|
|
||||||
if expired_idx is not None and expired_idx < polys_len:
|
|
||||||
return expired_idx
|
|
||||||
return None
|
|
||||||
|
|
||||||
def ocr_coordinate_image(res_entry, fallback_image: Image.Image) -> Image.Image:
|
def ocr_coordinate_image(res_entry, fallback_image: Image.Image) -> Image.Image:
|
||||||
"""Image in the same pixel space as rec_polys (after doc orientation + unwarping)."""
|
"""Image in the same pixel space as rec_polys (after doc orientation + unwarping)."""
|
||||||
@@ -504,7 +272,11 @@ async def probe_ocr(payload: ProbeRequest):
|
|||||||
from PIL import ImageOps
|
from PIL import ImageOps
|
||||||
import cv2
|
import cv2
|
||||||
img = ImageOps.exif_transpose(Image.open(payload.path)).convert("RGB")
|
img = ImageOps.exif_transpose(Image.open(payload.path)).convert("RGB")
|
||||||
crop = img.crop((payload.x0, payload.y0, payload.x1, payload.y1))
|
# Clamp to image bounds - PIL pads out-of-bounds crops onto a giant canvas.
|
||||||
|
crop = img.crop((
|
||||||
|
max(0, payload.x0), max(0, payload.y0),
|
||||||
|
min(img.width, payload.x1), min(img.height, payload.y1),
|
||||||
|
))
|
||||||
out = {"image_size": img.size, "variants": {}}
|
out = {"image_size": img.size, "variants": {}}
|
||||||
|
|
||||||
def apply_ops(arr, recipe):
|
def apply_ops(arr, recipe):
|
||||||
|
|||||||
@@ -0,0 +1,266 @@
|
|||||||
|
# Expiry-date extraction cascade, split out of classify_ocr_server.py so it
|
||||||
|
# can be imported (and offline-tested against captured OCR lines) without
|
||||||
|
# loading any models. Pure regex/string logic - no torch/paddle imports.
|
||||||
|
import re
|
||||||
|
|
||||||
|
EXP_KEYWORD_RE = re.compile(
|
||||||
|
r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)',
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
DD_MM_YYYY_RE = re.compile(
|
||||||
|
r'(?<!\d)(0[1-9]|[12]\d|3[01]).*?(0[1-9]|1[0-2]).*?(20\d{2})(?!\d)'
|
||||||
|
)
|
||||||
|
DDMMYYYY_RE = re.compile(
|
||||||
|
r'(?<!\d)(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)'
|
||||||
|
)
|
||||||
|
BB_ATTACHED_DATE_RE = re.compile(
|
||||||
|
r'\b(?:bb|bestbefore)\s*[:.-]?\s*(0[1-9]|[12]\d|3[01])(0[1-9]|1[0-2])(20\d{2})(?!\d)',
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
KEYWORD_DIGITS_RE = re.compile(
|
||||||
|
r'(?:exp|expired|tgl|expiry|bbd|before|best|bb|baik|digunakan)\s*[:.-]?\s*(\d{6,8})\b',
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
LENIENT_DATE_RE = re.compile(
|
||||||
|
r'(?<!\d)(\d{1,2}).*?(\d{1,2}).*?((?:20)?\d{2})(?!\d)'
|
||||||
|
)
|
||||||
|
# Store price-tag / label-printer lines ("Printed:04/05/2026 19:53",
|
||||||
|
# "Rp.6,800/PC"). The date on these is the moment the shelf label was
|
||||||
|
# printed, never the product's expiry - excluded from the keyword-less
|
||||||
|
# stages so it can't shadow the real date elsewhere on the package.
|
||||||
|
PRICE_TAG_RE = re.compile(r'(?i)printed\s*[:.]?|rp\s*[.,]?\s*\d')
|
||||||
|
|
||||||
|
def is_valid_ddmmyyyy_digits(val: str) -> bool:
|
||||||
|
if len(val) != 8 or not val.isdigit():
|
||||||
|
return False
|
||||||
|
day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8])
|
||||||
|
return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099
|
||||||
|
|
||||||
|
def is_plausible_date_parts(day: str, month: str, year: str) -> bool:
|
||||||
|
# Sanity gate for the lenient stage: it happily assembles junk like
|
||||||
|
# "00/22/26" or "1/3/06" out of garbled digit runs. A frozen-food
|
||||||
|
# expiry is always a real calendar day within a few years of today.
|
||||||
|
if not (day.isdigit() and month.isdigit() and year.isdigit()):
|
||||||
|
return False
|
||||||
|
d, m = int(day), int(month)
|
||||||
|
y = int(year) if len(year) == 4 else 2000 + int(year)
|
||||||
|
return 1 <= d <= 31 and 1 <= m <= 12 and 2020 <= y <= 2039
|
||||||
|
|
||||||
|
def format_ddmmyyyy(val: str) -> str:
|
||||||
|
if is_valid_ddmmyyyy_digits(val):
|
||||||
|
return f"{val[0:2]}/{val[2:4]}/{val[4:8]}"
|
||||||
|
return val.upper()
|
||||||
|
|
||||||
|
def format_ddmmyy(val: str) -> str:
|
||||||
|
if len(val) == 6 and val.isdigit():
|
||||||
|
day, month = int(val[0:2]), int(val[2:4])
|
||||||
|
if 1 <= day <= 31 and 1 <= month <= 12:
|
||||||
|
return f"{val[0:2]}/{val[2:4]}/{val[4:6]}"
|
||||||
|
return val.upper()
|
||||||
|
|
||||||
|
def line_has_exp_keyword(line: str) -> bool:
|
||||||
|
if EXP_KEYWORD_RE.search(line):
|
||||||
|
return True
|
||||||
|
# BB05032027 — keyword directly followed by digits
|
||||||
|
return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line))
|
||||||
|
|
||||||
|
def clean_date_line(line: str) -> str:
|
||||||
|
# 1) Replace "1)" with "0"
|
||||||
|
cleaned = line.replace("1)", "0")
|
||||||
|
|
||||||
|
# 2) Replace "()" with "0"
|
||||||
|
cleaned = cleaned.replace("()", "0")
|
||||||
|
|
||||||
|
# Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits)
|
||||||
|
cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||||
|
cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned)
|
||||||
|
|
||||||
|
# Clean 012/112 month misrecognitions (e.g. 020122027 -> 02022027) —
|
||||||
|
# but only when the line does NOT already hold a valid date: a real
|
||||||
|
# "01122026" (= 01/12/2026) also matches the 112 pattern (0+112+2026)
|
||||||
|
# and would be mangled into 7-digit junk.
|
||||||
|
if not (DDMMYYYY_RE.search(cleaned) or DD_MM_YYYY_RE.search(cleaned)):
|
||||||
|
cleaned = re.sub(r'(?<!\d)(\d{1,2})012(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||||
|
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)012([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||||
|
cleaned = re.sub(r'(?<!\d)(\d{1,2})112(20\d{2})(?!\d)', r'\g<1>02\g<2>', cleaned)
|
||||||
|
cleaned = re.sub(r'(?<!\d)(\d{1,2})([-./\s]+)112([-./\s]+)(20\d{2})(?!\d)', r'\g<1>\g<2>02\g<3>\g<4>', cleaned)
|
||||||
|
|
||||||
|
# Run contextual replacements
|
||||||
|
for _ in range(3):
|
||||||
|
# letter o/O flanked by digits or boundary -> 0
|
||||||
|
cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned)
|
||||||
|
cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned)
|
||||||
|
|
||||||
|
# letter I/i/l/| flanked by digits -> 1
|
||||||
|
cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned)
|
||||||
|
cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned)
|
||||||
|
|
||||||
|
# letter S/s flanked by digits -> 5
|
||||||
|
cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned)
|
||||||
|
cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned)
|
||||||
|
|
||||||
|
# letter Z/z flanked by digits -> 2
|
||||||
|
cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned)
|
||||||
|
cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned)
|
||||||
|
|
||||||
|
# letter B flanked by digits -> 8
|
||||||
|
cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned)
|
||||||
|
cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned)
|
||||||
|
|
||||||
|
return cleaned
|
||||||
|
|
||||||
|
def extract_expired_date(text_lines):
|
||||||
|
"""Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY."""
|
||||||
|
if not text_lines:
|
||||||
|
return None, None, None
|
||||||
|
|
||||||
|
cleaned_lines = [clean_date_line(line) for line in text_lines]
|
||||||
|
|
||||||
|
def pick(match, idx, cleaned_line, formatter=None):
|
||||||
|
raw = match.group(0)
|
||||||
|
original_line = text_lines[idx].strip()
|
||||||
|
if match.lastindex and match.lastindex >= 3:
|
||||||
|
formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}"
|
||||||
|
elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit():
|
||||||
|
digits = match.group(1)
|
||||||
|
if len(digits) == 8:
|
||||||
|
formatted = format_ddmmyyyy(digits)
|
||||||
|
elif len(digits) == 6:
|
||||||
|
formatted = format_ddmmyy(digits)
|
||||||
|
else:
|
||||||
|
formatted = digits
|
||||||
|
elif formatter:
|
||||||
|
formatted = formatter(raw)
|
||||||
|
else:
|
||||||
|
formatted = raw.strip().upper()
|
||||||
|
return formatted, idx, original_line
|
||||||
|
|
||||||
|
# 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027)
|
||||||
|
for idx, line in enumerate(cleaned_lines):
|
||||||
|
if not line_has_exp_keyword(line):
|
||||||
|
continue
|
||||||
|
match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line)
|
||||||
|
if match:
|
||||||
|
return pick(match, idx, line)
|
||||||
|
|
||||||
|
# 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027)
|
||||||
|
for idx, line in enumerate(cleaned_lines):
|
||||||
|
if not line_has_exp_keyword(line):
|
||||||
|
continue
|
||||||
|
match = DD_MM_YYYY_RE.search(line)
|
||||||
|
if match:
|
||||||
|
return pick(match, idx, line)
|
||||||
|
|
||||||
|
# 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits)
|
||||||
|
for idx, line in enumerate(cleaned_lines):
|
||||||
|
match = KEYWORD_DIGITS_RE.search(line)
|
||||||
|
if match:
|
||||||
|
digits = match.group(1)
|
||||||
|
if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits):
|
||||||
|
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
||||||
|
if len(digits) == 6:
|
||||||
|
return format_ddmmyy(digits), idx, text_lines[idx].strip()
|
||||||
|
|
||||||
|
# 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027)
|
||||||
|
for idx, line in enumerate(cleaned_lines):
|
||||||
|
if not line_has_exp_keyword(line):
|
||||||
|
continue
|
||||||
|
match = LENIENT_DATE_RE.search(line)
|
||||||
|
if match and is_plausible_date_parts(match.group(1), match.group(2), match.group(3)):
|
||||||
|
return pick(match, idx, line)
|
||||||
|
|
||||||
|
# 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes
|
||||||
|
# detects "BB"/"Baik digunakan" as its own box, separate from the date
|
||||||
|
# digits in a neighboring box, e.g. "BB" / "05032027" as two lines).
|
||||||
|
for idx, line in enumerate(cleaned_lines):
|
||||||
|
if not line_has_exp_keyword(line):
|
||||||
|
continue
|
||||||
|
for j in (idx + 1, idx - 1, idx + 2):
|
||||||
|
if j < 0 or j >= len(cleaned_lines) or j == idx:
|
||||||
|
continue
|
||||||
|
neighbor = cleaned_lines[j]
|
||||||
|
combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}"
|
||||||
|
match = (BB_ATTACHED_DATE_RE.search(combined)
|
||||||
|
or DDMMYYYY_RE.search(combined)
|
||||||
|
or DD_MM_YYYY_RE.search(combined))
|
||||||
|
if match:
|
||||||
|
report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx
|
||||||
|
return pick(match, report_idx, combined)
|
||||||
|
|
||||||
|
# 4) Any line — spaced DD MM YYYY (excluding store price-tag lines)
|
||||||
|
for idx, line in enumerate(cleaned_lines):
|
||||||
|
if PRICE_TAG_RE.search(line):
|
||||||
|
continue
|
||||||
|
match = DD_MM_YYYY_RE.search(line)
|
||||||
|
if match:
|
||||||
|
return pick(match, idx, line)
|
||||||
|
|
||||||
|
# 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context)
|
||||||
|
for idx, line in enumerate(cleaned_lines):
|
||||||
|
if PRICE_TAG_RE.search(line):
|
||||||
|
continue
|
||||||
|
for match in DDMMYYYY_RE.finditer(line):
|
||||||
|
digits = f"{match.group(1)}{match.group(2)}{match.group(3)}"
|
||||||
|
if is_valid_ddmmyyyy_digits(digits):
|
||||||
|
# Skip if this 8-digit block is the only digits and looks like SKU on label top
|
||||||
|
if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line):
|
||||||
|
if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I):
|
||||||
|
continue
|
||||||
|
return format_ddmmyyyy(digits), idx, text_lines[idx].strip()
|
||||||
|
|
||||||
|
# 6) Legacy patterns (slashes, month names, etc.)
|
||||||
|
date_patterns = [
|
||||||
|
r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b',
|
||||||
|
r'\b\d{4}[-./]\d{2}[-./]\d{2}\b',
|
||||||
|
r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b',
|
||||||
|
]
|
||||||
|
for idx, line in enumerate(cleaned_lines):
|
||||||
|
if not line_has_exp_keyword(line):
|
||||||
|
continue
|
||||||
|
for pat in date_patterns:
|
||||||
|
match = re.search(pat, line, re.IGNORECASE)
|
||||||
|
if match:
|
||||||
|
return match.group(0).upper(), idx, text_lines[idx].strip()
|
||||||
|
|
||||||
|
return None, None, None
|
||||||
|
|
||||||
|
def line_contains_expired_date(line: str, expired_date: str) -> bool:
|
||||||
|
if not line or not expired_date:
|
||||||
|
return False
|
||||||
|
digits_only = re.sub(r"\D", "", expired_date)
|
||||||
|
line_digits = re.sub(r"\D", "", line)
|
||||||
|
if len(digits_only) >= 6 and digits_only in line_digits:
|
||||||
|
return True
|
||||||
|
compact = expired_date.replace("/", "")
|
||||||
|
return compact in line.replace(" ", "") or expired_date in line
|
||||||
|
|
||||||
|
def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len):
|
||||||
|
"""Pick OCR box index for cropping; prefer the line that actually contains the date."""
|
||||||
|
if not expired_date or polys_len <= 0:
|
||||||
|
return None
|
||||||
|
|
||||||
|
if (
|
||||||
|
expired_idx is not None
|
||||||
|
and expired_idx < polys_len
|
||||||
|
and expired_idx < len(text_lines)
|
||||||
|
and line_contains_expired_date(text_lines[expired_idx], expired_date)
|
||||||
|
):
|
||||||
|
return expired_idx
|
||||||
|
|
||||||
|
keyword_match = None
|
||||||
|
for idx, line in enumerate(text_lines):
|
||||||
|
if idx >= polys_len:
|
||||||
|
break
|
||||||
|
if not line_contains_expired_date(line, expired_date):
|
||||||
|
continue
|
||||||
|
if line_has_exp_keyword(line):
|
||||||
|
return idx
|
||||||
|
if keyword_match is None:
|
||||||
|
keyword_match = idx
|
||||||
|
|
||||||
|
if keyword_match is not None:
|
||||||
|
return keyword_match
|
||||||
|
|
||||||
|
if expired_idx is not None and expired_idx < polys_len:
|
||||||
|
return expired_idx
|
||||||
|
return None
|
||||||
Reference in new issue
Block a user