diff --git a/CLAUDE.md b/CLAUDE.md index c5e0722..abf301d 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -79,11 +79,12 @@ run with `--no-label` to skip the optional LLM community-naming step). - **Still Read the actual file** before editing it, or whenever the question depends on exact logic/values — the graph captures structure (nodes/edges/call relationships), not full source text. -- Not wired in automatically: the global skill+hook (`graphify install --platform - claude`) that would nudge every Claude Code session toward the graph is - deliberately not installed — it would write into `~/.claude/skills/`, which - auto-loads as instructions across *all* future sessions/projects, sourced from a - third-party pip package. Using the graph here is a deliberate per-task choice. +- The global Claude Code integration (`graphify install --platform claude`) **was + installed 2026-07-15 at the user's explicit request** (it had previously been + deferred). It wrote `~/.claude/skills/graphify/SKILL.md` (678 lines, auto-loads + across all projects, fires on codebase/architecture questions and `/graphify`) and + a 3-line `~/.claude/CLAUDE.md` pointer. Content comes from the third-party + `graphify` pip package — re-review after any `pip install -U graphify`. - `graphify.exe` is not on PATH (Windows user pip install) — invoke via full path `%APPDATA%\Roaming\Python\Python312\Scripts\graphify.exe` or add that dir to PATH. diff --git a/backend/config/classify_ocr_server.py b/backend/config/classify_ocr_server.py index df7ee32..ed51599 100644 --- a/backend/config/classify_ocr_server.py +++ b/backend/config/classify_ocr_server.py @@ -180,246 +180,14 @@ class ProbeRequest(BaseModel): def clean_ocr_text(text: str) -> str: return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip() -EXP_KEYWORD_RE = re.compile( - r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)', - re.IGNORECASE, +# Expiry-date extraction cascade lives in date_extract.py (same dir) so it +# can be offline-tested without loading models. +from date_extract import ( + clean_date_line, + extract_expired_date, + find_expired_crop_index, + line_has_exp_keyword, ) -DD_MM_YYYY_RE = re.compile( - r'(? bool: - if len(val) != 8 or not val.isdigit(): - return False - day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8]) - return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099 - -def format_ddmmyyyy(val: str) -> str: - if is_valid_ddmmyyyy_digits(val): - return f"{val[0:2]}/{val[2:4]}/{val[4:8]}" - return val.upper() - -def format_ddmmyy(val: str) -> str: - if len(val) == 6 and val.isdigit(): - day, month = int(val[0:2]), int(val[2:4]) - if 1 <= day <= 31 and 1 <= month <= 12: - return f"{val[0:2]}/{val[2:4]}/{val[4:6]}" - return val.upper() - -def line_has_exp_keyword(line: str) -> bool: - if EXP_KEYWORD_RE.search(line): - return True - # BB05032027 — keyword directly followed by digits - return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line)) - -def clean_date_line(line: str) -> str: - # 1) Replace "1)" with "0" - cleaned = line.replace("1)", "0") - - # 2) Replace "()" with "0" - cleaned = cleaned.replace("()", "0") - - # Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits) - cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned) - cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned) - - # Clean 012 month misrecognition (e.g. 020122027 -> 02022027) - cleaned = re.sub(r'(?02\g<2>', cleaned) - cleaned = re.sub(r'(?\g<2>02\g<3>\g<4>', cleaned) - - # Clean 112 month misrecognition (e.g. 021122027 -> 02022027) - cleaned = re.sub(r'(?02\g<2>', cleaned) - cleaned = re.sub(r'(?\g<2>02\g<3>\g<4>', cleaned) - - # Run contextual replacements - for _ in range(3): - # letter o/O flanked by digits or boundary -> 0 - cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned) - cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned) - - # letter I/i/l/| flanked by digits -> 1 - cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned) - cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned) - - # letter S/s flanked by digits -> 5 - cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned) - cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned) - - # letter Z/z flanked by digits -> 2 - cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned) - cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned) - - # letter B flanked by digits -> 8 - cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned) - cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned) - - return cleaned - -def extract_expired_date(text_lines): - """Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY.""" - if not text_lines: - return None, None, None - - cleaned_lines = [clean_date_line(line) for line in text_lines] - - def pick(match, idx, cleaned_line, formatter=None): - raw = match.group(0) - original_line = text_lines[idx].strip() - if match.lastindex and match.lastindex >= 3: - formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}" - elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit(): - digits = match.group(1) - if len(digits) == 8: - formatted = format_ddmmyyyy(digits) - elif len(digits) == 6: - formatted = format_ddmmyy(digits) - else: - formatted = digits - elif formatter: - formatted = formatter(raw) - else: - formatted = raw.strip().upper() - return formatted, idx, original_line - - # 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027) - for idx, line in enumerate(cleaned_lines): - if not line_has_exp_keyword(line): - continue - match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line) - if match: - return pick(match, idx, line) - - # 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027) - for idx, line in enumerate(cleaned_lines): - if not line_has_exp_keyword(line): - continue - match = DD_MM_YYYY_RE.search(line) - if match: - return pick(match, idx, line) - - # 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits) - for idx, line in enumerate(cleaned_lines): - match = KEYWORD_DIGITS_RE.search(line) - if match: - digits = match.group(1) - if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits): - return format_ddmmyyyy(digits), idx, text_lines[idx].strip() - if len(digits) == 6: - return format_ddmmyy(digits), idx, text_lines[idx].strip() - - # 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027) - for idx, line in enumerate(cleaned_lines): - if not line_has_exp_keyword(line): - continue - match = LENIENT_DATE_RE.search(line) - if match: - return pick(match, idx, line) - - # 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes - # detects "BB"/"Baik digunakan" as its own box, separate from the date - # digits in a neighboring box, e.g. "BB" / "05032027" as two lines). - for idx, line in enumerate(cleaned_lines): - if not line_has_exp_keyword(line): - continue - for j in (idx + 1, idx - 1, idx + 2): - if j < 0 or j >= len(cleaned_lines) or j == idx: - continue - neighbor = cleaned_lines[j] - combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}" - match = (BB_ATTACHED_DATE_RE.search(combined) - or DDMMYYYY_RE.search(combined) - or DD_MM_YYYY_RE.search(combined)) - if match: - report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx - return pick(match, report_idx, combined) - - # 4) Any line — spaced DD MM YYYY - for idx, line in enumerate(cleaned_lines): - match = DD_MM_YYYY_RE.search(line) - if match: - return pick(match, idx, line) - - # 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context) - for idx, line in enumerate(cleaned_lines): - for match in DDMMYYYY_RE.finditer(line): - digits = f"{match.group(1)}{match.group(2)}{match.group(3)}" - if is_valid_ddmmyyyy_digits(digits): - # Skip if this 8-digit block is the only digits and looks like SKU on label top - if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line): - if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I): - continue - return format_ddmmyyyy(digits), idx, text_lines[idx].strip() - - # 6) Legacy patterns (slashes, month names, etc.) - date_patterns = [ - r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b', - r'\b\d{4}[-./]\d{2}[-./]\d{2}\b', - r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b', - ] - for idx, line in enumerate(cleaned_lines): - if not line_has_exp_keyword(line): - continue - for pat in date_patterns: - match = re.search(pat, line, re.IGNORECASE) - if match: - return match.group(0).upper(), idx, text_lines[idx].strip() - - return None, None, None - -def line_contains_expired_date(line: str, expired_date: str) -> bool: - if not line or not expired_date: - return False - digits_only = re.sub(r"\D", "", expired_date) - line_digits = re.sub(r"\D", "", line) - if len(digits_only) >= 6 and digits_only in line_digits: - return True - compact = expired_date.replace("/", "") - return compact in line.replace(" ", "") or expired_date in line - -def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len): - """Pick OCR box index for cropping; prefer the line that actually contains the date.""" - if not expired_date or polys_len <= 0: - return None - - if ( - expired_idx is not None - and expired_idx < polys_len - and expired_idx < len(text_lines) - and line_contains_expired_date(text_lines[expired_idx], expired_date) - ): - return expired_idx - - keyword_match = None - for idx, line in enumerate(text_lines): - if idx >= polys_len: - break - if not line_contains_expired_date(line, expired_date): - continue - if line_has_exp_keyword(line): - return idx - if keyword_match is None: - keyword_match = idx - - if keyword_match is not None: - return keyword_match - - if expired_idx is not None and expired_idx < polys_len: - return expired_idx - return None def ocr_coordinate_image(res_entry, fallback_image: Image.Image) -> Image.Image: """Image in the same pixel space as rec_polys (after doc orientation + unwarping).""" @@ -504,7 +272,11 @@ async def probe_ocr(payload: ProbeRequest): from PIL import ImageOps import cv2 img = ImageOps.exif_transpose(Image.open(payload.path)).convert("RGB") - crop = img.crop((payload.x0, payload.y0, payload.x1, payload.y1)) + # Clamp to image bounds - PIL pads out-of-bounds crops onto a giant canvas. + crop = img.crop(( + max(0, payload.x0), max(0, payload.y0), + min(img.width, payload.x1), min(img.height, payload.y1), + )) out = {"image_size": img.size, "variants": {}} def apply_ops(arr, recipe): diff --git a/backend/config/date_extract.py b/backend/config/date_extract.py new file mode 100644 index 0000000..9357c29 --- /dev/null +++ b/backend/config/date_extract.py @@ -0,0 +1,266 @@ +# Expiry-date extraction cascade, split out of classify_ocr_server.py so it +# can be imported (and offline-tested against captured OCR lines) without +# loading any models. Pure regex/string logic - no torch/paddle imports. +import re + +EXP_KEYWORD_RE = re.compile( + r'(?:exp(?:\.|ired)?|tgl(?:\s*exp)?|expiry|bbd|best\s*before|before|best|baik\s*digunakan|\bbb\b)', + re.IGNORECASE, +) +DD_MM_YYYY_RE = re.compile( + r'(? bool: + if len(val) != 8 or not val.isdigit(): + return False + day, month, year = int(val[0:2]), int(val[2:4]), int(val[4:8]) + return 1 <= day <= 31 and 1 <= month <= 12 and 2000 <= year <= 2099 + +def is_plausible_date_parts(day: str, month: str, year: str) -> bool: + # Sanity gate for the lenient stage: it happily assembles junk like + # "00/22/26" or "1/3/06" out of garbled digit runs. A frozen-food + # expiry is always a real calendar day within a few years of today. + if not (day.isdigit() and month.isdigit() and year.isdigit()): + return False + d, m = int(day), int(month) + y = int(year) if len(year) == 4 else 2000 + int(year) + return 1 <= d <= 31 and 1 <= m <= 12 and 2020 <= y <= 2039 + +def format_ddmmyyyy(val: str) -> str: + if is_valid_ddmmyyyy_digits(val): + return f"{val[0:2]}/{val[2:4]}/{val[4:8]}" + return val.upper() + +def format_ddmmyy(val: str) -> str: + if len(val) == 6 and val.isdigit(): + day, month = int(val[0:2]), int(val[2:4]) + if 1 <= day <= 31 and 1 <= month <= 12: + return f"{val[0:2]}/{val[2:4]}/{val[4:6]}" + return val.upper() + +def line_has_exp_keyword(line: str) -> bool: + if EXP_KEYWORD_RE.search(line): + return True + # BB05032027 — keyword directly followed by digits + return bool(re.search(r'(?i)\b(?:bb|bestbefore)(?:\s*[:.-]?\s*)?\d', line)) + +def clean_date_line(line: str) -> str: + # 1) Replace "1)" with "0" + cleaned = line.replace("1)", "0") + + # 2) Replace "()" with "0" + cleaned = cleaned.replace("()", "0") + + # Clean BB misrecognitions (convert B8, 8B, 88 to BB when followed by digits) + cleaned = re.sub(r'\b(?:B8|8B|88)(?=\d)', 'BB', cleaned) + cleaned = re.sub(r'^(?:B8|8B|88)(?=\d)', 'BB', cleaned) + + # Clean 012/112 month misrecognitions (e.g. 020122027 -> 02022027) — + # but only when the line does NOT already hold a valid date: a real + # "01122026" (= 01/12/2026) also matches the 112 pattern (0+112+2026) + # and would be mangled into 7-digit junk. + if not (DDMMYYYY_RE.search(cleaned) or DD_MM_YYYY_RE.search(cleaned)): + cleaned = re.sub(r'(?02\g<2>', cleaned) + cleaned = re.sub(r'(?\g<2>02\g<3>\g<4>', cleaned) + cleaned = re.sub(r'(?02\g<2>', cleaned) + cleaned = re.sub(r'(?\g<2>02\g<3>\g<4>', cleaned) + + # Run contextual replacements + for _ in range(3): + # letter o/O flanked by digits or boundary -> 0 + cleaned = re.compile(r'(\d)[oO](\d|\b)').sub(r'\g<1>0\g<2>', cleaned) + cleaned = re.compile(r'(\b|\d)[oO](\d)').sub(r'\g<1>0\g<2>', cleaned) + + # letter I/i/l/| flanked by digits -> 1 + cleaned = re.compile(r'(\d)[Ii|l](\d|\b)').sub(r'\g<1>1\g<2>', cleaned) + cleaned = re.compile(r'(\b|\d)[Ii|l](\d)').sub(r'\g<1>1\g<2>', cleaned) + + # letter S/s flanked by digits -> 5 + cleaned = re.compile(r'(\d)[Ss](\d|\b)').sub(r'\g<1>5\g<2>', cleaned) + cleaned = re.compile(r'(\b|\d)[Ss](\d)').sub(r'\g<1>5\g<2>', cleaned) + + # letter Z/z flanked by digits -> 2 + cleaned = re.compile(r'(\d)[Zz](\d|\b)').sub(r'\g<1>2\g<2>', cleaned) + cleaned = re.compile(r'(\b|\d)[Zz](\d)').sub(r'\g<1>2\g<2>', cleaned) + + # letter B flanked by digits -> 8 + cleaned = re.compile(r'(\d)B(\d|\b)').sub(r'\g<1>8\g<2>', cleaned) + cleaned = re.compile(r'(\b|\d)B(\d)').sub(r'\g<1>8\g<2>', cleaned) + + return cleaned + +def extract_expired_date(text_lines): + """Return (formatted_date, line_index, source_line). Prioritises BB/EXP + DDMMYYYY or DD MM YYYY.""" + if not text_lines: + return None, None, None + + cleaned_lines = [clean_date_line(line) for line in text_lines] + + def pick(match, idx, cleaned_line, formatter=None): + raw = match.group(0) + original_line = text_lines[idx].strip() + if match.lastindex and match.lastindex >= 3: + formatted = f"{match.group(1)}/{match.group(2)}/{match.group(3)}" + elif match.lastindex and match.lastindex >= 1 and match.group(1).isdigit(): + digits = match.group(1) + if len(digits) == 8: + formatted = format_ddmmyyyy(digits) + elif len(digits) == 6: + formatted = format_ddmmyy(digits) + else: + formatted = digits + elif formatter: + formatted = formatter(raw) + else: + formatted = raw.strip().upper() + return formatted, idx, original_line + + # 1) BB/EXP keyword lines — compact DDMMYYYY (e.g. BB05032027, EXP 05032027) + for idx, line in enumerate(cleaned_lines): + if not line_has_exp_keyword(line): + continue + match = BB_ATTACHED_DATE_RE.search(line) or DDMMYYYY_RE.search(line) + if match: + return pick(match, idx, line) + + # 2) BB/EXP keyword lines — spaced DD MM YYYY (e.g. BB 05 03 2027) + for idx, line in enumerate(cleaned_lines): + if not line_has_exp_keyword(line): + continue + match = DD_MM_YYYY_RE.search(line) + if match: + return pick(match, idx, line) + + # 3) Keyword + 6–8 digit run (BB05032027 via keyword_digits) + for idx, line in enumerate(cleaned_lines): + match = KEYWORD_DIGITS_RE.search(line) + if match: + digits = match.group(1) + if len(digits) == 8 and is_valid_ddmmyyyy_digits(digits): + return format_ddmmyyyy(digits), idx, text_lines[idx].strip() + if len(digits) == 6: + return format_ddmmyy(digits), idx, text_lines[idx].strip() + + # 3.5) BB/EXP keyword lines — lenient check for unclear/noisy date formats (e.g. BB 02J 132027) + for idx, line in enumerate(cleaned_lines): + if not line_has_exp_keyword(line): + continue + match = LENIENT_DATE_RE.search(line) + if match and is_plausible_date_parts(match.group(1), match.group(2), match.group(3)): + return pick(match, idx, line) + + # 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes + # detects "BB"/"Baik digunakan" as its own box, separate from the date + # digits in a neighboring box, e.g. "BB" / "05032027" as two lines). + for idx, line in enumerate(cleaned_lines): + if not line_has_exp_keyword(line): + continue + for j in (idx + 1, idx - 1, idx + 2): + if j < 0 or j >= len(cleaned_lines) or j == idx: + continue + neighbor = cleaned_lines[j] + combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}" + match = (BB_ATTACHED_DATE_RE.search(combined) + or DDMMYYYY_RE.search(combined) + or DD_MM_YYYY_RE.search(combined)) + if match: + report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx + return pick(match, report_idx, combined) + + # 4) Any line — spaced DD MM YYYY (excluding store price-tag lines) + for idx, line in enumerate(cleaned_lines): + if PRICE_TAG_RE.search(line): + continue + match = DD_MM_YYYY_RE.search(line) + if match: + return pick(match, idx, line) + + # 5) Any line — compact DDMMYYYY (skip likely SKU: same line has 8-digit product code context) + for idx, line in enumerate(cleaned_lines): + if PRICE_TAG_RE.search(line): + continue + for match in DDMMYYYY_RE.finditer(line): + digits = f"{match.group(1)}{match.group(2)}{match.group(3)}" + if is_valid_ddmmyyyy_digits(digits): + # Skip if this 8-digit block is the only digits and looks like SKU on label top + if re.search(r'\b\d{8}\b', line) and not line_has_exp_keyword(line): + if re.search(r'(?:nugget|chicken|fiesta|champ|okey|akumo|frozen|gr)', line, re.I): + continue + return format_ddmmyyyy(digits), idx, text_lines[idx].strip() + + # 6) Legacy patterns (slashes, month names, etc.) + date_patterns = [ + r'\b\d{2}[-./]\d{2}[-./]\d{2,4}\b', + r'\b\d{4}[-./]\d{2}[-./]\d{2}\b', + r'\b\d{2}\s+(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)[a-zA-Z]*\s+\d{2,4}\b', + ] + for idx, line in enumerate(cleaned_lines): + if not line_has_exp_keyword(line): + continue + for pat in date_patterns: + match = re.search(pat, line, re.IGNORECASE) + if match: + return match.group(0).upper(), idx, text_lines[idx].strip() + + return None, None, None + +def line_contains_expired_date(line: str, expired_date: str) -> bool: + if not line or not expired_date: + return False + digits_only = re.sub(r"\D", "", expired_date) + line_digits = re.sub(r"\D", "", line) + if len(digits_only) >= 6 and digits_only in line_digits: + return True + compact = expired_date.replace("/", "") + return compact in line.replace(" ", "") or expired_date in line + +def find_expired_crop_index(text_lines, expired_idx, expired_date, polys_len): + """Pick OCR box index for cropping; prefer the line that actually contains the date.""" + if not expired_date or polys_len <= 0: + return None + + if ( + expired_idx is not None + and expired_idx < polys_len + and expired_idx < len(text_lines) + and line_contains_expired_date(text_lines[expired_idx], expired_date) + ): + return expired_idx + + keyword_match = None + for idx, line in enumerate(text_lines): + if idx >= polys_len: + break + if not line_contains_expired_date(line, expired_date): + continue + if line_has_exp_keyword(line): + return idx + if keyword_match is None: + keyword_match = idx + + if keyword_match is not None: + return keyword_match + + if expired_idx is not None and expired_idx < polys_len: + return expired_idx + return None