feat(backend): scan-product accuracy 66.2% -> 79.7% + frozen validation benchmark

Accuracy work on the 79-image product-scan validation set (user goal: 90%):
- classify_ocr_server.py: 0/90/180/270-degree expiry-date search (stops at
  first hit, 0-degree fallback); classification decoupled onto the upright
  image (rotated frames regressed DINOv2 -6pts until this); cross-line date
  stitching; tiled full-res OCR pass (defeats the 4000px downscale that
  killed small inkjet dates); VL-pipeline expiry fallback with
  keyword-anchored anti-hallucination guard; VL text lines merged into
  text_lines + VL SKU retry. Visualization endpoints removed entirely
  (Visual/Spotting grids - unused by frontend, 3x per-scan GPU cost).
- product-scan.ts: coverage-normalized OCR-evidence re-ranking of DINOv2
  top-K (tuned offline: +8/-0 on top-1 misses), re-ranked class mapped to
  sku_master by SKU prefix; classifier timeout 90s->240s for fallback paths.
- Frozen benchmark: product-test-images-fixed/ (79 renamed images) +
  freeze/seed/build-undetected/capture/experiment scripts; labels trimmed to
  the 79 validation entries (training rows kept in .bak-with-training);
  5 TRAINED-ON SKUs replaced with fresh held-out photos.
- manual-label-scan page: shows last batch-test AI prediction under every
  field by default (new /api/product-scan-results); serves the fixed folder;
  fixed total hydration failure via allowedDevOrigins 127.0.0.1.
- Measured (all-79, zero failures): sku/name 87.3%, expiry 64.6%, overall
  79.7%. Tiles/VL-evidence/VL-SKU deployed but not yet batch-measured.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
Rafhan Mazaya FathurrahmanandClaude Fable 5 committed 2026-07-14 19:55:17 +07:00
1 parent 19f1facf9b
commit e76ccb60a6
156 files changed
+17150 -1405

No files matched your search

+202 -110
View File
@@ -316,6 +316,24 @@ def extract_expired_date(text_lines):
if match:
return pick(match, idx, line)
# 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes
# detects "BB"/"Baik digunakan" as its own box, separate from the date
# digits in a neighboring box, e.g. "BB" / "05032027" as two lines).
for idx, line in enumerate(cleaned_lines):
if not line_has_exp_keyword(line):
continue
for j in (idx + 1, idx - 1, idx + 2):
if j < 0 or j >= len(cleaned_lines) or j == idx:
continue
neighbor = cleaned_lines[j]
combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}"
match = (BB_ATTACHED_DATE_RE.search(combined)
or DDMMYYYY_RE.search(combined)
or DD_MM_YYYY_RE.search(combined))
if match:
report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx
return pick(match, report_idx, combined)
# 4) Any line — spaced DD MM YYYY
for idx, line in enumerate(cleaned_lines):
match = DD_MM_YYYY_RE.search(line)
@@ -471,8 +489,24 @@ async def classify_ocr(payload: ScanRequest):
img_data = base64.b64decode(payload.image_base64.split(",")[-1])
raw_image = Image.open(io.BytesIO(img_data))
image = ImageOps.exif_transpose(raw_image).convert("RGB")
# First-pass PaddleOCR to check orientation based on Expiry Date
# Classification always sees the original upright orientation - the
# 90-degree expiry-date search below may rotate `image` to a
# sideways/upside-down orientation that DINOv2/YOLO were never
# trained on (their reference photos are all shot upright), so using
# a rotated frame there would hurt classification, not help it.
classification_image = image
# Multi-orientation expiry-date search: some photos are captured with
# the whole frame rotated ~90 degrees from upright (e.g. staff held
# the phone in portrait for a package whose printed date runs
# horizontally), so the expiry stamp - and the product framing -
# ends up sideways. Try 0/90/180/270 degree rotations in order and
# stop at the first one where PaddleOCR actually finds an expiry
# date; if none of the four find one, fall back to the 0-degree
# result so behaviour for genuinely-undetectable photos is unchanged.
# This costs extra OCR passes (up to 4x) only on images where the
# first pass found nothing - already-working images stay on the fast
# single-pass path below.
rotated_image_used = False
res_list = []
text_lines = []
@@ -480,35 +514,160 @@ async def classify_ocr(payload: ScanRequest):
expired_date = None
expired_idx = None
expired_source_line = None
if ocr:
base_image = image
for step_angle in (0, 90, 180, 270):
try:
candidate_image = (
base_image.rotate(step_angle, resample=Image.BICUBIC, expand=True)
if step_angle else base_image
)
img_arr = np.array(candidate_image)
candidate_res_list = list(ocr.predict(img_arr))
candidate_res_entry = candidate_res_list[0] if candidate_res_list else {}
candidate_text_lines = candidate_res_entry.get("rec_texts", [])
candidate_text_polys = ocr_text_polys(candidate_res_entry)
candidate_expired_date, candidate_expired_idx, candidate_expired_source_line = (
extract_expired_date(candidate_text_lines)
)
if step_angle == 0:
# Always keep the 0-degree pass as the fallback result.
image, res_list, text_lines, text_polys = (
candidate_image, candidate_res_list, candidate_text_lines, candidate_text_polys
)
expired_date, expired_idx, expired_source_line = (
candidate_expired_date, candidate_expired_idx, candidate_expired_source_line
)
if candidate_expired_date is not None:
if step_angle != 0:
print(f"[Auto-Rotate-90] Expiry date found after rotating {step_angle} degrees.")
image, res_list, text_lines, text_polys = (
candidate_image, candidate_res_list, candidate_text_lines, candidate_text_polys
)
rotated_image_used = True
expired_date, expired_idx, expired_source_line = (
candidate_expired_date, candidate_expired_idx, candidate_expired_source_line
)
break
except Exception as rot_err:
print(f"Error during {step_angle}-degree OCR pass: {rot_err}")
traceback.print_exc()
# Tiled full-resolution pass: PaddleOCR downscales anything over
# its 4000px max_side_limit, which is exactly what kills small
# inkjet date stamps on these ~3200x5700 phone photos. Split the
# original image into overlapping tiles that each fit under the
# limit (so the date region is OCR'd at native resolution) and
# run the cascade per tile. Failure-path only, keyword-anchored
# acceptance like the VL fallback below.
if expired_date is None and max(base_image.size) > 2600:
TILE, OVERLAP = 2400, 400
W, H = base_image.size
step = TILE - OVERLAP
try:
found = False
for y0 in range(0, H, step):
if found:
break
for x0 in range(0, W, step):
tile = base_image.crop((x0, y0, min(x0 + TILE, W), min(y0 + TILE, H)))
if tile.width < 300 or tile.height < 300:
continue
tile_res = list(ocr.predict(np.array(tile)))
tile_lines = tile_res[0].get("rec_texts", []) if tile_res else []
if not tile_lines:
continue
t_date, _t_idx, t_source = extract_expired_date(tile_lines)
if t_date is not None and t_source and line_has_exp_keyword(
clean_date_line(t_source)
):
print(f"[Tile-Pass] Expiry date {t_date} found in full-res tile ({x0},{y0}) (line: {t_source!r})")
expired_date = t_date
expired_idx = None # tile polys don't map to the full image
expired_source_line = t_source
found = True
break
except Exception as tile_err:
print(f"[Tile-Pass] failed: {tile_err}")
traceback.print_exc()
# VL fallback: the lightweight PP-OCRv6 detector missed the date
# at every orientation. The vLLM-backed PaddleOCR-VL pipeline
# (:8090, same container) is a much stronger reader of small,
# low-contrast inkjet codes - ask it to read the whole package
# and run the same date cascade over its text output. Only fires
# on already-failed images, so the happy path stays single-pass.
# Acceptance is stricter than the local cascade: the matched
# line must carry an expiry keyword (BB/EXP/Baik digunakan...),
# so a bare number elsewhere on the package can't be
# hallucinated into a date on photos where none is visible.
# Even when no date is found, the VL's (much cleaner) text lines
# are kept and appended to text_lines below - they feed the
# gateway's OCR-evidence classification re-ranking.
vl_text_lines = []
if expired_date is None:
try:
vl_url = os.environ.get(
"VL_PIPELINE_URL", "http://localhost:8090/layout-parsing"
)
buffered = io.BytesIO()
base_image.save(buffered, format="JPEG")
vl_payload = {
"file": base64.b64encode(buffered.getvalue()).decode("utf-8"),
"matchHistoryJob": False,
"useLayoutDetection": True,
"fileType": 1,
"useDocUnwarping": False,
"useDocOrientationClassify": True,
}
vl_resp = requests.post(vl_url, json=vl_payload, timeout=120)
if vl_resp.status_code == 200:
vl_data = vl_resp.json()
if vl_data.get("errorCode") == 0:
layout_results = vl_data.get("result", {}).get("layoutParsingResults", [])
md_text = ""
if layout_results:
md_text = (layout_results[0].get("markdown") or {}).get("text", "") or ""
vl_lines = [ln.strip() for ln in md_text.splitlines() if ln.strip()]
vl_text_lines = vl_lines
if vl_lines:
vl_date, vl_idx, vl_source_line = extract_expired_date(vl_lines)
if vl_date is not None and vl_source_line and line_has_exp_keyword(
clean_date_line(vl_source_line)
):
print(f"[VL-Fallback] Expiry date {vl_date} found by VL pipeline (line: {vl_source_line!r})")
expired_date = vl_date
expired_idx = None # no OCR polys for VL text; skip crop
expired_source_line = vl_source_line
else:
print(f"[VL-Fallback] pipeline error: {vl_resp.status_code} {vl_resp.text[:200]}")
except Exception as vl_err:
print(f"[VL-Fallback] failed: {vl_err}")
traceback.print_exc()
# Fine tilt-straighten correction (<90 degrees), applied on top of
# whichever 90-degree orientation the search above landed on.
try:
img_arr = np.array(image)
res_list = list(ocr.predict(img_arr))
if res_list and len(res_list) > 0:
res_entry = res_list[0]
text_lines = res_entry.get("rec_texts", [])
text_polys = ocr_text_polys(res_entry)
expired_date, expired_idx, expired_source_line = extract_expired_date(text_lines)
if expired_idx is not None and expired_idx < len(text_polys):
poly = text_polys[expired_idx]
if len(poly) >= 2:
p0 = poly[0]
p1 = poly[1]
dx = float(p1[0]) - float(p0[0])
dy = float(p1[1]) - float(p0[1])
angle_rad = math.atan2(dy, dx)
angle_deg = math.degrees(angle_rad)
# Standardize tilt rotation
if abs(angle_deg) > 3.0:
print(f"[Auto-Rotate] Detected Expiry Date text line angle: {angle_deg:.2f} degrees. Rotating image...")
image = image.rotate(angle_deg, resample=Image.BICUBIC, expand=True)
rotated_image_used = True
if expired_idx is not None and expired_idx < len(text_polys):
poly = text_polys[expired_idx]
if len(poly) >= 2:
p0 = poly[0]
p1 = poly[1]
dx = float(p1[0]) - float(p0[0])
dy = float(p1[1]) - float(p0[1])
angle_rad = math.atan2(dy, dx)
angle_deg = math.degrees(angle_rad)
if abs(angle_deg) > 3.0:
print(f"[Auto-Rotate] Detected Expiry Date text line angle: {angle_deg:.2f} degrees. Rotating image...")
image = image.rotate(angle_deg, resample=Image.BICUBIC, expand=True)
rotated_image_used = True
except Exception as pre_ocr_err:
print(f"Error in pre-pass OCR: {pre_ocr_err}")
print(f"Error in fine tilt-straighten pass: {pre_ocr_err}")
traceback.print_exc()
# 1. Run DINOv2 Similarity Search or YOLO Classification
@@ -521,7 +680,7 @@ async def classify_ocr(payload: ScanRequest):
device = "cuda" if torch.cuda.is_available() else "cpu"
# Preprocess image
preprocessed = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
preprocessed = DINOV2_TRANSFORMS(classification_image).unsqueeze(0).to(device)
# Extract query embedding
with torch.no_grad():
@@ -572,7 +731,7 @@ async def classify_ocr(payload: ScanRequest):
# Fallback to YOLO if DINOv2 was not run or failed
if not top1_name:
if yolo_model:
results = yolo_model(image)
results = yolo_model(classification_image)
probs = results[0].probs
top1_idx = probs.top1
top1_conf = float(probs.top1conf)
@@ -628,47 +787,19 @@ async def classify_ocr(payload: ScanRequest):
crop_idx = find_expired_crop_index(
text_lines, expired_idx, expired_date, len(text_polys)
)
# Create visual OCR image with bounding boxes
vis_image_b64 = None
try:
vis_image = coord_image.copy()
from PIL import ImageDraw, ImageFont
draw = ImageDraw.Draw(vis_image)
try:
font = ImageFont.load_default()
except:
font = None
for idx, poly in enumerate(text_polys):
is_expired = (crop_idx is not None and idx == crop_idx)
pts = [(float(p[0]), float(p[1])) for p in poly]
if is_expired:
color = (245, 158, 11) # Amber
label = "EXP"
else:
color = (13, 148, 136) # Teal
label = "TEXT"
draw.polygon(pts, outline=color, width=3)
x0, y0 = pts[0]
label_w = 32 if label == "EXP" else 38
draw.rectangle([x0, y0 - 15, x0 + label_w, y0], fill=color)
if font:
draw.text((x0 + 4, y0 - 14), label, fill=(255, 255, 255), font=font)
else:
draw.text((x0 + 4, y0 - 14), label, fill=(255, 255, 255))
buffered = io.BytesIO()
vis_image.save(buffered, format="JPEG")
vis_image_b64 = "data:image/jpeg;base64," + base64.b64encode(buffered.getvalue()).decode("utf-8")
except Exception as draw_err:
print(f"Error drawing visual OCR: {draw_err}")
traceback.print_exc()
# Merge the VL pipeline's text lines (when its fallback ran) into
# the returned text_lines: the gateway's classification re-ranking
# feeds on them, and they're much cleaner than local OCR on hard
# photos. Appended after all poly-aligned work above, so rec_polys
# indexing is unaffected. Also retry SKU extraction over them -
# a VL-read 8-digit SKU enables the gateway's exact-match pin.
if vl_text_lines:
text_lines = list(text_lines) + vl_text_lines
if not sku:
sku = extract_sku(vl_text_lines)
if sku:
print(f"[VL-Fallback] SKU {sku} extracted from VL text lines.")
# Crop expired date OCR region for summary verification
expired_date_crop_b64 = None
@@ -679,43 +810,6 @@ async def classify_ocr(payload: ScanRequest):
print(f"Error cropping expired date image: {crop_err}")
traceback.print_exc()
# 3. Call Spotting API
spotting_image_b64 = None
try:
if rotated_image_used:
buffered = io.BytesIO()
image.save(buffered, format="JPEG")
img_b64_only = base64.b64encode(buffered.getvalue()).decode("utf-8")
else:
img_b64_only = payload.image_base64.split(",")[-1]
spotting_payload = {
"file": img_b64_only,
"matchHistoryJob": False,
"useLayoutDetection": False,
"fileType": 1,
"useDocUnwarping": False,
"useDocOrientationClassify": False,
"promptLabel": "spotting"
}
spotting_url = "http://localhost:8090/layout-parsing"
spotting_resp = requests.post(spotting_url, json=spotting_payload, timeout=60)
if spotting_resp.status_code == 200:
spotting_data = spotting_resp.json()
if spotting_data.get("errorCode") == 0:
layout_results = spotting_data.get("result", {}).get("layoutParsingResults", [])
if layout_results:
page0 = layout_results[0]
out_imgs = page0.get("outputImages", {})
spotting_img = out_imgs.get("spotting_res_img")
if spotting_img:
spotting_image_b64 = "data:image/jpeg;base64," + spotting_img
else:
print(f"Spotting API error: {spotting_resp.text}")
except Exception as spotting_err:
print(f"Error calling spotting API: {spotting_err}")
traceback.print_exc()
ocr_result = {
"text_lines": text_lines,
"extracted_product_name": product_name,
@@ -723,9 +817,7 @@ async def classify_ocr(payload: ScanRequest):
"extracted_expired_date": expired_date,
"expired_line_index": crop_idx,
"expired_source_line": expired_source_line,
"expired_date_crop_base64": expired_date_crop_b64,
"vis_image_base64": vis_image_b64,
"spotting_image_base64": spotting_image_b64
"expired_date_crop_base64": expired_date_crop_b64
}
else:
ocr_result = {