feat(backend): scan-product accuracy 66.2% -> 79.7% + frozen validation benchmark
Accuracy work on the 79-image product-scan validation set (user goal: 90%): - classify_ocr_server.py: 0/90/180/270-degree expiry-date search (stops at first hit, 0-degree fallback); classification decoupled onto the upright image (rotated frames regressed DINOv2 -6pts until this); cross-line date stitching; tiled full-res OCR pass (defeats the 4000px downscale that killed small inkjet dates); VL-pipeline expiry fallback with keyword-anchored anti-hallucination guard; VL text lines merged into text_lines + VL SKU retry. Visualization endpoints removed entirely (Visual/Spotting grids - unused by frontend, 3x per-scan GPU cost). - product-scan.ts: coverage-normalized OCR-evidence re-ranking of DINOv2 top-K (tuned offline: +8/-0 on top-1 misses), re-ranked class mapped to sku_master by SKU prefix; classifier timeout 90s->240s for fallback paths. - Frozen benchmark: product-test-images-fixed/ (79 renamed images) + freeze/seed/build-undetected/capture/experiment scripts; labels trimmed to the 79 validation entries (training rows kept in .bak-with-training); 5 TRAINED-ON SKUs replaced with fresh held-out photos. - manual-label-scan page: shows last batch-test AI prediction under every field by default (new /api/product-scan-results); serves the fixed folder; fixed total hydration failure via allowedDevOrigins 127.0.0.1. - Measured (all-79, zero failures): sku/name 87.3%, expiry 64.6%, overall 79.7%. Tiles/VL-evidence/VL-SKU deployed but not yet batch-measured. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
1 parent
19f1facf9b
commit
e76ccb60a6
156 files changed
+17150
-1405
No files matched your search
@@ -316,6 +316,24 @@ def extract_expired_date(text_lines):
|
||||
if match:
|
||||
return pick(match, idx, line)
|
||||
|
||||
# 3.6) Keyword line + date split onto an adjacent line (PaddleOCR sometimes
|
||||
# detects "BB"/"Baik digunakan" as its own box, separate from the date
|
||||
# digits in a neighboring box, e.g. "BB" / "05032027" as two lines).
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
if not line_has_exp_keyword(line):
|
||||
continue
|
||||
for j in (idx + 1, idx - 1, idx + 2):
|
||||
if j < 0 or j >= len(cleaned_lines) or j == idx:
|
||||
continue
|
||||
neighbor = cleaned_lines[j]
|
||||
combined = f"{line} {neighbor}" if j > idx else f"{neighbor} {line}"
|
||||
match = (BB_ATTACHED_DATE_RE.search(combined)
|
||||
or DDMMYYYY_RE.search(combined)
|
||||
or DD_MM_YYYY_RE.search(combined))
|
||||
if match:
|
||||
report_idx = j if sum(c.isdigit() for c in neighbor) > sum(c.isdigit() for c in line) else idx
|
||||
return pick(match, report_idx, combined)
|
||||
|
||||
# 4) Any line — spaced DD MM YYYY
|
||||
for idx, line in enumerate(cleaned_lines):
|
||||
match = DD_MM_YYYY_RE.search(line)
|
||||
@@ -471,8 +489,24 @@ async def classify_ocr(payload: ScanRequest):
|
||||
img_data = base64.b64decode(payload.image_base64.split(",")[-1])
|
||||
raw_image = Image.open(io.BytesIO(img_data))
|
||||
image = ImageOps.exif_transpose(raw_image).convert("RGB")
|
||||
|
||||
# First-pass PaddleOCR to check orientation based on Expiry Date
|
||||
# Classification always sees the original upright orientation - the
|
||||
# 90-degree expiry-date search below may rotate `image` to a
|
||||
# sideways/upside-down orientation that DINOv2/YOLO were never
|
||||
# trained on (their reference photos are all shot upright), so using
|
||||
# a rotated frame there would hurt classification, not help it.
|
||||
classification_image = image
|
||||
|
||||
# Multi-orientation expiry-date search: some photos are captured with
|
||||
# the whole frame rotated ~90 degrees from upright (e.g. staff held
|
||||
# the phone in portrait for a package whose printed date runs
|
||||
# horizontally), so the expiry stamp - and the product framing -
|
||||
# ends up sideways. Try 0/90/180/270 degree rotations in order and
|
||||
# stop at the first one where PaddleOCR actually finds an expiry
|
||||
# date; if none of the four find one, fall back to the 0-degree
|
||||
# result so behaviour for genuinely-undetectable photos is unchanged.
|
||||
# This costs extra OCR passes (up to 4x) only on images where the
|
||||
# first pass found nothing - already-working images stay on the fast
|
||||
# single-pass path below.
|
||||
rotated_image_used = False
|
||||
res_list = []
|
||||
text_lines = []
|
||||
@@ -480,35 +514,160 @@ async def classify_ocr(payload: ScanRequest):
|
||||
expired_date = None
|
||||
expired_idx = None
|
||||
expired_source_line = None
|
||||
|
||||
|
||||
if ocr:
|
||||
base_image = image
|
||||
for step_angle in (0, 90, 180, 270):
|
||||
try:
|
||||
candidate_image = (
|
||||
base_image.rotate(step_angle, resample=Image.BICUBIC, expand=True)
|
||||
if step_angle else base_image
|
||||
)
|
||||
img_arr = np.array(candidate_image)
|
||||
candidate_res_list = list(ocr.predict(img_arr))
|
||||
candidate_res_entry = candidate_res_list[0] if candidate_res_list else {}
|
||||
candidate_text_lines = candidate_res_entry.get("rec_texts", [])
|
||||
candidate_text_polys = ocr_text_polys(candidate_res_entry)
|
||||
candidate_expired_date, candidate_expired_idx, candidate_expired_source_line = (
|
||||
extract_expired_date(candidate_text_lines)
|
||||
)
|
||||
|
||||
if step_angle == 0:
|
||||
# Always keep the 0-degree pass as the fallback result.
|
||||
image, res_list, text_lines, text_polys = (
|
||||
candidate_image, candidate_res_list, candidate_text_lines, candidate_text_polys
|
||||
)
|
||||
expired_date, expired_idx, expired_source_line = (
|
||||
candidate_expired_date, candidate_expired_idx, candidate_expired_source_line
|
||||
)
|
||||
|
||||
if candidate_expired_date is not None:
|
||||
if step_angle != 0:
|
||||
print(f"[Auto-Rotate-90] Expiry date found after rotating {step_angle} degrees.")
|
||||
image, res_list, text_lines, text_polys = (
|
||||
candidate_image, candidate_res_list, candidate_text_lines, candidate_text_polys
|
||||
)
|
||||
rotated_image_used = True
|
||||
expired_date, expired_idx, expired_source_line = (
|
||||
candidate_expired_date, candidate_expired_idx, candidate_expired_source_line
|
||||
)
|
||||
break
|
||||
except Exception as rot_err:
|
||||
print(f"Error during {step_angle}-degree OCR pass: {rot_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
# Tiled full-resolution pass: PaddleOCR downscales anything over
|
||||
# its 4000px max_side_limit, which is exactly what kills small
|
||||
# inkjet date stamps on these ~3200x5700 phone photos. Split the
|
||||
# original image into overlapping tiles that each fit under the
|
||||
# limit (so the date region is OCR'd at native resolution) and
|
||||
# run the cascade per tile. Failure-path only, keyword-anchored
|
||||
# acceptance like the VL fallback below.
|
||||
if expired_date is None and max(base_image.size) > 2600:
|
||||
TILE, OVERLAP = 2400, 400
|
||||
W, H = base_image.size
|
||||
step = TILE - OVERLAP
|
||||
try:
|
||||
found = False
|
||||
for y0 in range(0, H, step):
|
||||
if found:
|
||||
break
|
||||
for x0 in range(0, W, step):
|
||||
tile = base_image.crop((x0, y0, min(x0 + TILE, W), min(y0 + TILE, H)))
|
||||
if tile.width < 300 or tile.height < 300:
|
||||
continue
|
||||
tile_res = list(ocr.predict(np.array(tile)))
|
||||
tile_lines = tile_res[0].get("rec_texts", []) if tile_res else []
|
||||
if not tile_lines:
|
||||
continue
|
||||
t_date, _t_idx, t_source = extract_expired_date(tile_lines)
|
||||
if t_date is not None and t_source and line_has_exp_keyword(
|
||||
clean_date_line(t_source)
|
||||
):
|
||||
print(f"[Tile-Pass] Expiry date {t_date} found in full-res tile ({x0},{y0}) (line: {t_source!r})")
|
||||
expired_date = t_date
|
||||
expired_idx = None # tile polys don't map to the full image
|
||||
expired_source_line = t_source
|
||||
found = True
|
||||
break
|
||||
except Exception as tile_err:
|
||||
print(f"[Tile-Pass] failed: {tile_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
# VL fallback: the lightweight PP-OCRv6 detector missed the date
|
||||
# at every orientation. The vLLM-backed PaddleOCR-VL pipeline
|
||||
# (:8090, same container) is a much stronger reader of small,
|
||||
# low-contrast inkjet codes - ask it to read the whole package
|
||||
# and run the same date cascade over its text output. Only fires
|
||||
# on already-failed images, so the happy path stays single-pass.
|
||||
# Acceptance is stricter than the local cascade: the matched
|
||||
# line must carry an expiry keyword (BB/EXP/Baik digunakan...),
|
||||
# so a bare number elsewhere on the package can't be
|
||||
# hallucinated into a date on photos where none is visible.
|
||||
# Even when no date is found, the VL's (much cleaner) text lines
|
||||
# are kept and appended to text_lines below - they feed the
|
||||
# gateway's OCR-evidence classification re-ranking.
|
||||
vl_text_lines = []
|
||||
if expired_date is None:
|
||||
try:
|
||||
vl_url = os.environ.get(
|
||||
"VL_PIPELINE_URL", "http://localhost:8090/layout-parsing"
|
||||
)
|
||||
buffered = io.BytesIO()
|
||||
base_image.save(buffered, format="JPEG")
|
||||
vl_payload = {
|
||||
"file": base64.b64encode(buffered.getvalue()).decode("utf-8"),
|
||||
"matchHistoryJob": False,
|
||||
"useLayoutDetection": True,
|
||||
"fileType": 1,
|
||||
"useDocUnwarping": False,
|
||||
"useDocOrientationClassify": True,
|
||||
}
|
||||
vl_resp = requests.post(vl_url, json=vl_payload, timeout=120)
|
||||
if vl_resp.status_code == 200:
|
||||
vl_data = vl_resp.json()
|
||||
if vl_data.get("errorCode") == 0:
|
||||
layout_results = vl_data.get("result", {}).get("layoutParsingResults", [])
|
||||
md_text = ""
|
||||
if layout_results:
|
||||
md_text = (layout_results[0].get("markdown") or {}).get("text", "") or ""
|
||||
vl_lines = [ln.strip() for ln in md_text.splitlines() if ln.strip()]
|
||||
vl_text_lines = vl_lines
|
||||
if vl_lines:
|
||||
vl_date, vl_idx, vl_source_line = extract_expired_date(vl_lines)
|
||||
if vl_date is not None and vl_source_line and line_has_exp_keyword(
|
||||
clean_date_line(vl_source_line)
|
||||
):
|
||||
print(f"[VL-Fallback] Expiry date {vl_date} found by VL pipeline (line: {vl_source_line!r})")
|
||||
expired_date = vl_date
|
||||
expired_idx = None # no OCR polys for VL text; skip crop
|
||||
expired_source_line = vl_source_line
|
||||
else:
|
||||
print(f"[VL-Fallback] pipeline error: {vl_resp.status_code} {vl_resp.text[:200]}")
|
||||
except Exception as vl_err:
|
||||
print(f"[VL-Fallback] failed: {vl_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
# Fine tilt-straighten correction (<90 degrees), applied on top of
|
||||
# whichever 90-degree orientation the search above landed on.
|
||||
try:
|
||||
img_arr = np.array(image)
|
||||
res_list = list(ocr.predict(img_arr))
|
||||
if res_list and len(res_list) > 0:
|
||||
res_entry = res_list[0]
|
||||
text_lines = res_entry.get("rec_texts", [])
|
||||
text_polys = ocr_text_polys(res_entry)
|
||||
expired_date, expired_idx, expired_source_line = extract_expired_date(text_lines)
|
||||
|
||||
if expired_idx is not None and expired_idx < len(text_polys):
|
||||
poly = text_polys[expired_idx]
|
||||
if len(poly) >= 2:
|
||||
p0 = poly[0]
|
||||
p1 = poly[1]
|
||||
dx = float(p1[0]) - float(p0[0])
|
||||
dy = float(p1[1]) - float(p0[1])
|
||||
|
||||
angle_rad = math.atan2(dy, dx)
|
||||
angle_deg = math.degrees(angle_rad)
|
||||
|
||||
# Standardize tilt rotation
|
||||
if abs(angle_deg) > 3.0:
|
||||
print(f"[Auto-Rotate] Detected Expiry Date text line angle: {angle_deg:.2f} degrees. Rotating image...")
|
||||
image = image.rotate(angle_deg, resample=Image.BICUBIC, expand=True)
|
||||
rotated_image_used = True
|
||||
if expired_idx is not None and expired_idx < len(text_polys):
|
||||
poly = text_polys[expired_idx]
|
||||
if len(poly) >= 2:
|
||||
p0 = poly[0]
|
||||
p1 = poly[1]
|
||||
dx = float(p1[0]) - float(p0[0])
|
||||
dy = float(p1[1]) - float(p0[1])
|
||||
|
||||
angle_rad = math.atan2(dy, dx)
|
||||
angle_deg = math.degrees(angle_rad)
|
||||
|
||||
if abs(angle_deg) > 3.0:
|
||||
print(f"[Auto-Rotate] Detected Expiry Date text line angle: {angle_deg:.2f} degrees. Rotating image...")
|
||||
image = image.rotate(angle_deg, resample=Image.BICUBIC, expand=True)
|
||||
rotated_image_used = True
|
||||
except Exception as pre_ocr_err:
|
||||
print(f"Error in pre-pass OCR: {pre_ocr_err}")
|
||||
print(f"Error in fine tilt-straighten pass: {pre_ocr_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
# 1. Run DINOv2 Similarity Search or YOLO Classification
|
||||
@@ -521,7 +680,7 @@ async def classify_ocr(payload: ScanRequest):
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
|
||||
# Preprocess image
|
||||
preprocessed = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
|
||||
preprocessed = DINOV2_TRANSFORMS(classification_image).unsqueeze(0).to(device)
|
||||
|
||||
# Extract query embedding
|
||||
with torch.no_grad():
|
||||
@@ -572,7 +731,7 @@ async def classify_ocr(payload: ScanRequest):
|
||||
# Fallback to YOLO if DINOv2 was not run or failed
|
||||
if not top1_name:
|
||||
if yolo_model:
|
||||
results = yolo_model(image)
|
||||
results = yolo_model(classification_image)
|
||||
probs = results[0].probs
|
||||
top1_idx = probs.top1
|
||||
top1_conf = float(probs.top1conf)
|
||||
@@ -628,47 +787,19 @@ async def classify_ocr(payload: ScanRequest):
|
||||
crop_idx = find_expired_crop_index(
|
||||
text_lines, expired_idx, expired_date, len(text_polys)
|
||||
)
|
||||
|
||||
# Create visual OCR image with bounding boxes
|
||||
vis_image_b64 = None
|
||||
try:
|
||||
vis_image = coord_image.copy()
|
||||
from PIL import ImageDraw, ImageFont
|
||||
draw = ImageDraw.Draw(vis_image)
|
||||
|
||||
try:
|
||||
font = ImageFont.load_default()
|
||||
except:
|
||||
font = None
|
||||
|
||||
for idx, poly in enumerate(text_polys):
|
||||
is_expired = (crop_idx is not None and idx == crop_idx)
|
||||
pts = [(float(p[0]), float(p[1])) for p in poly]
|
||||
|
||||
if is_expired:
|
||||
color = (245, 158, 11) # Amber
|
||||
label = "EXP"
|
||||
else:
|
||||
color = (13, 148, 136) # Teal
|
||||
label = "TEXT"
|
||||
|
||||
draw.polygon(pts, outline=color, width=3)
|
||||
|
||||
x0, y0 = pts[0]
|
||||
label_w = 32 if label == "EXP" else 38
|
||||
draw.rectangle([x0, y0 - 15, x0 + label_w, y0], fill=color)
|
||||
|
||||
if font:
|
||||
draw.text((x0 + 4, y0 - 14), label, fill=(255, 255, 255), font=font)
|
||||
else:
|
||||
draw.text((x0 + 4, y0 - 14), label, fill=(255, 255, 255))
|
||||
|
||||
buffered = io.BytesIO()
|
||||
vis_image.save(buffered, format="JPEG")
|
||||
vis_image_b64 = "data:image/jpeg;base64," + base64.b64encode(buffered.getvalue()).decode("utf-8")
|
||||
except Exception as draw_err:
|
||||
print(f"Error drawing visual OCR: {draw_err}")
|
||||
traceback.print_exc()
|
||||
# Merge the VL pipeline's text lines (when its fallback ran) into
|
||||
# the returned text_lines: the gateway's classification re-ranking
|
||||
# feeds on them, and they're much cleaner than local OCR on hard
|
||||
# photos. Appended after all poly-aligned work above, so rec_polys
|
||||
# indexing is unaffected. Also retry SKU extraction over them -
|
||||
# a VL-read 8-digit SKU enables the gateway's exact-match pin.
|
||||
if vl_text_lines:
|
||||
text_lines = list(text_lines) + vl_text_lines
|
||||
if not sku:
|
||||
sku = extract_sku(vl_text_lines)
|
||||
if sku:
|
||||
print(f"[VL-Fallback] SKU {sku} extracted from VL text lines.")
|
||||
|
||||
# Crop expired date OCR region for summary verification
|
||||
expired_date_crop_b64 = None
|
||||
@@ -679,43 +810,6 @@ async def classify_ocr(payload: ScanRequest):
|
||||
print(f"Error cropping expired date image: {crop_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
# 3. Call Spotting API
|
||||
spotting_image_b64 = None
|
||||
try:
|
||||
if rotated_image_used:
|
||||
buffered = io.BytesIO()
|
||||
image.save(buffered, format="JPEG")
|
||||
img_b64_only = base64.b64encode(buffered.getvalue()).decode("utf-8")
|
||||
else:
|
||||
img_b64_only = payload.image_base64.split(",")[-1]
|
||||
|
||||
spotting_payload = {
|
||||
"file": img_b64_only,
|
||||
"matchHistoryJob": False,
|
||||
"useLayoutDetection": False,
|
||||
"fileType": 1,
|
||||
"useDocUnwarping": False,
|
||||
"useDocOrientationClassify": False,
|
||||
"promptLabel": "spotting"
|
||||
}
|
||||
spotting_url = "http://localhost:8090/layout-parsing"
|
||||
spotting_resp = requests.post(spotting_url, json=spotting_payload, timeout=60)
|
||||
if spotting_resp.status_code == 200:
|
||||
spotting_data = spotting_resp.json()
|
||||
if spotting_data.get("errorCode") == 0:
|
||||
layout_results = spotting_data.get("result", {}).get("layoutParsingResults", [])
|
||||
if layout_results:
|
||||
page0 = layout_results[0]
|
||||
out_imgs = page0.get("outputImages", {})
|
||||
spotting_img = out_imgs.get("spotting_res_img")
|
||||
if spotting_img:
|
||||
spotting_image_b64 = "data:image/jpeg;base64," + spotting_img
|
||||
else:
|
||||
print(f"Spotting API error: {spotting_resp.text}")
|
||||
except Exception as spotting_err:
|
||||
print(f"Error calling spotting API: {spotting_err}")
|
||||
traceback.print_exc()
|
||||
|
||||
ocr_result = {
|
||||
"text_lines": text_lines,
|
||||
"extracted_product_name": product_name,
|
||||
@@ -723,9 +817,7 @@ async def classify_ocr(payload: ScanRequest):
|
||||
"extracted_expired_date": expired_date,
|
||||
"expired_line_index": crop_idx,
|
||||
"expired_source_line": expired_source_line,
|
||||
"expired_date_crop_base64": expired_date_crop_b64,
|
||||
"vis_image_base64": vis_image_b64,
|
||||
"spotting_image_base64": spotting_image_b64
|
||||
"expired_date_crop_base64": expired_date_crop_b64
|
||||
}
|
||||
else:
|
||||
ocr_result = {
|
||||
|
||||
Reference in new issue
Block a user