chore(backend): measure tiles+VL-evidence (zero net change, 79.7% holds) + add /probe-ocr debug endpoint

2026-07-15 re-run of the 79-image benchmark: the tiled full-res pass, VL
text-line evidence, and VL SKU retry deployed at end of 2026-07-14 produce
ZERO flips vs the 79.7% baseline (only img1's expiry pred changed to a
lenient-parse garble). Detail dump: product_scan_detail_20260715.json.

Probe endpoint POST :8120/probe-ocr (temporary, remove before production):
OCRs a crop of a container-local image through arbitrary preprocessing
recipes (scale/blur/close/CLAHE/threshold) and alternate readers - VL
layout-parsing with/without layout detection, direct VLM chat on :8118,
lazily-loaded PP-OCRv5 server det/rec. Probing the dot-matrix expiry crops
of img2/img41 with every combination shows none of the stack's models can
read these prints (VL tags them as pictures, VLM hallucinates, v5-server
skips them) - the 21 expiry non-detections are a model capability ceiling,
not a pipeline bug.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
Rafhan Mazaya FathurrahmanandClaude Fable 5 committed 2026-07-14 21:45:44 +07:00
1 parent e76ccb60a6
commit 48334ded8f
3 files changed
+2127

No files matched your search

+145
View File
@@ -164,6 +164,19 @@ except Exception as e:
class ScanRequest(BaseModel):
image_base64: str
class ProbeRequest(BaseModel):
# Temporary debug endpoint input: container-local image path + crop box.
path: str
x0: int
y0: int
x1: int
y1: int
# Each recipe is a comma-separated op chain applied left to right, e.g.
# "s2,blur5" = upscale 2x then Gaussian-blur k=5. Ops: sN (scale xN,
# floats ok), blurN, closeN (morph close), clahe, gray, inv, thrN
# (adaptive threshold, block N).
recipes: list = ["none", "s2", "blur5", "s2,blur5"]
def clean_ocr_text(text: str) -> str:
return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip()
@@ -481,6 +494,138 @@ def extract_product_name(text_lines, classified_name=None):
return "Unknown Product"
@app.post("/probe-ocr")
async def probe_ocr(payload: ProbeRequest):
# Temporary debug endpoint: OCR a crop of a container-local image at
# several upscale factors (optionally CLAHE-enhanced) using the
# already-loaded GPU OCR model, and run the expiry cascade on each
# variant's lines. Lets us test enhancement recipes without loading a
# second model instance (GPU is full) - remove once tuning is done.
from PIL import ImageOps
import cv2
img = ImageOps.exif_transpose(Image.open(payload.path)).convert("RGB")
crop = img.crop((payload.x0, payload.y0, payload.x1, payload.y1))
out = {"image_size": img.size, "variants": {}}
def apply_ops(arr, recipe):
for op in recipe.split(","):
op = op.strip().lower()
if not op or op == "none":
continue
if op.startswith("s"):
f = float(op[1:])
arr = cv2.resize(arr, None, fx=f, fy=f, interpolation=cv2.INTER_LANCZOS4)
elif op.startswith("blur"):
k = int(op[4:]) | 1
arr = cv2.GaussianBlur(arr, (k, k), 0)
elif op.startswith("close"):
k = int(op[5:])
kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (k, k))
arr = cv2.morphologyEx(arr, cv2.MORPH_CLOSE, kernel)
elif op == "clahe":
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
cl = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8, 8)).apply(g)
arr = cv2.cvtColor(cl, cv2.COLOR_GRAY2RGB)
elif op == "gray":
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
arr = cv2.cvtColor(g, cv2.COLOR_GRAY2RGB)
elif op == "inv":
arr = 255 - arr
elif op.startswith("thr"):
b = int(op[3:]) | 1
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
t = cv2.adaptiveThreshold(g, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C,
cv2.THRESH_BINARY, b, 10)
arr = cv2.cvtColor(t, cv2.COLOR_GRAY2RGB)
return arr
def vl_read(arr, use_layout=True):
# Route the crop through the vLLM-backed VL pipeline instead of the
# local PP-OCR model. Returns text lines from its markdown output.
buffered = io.BytesIO()
Image.fromarray(arr).save(buffered, format="JPEG")
resp = requests.post(
os.environ.get("VL_PIPELINE_URL", "http://localhost:8090/layout-parsing"),
json={
"file": base64.b64encode(buffered.getvalue()).decode("utf-8"),
"matchHistoryJob": False,
"useLayoutDetection": use_layout,
"fileType": 1,
"useDocUnwarping": False,
"useDocOrientationClassify": True,
},
timeout=120,
)
resp.raise_for_status()
data = resp.json()
results = data.get("result", {}).get("layoutParsingResults", [])
md = (results[0].get("markdown") or {}).get("text", "") if results else ""
return [ln.strip() for ln in md.splitlines() if ln.strip()]
def vlm_read(arr):
# Ask the PaddleOCR-VL model on the vLLM genai server (:8118) to OCR
# the crop directly, skipping the pipeline's layout detection (which
# tags dot-matrix prints as pictures and refuses to read them).
buffered = io.BytesIO()
Image.fromarray(arr).save(buffered, format="JPEG")
b64 = base64.b64encode(buffered.getvalue()).decode("utf-8")
resp = requests.post(
os.environ.get("VLM_CHAT_URL", "http://paddleocr-vllm-server:8118/v1/chat/completions"),
json={
"model": os.environ.get("VLM_MODEL", "PaddleOCR-VL-1.6-0.9B"),
"messages": [{
"role": "user",
"content": [
{"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{b64}"}},
{"type": "text", "text": "OCR:"},
],
}],
"temperature": 0.0,
"max_tokens": 256,
},
timeout=120,
)
resp.raise_for_status()
content = resp.json()["choices"][0]["message"]["content"] or ""
return [ln.strip() for ln in content.splitlines() if ln.strip()]
def v5s_ocr():
# Lazily load the heavier PP-OCRv5 server det/rec pair (the main
# pipeline runs PP-OCRv6_medium). Cached on the app object so
# repeated probes don't reload weights.
if not hasattr(app.state, "v5s_ocr"):
app.state.v5s_ocr = PaddleOCR(
text_detection_model_name="PP-OCRv5_server_det",
text_recognition_model_name="PP-OCRv5_server_rec",
use_textline_orientation=True,
)
return app.state.v5s_ocr
base = np.array(crop)
readers = ("vl", "vlnl", "vlm", "v5s")
for recipe in payload.recipes:
try:
ops = [op.strip().lower() for op in recipe.split(",")]
reader = next((o for o in ops if o in readers), None)
arr = apply_ops(base.copy(), ",".join(o for o in ops if o not in readers))
if reader == "vl":
lines = vl_read(arr, use_layout=True)
elif reader == "vlnl":
lines = vl_read(arr, use_layout=False)
elif reader == "vlm":
lines = vlm_read(arr)
elif reader == "v5s":
res = list(v5s_ocr().predict(arr))
lines = res[0].get("rec_texts", []) if res else []
else:
res = list(ocr.predict(arr))
lines = res[0].get("rec_texts", []) if res else []
d, _i, src = extract_expired_date(lines)
out["variants"][recipe] = {"size": [arr.shape[1], arr.shape[0]], "lines": lines, "date": d, "source": src}
except Exception as e:
out["variants"][recipe] = {"error": str(e)}
return out
@app.post("/classify-ocr")
async def classify_ocr(payload: ScanRequest):
try: