chore(backend): measure tiles+VL-evidence (zero net change, 79.7% holds) + add /probe-ocr debug endpoint
2026-07-15 re-run of the 79-image benchmark: the tiled full-res pass, VL text-line evidence, and VL SKU retry deployed at end of 2026-07-14 produce ZERO flips vs the 79.7% baseline (only img1's expiry pred changed to a lenient-parse garble). Detail dump: product_scan_detail_20260715.json. Probe endpoint POST :8120/probe-ocr (temporary, remove before production): OCRs a crop of a container-local image through arbitrary preprocessing recipes (scale/blur/close/CLAHE/threshold) and alternate readers - VL layout-parsing with/without layout detection, direct VLM chat on :8118, lazily-loaded PP-OCRv5 server det/rec. Probing the dot-matrix expiry crops of img2/img41 with every combination shows none of the stack's models can read these prints (VL tags them as pictures, VLM hallucinates, v5-server skips them) - the 21 expiry non-detections are a model capability ceiling, not a pipeline bug. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
1 parent
e76ccb60a6
commit
48334ded8f
3 files changed
+2127
No files matched your search
@@ -164,6 +164,19 @@ except Exception as e:
|
||||
class ScanRequest(BaseModel):
|
||||
image_base64: str
|
||||
|
||||
class ProbeRequest(BaseModel):
|
||||
# Temporary debug endpoint input: container-local image path + crop box.
|
||||
path: str
|
||||
x0: int
|
||||
y0: int
|
||||
x1: int
|
||||
y1: int
|
||||
# Each recipe is a comma-separated op chain applied left to right, e.g.
|
||||
# "s2,blur5" = upscale 2x then Gaussian-blur k=5. Ops: sN (scale xN,
|
||||
# floats ok), blurN, closeN (morph close), clahe, gray, inv, thrN
|
||||
# (adaptive threshold, block N).
|
||||
recipes: list = ["none", "s2", "blur5", "s2,blur5"]
|
||||
|
||||
def clean_ocr_text(text: str) -> str:
|
||||
return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip()
|
||||
|
||||
@@ -481,6 +494,138 @@ def extract_product_name(text_lines, classified_name=None):
|
||||
|
||||
return "Unknown Product"
|
||||
|
||||
@app.post("/probe-ocr")
|
||||
async def probe_ocr(payload: ProbeRequest):
|
||||
# Temporary debug endpoint: OCR a crop of a container-local image at
|
||||
# several upscale factors (optionally CLAHE-enhanced) using the
|
||||
# already-loaded GPU OCR model, and run the expiry cascade on each
|
||||
# variant's lines. Lets us test enhancement recipes without loading a
|
||||
# second model instance (GPU is full) - remove once tuning is done.
|
||||
from PIL import ImageOps
|
||||
import cv2
|
||||
img = ImageOps.exif_transpose(Image.open(payload.path)).convert("RGB")
|
||||
crop = img.crop((payload.x0, payload.y0, payload.x1, payload.y1))
|
||||
out = {"image_size": img.size, "variants": {}}
|
||||
|
||||
def apply_ops(arr, recipe):
|
||||
for op in recipe.split(","):
|
||||
op = op.strip().lower()
|
||||
if not op or op == "none":
|
||||
continue
|
||||
if op.startswith("s"):
|
||||
f = float(op[1:])
|
||||
arr = cv2.resize(arr, None, fx=f, fy=f, interpolation=cv2.INTER_LANCZOS4)
|
||||
elif op.startswith("blur"):
|
||||
k = int(op[4:]) | 1
|
||||
arr = cv2.GaussianBlur(arr, (k, k), 0)
|
||||
elif op.startswith("close"):
|
||||
k = int(op[5:])
|
||||
kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (k, k))
|
||||
arr = cv2.morphologyEx(arr, cv2.MORPH_CLOSE, kernel)
|
||||
elif op == "clahe":
|
||||
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
|
||||
cl = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8, 8)).apply(g)
|
||||
arr = cv2.cvtColor(cl, cv2.COLOR_GRAY2RGB)
|
||||
elif op == "gray":
|
||||
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
|
||||
arr = cv2.cvtColor(g, cv2.COLOR_GRAY2RGB)
|
||||
elif op == "inv":
|
||||
arr = 255 - arr
|
||||
elif op.startswith("thr"):
|
||||
b = int(op[3:]) | 1
|
||||
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
|
||||
t = cv2.adaptiveThreshold(g, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C,
|
||||
cv2.THRESH_BINARY, b, 10)
|
||||
arr = cv2.cvtColor(t, cv2.COLOR_GRAY2RGB)
|
||||
return arr
|
||||
|
||||
def vl_read(arr, use_layout=True):
|
||||
# Route the crop through the vLLM-backed VL pipeline instead of the
|
||||
# local PP-OCR model. Returns text lines from its markdown output.
|
||||
buffered = io.BytesIO()
|
||||
Image.fromarray(arr).save(buffered, format="JPEG")
|
||||
resp = requests.post(
|
||||
os.environ.get("VL_PIPELINE_URL", "http://localhost:8090/layout-parsing"),
|
||||
json={
|
||||
"file": base64.b64encode(buffered.getvalue()).decode("utf-8"),
|
||||
"matchHistoryJob": False,
|
||||
"useLayoutDetection": use_layout,
|
||||
"fileType": 1,
|
||||
"useDocUnwarping": False,
|
||||
"useDocOrientationClassify": True,
|
||||
},
|
||||
timeout=120,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
results = data.get("result", {}).get("layoutParsingResults", [])
|
||||
md = (results[0].get("markdown") or {}).get("text", "") if results else ""
|
||||
return [ln.strip() for ln in md.splitlines() if ln.strip()]
|
||||
|
||||
def vlm_read(arr):
|
||||
# Ask the PaddleOCR-VL model on the vLLM genai server (:8118) to OCR
|
||||
# the crop directly, skipping the pipeline's layout detection (which
|
||||
# tags dot-matrix prints as pictures and refuses to read them).
|
||||
buffered = io.BytesIO()
|
||||
Image.fromarray(arr).save(buffered, format="JPEG")
|
||||
b64 = base64.b64encode(buffered.getvalue()).decode("utf-8")
|
||||
resp = requests.post(
|
||||
os.environ.get("VLM_CHAT_URL", "http://paddleocr-vllm-server:8118/v1/chat/completions"),
|
||||
json={
|
||||
"model": os.environ.get("VLM_MODEL", "PaddleOCR-VL-1.6-0.9B"),
|
||||
"messages": [{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{b64}"}},
|
||||
{"type": "text", "text": "OCR:"},
|
||||
],
|
||||
}],
|
||||
"temperature": 0.0,
|
||||
"max_tokens": 256,
|
||||
},
|
||||
timeout=120,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
content = resp.json()["choices"][0]["message"]["content"] or ""
|
||||
return [ln.strip() for ln in content.splitlines() if ln.strip()]
|
||||
|
||||
def v5s_ocr():
|
||||
# Lazily load the heavier PP-OCRv5 server det/rec pair (the main
|
||||
# pipeline runs PP-OCRv6_medium). Cached on the app object so
|
||||
# repeated probes don't reload weights.
|
||||
if not hasattr(app.state, "v5s_ocr"):
|
||||
app.state.v5s_ocr = PaddleOCR(
|
||||
text_detection_model_name="PP-OCRv5_server_det",
|
||||
text_recognition_model_name="PP-OCRv5_server_rec",
|
||||
use_textline_orientation=True,
|
||||
)
|
||||
return app.state.v5s_ocr
|
||||
|
||||
base = np.array(crop)
|
||||
readers = ("vl", "vlnl", "vlm", "v5s")
|
||||
for recipe in payload.recipes:
|
||||
try:
|
||||
ops = [op.strip().lower() for op in recipe.split(",")]
|
||||
reader = next((o for o in ops if o in readers), None)
|
||||
arr = apply_ops(base.copy(), ",".join(o for o in ops if o not in readers))
|
||||
if reader == "vl":
|
||||
lines = vl_read(arr, use_layout=True)
|
||||
elif reader == "vlnl":
|
||||
lines = vl_read(arr, use_layout=False)
|
||||
elif reader == "vlm":
|
||||
lines = vlm_read(arr)
|
||||
elif reader == "v5s":
|
||||
res = list(v5s_ocr().predict(arr))
|
||||
lines = res[0].get("rec_texts", []) if res else []
|
||||
else:
|
||||
res = list(ocr.predict(arr))
|
||||
lines = res[0].get("rec_texts", []) if res else []
|
||||
d, _i, src = extract_expired_date(lines)
|
||||
out["variants"][recipe] = {"size": [arr.shape[1], arr.shape[0]], "lines": lines, "date": d, "source": src}
|
||||
except Exception as e:
|
||||
out["variants"][recipe] = {"error": str(e)}
|
||||
return out
|
||||
|
||||
@app.post("/classify-ocr")
|
||||
async def classify_ocr(payload: ScanRequest):
|
||||
try:
|
||||
|
||||
Reference in new issue
Block a user