feat(backend): scan-product accuracy 66.2% -> 79.7% + frozen validation benchmark

Accuracy work on the 79-image product-scan validation set (user goal: 90%):
- classify_ocr_server.py: 0/90/180/270-degree expiry-date search (stops at
  first hit, 0-degree fallback); classification decoupled onto the upright
  image (rotated frames regressed DINOv2 -6pts until this); cross-line date
  stitching; tiled full-res OCR pass (defeats the 4000px downscale that
  killed small inkjet dates); VL-pipeline expiry fallback with
  keyword-anchored anti-hallucination guard; VL text lines merged into
  text_lines + VL SKU retry. Visualization endpoints removed entirely
  (Visual/Spotting grids - unused by frontend, 3x per-scan GPU cost).
- product-scan.ts: coverage-normalized OCR-evidence re-ranking of DINOv2
  top-K (tuned offline: +8/-0 on top-1 misses), re-ranked class mapped to
  sku_master by SKU prefix; classifier timeout 90s->240s for fallback paths.
- Frozen benchmark: product-test-images-fixed/ (79 renamed images) +
  freeze/seed/build-undetected/capture/experiment scripts; labels trimmed to
  the 79 validation entries (training rows kept in .bak-with-training);
  5 TRAINED-ON SKUs replaced with fresh held-out photos.
- manual-label-scan page: shows last batch-test AI prediction under every
  field by default (new /api/product-scan-results); serves the fixed folder;
  fixed total hydration failure via allowedDevOrigins 127.0.0.1.
- Measured (all-79, zero failures): sku/name 87.3%, expiry 64.6%, overall
  79.7%. Tiles/VL-evidence/VL-SKU deployed but not yet batch-measured.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
Rafhan Mazaya FathurrahmanandClaude Fable 5 committed 2026-07-14 19:55:17 +07:00
1 parent 19f1facf9b
commit e76ccb60a6
156 files changed
+17150 -1405

No files matched your search

+129 -7
View File
@@ -1,10 +1,12 @@
import { query } from "../db";
// Bounds the classifier call so a wedged GPU container fails fast instead of
// hanging indefinitely - matches the bound `api/parse/route.ts` used to apply
// to its own separate inline classify call before it started sharing this
// function (see docs/api-contract-map.md G3).
const PIPELINE_TIMEOUT_MS = 90_000;
// hanging indefinitely. Raised from 90s (2026-07-14): hard images now
// legitimately take up to ~3 min - a 4-orientation OCR search plus a VL
// pipeline fallback when no expiry date is found (see
// config/classify_ocr_server.py) - and the old bound was killing exactly
// the images those fallbacks exist to save.
const PIPELINE_TIMEOUT_MS = 240_000;
// Thrown when the Python classifier service itself returns a non-2xx response,
// so callers can forward its actual status instead of collapsing everything to 500.
@@ -60,6 +62,115 @@ function getStringSimilarity(s1: string, s2: string): number {
return (maxLength - distance) / maxLength;
}
// --- OCR-evidence re-ranking of the classifier's top-K candidates ---
//
// DINOv2's misses are near-twin confusions (same brand line, different
// flavor/size) - exactly the cases where the printed variant words differ,
// and PaddleOCR usually reads some of them. Within a narrow similarity band
// of the top-1 candidate, prefer the one whose distinctive name tokens
// actually appear in the OCR'd text. Coverage-normalized so generic
// packaging words (e.g. "French Fries", "Ayam") that happen to be unique to
// one candidate's *name* can't hijack the ranking. Parameters tuned offline
// against the 79-image validation set (scripts/experiment-rerank.mjs,
// 2026-07-14: fixes 8 of 18 top-1 misses, breaks 0 of 61 correct).
const RERANK_TOP_K = 12;
const RERANK_SIM_BAND = 0.12;
const RERANK_COVERAGE_MARGIN = 0.25;
function classNameSku(className: string): string {
// foto-kemasan-v2 class names are "<SKU> <NAME...>"
return (className || "").trim().split(/\s+/)[0] || "";
}
function tokenizeName(name: string): string[] {
return name.toUpperCase().split(/[^A-Z0-9]+/).filter(t => t.length >= 2);
}
function withinEditDistance1(a: string, b: string): boolean {
if (a === b) return true;
const la = a.length, lb = b.length;
if (Math.abs(la - lb) > 1) return false;
if (la === lb) {
let diff = 0;
for (let i = 0; i < la; i++) if (a[i] !== b[i]) diff++;
return diff <= 1;
}
const [s, l] = la < lb ? [a, b] : [b, a];
let i = 0, j = 0, skipped = false;
while (i < s.length && j < l.length) {
if (s[i] === l[j]) { i++; j++; }
else if (!skipped) { skipped = true; j++; }
else return false;
}
return true;
}
interface OcrTextIndex { squashed: string; tokens: Set<string>; }
function buildOcrTextIndex(textLines: string[]): OcrTextIndex {
const joined = textLines.join(" ").toUpperCase();
return {
squashed: joined.replace(/[^A-Z0-9]/g, ""),
tokens: new Set(tokenizeName(joined))
};
}
function tokenFoundInOcr(token: string, ocr: OcrTextIndex): boolean {
if (token.length >= 4 && ocr.squashed.includes(token)) return true;
if (ocr.tokens.has(token)) return true;
if (token.length >= 5) {
for (const t of ocr.tokens) {
if (Math.abs(t.length - token.length) <= 1 && withinEditDistance1(token, t)) return true;
}
}
return false;
}
// Returns the class name of the best candidate after OCR-evidence
// re-ranking (the classifier's top-1 unless a close band-mate has clearly
// stronger printed-text evidence).
function rerankClassCandidates(
allProbabilities: Array<{ name: string; confidence: number }>,
textLines: string[]
): string {
if (!allProbabilities.length) return "";
const top1Sim = allProbabilities[0].confidence;
const band = allProbabilities
.slice(0, RERANK_TOP_K)
.filter(p => p.confidence >= top1Sim - RERANK_SIM_BAND);
if (band.length <= 1 || !textLines.length) return allProbabilities[0].name;
const ocrIdx = buildOcrTextIndex(textLines);
const cands = band.map(p => {
const sku = classNameSku(p.name);
return { name: p.name, tokens: new Set(tokenizeName(p.name.replace(sku, ""))), coverage: 0 };
});
const tokenCounts = new Map<string, number>();
for (const c of cands) {
for (const tok of c.tokens) tokenCounts.set(tok, (tokenCounts.get(tok) || 0) + 1);
}
for (const c of cands) {
let matched = 0, total = 0;
for (const tok of c.tokens) {
const nWith = tokenCounts.get(tok) || 1;
if (nWith >= cands.length) continue; // shared by all band-mates -> no signal
const w = 1 / nWith;
total += w;
if (tokenFoundInOcr(tok, ocrIdx)) matched += w;
}
c.coverage = total > 0 ? matched / total : 0;
}
let chosen = cands[0];
for (const c of cands.slice(1)) {
if (c.coverage >= chosen.coverage + RERANK_COVERAGE_MARGIN) chosen = c;
}
if (chosen !== cands[0]) {
console.log(`[Rerank] OCR evidence overrode classifier top-1 "${cands[0].name}" -> "${chosen.name}" (coverage ${cands[0].coverage.toFixed(2)} vs ${chosen.coverage.toFixed(2)})`);
}
return chosen.name;
}
// Shared by the classic /api/scan-pfm dev route and the authenticated
// /api/v1/scan-product route: calls the Python classifier, then matches the
// result against sku_master, returning the top-5 candidates.
@@ -86,17 +197,28 @@ export async function classifyAndMatchProduct(imageBase64: string): Promise<Prod
nama_item: row.nama_item
}));
const top1Name = data.classification?.top1_name || "";
const extractedSku = data.ocr?.extracted_sku || "";
// Re-rank the classifier's close candidates using OCR'd package text, then
// map the winner straight to its sku_master row by the SKU prefix embedded
// in the class name. The old approach (Levenshtein between top-1 class name
// and every master nama_item) lost classifier-correct results whenever a
// *different* SKU's master name happened to be textually closer.
const rerankedName = rerankClassCandidates(
data.classification?.all_probabilities || [],
data.ocr?.text_lines || []
) || data.classification?.top1_name || "";
const rerankedSku = classNameSku(rerankedName);
const matchedList: SkuMatch[] = skuMasterList.map(sku => {
const yoloSim = top1Name ? getStringSimilarity(sku.nama_item, top1Name) : 0;
const yoloSim = rerankedName ? getStringSimilarity(sku.nama_item, rerankedName) : 0;
const cleanMasterSku = sku.no_sku.trim();
const cleanExtractedSku = extractedSku.trim();
const isSkuMatch = cleanExtractedSku && cleanMasterSku === cleanExtractedSku;
const isClassifierPick = rerankedSku && cleanMasterSku === rerankedSku;
const score = isSkuMatch ? 1.0 : yoloSim;
const score = isSkuMatch ? 1.0 : isClassifierPick ? 0.995 : yoloSim;
return {
no_sku: sku.no_sku,