Accuracy work on the 79-image product-scan validation set (user goal: 90%): - classify_ocr_server.py: 0/90/180/270-degree expiry-date search (stops at first hit, 0-degree fallback); classification decoupled onto the upright image (rotated frames regressed DINOv2 -6pts until this); cross-line date stitching; tiled full-res OCR pass (defeats the 4000px downscale that killed small inkjet dates); VL-pipeline expiry fallback with keyword-anchored anti-hallucination guard; VL text lines merged into text_lines + VL SKU retry. Visualization endpoints removed entirely (Visual/Spotting grids - unused by frontend, 3x per-scan GPU cost). - product-scan.ts: coverage-normalized OCR-evidence re-ranking of DINOv2 top-K (tuned offline: +8/-0 on top-1 misses), re-ranked class mapped to sku_master by SKU prefix; classifier timeout 90s->240s for fallback paths. - Frozen benchmark: product-test-images-fixed/ (79 renamed images) + freeze/seed/build-undetected/capture/experiment scripts; labels trimmed to the 79 validation entries (training rows kept in .bak-with-training); 5 TRAINED-ON SKUs replaced with fresh held-out photos. - manual-label-scan page: shows last batch-test AI prediction under every field by default (new /api/product-scan-results); serves the fixed folder; fixed total hydration failure via allowedDevOrigins 127.0.0.1. - Measured (all-79, zero failures): sku/name 87.3%, expiry 64.6%, overall 79.7%. Tiles/VL-evidence/VL-SKU deployed but not yet batch-measured. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
179 lines
6.9 KiB
JavaScript
179 lines
6.9 KiB
JavaScript
// Offline experiment: OCR-evidence re-ranking of DINOv2 top-K candidates.
|
|
//
|
|
// Reads sources/product_scan_fullcap.json (captured live responses, see
|
|
// capture-scan-responses.mjs) + sources/product_manual_labels.json (ground
|
|
// truth) and simulates candidate re-ranking without touching the GPU stack,
|
|
// reporting fixed-vs-broken counts per parameter combination. The winning
|
|
// parameters get ported into pfm-web-app/src/utils/product-scan.ts.
|
|
//
|
|
// Idea: DINOv2's near-twin confusions (same brand, different flavor/size)
|
|
// are exactly the cases where the *printed variant words* differ - and
|
|
// PaddleOCR usually reads some of them. So within a narrow similarity band
|
|
// of the top-1, prefer the candidate whose distinguishing name tokens
|
|
// actually appear in the OCR'd text.
|
|
//
|
|
// Usage: node scripts/experiment-rerank.mjs
|
|
import fs from "fs";
|
|
import path from "path";
|
|
|
|
const cap = JSON.parse(fs.readFileSync(path.join("sources", "product_scan_fullcap.json"), "utf8"));
|
|
const labels = JSON.parse(fs.readFileSync(path.join("sources", "product_manual_labels.json"), "utf8"));
|
|
const gtBySku = new Map(labels.map((l) => [l.filename, l.no_sku]));
|
|
|
|
function classSku(className) {
|
|
// Class names are foto-kemasan-v2 folder names: "<SKU> <NAME...>"
|
|
return (className || "").trim().split(/\s+/)[0] || "";
|
|
}
|
|
|
|
function tokenize(name) {
|
|
return name
|
|
.toUpperCase()
|
|
.split(/[^A-Z0-9]+/)
|
|
.filter((t) => t.length >= 2);
|
|
}
|
|
|
|
function editDistance1(a, b) {
|
|
// true if edit distance <= 1 (same length: 1 substitution; off-by-one: 1 indel)
|
|
if (a === b) return true;
|
|
const la = a.length, lb = b.length;
|
|
if (Math.abs(la - lb) > 1) return false;
|
|
if (la === lb) {
|
|
let diff = 0;
|
|
for (let i = 0; i < la; i++) if (a[i] !== b[i]) diff++;
|
|
return diff <= 1;
|
|
}
|
|
const [s, l] = la < lb ? [a, b] : [b, a];
|
|
let i = 0, j = 0, skipped = false;
|
|
while (i < s.length && j < l.length) {
|
|
if (s[i] === l[j]) { i++; j++; }
|
|
else if (!skipped) { skipped = true; j++; }
|
|
else return false;
|
|
}
|
|
return true;
|
|
}
|
|
|
|
function buildOcrIndex(textLines) {
|
|
const joined = textLines.join(" ").toUpperCase();
|
|
const squashed = joined.replace(/[^A-Z0-9]/g, "");
|
|
const tokens = new Set(tokenize(joined));
|
|
return { squashed, tokens };
|
|
}
|
|
|
|
function tokenInOcr(token, ocrIdx, fuzzy) {
|
|
if (token.length >= 4 && ocrIdx.squashed.includes(token)) return true;
|
|
if (ocrIdx.tokens.has(token)) return true;
|
|
if (fuzzy && token.length >= 5) {
|
|
for (const t of ocrIdx.tokens) {
|
|
if (Math.abs(t.length - token.length) <= 1 && editDistance1(token, t)) return true;
|
|
}
|
|
}
|
|
return false;
|
|
}
|
|
|
|
function ocrEvidenceScore(candTokens, bandTokenCounts, bandSize, ocrIdx, fuzzy) {
|
|
// Coverage-normalized, rarity-weighted evidence: fraction of this
|
|
// candidate's *distinctive* name tokens (weighted by band rarity) that
|
|
// actually appear in the OCR'd text. Normalizing by the candidate's own
|
|
// distinctive-token mass is what stops generic packaging words from
|
|
// hijacking the ranking - a candidate whose name promises FRENCH +
|
|
// INSTITUSI + 2KG but whose package shows only "French Fries" scores
|
|
// 1/3, losing to a candidate whose 2 distinctive tokens both appear.
|
|
let matched = 0;
|
|
let total = 0;
|
|
for (const tok of new Set(candTokens)) {
|
|
const nWith = bandTokenCounts.get(tok) || 1;
|
|
if (nWith >= bandSize) continue; // shared by all -> no signal
|
|
const w = 1 / nWith;
|
|
total += w;
|
|
if (tokenInOcr(tok, ocrIdx, fuzzy)) matched += w;
|
|
}
|
|
return total > 0 ? matched / total : 0;
|
|
}
|
|
|
|
function skuFuzzyBoost(extractedSku, candidateSku) {
|
|
if (!extractedSku || extractedSku.length < 7) return 0;
|
|
if (extractedSku === candidateSku) return 10; // exact (normally pinned upstream anyway)
|
|
return editDistance1(extractedSku, candidateSku) ? 1 : 0;
|
|
}
|
|
|
|
function runConfig({ K, BAND, MARGIN, FUZZY, SKU_BOOST_W }) {
|
|
let baselineCorrect = 0, rerankCorrect = 0, fixed = [], broken = [];
|
|
for (const item of cap) {
|
|
if (item.error) continue;
|
|
const gt = gtBySku.get(item.filename);
|
|
if (!gt) continue;
|
|
const probs = item.classification?.all_probabilities || [];
|
|
if (!probs.length) continue;
|
|
|
|
const top1Sku = classSku(probs[0].name);
|
|
const baselineRight = top1Sku === gt;
|
|
if (baselineRight) baselineCorrect++;
|
|
|
|
// Candidate band: within BAND of top-1 similarity, capped at K
|
|
const top1Sim = probs[0].confidence;
|
|
const band = probs.slice(0, K).filter((p) => p.confidence >= top1Sim - BAND);
|
|
|
|
const ocrIdx = buildOcrIndex(item.ocr?.text_lines || []);
|
|
const candInfos = band.map((p) => {
|
|
const sku = classSku(p.name);
|
|
const tokens = tokenize(p.name.replace(sku, ""));
|
|
return { sku, sim: p.confidence, tokens };
|
|
});
|
|
const bandTokenCounts = new Map();
|
|
for (const c of candInfos) {
|
|
for (const tok of new Set(c.tokens)) {
|
|
bandTokenCounts.set(tok, (bandTokenCounts.get(tok) || 0) + 1);
|
|
}
|
|
}
|
|
for (const c of candInfos) {
|
|
c.ocrScore = ocrEvidenceScore(c.tokens, bandTokenCounts, candInfos.length, ocrIdx, FUZZY)
|
|
+ SKU_BOOST_W * skuFuzzyBoost(item.ocr?.extracted_sku || "", c.sku);
|
|
}
|
|
|
|
// Switch away from top-1 only when a band-mate has clearly stronger OCR evidence
|
|
let chosen = candInfos[0];
|
|
for (const c of candInfos.slice(1)) {
|
|
if (c.ocrScore >= chosen.ocrScore + MARGIN) chosen = c;
|
|
}
|
|
|
|
const rerankRight = chosen.sku === gt;
|
|
if (rerankRight) rerankCorrect++;
|
|
if (!baselineRight && rerankRight) fixed.push(item.filename);
|
|
if (baselineRight && !rerankRight) broken.push(item.filename);
|
|
}
|
|
return { baselineCorrect, rerankCorrect, fixed, broken };
|
|
}
|
|
|
|
const grid = [];
|
|
for (const K of [5, 8, 12]) {
|
|
for (const BAND of [0.04, 0.06, 0.08, 0.12]) {
|
|
// Coverage scores live in [0, 1]; margin is the minimum coverage lead a
|
|
// band-mate needs over the current pick before we switch away from it.
|
|
for (const MARGIN of [0.15, 0.25, 0.35, 0.5]) {
|
|
for (const FUZZY of [true, false]) {
|
|
for (const SKU_BOOST_W of [0, 2]) {
|
|
grid.push({ K, BAND, MARGIN, FUZZY, SKU_BOOST_W });
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
const results = grid.map((cfg) => ({ cfg, ...runConfig(cfg) }));
|
|
results.sort((a, b) => (b.rerankCorrect - b.broken.length * 0.01) - (a.rerankCorrect - a.broken.length * 0.01));
|
|
|
|
console.log(`Images evaluated: ${cap.filter((i) => !i.error && gtBySku.has(i.filename)).length}`);
|
|
console.log(`Baseline (DINOv2 top-1) correct: ${results[0].baselineCorrect}\n`);
|
|
console.log("Top 12 configs by re-ranked correct count:");
|
|
for (const r of results.slice(0, 12)) {
|
|
console.log(
|
|
` correct=${r.rerankCorrect} (+${r.fixed.length}/-${r.broken.length}) ` +
|
|
`K=${r.cfg.K} BAND=${r.cfg.BAND} MARGIN=${r.cfg.MARGIN} FUZZY=${r.cfg.FUZZY} SKUW=${r.cfg.SKU_BOOST_W}`
|
|
);
|
|
}
|
|
|
|
const best = results[0];
|
|
console.log(`\nBest config detail: ${JSON.stringify(best.cfg)}`);
|
|
console.log(` fixed (${best.fixed.length}): ${best.fixed.join(", ")}`);
|
|
console.log(` broken (${best.broken.length}): ${best.broken.join(", ")}`);
|