// Offline experiment: OCR-evidence re-ranking of DINOv2 top-K candidates. // // Reads sources/product_scan_fullcap.json (captured live responses, see // capture-scan-responses.mjs) + sources/product_manual_labels.json (ground // truth) and simulates candidate re-ranking without touching the GPU stack, // reporting fixed-vs-broken counts per parameter combination. The winning // parameters get ported into pfm-web-app/src/utils/product-scan.ts. // // Idea: DINOv2's near-twin confusions (same brand, different flavor/size) // are exactly the cases where the *printed variant words* differ - and // PaddleOCR usually reads some of them. So within a narrow similarity band // of the top-1, prefer the candidate whose distinguishing name tokens // actually appear in the OCR'd text. // // Usage: node scripts/experiment-rerank.mjs import fs from "fs"; import path from "path"; const cap = JSON.parse(fs.readFileSync(path.join("sources", "product_scan_fullcap.json"), "utf8")); const labels = JSON.parse(fs.readFileSync(path.join("sources", "product_manual_labels.json"), "utf8")); const gtBySku = new Map(labels.map((l) => [l.filename, l.no_sku])); function classSku(className) { // Class names are foto-kemasan-v2 folder names: " " return (className || "").trim().split(/\s+/)[0] || ""; } function tokenize(name) { return name .toUpperCase() .split(/[^A-Z0-9]+/) .filter((t) => t.length >= 2); } function editDistance1(a, b) { // true if edit distance <= 1 (same length: 1 substitution; off-by-one: 1 indel) if (a === b) return true; const la = a.length, lb = b.length; if (Math.abs(la - lb) > 1) return false; if (la === lb) { let diff = 0; for (let i = 0; i < la; i++) if (a[i] !== b[i]) diff++; return diff <= 1; } const [s, l] = la < lb ? [a, b] : [b, a]; let i = 0, j = 0, skipped = false; while (i < s.length && j < l.length) { if (s[i] === l[j]) { i++; j++; } else if (!skipped) { skipped = true; j++; } else return false; } return true; } function buildOcrIndex(textLines) { const joined = textLines.join(" ").toUpperCase(); const squashed = joined.replace(/[^A-Z0-9]/g, ""); const tokens = new Set(tokenize(joined)); return { squashed, tokens }; } function tokenInOcr(token, ocrIdx, fuzzy) { if (token.length >= 4 && ocrIdx.squashed.includes(token)) return true; if (ocrIdx.tokens.has(token)) return true; if (fuzzy && token.length >= 5) { for (const t of ocrIdx.tokens) { if (Math.abs(t.length - token.length) <= 1 && editDistance1(token, t)) return true; } } return false; } function ocrEvidenceScore(candTokens, bandTokenCounts, bandSize, ocrIdx, fuzzy) { // Coverage-normalized, rarity-weighted evidence: fraction of this // candidate's *distinctive* name tokens (weighted by band rarity) that // actually appear in the OCR'd text. Normalizing by the candidate's own // distinctive-token mass is what stops generic packaging words from // hijacking the ranking - a candidate whose name promises FRENCH + // INSTITUSI + 2KG but whose package shows only "French Fries" scores // 1/3, losing to a candidate whose 2 distinctive tokens both appear. let matched = 0; let total = 0; for (const tok of new Set(candTokens)) { const nWith = bandTokenCounts.get(tok) || 1; if (nWith >= bandSize) continue; // shared by all -> no signal const w = 1 / nWith; total += w; if (tokenInOcr(tok, ocrIdx, fuzzy)) matched += w; } return total > 0 ? matched / total : 0; } function skuFuzzyBoost(extractedSku, candidateSku) { if (!extractedSku || extractedSku.length < 7) return 0; if (extractedSku === candidateSku) return 10; // exact (normally pinned upstream anyway) return editDistance1(extractedSku, candidateSku) ? 1 : 0; } function runConfig({ K, BAND, MARGIN, FUZZY, SKU_BOOST_W }) { let baselineCorrect = 0, rerankCorrect = 0, fixed = [], broken = []; for (const item of cap) { if (item.error) continue; const gt = gtBySku.get(item.filename); if (!gt) continue; const probs = item.classification?.all_probabilities || []; if (!probs.length) continue; const top1Sku = classSku(probs[0].name); const baselineRight = top1Sku === gt; if (baselineRight) baselineCorrect++; // Candidate band: within BAND of top-1 similarity, capped at K const top1Sim = probs[0].confidence; const band = probs.slice(0, K).filter((p) => p.confidence >= top1Sim - BAND); const ocrIdx = buildOcrIndex(item.ocr?.text_lines || []); const candInfos = band.map((p) => { const sku = classSku(p.name); const tokens = tokenize(p.name.replace(sku, "")); return { sku, sim: p.confidence, tokens }; }); const bandTokenCounts = new Map(); for (const c of candInfos) { for (const tok of new Set(c.tokens)) { bandTokenCounts.set(tok, (bandTokenCounts.get(tok) || 0) + 1); } } for (const c of candInfos) { c.ocrScore = ocrEvidenceScore(c.tokens, bandTokenCounts, candInfos.length, ocrIdx, FUZZY) + SKU_BOOST_W * skuFuzzyBoost(item.ocr?.extracted_sku || "", c.sku); } // Switch away from top-1 only when a band-mate has clearly stronger OCR evidence let chosen = candInfos[0]; for (const c of candInfos.slice(1)) { if (c.ocrScore >= chosen.ocrScore + MARGIN) chosen = c; } const rerankRight = chosen.sku === gt; if (rerankRight) rerankCorrect++; if (!baselineRight && rerankRight) fixed.push(item.filename); if (baselineRight && !rerankRight) broken.push(item.filename); } return { baselineCorrect, rerankCorrect, fixed, broken }; } const grid = []; for (const K of [5, 8, 12]) { for (const BAND of [0.04, 0.06, 0.08, 0.12]) { // Coverage scores live in [0, 1]; margin is the minimum coverage lead a // band-mate needs over the current pick before we switch away from it. for (const MARGIN of [0.15, 0.25, 0.35, 0.5]) { for (const FUZZY of [true, false]) { for (const SKU_BOOST_W of [0, 2]) { grid.push({ K, BAND, MARGIN, FUZZY, SKU_BOOST_W }); } } } } } const results = grid.map((cfg) => ({ cfg, ...runConfig(cfg) })); results.sort((a, b) => (b.rerankCorrect - b.broken.length * 0.01) - (a.rerankCorrect - a.broken.length * 0.01)); console.log(`Images evaluated: ${cap.filter((i) => !i.error && gtBySku.has(i.filename)).length}`); console.log(`Baseline (DINOv2 top-1) correct: ${results[0].baselineCorrect}\n`); console.log("Top 12 configs by re-ranked correct count:"); for (const r of results.slice(0, 12)) { console.log( ` correct=${r.rerankCorrect} (+${r.fixed.length}/-${r.broken.length}) ` + `K=${r.cfg.K} BAND=${r.cfg.BAND} MARGIN=${r.cfg.MARGIN} FUZZY=${r.cfg.FUZZY} SKUW=${r.cfg.SKU_BOOST_W}` ); } const best = results[0]; console.log(`\nBest config detail: ${JSON.stringify(best.cfg)}`); console.log(` fixed (${best.fixed.length}): ${best.fixed.join(", ")}`); console.log(` broken (${best.broken.length}): ${best.broken.join(", ")}`);