feat(backend): scan-product accuracy 66.2% -> 79.7% + frozen validation benchmark

Accuracy work on the 79-image product-scan validation set (user goal: 90%):
- classify_ocr_server.py: 0/90/180/270-degree expiry-date search (stops at
  first hit, 0-degree fallback); classification decoupled onto the upright
  image (rotated frames regressed DINOv2 -6pts until this); cross-line date
  stitching; tiled full-res OCR pass (defeats the 4000px downscale that
  killed small inkjet dates); VL-pipeline expiry fallback with
  keyword-anchored anti-hallucination guard; VL text lines merged into
  text_lines + VL SKU retry. Visualization endpoints removed entirely
  (Visual/Spotting grids - unused by frontend, 3x per-scan GPU cost).
- product-scan.ts: coverage-normalized OCR-evidence re-ranking of DINOv2
  top-K (tuned offline: +8/-0 on top-1 misses), re-ranked class mapped to
  sku_master by SKU prefix; classifier timeout 90s->240s for fallback paths.
- Frozen benchmark: product-test-images-fixed/ (79 renamed images) +
  freeze/seed/build-undetected/capture/experiment scripts; labels trimmed to
  the 79 validation entries (training rows kept in .bak-with-training);
  5 TRAINED-ON SKUs replaced with fresh held-out photos.
- manual-label-scan page: shows last batch-test AI prediction under every
  field by default (new /api/product-scan-results); serves the fixed folder;
  fixed total hydration failure via allowedDevOrigins 127.0.0.1.
- Measured (all-79, zero failures): sku/name 87.3%, expiry 64.6%, overall
  79.7%. Tiles/VL-evidence/VL-SKU deployed but not yet batch-measured.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
Rafhan Mazaya FathurrahmanandClaude Fable 5 committed 2026-07-14 19:55:17 +07:00
1 parent 19f1facf9b
commit e76ccb60a6
156 files changed
+17150 -1405

No files matched your search

+56 -10
View File
@@ -4,8 +4,9 @@
// backend/sources/product_manual_labels.json, checks 3 fields (no_sku,
// nama_item, expiry_date) against ground truth, splits results into a
// Training Set (gallery photos under foto-kemasan-v2/ that trained the
// classifier itself) vs a Validation Set (flat filenames dropped in
// backend/sources/product-test-images/), and appends a summary to
// classifier itself) vs a Validation Set (flat filenames, scored from the
// frozen backend/sources/product-test-images-fixed/ snapshot so reruns always
// grade the exact same images), and appends a summary to
// backend/sources/product_accuracy_history.jsonl. Every run auto-diffs
// against the last history entry and flags field/image regressions or
// improvements, so a tuning change to classify_ocr_server.py shows its
@@ -30,7 +31,10 @@ const SOURCES_DIR = path.join(__dirname, "..", "sources");
const LABELS_PATH = process.env.ACCURACY_LABELS_PATH || path.join(SOURCES_DIR, "product_manual_labels.json");
const HISTORY_PATH = process.env.ACCURACY_HISTORY_PATH || path.join(SOURCES_DIR, "product_accuracy_history.jsonl");
const FETCH_TIMEOUT_MS = 120_000;
// Hard images legitimately take up to ~3 min now (4-orientation OCR search +
// VL pipeline fallback for missing expiry dates); must exceed the gateway's
// own PIPELINE_TIMEOUT_MS (240s) so slow scans fail there, not here.
const FETCH_TIMEOUT_MS = 300_000;
const FIELDS = ["no_sku", "nama_item", "expiry_date"] as const;
type Field = typeof FIELDS[number];
type Split = "training" | "validation";
@@ -61,6 +65,7 @@ interface ScanResponse {
interface Check {
field: Field;
match: boolean;
predicted: string;
}
interface ResultItem {
@@ -70,6 +75,12 @@ interface ResultItem {
confidence?: number;
}
// Optional: set ACCURACY_DETAIL_DUMP_PATH to write full per-image,
// per-field ground-truth-vs-predicted detail (plus failures) as JSON —
// used to triage which images to pull into an "undetected" folder for
// visual inspection instead of just the aggregate percentages.
const DETAIL_DUMP_PATH = process.env.ACCURACY_DETAIL_DUMP_PATH;
interface ClassificationStats {
methodCounts: Record<string, number>;
avgConfidence: number;
@@ -104,7 +115,12 @@ function getImagePath(filename: string): string {
if (filename.includes("/")) {
return path.join(APP_ROOT, "public", "produk-pfm", "foto-kemasan-v2", filename);
}
return path.join(SOURCES_DIR, "product-test-images", filename);
// Validation Set images are scored from the frozen, sequentially-renamed
// copy in product-test-images-fixed/ (built by scripts/freeze-validation-set.mjs)
// rather than the live-intake product-test-images/ folder, so a rerun always
// scores the exact same image set regardless of what's since been dropped
// into the live folder for future curation.
return path.join(SOURCES_DIR, "product-test-images-fixed", filename);
}
async function checkServerReachable(baseUrl: string) {
@@ -338,6 +354,7 @@ async function main() {
validation: [] as ResultItem[],
failed: [] as string[]
};
const failedDetail: Array<{ filename: string; error: string }> = [];
const perImagePct: Record<string, number> = {};
for (const gt of labels) {
@@ -353,14 +370,21 @@ async function main() {
const b64 = "data:image/jpeg;base64," + fs.readFileSync(imgPath, "base64");
const parsed = await fetchScan(args.baseUrl, b64);
const bestMatchSku = parsed.possibleMatches?.find(m => m.isBestMatch)?.no_sku || "";
const predictedItemName = parsed.classification?.top1_name || "";
const bestMatch = parsed.possibleMatches?.find(m => m.isBestMatch);
const bestMatchSku = bestMatch?.no_sku || "";
// Compare against the sku_master-resolved name (what the app actually
// shows/saves as nama_item), not classification.top1_name — that's the
// classifier's raw internal class label (e.g. "11110059 CEKER BERKUKU
// FROZEN PACK 1 KG", literally the foto-kemasan-v2 folder name), which
// structurally never matches a sku_master-style ground truth string
// even when the classification itself is correct.
const predictedItemName = bestMatch?.nama_item || "";
const predictedExpiry = parsed.ocr?.extracted_expired_date || "";
const checks: Check[] = [
{ field: "no_sku", match: isMatch(gt.no_sku, bestMatchSku) },
{ field: "nama_item", match: isMatch(gt.nama_item, predictedItemName) },
{ field: "expiry_date", match: isMatch(gt.expiry_date, predictedExpiry) }
{ field: "no_sku", match: isMatch(gt.no_sku, bestMatchSku), predicted: bestMatchSku },
{ field: "nama_item", match: isMatch(gt.nama_item, predictedItemName), predicted: predictedItemName },
{ field: "expiry_date", match: isMatch(gt.expiry_date, predictedExpiry), predicted: predictedExpiry }
];
const item: ResultItem = {
@@ -380,8 +404,10 @@ async function main() {
perImagePct[gt.filename] = (score / checks.length) * 100;
console.log(`done (${score}/3)`);
} catch (err) {
console.log(`FAILED (${(err as Error).message})`);
const message = (err as Error).message;
console.log(`FAILED (${message})`);
results.failed.push(gt.filename);
failedDetail.push({ filename: gt.filename, error: message });
}
}
@@ -403,6 +429,26 @@ async function main() {
perImage: perImagePct
};
fs.appendFileSync(HISTORY_PATH, JSON.stringify(entry) + "\n");
if (DETAIL_DUMP_PATH) {
const detail = {
timestamp: entry.timestamp,
validation: results.validation.map(r => ({
filename: r.gt.filename,
method: r.method,
confidence: r.confidence,
checks: r.checks.map(c => ({
field: c.field,
match: c.match,
expected: r.gt[c.field],
predicted: c.predicted
}))
})),
failed: failedDetail
};
fs.writeFileSync(DETAIL_DUMP_PATH, JSON.stringify(detail, null, 2), "utf8");
console.log(`\nDetail dump written to ${DETAIL_DUMP_PATH}`);
}
}
main().catch(err => {