feat(backend): scan-product accuracy 66.2% -> 79.7% + frozen validation benchmark
Accuracy work on the 79-image product-scan validation set (user goal: 90%): - classify_ocr_server.py: 0/90/180/270-degree expiry-date search (stops at first hit, 0-degree fallback); classification decoupled onto the upright image (rotated frames regressed DINOv2 -6pts until this); cross-line date stitching; tiled full-res OCR pass (defeats the 4000px downscale that killed small inkjet dates); VL-pipeline expiry fallback with keyword-anchored anti-hallucination guard; VL text lines merged into text_lines + VL SKU retry. Visualization endpoints removed entirely (Visual/Spotting grids - unused by frontend, 3x per-scan GPU cost). - product-scan.ts: coverage-normalized OCR-evidence re-ranking of DINOv2 top-K (tuned offline: +8/-0 on top-1 misses), re-ranked class mapped to sku_master by SKU prefix; classifier timeout 90s->240s for fallback paths. - Frozen benchmark: product-test-images-fixed/ (79 renamed images) + freeze/seed/build-undetected/capture/experiment scripts; labels trimmed to the 79 validation entries (training rows kept in .bak-with-training); 5 TRAINED-ON SKUs replaced with fresh held-out photos. - manual-label-scan page: shows last batch-test AI prediction under every field by default (new /api/product-scan-results); serves the fixed folder; fixed total hydration failure via allowedDevOrigins 127.0.0.1. - Measured (all-79, zero failures): sku/name 87.3%, expiry 64.6%, overall 79.7%. Tiles/VL-evidence/VL-SKU deployed but not yet batch-measured. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
1 parent
19f1facf9b
commit
e76ccb60a6
156 files changed
+17150
-1405
No files matched your search
@@ -4,8 +4,9 @@
|
||||
// backend/sources/product_manual_labels.json, checks 3 fields (no_sku,
|
||||
// nama_item, expiry_date) against ground truth, splits results into a
|
||||
// Training Set (gallery photos under foto-kemasan-v2/ that trained the
|
||||
// classifier itself) vs a Validation Set (flat filenames dropped in
|
||||
// backend/sources/product-test-images/), and appends a summary to
|
||||
// classifier itself) vs a Validation Set (flat filenames, scored from the
|
||||
// frozen backend/sources/product-test-images-fixed/ snapshot so reruns always
|
||||
// grade the exact same images), and appends a summary to
|
||||
// backend/sources/product_accuracy_history.jsonl. Every run auto-diffs
|
||||
// against the last history entry and flags field/image regressions or
|
||||
// improvements, so a tuning change to classify_ocr_server.py shows its
|
||||
@@ -30,7 +31,10 @@ const SOURCES_DIR = path.join(__dirname, "..", "sources");
|
||||
const LABELS_PATH = process.env.ACCURACY_LABELS_PATH || path.join(SOURCES_DIR, "product_manual_labels.json");
|
||||
const HISTORY_PATH = process.env.ACCURACY_HISTORY_PATH || path.join(SOURCES_DIR, "product_accuracy_history.jsonl");
|
||||
|
||||
const FETCH_TIMEOUT_MS = 120_000;
|
||||
// Hard images legitimately take up to ~3 min now (4-orientation OCR search +
|
||||
// VL pipeline fallback for missing expiry dates); must exceed the gateway's
|
||||
// own PIPELINE_TIMEOUT_MS (240s) so slow scans fail there, not here.
|
||||
const FETCH_TIMEOUT_MS = 300_000;
|
||||
const FIELDS = ["no_sku", "nama_item", "expiry_date"] as const;
|
||||
type Field = typeof FIELDS[number];
|
||||
type Split = "training" | "validation";
|
||||
@@ -61,6 +65,7 @@ interface ScanResponse {
|
||||
interface Check {
|
||||
field: Field;
|
||||
match: boolean;
|
||||
predicted: string;
|
||||
}
|
||||
|
||||
interface ResultItem {
|
||||
@@ -70,6 +75,12 @@ interface ResultItem {
|
||||
confidence?: number;
|
||||
}
|
||||
|
||||
// Optional: set ACCURACY_DETAIL_DUMP_PATH to write full per-image,
|
||||
// per-field ground-truth-vs-predicted detail (plus failures) as JSON —
|
||||
// used to triage which images to pull into an "undetected" folder for
|
||||
// visual inspection instead of just the aggregate percentages.
|
||||
const DETAIL_DUMP_PATH = process.env.ACCURACY_DETAIL_DUMP_PATH;
|
||||
|
||||
interface ClassificationStats {
|
||||
methodCounts: Record<string, number>;
|
||||
avgConfidence: number;
|
||||
@@ -104,7 +115,12 @@ function getImagePath(filename: string): string {
|
||||
if (filename.includes("/")) {
|
||||
return path.join(APP_ROOT, "public", "produk-pfm", "foto-kemasan-v2", filename);
|
||||
}
|
||||
return path.join(SOURCES_DIR, "product-test-images", filename);
|
||||
// Validation Set images are scored from the frozen, sequentially-renamed
|
||||
// copy in product-test-images-fixed/ (built by scripts/freeze-validation-set.mjs)
|
||||
// rather than the live-intake product-test-images/ folder, so a rerun always
|
||||
// scores the exact same image set regardless of what's since been dropped
|
||||
// into the live folder for future curation.
|
||||
return path.join(SOURCES_DIR, "product-test-images-fixed", filename);
|
||||
}
|
||||
|
||||
async function checkServerReachable(baseUrl: string) {
|
||||
@@ -338,6 +354,7 @@ async function main() {
|
||||
validation: [] as ResultItem[],
|
||||
failed: [] as string[]
|
||||
};
|
||||
const failedDetail: Array<{ filename: string; error: string }> = [];
|
||||
const perImagePct: Record<string, number> = {};
|
||||
|
||||
for (const gt of labels) {
|
||||
@@ -353,14 +370,21 @@ async function main() {
|
||||
const b64 = "data:image/jpeg;base64," + fs.readFileSync(imgPath, "base64");
|
||||
const parsed = await fetchScan(args.baseUrl, b64);
|
||||
|
||||
const bestMatchSku = parsed.possibleMatches?.find(m => m.isBestMatch)?.no_sku || "";
|
||||
const predictedItemName = parsed.classification?.top1_name || "";
|
||||
const bestMatch = parsed.possibleMatches?.find(m => m.isBestMatch);
|
||||
const bestMatchSku = bestMatch?.no_sku || "";
|
||||
// Compare against the sku_master-resolved name (what the app actually
|
||||
// shows/saves as nama_item), not classification.top1_name — that's the
|
||||
// classifier's raw internal class label (e.g. "11110059 CEKER BERKUKU
|
||||
// FROZEN PACK 1 KG", literally the foto-kemasan-v2 folder name), which
|
||||
// structurally never matches a sku_master-style ground truth string
|
||||
// even when the classification itself is correct.
|
||||
const predictedItemName = bestMatch?.nama_item || "";
|
||||
const predictedExpiry = parsed.ocr?.extracted_expired_date || "";
|
||||
|
||||
const checks: Check[] = [
|
||||
{ field: "no_sku", match: isMatch(gt.no_sku, bestMatchSku) },
|
||||
{ field: "nama_item", match: isMatch(gt.nama_item, predictedItemName) },
|
||||
{ field: "expiry_date", match: isMatch(gt.expiry_date, predictedExpiry) }
|
||||
{ field: "no_sku", match: isMatch(gt.no_sku, bestMatchSku), predicted: bestMatchSku },
|
||||
{ field: "nama_item", match: isMatch(gt.nama_item, predictedItemName), predicted: predictedItemName },
|
||||
{ field: "expiry_date", match: isMatch(gt.expiry_date, predictedExpiry), predicted: predictedExpiry }
|
||||
];
|
||||
|
||||
const item: ResultItem = {
|
||||
@@ -380,8 +404,10 @@ async function main() {
|
||||
perImagePct[gt.filename] = (score / checks.length) * 100;
|
||||
console.log(`done (${score}/3)`);
|
||||
} catch (err) {
|
||||
console.log(`FAILED (${(err as Error).message})`);
|
||||
const message = (err as Error).message;
|
||||
console.log(`FAILED (${message})`);
|
||||
results.failed.push(gt.filename);
|
||||
failedDetail.push({ filename: gt.filename, error: message });
|
||||
}
|
||||
}
|
||||
|
||||
@@ -403,6 +429,26 @@ async function main() {
|
||||
perImage: perImagePct
|
||||
};
|
||||
fs.appendFileSync(HISTORY_PATH, JSON.stringify(entry) + "\n");
|
||||
|
||||
if (DETAIL_DUMP_PATH) {
|
||||
const detail = {
|
||||
timestamp: entry.timestamp,
|
||||
validation: results.validation.map(r => ({
|
||||
filename: r.gt.filename,
|
||||
method: r.method,
|
||||
confidence: r.confidence,
|
||||
checks: r.checks.map(c => ({
|
||||
field: c.field,
|
||||
match: c.match,
|
||||
expected: r.gt[c.field],
|
||||
predicted: c.predicted
|
||||
}))
|
||||
})),
|
||||
failed: failedDetail
|
||||
};
|
||||
fs.writeFileSync(DETAIL_DUMP_PATH, JSON.stringify(detail, null, 2), "utf8");
|
||||
console.log(`\nDetail dump written to ${DETAIL_DUMP_PATH}`);
|
||||
}
|
||||
}
|
||||
|
||||
main().catch(err => {
|
||||
|
||||
Reference in new issue
Block a user