feat(backend): scan-product accuracy 66.2% -> 79.7% + frozen validation benchmark

Accuracy work on the 79-image product-scan validation set (user goal: 90%):
- classify_ocr_server.py: 0/90/180/270-degree expiry-date search (stops at
  first hit, 0-degree fallback); classification decoupled onto the upright
  image (rotated frames regressed DINOv2 -6pts until this); cross-line date
  stitching; tiled full-res OCR pass (defeats the 4000px downscale that
  killed small inkjet dates); VL-pipeline expiry fallback with
  keyword-anchored anti-hallucination guard; VL text lines merged into
  text_lines + VL SKU retry. Visualization endpoints removed entirely
  (Visual/Spotting grids - unused by frontend, 3x per-scan GPU cost).
- product-scan.ts: coverage-normalized OCR-evidence re-ranking of DINOv2
  top-K (tuned offline: +8/-0 on top-1 misses), re-ranked class mapped to
  sku_master by SKU prefix; classifier timeout 90s->240s for fallback paths.
- Frozen benchmark: product-test-images-fixed/ (79 renamed images) +
  freeze/seed/build-undetected/capture/experiment scripts; labels trimmed to
  the 79 validation entries (training rows kept in .bak-with-training);
  5 TRAINED-ON SKUs replaced with fresh held-out photos.
- manual-label-scan page: shows last batch-test AI prediction under every
  field by default (new /api/product-scan-results); serves the fixed folder;
  fixed total hydration failure via allowedDevOrigins 127.0.0.1.
- Measured (all-79, zero failures): sku/name 87.3%, expiry 64.6%, overall
  79.7%. Tiles/VL-evidence/VL-SKU deployed but not yet batch-measured.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
Rafhan Mazaya FathurrahmanandClaude Fable 5 committed 2026-07-14 19:55:17 +07:00
1 parent 19f1facf9b
commit e76ccb60a6
156 files changed
+17150 -1405

No files matched your search

+56 -10
View File
@@ -4,8 +4,9 @@
// backend/sources/product_manual_labels.json, checks 3 fields (no_sku,
// nama_item, expiry_date) against ground truth, splits results into a
// Training Set (gallery photos under foto-kemasan-v2/ that trained the
// classifier itself) vs a Validation Set (flat filenames dropped in
// backend/sources/product-test-images/), and appends a summary to
// classifier itself) vs a Validation Set (flat filenames, scored from the
// frozen backend/sources/product-test-images-fixed/ snapshot so reruns always
// grade the exact same images), and appends a summary to
// backend/sources/product_accuracy_history.jsonl. Every run auto-diffs
// against the last history entry and flags field/image regressions or
// improvements, so a tuning change to classify_ocr_server.py shows its
@@ -30,7 +31,10 @@ const SOURCES_DIR = path.join(__dirname, "..", "sources");
const LABELS_PATH = process.env.ACCURACY_LABELS_PATH || path.join(SOURCES_DIR, "product_manual_labels.json");
const HISTORY_PATH = process.env.ACCURACY_HISTORY_PATH || path.join(SOURCES_DIR, "product_accuracy_history.jsonl");
const FETCH_TIMEOUT_MS = 120_000;
// Hard images legitimately take up to ~3 min now (4-orientation OCR search +
// VL pipeline fallback for missing expiry dates); must exceed the gateway's
// own PIPELINE_TIMEOUT_MS (240s) so slow scans fail there, not here.
const FETCH_TIMEOUT_MS = 300_000;
const FIELDS = ["no_sku", "nama_item", "expiry_date"] as const;
type Field = typeof FIELDS[number];
type Split = "training" | "validation";
@@ -61,6 +65,7 @@ interface ScanResponse {
interface Check {
field: Field;
match: boolean;
predicted: string;
}
interface ResultItem {
@@ -70,6 +75,12 @@ interface ResultItem {
confidence?: number;
}
// Optional: set ACCURACY_DETAIL_DUMP_PATH to write full per-image,
// per-field ground-truth-vs-predicted detail (plus failures) as JSON —
// used to triage which images to pull into an "undetected" folder for
// visual inspection instead of just the aggregate percentages.
const DETAIL_DUMP_PATH = process.env.ACCURACY_DETAIL_DUMP_PATH;
interface ClassificationStats {
methodCounts: Record<string, number>;
avgConfidence: number;
@@ -104,7 +115,12 @@ function getImagePath(filename: string): string {
if (filename.includes("/")) {
return path.join(APP_ROOT, "public", "produk-pfm", "foto-kemasan-v2", filename);
}
return path.join(SOURCES_DIR, "product-test-images", filename);
// Validation Set images are scored from the frozen, sequentially-renamed
// copy in product-test-images-fixed/ (built by scripts/freeze-validation-set.mjs)
// rather than the live-intake product-test-images/ folder, so a rerun always
// scores the exact same image set regardless of what's since been dropped
// into the live folder for future curation.
return path.join(SOURCES_DIR, "product-test-images-fixed", filename);
}
async function checkServerReachable(baseUrl: string) {
@@ -338,6 +354,7 @@ async function main() {
validation: [] as ResultItem[],
failed: [] as string[]
};
const failedDetail: Array<{ filename: string; error: string }> = [];
const perImagePct: Record<string, number> = {};
for (const gt of labels) {
@@ -353,14 +370,21 @@ async function main() {
const b64 = "data:image/jpeg;base64," + fs.readFileSync(imgPath, "base64");
const parsed = await fetchScan(args.baseUrl, b64);
const bestMatchSku = parsed.possibleMatches?.find(m => m.isBestMatch)?.no_sku || "";
const predictedItemName = parsed.classification?.top1_name || "";
const bestMatch = parsed.possibleMatches?.find(m => m.isBestMatch);
const bestMatchSku = bestMatch?.no_sku || "";
// Compare against the sku_master-resolved name (what the app actually
// shows/saves as nama_item), not classification.top1_name — that's the
// classifier's raw internal class label (e.g. "11110059 CEKER BERKUKU
// FROZEN PACK 1 KG", literally the foto-kemasan-v2 folder name), which
// structurally never matches a sku_master-style ground truth string
// even when the classification itself is correct.
const predictedItemName = bestMatch?.nama_item || "";
const predictedExpiry = parsed.ocr?.extracted_expired_date || "";
const checks: Check[] = [
{ field: "no_sku", match: isMatch(gt.no_sku, bestMatchSku) },
{ field: "nama_item", match: isMatch(gt.nama_item, predictedItemName) },
{ field: "expiry_date", match: isMatch(gt.expiry_date, predictedExpiry) }
{ field: "no_sku", match: isMatch(gt.no_sku, bestMatchSku), predicted: bestMatchSku },
{ field: "nama_item", match: isMatch(gt.nama_item, predictedItemName), predicted: predictedItemName },
{ field: "expiry_date", match: isMatch(gt.expiry_date, predictedExpiry), predicted: predictedExpiry }
];
const item: ResultItem = {
@@ -380,8 +404,10 @@ async function main() {
perImagePct[gt.filename] = (score / checks.length) * 100;
console.log(`done (${score}/3)`);
} catch (err) {
console.log(`FAILED (${(err as Error).message})`);
const message = (err as Error).message;
console.log(`FAILED (${message})`);
results.failed.push(gt.filename);
failedDetail.push({ filename: gt.filename, error: message });
}
}
@@ -403,6 +429,26 @@ async function main() {
perImage: perImagePct
};
fs.appendFileSync(HISTORY_PATH, JSON.stringify(entry) + "\n");
if (DETAIL_DUMP_PATH) {
const detail = {
timestamp: entry.timestamp,
validation: results.validation.map(r => ({
filename: r.gt.filename,
method: r.method,
confidence: r.confidence,
checks: r.checks.map(c => ({
field: c.field,
match: c.match,
expected: r.gt[c.field],
predicted: c.predicted
}))
})),
failed: failedDetail
};
fs.writeFileSync(DETAIL_DUMP_PATH, JSON.stringify(detail, null, 2), "utf8");
console.log(`\nDetail dump written to ${DETAIL_DUMP_PATH}`);
}
}
main().catch(err => {
+76
View File
@@ -0,0 +1,76 @@
// One-off script: pull every Validation Set image that had at least one
// mismatched field (or failed to scan entirely) out of
// sources/product-test-images-fixed/ into sources/product-test-images-undetected/,
// renamed to show which field(s) missed, plus a manifest.md with
// expected-vs-predicted per field — so a human can tell at a glance whether
// a miss is an OCR problem, a classification problem, or the photo itself
// lacking the data (e.g. expiry code out of frame).
// Usage: node scripts/build-undetected-set.mjs <path-to-detail-dump.json>
import fs from "fs";
import path from "path";
const detailPath = process.argv[2];
if (!detailPath) {
console.error("Usage: node scripts/build-undetected-set.mjs <detail-dump.json>");
process.exit(1);
}
const FIXED_DIR = path.join("sources", "product-test-images-fixed");
const OUT_DIR = path.join("sources", "product-test-images-undetected");
const detail = JSON.parse(fs.readFileSync(detailPath, "utf8"));
if (fs.existsSync(OUT_DIR)) fs.rmSync(OUT_DIR, { recursive: true });
fs.mkdirSync(OUT_DIR, { recursive: true });
const manifestRows = [];
let copied = 0;
for (const item of detail.validation) {
const failedFields = item.checks.filter((c) => !c.match);
if (!failedFields.length) continue;
const ext = path.extname(item.filename);
const base = path.basename(item.filename, ext);
const tag = failedFields.map((c) => c.field).join(",");
const destName = `${base} [FAIL ${tag}]${ext}`;
fs.copyFileSync(path.join(FIXED_DIR, item.filename), path.join(OUT_DIR, destName));
copied++;
for (const c of item.checks) {
manifestRows.push({
image: destName,
field: c.field,
match: c.match,
expected: c.expected || "(empty)",
predicted: c.predicted || "(empty — not detected)"
});
}
}
for (const f of detail.failed) {
const srcPath = path.join(FIXED_DIR, f.filename);
if (!fs.existsSync(srcPath)) continue;
const ext = path.extname(f.filename);
const base = path.basename(f.filename, ext);
const destName = `${base} [FAILED-scan]${ext}`;
fs.copyFileSync(srcPath, path.join(OUT_DIR, destName));
copied++;
manifestRows.push({ image: destName, field: "(entire scan)", match: false, expected: "-", predicted: `ERROR: ${f.error}` });
}
const lines = [
"# Undetected / mismatched images",
"",
`Generated ${detail.timestamp} from the accuracy-check-scan.mts detail dump.`,
`${copied} of ${detail.validation.length + detail.failed.length} Validation Set images had at least one wrong/missing field or failed to scan.`,
"",
"| Image | Field | OK? | Expected | Predicted |",
"|---|---|---|---|---|"
];
for (const r of manifestRows) {
lines.push(`| ${r.image} | ${r.field} | ${r.match ? "✓" : "✗"} | ${r.expected} | ${r.predicted} |`);
}
fs.writeFileSync(path.join(OUT_DIR, "manifest.md"), lines.join("\n") + "\n", "utf8");
console.log(`Copied ${copied} images into ${OUT_DIR}, wrote manifest.md (${manifestRows.length} rows).`);
@@ -0,0 +1,59 @@
// One-off capture: hit /api/scan-pfm for every image in
// sources/product-test-images-fixed/ and save the full text-level response
// (classification all_probabilities, OCR text_lines, extracted fields,
// possibleMatches) minus base64 image blobs to
// sources/product_scan_fullcap.json. This lets classification re-ranking
// experiments run offline against ground truth in seconds instead of
// re-running the 11-minute GPU batch per iteration.
// Usage: node scripts/capture-scan-responses.mjs [baseUrl]
import fs from "fs";
import path from "path";
const BASE_URL = process.argv[2] || process.env.ACCURACY_BASE_URL || "http://127.0.0.1:3000";
const FIXED_DIR = path.join("sources", "product-test-images-fixed");
const OUT_PATH = path.join("sources", "product_scan_fullcap.json");
const files = fs.readdirSync(FIXED_DIR)
.filter((f) => /\.(jpe?g|png|webp)$/i.test(f))
.sort((a, b) => parseInt(a) - parseInt(b));
const results = [];
for (const filename of files) {
const b64 = "data:image/jpeg;base64," +
fs.readFileSync(path.join(FIXED_DIR, filename), "base64");
process.stdout.write(`Capturing ${filename}... `);
try {
const res = await fetch(`${BASE_URL}/api/scan-pfm`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ image_base64: b64 }),
signal: AbortSignal.timeout(120_000)
});
if (!res.ok) throw new Error(`HTTP ${res.status}`);
const d = await res.json();
results.push({
filename,
classification: {
top1_name: d.classification?.top1_name,
top1_confidence: d.classification?.top1_confidence,
method: d.classification?.method,
all_probabilities: (d.classification?.all_probabilities || []).slice(0, 15)
},
ocr: {
text_lines: d.ocr?.text_lines || [],
extracted_sku: d.ocr?.extracted_sku ?? null,
extracted_product_name: d.ocr?.extracted_product_name ?? null,
extracted_expired_date: d.ocr?.extracted_expired_date ?? null,
expired_source_line: d.ocr?.expired_source_line ?? null
},
possibleMatches: d.possibleMatches || []
});
console.log("ok");
} catch (err) {
console.log(`FAILED (${err.message})`);
results.push({ filename, error: err.message });
}
}
fs.writeFileSync(OUT_PATH, JSON.stringify(results, null, 2), "utf8");
console.log(`\nWrote ${results.length} captures to ${OUT_PATH}`);
+178
View File
@@ -0,0 +1,178 @@
// Offline experiment: OCR-evidence re-ranking of DINOv2 top-K candidates.
//
// Reads sources/product_scan_fullcap.json (captured live responses, see
// capture-scan-responses.mjs) + sources/product_manual_labels.json (ground
// truth) and simulates candidate re-ranking without touching the GPU stack,
// reporting fixed-vs-broken counts per parameter combination. The winning
// parameters get ported into pfm-web-app/src/utils/product-scan.ts.
//
// Idea: DINOv2's near-twin confusions (same brand, different flavor/size)
// are exactly the cases where the *printed variant words* differ - and
// PaddleOCR usually reads some of them. So within a narrow similarity band
// of the top-1, prefer the candidate whose distinguishing name tokens
// actually appear in the OCR'd text.
//
// Usage: node scripts/experiment-rerank.mjs
import fs from "fs";
import path from "path";
const cap = JSON.parse(fs.readFileSync(path.join("sources", "product_scan_fullcap.json"), "utf8"));
const labels = JSON.parse(fs.readFileSync(path.join("sources", "product_manual_labels.json"), "utf8"));
const gtBySku = new Map(labels.map((l) => [l.filename, l.no_sku]));
function classSku(className) {
// Class names are foto-kemasan-v2 folder names: "<SKU> <NAME...>"
return (className || "").trim().split(/\s+/)[0] || "";
}
function tokenize(name) {
return name
.toUpperCase()
.split(/[^A-Z0-9]+/)
.filter((t) => t.length >= 2);
}
function editDistance1(a, b) {
// true if edit distance <= 1 (same length: 1 substitution; off-by-one: 1 indel)
if (a === b) return true;
const la = a.length, lb = b.length;
if (Math.abs(la - lb) > 1) return false;
if (la === lb) {
let diff = 0;
for (let i = 0; i < la; i++) if (a[i] !== b[i]) diff++;
return diff <= 1;
}
const [s, l] = la < lb ? [a, b] : [b, a];
let i = 0, j = 0, skipped = false;
while (i < s.length && j < l.length) {
if (s[i] === l[j]) { i++; j++; }
else if (!skipped) { skipped = true; j++; }
else return false;
}
return true;
}
function buildOcrIndex(textLines) {
const joined = textLines.join(" ").toUpperCase();
const squashed = joined.replace(/[^A-Z0-9]/g, "");
const tokens = new Set(tokenize(joined));
return { squashed, tokens };
}
function tokenInOcr(token, ocrIdx, fuzzy) {
if (token.length >= 4 && ocrIdx.squashed.includes(token)) return true;
if (ocrIdx.tokens.has(token)) return true;
if (fuzzy && token.length >= 5) {
for (const t of ocrIdx.tokens) {
if (Math.abs(t.length - token.length) <= 1 && editDistance1(token, t)) return true;
}
}
return false;
}
function ocrEvidenceScore(candTokens, bandTokenCounts, bandSize, ocrIdx, fuzzy) {
// Coverage-normalized, rarity-weighted evidence: fraction of this
// candidate's *distinctive* name tokens (weighted by band rarity) that
// actually appear in the OCR'd text. Normalizing by the candidate's own
// distinctive-token mass is what stops generic packaging words from
// hijacking the ranking - a candidate whose name promises FRENCH +
// INSTITUSI + 2KG but whose package shows only "French Fries" scores
// 1/3, losing to a candidate whose 2 distinctive tokens both appear.
let matched = 0;
let total = 0;
for (const tok of new Set(candTokens)) {
const nWith = bandTokenCounts.get(tok) || 1;
if (nWith >= bandSize) continue; // shared by all -> no signal
const w = 1 / nWith;
total += w;
if (tokenInOcr(tok, ocrIdx, fuzzy)) matched += w;
}
return total > 0 ? matched / total : 0;
}
function skuFuzzyBoost(extractedSku, candidateSku) {
if (!extractedSku || extractedSku.length < 7) return 0;
if (extractedSku === candidateSku) return 10; // exact (normally pinned upstream anyway)
return editDistance1(extractedSku, candidateSku) ? 1 : 0;
}
function runConfig({ K, BAND, MARGIN, FUZZY, SKU_BOOST_W }) {
let baselineCorrect = 0, rerankCorrect = 0, fixed = [], broken = [];
for (const item of cap) {
if (item.error) continue;
const gt = gtBySku.get(item.filename);
if (!gt) continue;
const probs = item.classification?.all_probabilities || [];
if (!probs.length) continue;
const top1Sku = classSku(probs[0].name);
const baselineRight = top1Sku === gt;
if (baselineRight) baselineCorrect++;
// Candidate band: within BAND of top-1 similarity, capped at K
const top1Sim = probs[0].confidence;
const band = probs.slice(0, K).filter((p) => p.confidence >= top1Sim - BAND);
const ocrIdx = buildOcrIndex(item.ocr?.text_lines || []);
const candInfos = band.map((p) => {
const sku = classSku(p.name);
const tokens = tokenize(p.name.replace(sku, ""));
return { sku, sim: p.confidence, tokens };
});
const bandTokenCounts = new Map();
for (const c of candInfos) {
for (const tok of new Set(c.tokens)) {
bandTokenCounts.set(tok, (bandTokenCounts.get(tok) || 0) + 1);
}
}
for (const c of candInfos) {
c.ocrScore = ocrEvidenceScore(c.tokens, bandTokenCounts, candInfos.length, ocrIdx, FUZZY)
+ SKU_BOOST_W * skuFuzzyBoost(item.ocr?.extracted_sku || "", c.sku);
}
// Switch away from top-1 only when a band-mate has clearly stronger OCR evidence
let chosen = candInfos[0];
for (const c of candInfos.slice(1)) {
if (c.ocrScore >= chosen.ocrScore + MARGIN) chosen = c;
}
const rerankRight = chosen.sku === gt;
if (rerankRight) rerankCorrect++;
if (!baselineRight && rerankRight) fixed.push(item.filename);
if (baselineRight && !rerankRight) broken.push(item.filename);
}
return { baselineCorrect, rerankCorrect, fixed, broken };
}
const grid = [];
for (const K of [5, 8, 12]) {
for (const BAND of [0.04, 0.06, 0.08, 0.12]) {
// Coverage scores live in [0, 1]; margin is the minimum coverage lead a
// band-mate needs over the current pick before we switch away from it.
for (const MARGIN of [0.15, 0.25, 0.35, 0.5]) {
for (const FUZZY of [true, false]) {
for (const SKU_BOOST_W of [0, 2]) {
grid.push({ K, BAND, MARGIN, FUZZY, SKU_BOOST_W });
}
}
}
}
}
const results = grid.map((cfg) => ({ cfg, ...runConfig(cfg) }));
results.sort((a, b) => (b.rerankCorrect - b.broken.length * 0.01) - (a.rerankCorrect - a.broken.length * 0.01));
console.log(`Images evaluated: ${cap.filter((i) => !i.error && gtBySku.has(i.filename)).length}`);
console.log(`Baseline (DINOv2 top-1) correct: ${results[0].baselineCorrect}\n`);
console.log("Top 12 configs by re-ranked correct count:");
for (const r of results.slice(0, 12)) {
console.log(
` correct=${r.rerankCorrect} (+${r.fixed.length}/-${r.broken.length}) ` +
`K=${r.cfg.K} BAND=${r.cfg.BAND} MARGIN=${r.cfg.MARGIN} FUZZY=${r.cfg.FUZZY} SKUW=${r.cfg.SKU_BOOST_W}`
);
}
const best = results[0];
console.log(`\nBest config detail: ${JSON.stringify(best.cfg)}`);
console.log(` fixed (${best.fixed.length}): ${best.fixed.join(", ")}`);
console.log(` broken (${best.broken.length}): ${best.broken.join(", ")}`);
+47
View File
@@ -0,0 +1,47 @@
// One-off script: freeze the current 74-image product-scan Validation Set into
// a dedicated, stable folder (sources/product-test-images-fixed/) so re-running
// the accuracy harness always scores the exact same images, independent of
// whatever new photos get dropped into the live-intake folder
// (sources/product-test-images/, still fed by the /manual-label-scan page).
// Renames each image "<index> <no_sku>.<ext>" (index = its stable position,
// 1-based) and updates product_manual_labels.json's flat-filename entries to
// match. Run once from backend/: node scripts/freeze-validation-set.mjs
import fs from "fs";
import path from "path";
const LIVE_DIR = path.join("sources", "product-test-images");
const FIXED_DIR = path.join("sources", "product-test-images-fixed");
const LABELS_PATH = path.join("sources", "product_manual_labels.json");
const labels = JSON.parse(fs.readFileSync(LABELS_PATH, "utf8"));
const flatEntries = labels.filter((l) => !l.filename.includes("/"));
if (!fs.existsSync(FIXED_DIR)) fs.mkdirSync(FIXED_DIR, { recursive: true });
// Resolve by SKU prefix (not entry.filename directly) so this script is
// idempotent/rerunnable even after a previous run already renamed
// entry.filename to "<index> <sku>.<ext>" — the live-intake folder always
// keeps its original "<sku> <product>__<camera-filename>.<ext>" names.
const liveFiles = fs.readdirSync(LIVE_DIR);
function findSourceFile(no_sku) {
const match = liveFiles.find((f) => f.startsWith(`${no_sku} `) || f.startsWith(`${no_sku}__`));
if (!match) return null;
return path.join(LIVE_DIR, match);
}
let copied = 0;
flatEntries.forEach((entry, i) => {
const index = i + 1;
const srcPath = findSourceFile(entry.no_sku);
if (!srcPath) {
throw new Error(`Missing source image for ${entry.no_sku} in ${LIVE_DIR}`);
}
const ext = path.extname(srcPath);
const newFilename = `${index} ${entry.no_sku}${ext}`;
fs.copyFileSync(srcPath, path.join(FIXED_DIR, newFilename));
entry.filename = newFilename;
copied++;
});
fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8");
console.log(`Copied ${copied} images into ${FIXED_DIR} and updated ${LABELS_PATH}.`);
+130
View File
@@ -0,0 +1,130 @@
// One-off script: fill ground-truth labels for backend/sources/product-test-images/
// (the product-scan accuracy harness's Validation Set, previously 0 labeled images).
// Run once from backend/: node scripts/seed-validation-labels.mjs
import fs from "fs";
import path from "path";
const IMAGES_DIR = path.join("sources", "product-test-images");
const LABELS_PATH = path.join("sources", "product_manual_labels.json");
// [no_sku, nama_item, expiry_date ("" = not legible in photo, needs re-shoot)]
const DATA = [
["11110059", "CEKER BERKUKU FROZEN PACK 1 KG(*)", "13/06/2027"],
["11140051", "AMPELA FROZEN PACK 1 KG(*)", "26/11/2026"],
["11620056", "SBL (FILLET PAHA) 1 KG(*)", "27/02/2027"],
["11650053", "PAHA ATAS 1 KG(*)", "26/02/2027"],
["11660050", "PAHA BAWAH (1 KG)(*)", "22/06/2027"],
["11710051", "DADA UTUH (1 KG)(*)", ""],
["11818300", "CP-BEBEK GORENG 400GR/PAC", ""],
["1195008A", "RTC CHICKEN KALASAN 400 GR (PAC)", ""],
["11959937", "SATE AYAM FRESHMART 360 GR (PAC)", "24/11/2026"],
["12010111", "FIESTA CRISPY BUBBLE 400 GR/PAC", "07/05/2027"],
["12010115", "FIESTA NUGGET ZOO 400 GR/PAC", "01/12/2026"],
["12010117", "FIESTA NUGGET HAPPY STAR 400 GR/PAC", "27/08/2026"],
["12010119", "FIESTA NUGGET CHEESE 123 400 GR/PAC", "09/04/2027"],
["12010121", "FIESTA NUGGET PIZZABC 400 GR/PAC", "12/11/2026"],
["12010127", "FIESTA SPICY NUGGET 400 GR/PAC", "09/12/2027"],
["12010509", "CHAMP CRUNCHY NUGGET 450 GR/PAC", "15/04/2027"],
["12010515", "CHAMP KOIN KOMBINASI 450 GR/PAC", "15/04/2027"],
["12010519", "CHAMP NUGGET STICK 900 GR/PAC", "08/03/2027"],
["12012202", "ASIMO NUGGET KOMBINASI 1 KG/PAC", "13/05/2027"],
["12012501", "AKUMO CHICKEN NAGET 250 GR", "21/05/2027"],
["12012503", "AKUMO CHICKEN NUGGET 1000 GR", "17/06/2027"],
["12012505", "AKUMO KOIN 400 GR/PAC", "16/11/2026"],
["12020102", "FIESTA SPICY WING 400 GR/PAC", "22/05/2027"],
["12030102", "FIESTA STIKIE 200 GR/PAC", "26/02/2027"],
["12030403", "GOLDEN FIESTA STIKIE W/ SWEET CHILLI SAUCE 500GR", "07/04/2027"],
["12032502", "AKUMO CHICKEN STIK 500 GR", "09/03/2027"],
["12040101", "FIESTA SCHNITZEL 400 GR/PAC", "09/10/2026"],
["12040102", "FIESTA CRISPY BUBBLE KATSU 400 GR/PAC", "12/04/2027"],
["12060103", "FIESTA KARAGE 200 GR/PAC", ""],
["12060402", "GOLDEN FIESTA KARAGE CHILI SAUCE 500GR", "15/05/2027"],
["12080101", "FIESTA SPICY CHICK 400 GR/PAC", "26/04/2027"],
["12130102", "FIESTA CRISPY BURGER 360 GR (NEW)", "27/04/2027"],
["12150201", "FIESTA DS CRISPY CRUNCH 300 GR/PAC", "03/06/2027"],
["12150501", "CHAMP CRUNCHY HOTZZ 300 GR/PAC", ""],
["12190103", "FIESTA DELISTRIPE 400 GR/PAC", "07/04/2027"],
["12240103", "FIESTA YAKINIKU R/BITES 400 GR/PAC", "07/05/2027"],
["13010111", "FIESTA SOSIS BRATWURST 300 GR", "25/03/2027"],
["13010116", "FIESTA SSG ORIGINAL 300 GR", "23/06/2027"],
["13010118", "FIESTA RTG SSG 65 GR/PAC", ""],
["13010120", "FIESTA RTG C/CHEESY MELTS 65 GR/PAC", ""],
["13010122", "FIESTA RTG SAUSAGE WITH HOT LAVA 60G", ""],
["13010123", "FIESTA RTG SAUSAGE WITH CHEESE LAVA 60G", "25/10/2026"],
["13010125", "FIESTA RTG SAUSAGE WITH MENTAI LAVA 60GR", "05/11/2026"],
["13010518", "CHAMP SSG JUMBO BAKAR 500 GR/PAC", ""],
["13010524", "CHAMP SSG JUMBO BAKAR 500 GR/PAC (NEW)", ""],
["13012206", "ASIMO SOSIS AYAM KOMBINASI 500 GR", "15/03/2027"],
["13030501", "CHAMP CHICK MEATBALL 200 GR", ""],
["13070506", "CHAMP FRANKFURTER SSG 375GR", "13/03/2027"],
["13100512", "CHAMP CHICK SSG S/SANTAP ORIG 546GR (CAN)", ""],
["15010101", "FIESTA SHOESTRING 500 GR", "04/06/2027"],
["15010102", "FIESTA SHOESTRING 1000 GR", "05/03/2027"],
["15010107", "FIESTA FRENCH F SHOESTRING INSTITUSI 2KG", "17/06/2027"],
["15020101", "FIESTA STRAIGHT CUT 500 GR", "19/06/2027"],
["15020102", "FIESTA STRAIGHT CUT 1000 GR", "18/05/2027"],
["15030101", "FIESTA CRINKLE CUT 500 GR", ""],
["15030102", "FIESTA CRINKLE CUT 1000 GR", "07/04/2027"],
["16060113", "FIESTA CHICK SIOMAY 180GR (NEW)", "24/02/2027"],
["16060114", "FIESTA GYOZA 180 GR (NEW)", "18/05/2027"],
["17200109", "FIESTA RTS C/TERIYAKI 300GR/PAC", ""],
["17210106", "FIESTA RTS B/YAKINIKU 300GR/PAC", "19/05/2027"],
["17210107", "FIESTA RTS B/RENDANG 300GR/PAC", "23/06/2027"],
["17210108", "FIESTA RTS B/BLACKPEPPER 300GR/PAC", "09/05/2027"],
["17210109", "FIESTA RTS B/BULGOGI 300GR/PAC", "18/05/2027"],
["20040101", "FIESTA RAMEN BEKU 570 GR/PAC", "23/06/2026"],
["20120102", "FIESTA T/B AYAM GORENG 80 GR", ""],
["20120105", "FIESTA T/B SERBAGUNA Â 80 GR", "09/03/2027"],
["20120115", "FIESTA RACIK AYAM GORENG 20 GR/PAC", ""],
["20120116", "FIESTA RACIK NASI GORENG 20 GR/PAC", "13/10/2026"],
["21000123", "FIESTA RICE W/GEPREK CHICKEN 320GR/PAC", "30/04/2027"],
["21000126", "NEW FIESTA CHICK RENDANG W RICE 320GR (PAC)", "16/04/2027"],
["21000130", "NEW FIESTA RICE W/C CHEESE BULDAK 320GR (PAC)", "09/04/2027"],
["21000137", "FIESTA HAINAMESE CHICKEN RICE 320GR (PAC)", "12/02/2027"],
["21010101", "FIESTA TRUFFLE GYUDON 320 GR/PAC", "18/03/2027"],
["21200107", "NEW FIESTA SPAGHETTI CARBONARA 300GR (PAC)", "20/05/2027"],
];
const files = fs
.readdirSync(IMAGES_DIR)
.filter((f) => f !== "README.md" && f !== "filelist.txt")
.sort();
if (files.length !== DATA.length) {
throw new Error(`File count ${files.length} != DATA count ${DATA.length}`);
}
const existing = JSON.parse(fs.readFileSync(LABELS_PATH, "utf8"));
const now = new Date().toISOString();
let added = 0;
let skipped = 0;
let unreadable = 0;
for (let i = 0; i < files.length; i++) {
const filename = files[i];
const [no_sku, nama_item, expiry_date] = DATA[i];
if (existing.some((l) => l.filename === filename)) {
skipped++;
continue;
}
if (!expiry_date) unreadable++;
existing.push({
filename,
no_sku,
nama_item,
expiry_date,
top1_confidence: null,
notes: expiry_date
? ""
: "expiry date not legible in photo (cropped/blurry/out of frame) - needs re-shoot",
saved_at: now,
});
added++;
}
fs.writeFileSync(LABELS_PATH, JSON.stringify(existing, null, 2), "utf8");
console.log(`Added ${added} labels (${unreadable} flagged with no expiry_date), skipped ${skipped} already-labeled.`);