Ports the DO-harness's auto-diff-vs-previous-run reporting into accuracy-check-scan.mts: prints a per-field, per-split (Training/ Validation) delta against the last product_accuracy_history.jsonl entry and calls out regressions/improvements explicitly, plus classifier method distribution and average confidence as informational context. Also adds the real held-out validation photo set into sources/product-test-images/ (75 photos, one per current SKU class) with its README documenting the drop-photo -> label -> re-run workflow, so the harness's Validation Set split actually has images to score against. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Xsxk4ZkDQVVaLUcixDcqb5
412 lines
15 KiB
TypeScript
412 lines
15 KiB
TypeScript
// Product-scan (scan-pfm) accuracy regression tool.
|
|
//
|
|
// Hits the live /api/scan-pfm endpoint for every labeled image in
|
|
// backend/sources/product_manual_labels.json, checks 3 fields (no_sku,
|
|
// nama_item, expiry_date) against ground truth, splits results into a
|
|
// Training Set (gallery photos under foto-kemasan-v2/ that trained the
|
|
// classifier itself) vs a Validation Set (flat filenames dropped in
|
|
// backend/sources/product-test-images/), and appends a summary to
|
|
// backend/sources/product_accuracy_history.jsonl. Every run auto-diffs
|
|
// against the last history entry and flags field/image regressions or
|
|
// improvements, so a tuning change to classify_ocr_server.py shows its
|
|
// effect immediately instead of requiring manual before/after comparison.
|
|
//
|
|
// Usage:
|
|
// node scripts/accuracy-check-scan.mts
|
|
// node scripts/accuracy-check-scan.mts --base-url http://localhost:3000
|
|
|
|
import fs from "node:fs";
|
|
import path from "node:path";
|
|
import { fileURLToPath } from "node:url";
|
|
import { execSync } from "node:child_process";
|
|
|
|
const __filename = fileURLToPath(import.meta.url);
|
|
const __dirname = path.dirname(__filename);
|
|
|
|
const APP_ROOT = path.join(__dirname, "..", "pfm-web-app");
|
|
const SOURCES_DIR = path.join(__dirname, "..", "sources");
|
|
// Overridable for local smoke-testing against a scratch dataset without
|
|
// touching the real ground-truth/history files.
|
|
const LABELS_PATH = process.env.ACCURACY_LABELS_PATH || path.join(SOURCES_DIR, "product_manual_labels.json");
|
|
const HISTORY_PATH = process.env.ACCURACY_HISTORY_PATH || path.join(SOURCES_DIR, "product_accuracy_history.jsonl");
|
|
|
|
const FETCH_TIMEOUT_MS = 120_000;
|
|
const FIELDS = ["no_sku", "nama_item", "expiry_date"] as const;
|
|
type Field = typeof FIELDS[number];
|
|
type Split = "training" | "validation";
|
|
|
|
interface GroundTruth {
|
|
filename: string;
|
|
no_sku: string;
|
|
nama_item: string;
|
|
expiry_date: string;
|
|
}
|
|
|
|
interface ScanResponse {
|
|
classification?: {
|
|
top1_name: string;
|
|
top1_confidence: number;
|
|
method?: string;
|
|
};
|
|
ocr?: {
|
|
extracted_expired_date: string;
|
|
};
|
|
possibleMatches?: Array<{
|
|
no_sku: string;
|
|
nama_item: string;
|
|
isBestMatch: boolean;
|
|
}>;
|
|
}
|
|
|
|
interface Check {
|
|
field: Field;
|
|
match: boolean;
|
|
}
|
|
|
|
interface ResultItem {
|
|
gt: GroundTruth;
|
|
checks: Check[];
|
|
method?: string;
|
|
confidence?: number;
|
|
}
|
|
|
|
interface ClassificationStats {
|
|
methodCounts: Record<string, number>;
|
|
avgConfidence: number;
|
|
}
|
|
|
|
interface HistoryEntry {
|
|
timestamp: string;
|
|
commit: string;
|
|
imageCount: { training: number; validation: number };
|
|
failedImages: string[];
|
|
fields: Record<Split, Record<Field, { correct: number; total: number }>>;
|
|
classification: Record<Split, ClassificationStats>;
|
|
perImage: Record<string, number>;
|
|
}
|
|
|
|
function parseArgs(argv: string[]) {
|
|
return {
|
|
baseUrl: process.env.ACCURACY_BASE_URL || "http://localhost:3000",
|
|
};
|
|
}
|
|
|
|
function norm(v: unknown): string {
|
|
if (!v) return "";
|
|
return String(v).replace(/\s+/g, " ").trim().toUpperCase();
|
|
}
|
|
|
|
function isMatch(a: unknown, b: unknown): boolean {
|
|
return norm(a) === norm(b);
|
|
}
|
|
|
|
function getImagePath(filename: string): string {
|
|
if (filename.includes("/")) {
|
|
return path.join(APP_ROOT, "public", "produk-pfm", "foto-kemasan-v2", filename);
|
|
}
|
|
return path.join(SOURCES_DIR, "product-test-images", filename);
|
|
}
|
|
|
|
async function checkServerReachable(baseUrl: string) {
|
|
try {
|
|
const res = await fetch(`${baseUrl}/api/v1/health`, { signal: AbortSignal.timeout(5000) });
|
|
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
|
} catch (err) {
|
|
throw new Error(`Next dev server not reachable at ${baseUrl}. Ensure it's running.`);
|
|
}
|
|
}
|
|
|
|
async function fetchScan(baseUrl: string, base64: string): Promise<ScanResponse> {
|
|
const controller = new AbortController();
|
|
const timer = setTimeout(() => controller.abort(), FETCH_TIMEOUT_MS);
|
|
try {
|
|
const res = await fetch(`${baseUrl}/api/scan-pfm`, {
|
|
method: "POST",
|
|
headers: { "Content-Type": "application/json" },
|
|
body: JSON.stringify({ image_base64: base64 }),
|
|
signal: controller.signal
|
|
});
|
|
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
|
return await res.json();
|
|
} finally {
|
|
clearTimeout(timer);
|
|
}
|
|
}
|
|
|
|
function getGitCommit(): string {
|
|
try { return execSync("git rev-parse --short HEAD", { cwd: __dirname }).toString().trim(); }
|
|
catch { return "unknown"; }
|
|
}
|
|
|
|
function pct(c: number, t: number) { return t === 0 ? 0 : (c / t) * 100; }
|
|
function fmtPct(n: number) { return `${n.toFixed(1)}%`; }
|
|
|
|
function fmtDelta(curr: number, prev: number | undefined): string {
|
|
if (prev === undefined) return "";
|
|
const d = curr - prev;
|
|
if (Math.abs(d) < 0.05) return "±0.0";
|
|
const sign = d > 0 ? "+" : "";
|
|
return `${sign}${d.toFixed(1)}`;
|
|
}
|
|
|
|
function loadLastHistoryEntry(): HistoryEntry | null {
|
|
if (!fs.existsSync(HISTORY_PATH)) return null;
|
|
const lines = fs.readFileSync(HISTORY_PATH, "utf8").trim().split("\n").filter(Boolean);
|
|
if (lines.length === 0) return null;
|
|
try {
|
|
return JSON.parse(lines[lines.length - 1]);
|
|
} catch {
|
|
return null;
|
|
}
|
|
}
|
|
|
|
function aggregateFields(list: ResultItem[]): Record<Field, { correct: number; total: number }> {
|
|
const agg = {
|
|
no_sku: { correct: 0, total: 0 },
|
|
nama_item: { correct: 0, total: 0 },
|
|
expiry_date: { correct: 0, total: 0 }
|
|
} as Record<Field, { correct: number; total: number }>;
|
|
for (const item of list) {
|
|
for (const check of item.checks) {
|
|
agg[check.field].total++;
|
|
if (check.match) agg[check.field].correct++;
|
|
}
|
|
}
|
|
return agg;
|
|
}
|
|
|
|
function overallFromFields(fields: Record<Field, { correct: number; total: number }> | undefined) {
|
|
if (!fields) return undefined;
|
|
let correct = 0, total = 0;
|
|
for (const f of FIELDS) {
|
|
const s = fields[f];
|
|
if (s) { correct += s.correct; total += s.total; }
|
|
}
|
|
return { correct, total };
|
|
}
|
|
|
|
function aggregateClassification(list: ResultItem[]): ClassificationStats {
|
|
const methodCounts: Record<string, number> = {};
|
|
let confSum = 0;
|
|
let confCount = 0;
|
|
for (const item of list) {
|
|
if (item.method) methodCounts[item.method] = (methodCounts[item.method] || 0) + 1;
|
|
if (typeof item.confidence === "number") {
|
|
confSum += item.confidence;
|
|
confCount++;
|
|
}
|
|
}
|
|
return { methodCounts, avgConfidence: confCount ? confSum / confCount : 0 };
|
|
}
|
|
|
|
function printClassificationLine(label: string, stats: ClassificationStats, prevStats: ClassificationStats | undefined) {
|
|
const methodStr = Object.entries(stats.methodCounts).map(([m, c]) => `${m}=${c}`).join(", ") || "n/a";
|
|
const confStr = stats.avgConfidence ? stats.avgConfidence.toFixed(3) : "n/a";
|
|
let suffix = "";
|
|
if (prevStats && prevStats.avgConfidence) {
|
|
const delta = fmtDelta(stats.avgConfidence, prevStats.avgConfidence);
|
|
suffix = delta ? ` (Δ ${delta} vs prev)` : "";
|
|
}
|
|
console.log(` ${label.padEnd(11)}: ${methodStr.padEnd(28)} avg confidence ${confStr}${suffix}`);
|
|
}
|
|
|
|
function printSummary(
|
|
trainAgg: Record<Field, { correct: number; total: number }>,
|
|
valAgg: Record<Field, { correct: number; total: number }>,
|
|
trainClassStats: ClassificationStats,
|
|
valClassStats: ClassificationStats,
|
|
perImagePct: Record<string, number>,
|
|
prev: HistoryEntry | null,
|
|
imageCount: { training: number; validation: number },
|
|
failedImages: string[]
|
|
) {
|
|
console.log("\n=== Product Scan Accuracy Summary ===");
|
|
console.log(`Training Images: ${imageCount.training} | Validation Images: ${imageCount.validation} | Failed: ${failedImages.length}\n`);
|
|
|
|
const fieldCol = 14, numCol = 9, deltaCol = 8;
|
|
const header =
|
|
"Field".padEnd(fieldCol) +
|
|
"Training".padStart(numCol) + "Δ".padStart(deltaCol) + " " +
|
|
"Validation".padStart(numCol) + "Δ".padStart(deltaCol);
|
|
console.log(header);
|
|
console.log("-".repeat(header.length));
|
|
|
|
const overallTrain = { correct: 0, total: 0 };
|
|
const overallVal = { correct: 0, total: 0 };
|
|
|
|
for (const field of FIELDS) {
|
|
const t = trainAgg[field];
|
|
const v = valAgg[field];
|
|
overallTrain.correct += t.correct; overallTrain.total += t.total;
|
|
overallVal.correct += v.correct; overallVal.total += v.total;
|
|
|
|
const tPct = pct(t.correct, t.total);
|
|
const vPct = pct(v.correct, v.total);
|
|
const tPrevStat = prev?.fields?.training?.[field];
|
|
const vPrevStat = prev?.fields?.validation?.[field];
|
|
const tPrevPct = tPrevStat?.total ? pct(tPrevStat.correct, tPrevStat.total) : undefined;
|
|
const vPrevPct = vPrevStat?.total ? pct(vPrevStat.correct, vPrevStat.total) : undefined;
|
|
|
|
const tStr = t.total ? fmtPct(tPct) : "n/a";
|
|
const vStr = v.total ? fmtPct(vPct) : "n/a";
|
|
|
|
console.log(
|
|
field.padEnd(fieldCol) +
|
|
tStr.padStart(numCol) + fmtDelta(tPct, tPrevPct).padStart(deltaCol) + " " +
|
|
vStr.padStart(numCol) + fmtDelta(vPct, vPrevPct).padStart(deltaCol)
|
|
);
|
|
}
|
|
console.log("-".repeat(header.length));
|
|
|
|
const tOverallPct = pct(overallTrain.correct, overallTrain.total);
|
|
const vOverallPct = pct(overallVal.correct, overallVal.total);
|
|
const prevTrainOverall = overallFromFields(prev?.fields?.training);
|
|
const prevValOverall = overallFromFields(prev?.fields?.validation);
|
|
const tOverallPrevPct = prevTrainOverall?.total ? pct(prevTrainOverall.correct, prevTrainOverall.total) : undefined;
|
|
const vOverallPrevPct = prevValOverall?.total ? pct(prevValOverall.correct, prevValOverall.total) : undefined;
|
|
|
|
console.log(
|
|
"OVERALL".padEnd(fieldCol) +
|
|
fmtPct(tOverallPct).padStart(numCol) + fmtDelta(tOverallPct, tOverallPrevPct).padStart(deltaCol) + " " +
|
|
fmtPct(vOverallPct).padStart(numCol) + fmtDelta(vOverallPct, vOverallPrevPct).padStart(deltaCol)
|
|
);
|
|
|
|
if (failedImages.length) {
|
|
console.log(`\nFailed to parse: ${failedImages.join(", ")}`);
|
|
}
|
|
|
|
// Informational only - DINOv2 "confidence" is a raw cosine similarity, not a
|
|
// calibrated probability (see docs/scan-product.md), so a delta here doesn't
|
|
// by itself mean better/worse. Only the fields above drive regression flags.
|
|
console.log("\nClassification (informational, not scored as pass/fail):");
|
|
printClassificationLine("Training", trainClassStats, prev?.classification?.training);
|
|
printClassificationLine("Validation", valClassStats, prev?.classification?.validation);
|
|
|
|
if (prev) {
|
|
const fieldRegressions: string[] = [];
|
|
const fieldImprovements: string[] = [];
|
|
for (const split of ["training", "validation"] as Split[]) {
|
|
const agg = split === "training" ? trainAgg : valAgg;
|
|
for (const field of FIELDS) {
|
|
const stat = agg[field];
|
|
if (!stat.total) continue;
|
|
const currPct = pct(stat.correct, stat.total);
|
|
const prevStat = prev.fields?.[split]?.[field];
|
|
if (!prevStat?.total) continue;
|
|
const prevPct = pct(prevStat.correct, prevStat.total);
|
|
const d = currPct - prevPct;
|
|
const label = `${field} (${split})`;
|
|
if (d <= -0.05) fieldRegressions.push(`${label} ${fmtDelta(currPct, prevPct)}`);
|
|
else if (d >= 0.05) fieldImprovements.push(`${label} ${fmtDelta(currPct, prevPct)}`);
|
|
}
|
|
}
|
|
if (fieldRegressions.length) console.log(`\nField regressions: ${fieldRegressions.join(", ")}`);
|
|
if (fieldImprovements.length) console.log(`Field improvements: ${fieldImprovements.join(", ")}`);
|
|
|
|
const imageRegressions: string[] = [];
|
|
const imageImprovements: string[] = [];
|
|
for (const [filename, currPct] of Object.entries(perImagePct)) {
|
|
const prevPct = prev.perImage?.[filename];
|
|
if (prevPct === undefined) continue;
|
|
const d = currPct - prevPct;
|
|
if (d <= -0.5) imageRegressions.push(`${filename} ${fmtDelta(currPct, prevPct)}`);
|
|
else if (d >= 0.5) imageImprovements.push(`${filename} ${fmtDelta(currPct, prevPct)}`);
|
|
}
|
|
if (imageRegressions.length) console.log(`\nImage regressions: ${imageRegressions.join(", ")}`);
|
|
if (imageImprovements.length) console.log(`Image improvements: ${imageImprovements.join(", ")}`);
|
|
|
|
console.log(`\n(vs run at ${prev.timestamp}${prev.commit !== "unknown" ? `, commit ${prev.commit}` : ""})`);
|
|
} else {
|
|
console.log("\n(no previous run in product_accuracy_history.jsonl — this is the baseline)");
|
|
}
|
|
console.log("");
|
|
}
|
|
|
|
async function main() {
|
|
const args = parseArgs(process.argv.slice(2));
|
|
await checkServerReachable(args.baseUrl);
|
|
|
|
if (!fs.existsSync(LABELS_PATH)) {
|
|
console.error(`No labels found at ${LABELS_PATH}`);
|
|
process.exit(1);
|
|
}
|
|
const labels: GroundTruth[] = JSON.parse(fs.readFileSync(LABELS_PATH, "utf8"));
|
|
const prev = loadLastHistoryEntry();
|
|
|
|
const results = {
|
|
training: [] as ResultItem[],
|
|
validation: [] as ResultItem[],
|
|
failed: [] as string[]
|
|
};
|
|
const perImagePct: Record<string, number> = {};
|
|
|
|
for (const gt of labels) {
|
|
if (gt.filename.startsWith("uploaded-")) continue; // Skip phantom
|
|
const imgPath = getImagePath(gt.filename);
|
|
if (!fs.existsSync(imgPath)) {
|
|
console.warn(`Warning: Image missing from disk: ${imgPath}`);
|
|
continue;
|
|
}
|
|
|
|
process.stdout.write(`Scanning ${gt.filename}... `);
|
|
try {
|
|
const b64 = "data:image/jpeg;base64," + fs.readFileSync(imgPath, "base64");
|
|
const parsed = await fetchScan(args.baseUrl, b64);
|
|
|
|
const bestMatchSku = parsed.possibleMatches?.find(m => m.isBestMatch)?.no_sku || "";
|
|
const predictedItemName = parsed.classification?.top1_name || "";
|
|
const predictedExpiry = parsed.ocr?.extracted_expired_date || "";
|
|
|
|
const checks: Check[] = [
|
|
{ field: "no_sku", match: isMatch(gt.no_sku, bestMatchSku) },
|
|
{ field: "nama_item", match: isMatch(gt.nama_item, predictedItemName) },
|
|
{ field: "expiry_date", match: isMatch(gt.expiry_date, predictedExpiry) }
|
|
];
|
|
|
|
const item: ResultItem = {
|
|
gt,
|
|
checks,
|
|
method: parsed.classification?.method,
|
|
confidence: parsed.classification?.top1_confidence
|
|
};
|
|
|
|
if (gt.filename.includes("/")) {
|
|
results.training.push(item);
|
|
} else {
|
|
results.validation.push(item);
|
|
}
|
|
|
|
const score = checks.filter(c => c.match).length;
|
|
perImagePct[gt.filename] = (score / checks.length) * 100;
|
|
console.log(`done (${score}/3)`);
|
|
} catch (err) {
|
|
console.log(`FAILED (${(err as Error).message})`);
|
|
results.failed.push(gt.filename);
|
|
}
|
|
}
|
|
|
|
const trainAgg = aggregateFields(results.training);
|
|
const valAgg = aggregateFields(results.validation);
|
|
const trainClassStats = aggregateClassification(results.training);
|
|
const valClassStats = aggregateClassification(results.validation);
|
|
const imageCount = { training: results.training.length, validation: results.validation.length };
|
|
|
|
printSummary(trainAgg, valAgg, trainClassStats, valClassStats, perImagePct, prev, imageCount, results.failed);
|
|
|
|
const entry: HistoryEntry = {
|
|
timestamp: new Date().toISOString(),
|
|
commit: getGitCommit(),
|
|
imageCount,
|
|
failedImages: results.failed,
|
|
fields: { training: trainAgg, validation: valAgg },
|
|
classification: { training: trainClassStats, validation: valClassStats },
|
|
perImage: perImagePct
|
|
};
|
|
fs.appendFileSync(HISTORY_PATH, JSON.stringify(entry) + "\n");
|
|
}
|
|
|
|
main().catch(err => {
|
|
console.error("Fatal error:", err);
|
|
process.exit(1);
|
|
});
|