feat(backend): diff-vs-previous-run reporting for product-scan accuracy harness
Ports the DO-harness's auto-diff-vs-previous-run reporting into accuracy-check-scan.mts: prints a per-field, per-split (Training/ Validation) delta against the last product_accuracy_history.jsonl entry and calls out regressions/improvements explicitly, plus classifier method distribution and average confidence as informational context. Also adds the real held-out validation photo set into sources/product-test-images/ (75 photos, one per current SKU class) with its README documenting the drop-photo -> label -> re-run workflow, so the harness's Validation Set split actually has images to score against. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Xsxk4ZkDQVVaLUcixDcqb5
This commit is contained in:
1 parent
dc0dd81318
commit
3a17c28758
78 files changed
+306
-55
No files matched your search
@@ -1,3 +1,20 @@
|
||||
// Product-scan (scan-pfm) accuracy regression tool.
|
||||
//
|
||||
// Hits the live /api/scan-pfm endpoint for every labeled image in
|
||||
// backend/sources/product_manual_labels.json, checks 3 fields (no_sku,
|
||||
// nama_item, expiry_date) against ground truth, splits results into a
|
||||
// Training Set (gallery photos under foto-kemasan-v2/ that trained the
|
||||
// classifier itself) vs a Validation Set (flat filenames dropped in
|
||||
// backend/sources/product-test-images/), and appends a summary to
|
||||
// backend/sources/product_accuracy_history.jsonl. Every run auto-diffs
|
||||
// against the last history entry and flags field/image regressions or
|
||||
// improvements, so a tuning change to classify_ocr_server.py shows its
|
||||
// effect immediately instead of requiring manual before/after comparison.
|
||||
//
|
||||
// Usage:
|
||||
// node scripts/accuracy-check-scan.mts
|
||||
// node scripts/accuracy-check-scan.mts --base-url http://localhost:3000
|
||||
|
||||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
@@ -8,10 +25,15 @@ const __dirname = path.dirname(__filename);
|
||||
|
||||
const APP_ROOT = path.join(__dirname, "..", "pfm-web-app");
|
||||
const SOURCES_DIR = path.join(__dirname, "..", "sources");
|
||||
const LABELS_PATH = path.join(SOURCES_DIR, "product_manual_labels.json");
|
||||
const HISTORY_PATH = path.join(SOURCES_DIR, "product_accuracy_history.jsonl");
|
||||
// Overridable for local smoke-testing against a scratch dataset without
|
||||
// touching the real ground-truth/history files.
|
||||
const LABELS_PATH = process.env.ACCURACY_LABELS_PATH || path.join(SOURCES_DIR, "product_manual_labels.json");
|
||||
const HISTORY_PATH = process.env.ACCURACY_HISTORY_PATH || path.join(SOURCES_DIR, "product_accuracy_history.jsonl");
|
||||
|
||||
const FETCH_TIMEOUT_MS = 120_000;
|
||||
const FIELDS = ["no_sku", "nama_item", "expiry_date"] as const;
|
||||
type Field = typeof FIELDS[number];
|
||||
type Split = "training" | "validation";
|
||||
|
||||
interface GroundTruth {
|
||||
filename: string;
|
||||
@@ -24,6 +46,7 @@ interface ScanResponse {
|
||||
classification?: {
|
||||
top1_name: string;
|
||||
top1_confidence: number;
|
||||
method?: string;
|
||||
};
|
||||
ocr?: {
|
||||
extracted_expired_date: string;
|
||||
@@ -36,19 +59,30 @@ interface ScanResponse {
|
||||
}
|
||||
|
||||
interface Check {
|
||||
field: "no_sku" | "nama_item" | "expiry_date";
|
||||
field: Field;
|
||||
match: boolean;
|
||||
}
|
||||
|
||||
interface ResultItem {
|
||||
gt: GroundTruth;
|
||||
checks: Check[];
|
||||
method?: string;
|
||||
confidence?: number;
|
||||
}
|
||||
|
||||
interface ClassificationStats {
|
||||
methodCounts: Record<string, number>;
|
||||
avgConfidence: number;
|
||||
}
|
||||
|
||||
interface HistoryEntry {
|
||||
timestamp: string;
|
||||
commit: string;
|
||||
imageCount: { training: number; validation: number };
|
||||
failedImages: string[];
|
||||
fields: {
|
||||
training: Record<string, { correct: number; total: number }>;
|
||||
validation: Record<string, { correct: number; total: number }>;
|
||||
};
|
||||
fields: Record<Split, Record<Field, { correct: number; total: number }>>;
|
||||
classification: Record<Split, ClassificationStats>;
|
||||
perImage: Record<string, number>;
|
||||
}
|
||||
|
||||
function parseArgs(argv: string[]) {
|
||||
@@ -107,6 +141,187 @@ function getGitCommit(): string {
|
||||
function pct(c: number, t: number) { return t === 0 ? 0 : (c / t) * 100; }
|
||||
function fmtPct(n: number) { return `${n.toFixed(1)}%`; }
|
||||
|
||||
function fmtDelta(curr: number, prev: number | undefined): string {
|
||||
if (prev === undefined) return "";
|
||||
const d = curr - prev;
|
||||
if (Math.abs(d) < 0.05) return "±0.0";
|
||||
const sign = d > 0 ? "+" : "";
|
||||
return `${sign}${d.toFixed(1)}`;
|
||||
}
|
||||
|
||||
function loadLastHistoryEntry(): HistoryEntry | null {
|
||||
if (!fs.existsSync(HISTORY_PATH)) return null;
|
||||
const lines = fs.readFileSync(HISTORY_PATH, "utf8").trim().split("\n").filter(Boolean);
|
||||
if (lines.length === 0) return null;
|
||||
try {
|
||||
return JSON.parse(lines[lines.length - 1]);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function aggregateFields(list: ResultItem[]): Record<Field, { correct: number; total: number }> {
|
||||
const agg = {
|
||||
no_sku: { correct: 0, total: 0 },
|
||||
nama_item: { correct: 0, total: 0 },
|
||||
expiry_date: { correct: 0, total: 0 }
|
||||
} as Record<Field, { correct: number; total: number }>;
|
||||
for (const item of list) {
|
||||
for (const check of item.checks) {
|
||||
agg[check.field].total++;
|
||||
if (check.match) agg[check.field].correct++;
|
||||
}
|
||||
}
|
||||
return agg;
|
||||
}
|
||||
|
||||
function overallFromFields(fields: Record<Field, { correct: number; total: number }> | undefined) {
|
||||
if (!fields) return undefined;
|
||||
let correct = 0, total = 0;
|
||||
for (const f of FIELDS) {
|
||||
const s = fields[f];
|
||||
if (s) { correct += s.correct; total += s.total; }
|
||||
}
|
||||
return { correct, total };
|
||||
}
|
||||
|
||||
function aggregateClassification(list: ResultItem[]): ClassificationStats {
|
||||
const methodCounts: Record<string, number> = {};
|
||||
let confSum = 0;
|
||||
let confCount = 0;
|
||||
for (const item of list) {
|
||||
if (item.method) methodCounts[item.method] = (methodCounts[item.method] || 0) + 1;
|
||||
if (typeof item.confidence === "number") {
|
||||
confSum += item.confidence;
|
||||
confCount++;
|
||||
}
|
||||
}
|
||||
return { methodCounts, avgConfidence: confCount ? confSum / confCount : 0 };
|
||||
}
|
||||
|
||||
function printClassificationLine(label: string, stats: ClassificationStats, prevStats: ClassificationStats | undefined) {
|
||||
const methodStr = Object.entries(stats.methodCounts).map(([m, c]) => `${m}=${c}`).join(", ") || "n/a";
|
||||
const confStr = stats.avgConfidence ? stats.avgConfidence.toFixed(3) : "n/a";
|
||||
let suffix = "";
|
||||
if (prevStats && prevStats.avgConfidence) {
|
||||
const delta = fmtDelta(stats.avgConfidence, prevStats.avgConfidence);
|
||||
suffix = delta ? ` (Δ ${delta} vs prev)` : "";
|
||||
}
|
||||
console.log(` ${label.padEnd(11)}: ${methodStr.padEnd(28)} avg confidence ${confStr}${suffix}`);
|
||||
}
|
||||
|
||||
function printSummary(
|
||||
trainAgg: Record<Field, { correct: number; total: number }>,
|
||||
valAgg: Record<Field, { correct: number; total: number }>,
|
||||
trainClassStats: ClassificationStats,
|
||||
valClassStats: ClassificationStats,
|
||||
perImagePct: Record<string, number>,
|
||||
prev: HistoryEntry | null,
|
||||
imageCount: { training: number; validation: number },
|
||||
failedImages: string[]
|
||||
) {
|
||||
console.log("\n=== Product Scan Accuracy Summary ===");
|
||||
console.log(`Training Images: ${imageCount.training} | Validation Images: ${imageCount.validation} | Failed: ${failedImages.length}\n`);
|
||||
|
||||
const fieldCol = 14, numCol = 9, deltaCol = 8;
|
||||
const header =
|
||||
"Field".padEnd(fieldCol) +
|
||||
"Training".padStart(numCol) + "Δ".padStart(deltaCol) + " " +
|
||||
"Validation".padStart(numCol) + "Δ".padStart(deltaCol);
|
||||
console.log(header);
|
||||
console.log("-".repeat(header.length));
|
||||
|
||||
const overallTrain = { correct: 0, total: 0 };
|
||||
const overallVal = { correct: 0, total: 0 };
|
||||
|
||||
for (const field of FIELDS) {
|
||||
const t = trainAgg[field];
|
||||
const v = valAgg[field];
|
||||
overallTrain.correct += t.correct; overallTrain.total += t.total;
|
||||
overallVal.correct += v.correct; overallVal.total += v.total;
|
||||
|
||||
const tPct = pct(t.correct, t.total);
|
||||
const vPct = pct(v.correct, v.total);
|
||||
const tPrevStat = prev?.fields?.training?.[field];
|
||||
const vPrevStat = prev?.fields?.validation?.[field];
|
||||
const tPrevPct = tPrevStat?.total ? pct(tPrevStat.correct, tPrevStat.total) : undefined;
|
||||
const vPrevPct = vPrevStat?.total ? pct(vPrevStat.correct, vPrevStat.total) : undefined;
|
||||
|
||||
const tStr = t.total ? fmtPct(tPct) : "n/a";
|
||||
const vStr = v.total ? fmtPct(vPct) : "n/a";
|
||||
|
||||
console.log(
|
||||
field.padEnd(fieldCol) +
|
||||
tStr.padStart(numCol) + fmtDelta(tPct, tPrevPct).padStart(deltaCol) + " " +
|
||||
vStr.padStart(numCol) + fmtDelta(vPct, vPrevPct).padStart(deltaCol)
|
||||
);
|
||||
}
|
||||
console.log("-".repeat(header.length));
|
||||
|
||||
const tOverallPct = pct(overallTrain.correct, overallTrain.total);
|
||||
const vOverallPct = pct(overallVal.correct, overallVal.total);
|
||||
const prevTrainOverall = overallFromFields(prev?.fields?.training);
|
||||
const prevValOverall = overallFromFields(prev?.fields?.validation);
|
||||
const tOverallPrevPct = prevTrainOverall?.total ? pct(prevTrainOverall.correct, prevTrainOverall.total) : undefined;
|
||||
const vOverallPrevPct = prevValOverall?.total ? pct(prevValOverall.correct, prevValOverall.total) : undefined;
|
||||
|
||||
console.log(
|
||||
"OVERALL".padEnd(fieldCol) +
|
||||
fmtPct(tOverallPct).padStart(numCol) + fmtDelta(tOverallPct, tOverallPrevPct).padStart(deltaCol) + " " +
|
||||
fmtPct(vOverallPct).padStart(numCol) + fmtDelta(vOverallPct, vOverallPrevPct).padStart(deltaCol)
|
||||
);
|
||||
|
||||
if (failedImages.length) {
|
||||
console.log(`\nFailed to parse: ${failedImages.join(", ")}`);
|
||||
}
|
||||
|
||||
// Informational only - DINOv2 "confidence" is a raw cosine similarity, not a
|
||||
// calibrated probability (see docs/scan-product.md), so a delta here doesn't
|
||||
// by itself mean better/worse. Only the fields above drive regression flags.
|
||||
console.log("\nClassification (informational, not scored as pass/fail):");
|
||||
printClassificationLine("Training", trainClassStats, prev?.classification?.training);
|
||||
printClassificationLine("Validation", valClassStats, prev?.classification?.validation);
|
||||
|
||||
if (prev) {
|
||||
const fieldRegressions: string[] = [];
|
||||
const fieldImprovements: string[] = [];
|
||||
for (const split of ["training", "validation"] as Split[]) {
|
||||
const agg = split === "training" ? trainAgg : valAgg;
|
||||
for (const field of FIELDS) {
|
||||
const stat = agg[field];
|
||||
if (!stat.total) continue;
|
||||
const currPct = pct(stat.correct, stat.total);
|
||||
const prevStat = prev.fields?.[split]?.[field];
|
||||
if (!prevStat?.total) continue;
|
||||
const prevPct = pct(prevStat.correct, prevStat.total);
|
||||
const d = currPct - prevPct;
|
||||
const label = `${field} (${split})`;
|
||||
if (d <= -0.05) fieldRegressions.push(`${label} ${fmtDelta(currPct, prevPct)}`);
|
||||
else if (d >= 0.05) fieldImprovements.push(`${label} ${fmtDelta(currPct, prevPct)}`);
|
||||
}
|
||||
}
|
||||
if (fieldRegressions.length) console.log(`\nField regressions: ${fieldRegressions.join(", ")}`);
|
||||
if (fieldImprovements.length) console.log(`Field improvements: ${fieldImprovements.join(", ")}`);
|
||||
|
||||
const imageRegressions: string[] = [];
|
||||
const imageImprovements: string[] = [];
|
||||
for (const [filename, currPct] of Object.entries(perImagePct)) {
|
||||
const prevPct = prev.perImage?.[filename];
|
||||
if (prevPct === undefined) continue;
|
||||
const d = currPct - prevPct;
|
||||
if (d <= -0.5) imageRegressions.push(`${filename} ${fmtDelta(currPct, prevPct)}`);
|
||||
else if (d >= 0.5) imageImprovements.push(`${filename} ${fmtDelta(currPct, prevPct)}`);
|
||||
}
|
||||
if (imageRegressions.length) console.log(`\nImage regressions: ${imageRegressions.join(", ")}`);
|
||||
if (imageImprovements.length) console.log(`Image improvements: ${imageImprovements.join(", ")}`);
|
||||
|
||||
console.log(`\n(vs run at ${prev.timestamp}${prev.commit !== "unknown" ? `, commit ${prev.commit}` : ""})`);
|
||||
} else {
|
||||
console.log("\n(no previous run in product_accuracy_history.jsonl — this is the baseline)");
|
||||
}
|
||||
console.log("");
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const args = parseArgs(process.argv.slice(2));
|
||||
await checkServerReachable(args.baseUrl);
|
||||
@@ -116,12 +331,14 @@ async function main() {
|
||||
process.exit(1);
|
||||
}
|
||||
const labels: GroundTruth[] = JSON.parse(fs.readFileSync(LABELS_PATH, "utf8"));
|
||||
const prev = loadLastHistoryEntry();
|
||||
|
||||
const results = {
|
||||
training: [] as { gt: GroundTruth; checks: Check[] }[],
|
||||
validation: [] as { gt: GroundTruth; checks: Check[] }[],
|
||||
training: [] as ResultItem[],
|
||||
validation: [] as ResultItem[],
|
||||
failed: [] as string[]
|
||||
};
|
||||
const perImagePct: Record<string, number> = {};
|
||||
|
||||
for (const gt of labels) {
|
||||
if (gt.filename.startsWith("uploaded-")) continue; // Skip phantom
|
||||
@@ -135,7 +352,7 @@ async function main() {
|
||||
try {
|
||||
const b64 = "data:image/jpeg;base64," + fs.readFileSync(imgPath, "base64");
|
||||
const parsed = await fetchScan(args.baseUrl, b64);
|
||||
|
||||
|
||||
const bestMatchSku = parsed.possibleMatches?.find(m => m.isBestMatch)?.no_sku || "";
|
||||
const predictedItemName = parsed.classification?.top1_name || "";
|
||||
const predictedExpiry = parsed.ocr?.extracted_expired_date || "";
|
||||
@@ -146,13 +363,21 @@ async function main() {
|
||||
{ field: "expiry_date", match: isMatch(gt.expiry_date, predictedExpiry) }
|
||||
];
|
||||
|
||||
const item: ResultItem = {
|
||||
gt,
|
||||
checks,
|
||||
method: parsed.classification?.method,
|
||||
confidence: parsed.classification?.top1_confidence
|
||||
};
|
||||
|
||||
if (gt.filename.includes("/")) {
|
||||
results.training.push({ gt, checks });
|
||||
results.training.push(item);
|
||||
} else {
|
||||
results.validation.push({ gt, checks });
|
||||
results.validation.push(item);
|
||||
}
|
||||
|
||||
|
||||
const score = checks.filter(c => c.match).length;
|
||||
perImagePct[gt.filename] = (score / checks.length) * 100;
|
||||
console.log(`done (${score}/3)`);
|
||||
} catch (err) {
|
||||
console.log(`FAILED (${(err as Error).message})`);
|
||||
@@ -160,42 +385,22 @@ async function main() {
|
||||
}
|
||||
}
|
||||
|
||||
const aggregate = (list: { checks: Check[] }[]) => {
|
||||
const agg: Record<string, { correct: number; total: number }> = {
|
||||
no_sku: { correct: 0, total: 0 },
|
||||
nama_item: { correct: 0, total: 0 },
|
||||
expiry_date: { correct: 0, total: 0 }
|
||||
};
|
||||
for (const item of list) {
|
||||
for (const check of item.checks) {
|
||||
agg[check.field].total++;
|
||||
if (check.match) agg[check.field].correct++;
|
||||
}
|
||||
}
|
||||
return agg;
|
||||
};
|
||||
const trainAgg = aggregateFields(results.training);
|
||||
const valAgg = aggregateFields(results.validation);
|
||||
const trainClassStats = aggregateClassification(results.training);
|
||||
const valClassStats = aggregateClassification(results.validation);
|
||||
const imageCount = { training: results.training.length, validation: results.validation.length };
|
||||
|
||||
const trainAgg = aggregate(results.training);
|
||||
const valAgg = aggregate(results.validation);
|
||||
|
||||
console.log("\n=== Product Scan Accuracy Summary ===");
|
||||
console.log(`Training Images: ${results.training.length} | Validation Images: ${results.validation.length} | Failed: ${results.failed.length}\n`);
|
||||
|
||||
console.log("Field | Training Set | Validation Set");
|
||||
console.log("---------------|--------------|---------------");
|
||||
["no_sku", "nama_item", "expiry_date"].forEach(f => {
|
||||
const t = trainAgg[f].total ? fmtPct(pct(trainAgg[f].correct, trainAgg[f].total)) : "n/a";
|
||||
const v = valAgg[f].total ? fmtPct(pct(valAgg[f].correct, valAgg[f].total)) : "n/a";
|
||||
console.log(`${f.padEnd(14)} | ${t.padEnd(12)} | ${v.padEnd(14)}`);
|
||||
});
|
||||
console.log("");
|
||||
printSummary(trainAgg, valAgg, trainClassStats, valClassStats, perImagePct, prev, imageCount, results.failed);
|
||||
|
||||
const entry: HistoryEntry = {
|
||||
timestamp: new Date().toISOString(),
|
||||
commit: getGitCommit(),
|
||||
imageCount: { training: results.training.length, validation: results.validation.length },
|
||||
imageCount,
|
||||
failedImages: results.failed,
|
||||
fields: { training: trainAgg, validation: valAgg }
|
||||
fields: { training: trainAgg, validation: valAgg },
|
||||
classification: { training: trainClassStats, validation: valClassStats },
|
||||
perImage: perImagePct
|
||||
};
|
||||
fs.appendFileSync(HISTORY_PATH, JSON.stringify(entry) + "\n");
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user