Files
pfm-ocr/backend/pfm-web-app/scripts/accuracy-check.mts
T
fhanyuh caf8e98378 chore: normalize line endings (CRLF -> LF)
No content changes: git diff --ignore-all-space over these files is empty.
The churn came from editing on Windows against a repo checked out with LF.
2026-08-27 10:40:49 +07:00

667 lines
23 KiB
TypeScript

// OCR/parsing accuracy regression tool.
//
// Hits the live /api/parse endpoint for every image in backend/sources/test-images,
// diffs the result against backend/sources/manual_labels.json at all three
// post-processing stages (layer1RawRegex -> layer2Sanitized -> layer3Final) so a
// mismatch can be attributed to the stage that introduced it, and appends a summary
// to backend/sources/accuracy_history.jsonl for run-over-run regression tracking.
//
// Usage:
// node scripts/accuracy-check.mts # reuse cached OCR (fast)
// node scripts/accuracy-check.mts --refresh-ocr # force a fresh pipeline run for every image
// node scripts/accuracy-check.mts --detail <filename> # print full per-stage breakdown for one image
// node scripts/accuracy-check.mts --base-url http://localhost:3000
import fs from "node:fs";
import path from "node:path";
import { fileURLToPath } from "node:url";
import { execSync } from "node:child_process";
const __filename = fileURLToPath(import.meta.url);
const __dirname = path.dirname(__filename);
const APP_ROOT = path.join(__dirname, "..");
const SOURCES_DIR = path.join(APP_ROOT, "..", "sources");
const TEST_IMAGES_DIR = path.join(SOURCES_DIR, "test-images");
const LABELS_PATH = path.join(SOURCES_DIR, "manual_labels.json");
const HISTORY_PATH = path.join(SOURCES_DIR, "accuracy_history.jsonl");
// A fresh (non-cached) pipeline run does the whole layout+recognition pass twice for any
// image with >1deg tilt (see route.ts), and dense tables can push a single pass past 120s
// on modest hardware (one observed case took ~194s) - keep this comfortably above that.
const FETCH_TIMEOUT_MS = 300_000;
interface Item {
kodeBarang: string;
namaBarang: string;
banyak: string;
jumlah: string;
}
interface GroundTruth {
filename: string;
noPO: string;
noSO: string;
noDO: string;
tanggal: string;
customer: string;
store: string;
alamat: string;
plat: string;
items: Item[];
}
type StageName = "layer1RawRegex" | "layer2Sanitized" | "layer3Final";
const STAGES: StageName[] = ["layer1RawRegex", "layer2Sanitized", "layer3Final"];
const STAGE_LABEL: Record<StageName, string> = {
layer1RawRegex: "L1raw",
layer2Sanitized: "L2san",
layer3Final: "L3final",
};
// Header fields present at every stage (untouched or normalized by parseDOMetadata/sanitizeParsedMetadata).
const HEADER_FIELD_MAP: [gtKey: keyof GroundTruth, stageKey: string][] = [
["noPO", "noPO"],
["noSO", "noSO"],
["noDO", "noDO"],
["tanggal", "tanggal"],
["customer", "customerInfo"],
["plat", "platTruk"],
];
// Only resolved on layer3Final (store/address lookup against the store master DB happens after sanitize).
const STORE_FIELD_MAP: [gtKey: keyof GroundTruth, stageKey: string][] = [
["store", "orderUntuk"],
["alamat", "alamat"],
];
const ITEM_FIELDS: (keyof Item)[] = ["kodeBarang", "namaBarang", "banyak", "jumlah"];
const FIELD_ORDER = [
"noPO", "noSO", "noDO", "tanggal", "customer", "plat", "store", "alamat",
"itemCount", "kodeBarang", "namaBarang", "banyak", "jumlah",
];
interface Check {
field: string;
stage: StageName;
match: boolean;
}
interface DetailFieldRow {
field: string;
gt: string;
values: Record<StageName, string | null>;
matches: Record<StageName, boolean | null>;
}
interface DetailItemRow {
rowIndex: number;
field: string;
gt: string;
values: Record<StageName, string | null>;
matches: Record<StageName, boolean>;
}
interface ImageDetail {
headerRows: DetailFieldRow[];
itemRows: DetailItemRow[];
}
interface HistoryEntry {
timestamp: string;
commit: string;
refreshOcr: boolean;
imageCount: number;
failedImages: string[];
overall: Record<StageName, number>;
fields: Record<string, Record<StageName, number>>;
perImage: Record<string, number>;
}
interface CliArgs {
refreshOcr: boolean;
detail: string | null;
baseUrl: string;
dumpJson: string | null;
}
function parseArgs(argv: string[]): CliArgs {
const args: CliArgs = {
refreshOcr: false,
detail: null,
baseUrl: process.env.ACCURACY_BASE_URL || "http://localhost:3000",
dumpJson: null,
};
for (let i = 0; i < argv.length; i++) {
const a = argv[i];
if (a === "--refresh-ocr") {
args.refreshOcr = true;
} else if (a === "--detail") {
args.detail = argv[++i] ?? null;
} else if (a === "--base-url") {
args.baseUrl = argv[++i] ?? args.baseUrl;
} else if (a === "--dump-json") {
args.dumpJson = argv[++i] ?? null;
} else if (a === "--help" || a === "-h") {
printHelp();
process.exit(0);
}
}
return args;
}
function printHelp() {
console.log(`OCR accuracy regression tool
Usage:
node scripts/accuracy-check.mts [options]
Options:
--refresh-ocr Force a fresh pipeline run for every test image (requires the
pipeline-api service to be up), instead of reusing cached OCR.
--detail <filename> Print the full per-stage breakdown for one image.
--base-url <url> Base URL of the running Next dev server (default http://localhost:3000).
--dump-json <path> Write full per-image ground-truth-vs-AI detail (all stages) as JSON.
--help Show this message.
`);
}
function norm(v: unknown): string {
if (v === null || v === undefined) return "";
// Spacing around punctuation ("PT. PRIMAFOOD" vs "PT.PRIMAFOOD") is a formatting
// difference, not an extraction error - the ground-truth labels themselves are
// inconsistent about it, so neutralize it before comparing.
return String(v)
.replace(/\s*([.,:;\/])\s*/g, "$1")
.replace(/\s+/g, " ")
.trim()
.toUpperCase();
}
function isMatch(a: unknown, b: unknown): boolean {
return norm(a) === norm(b);
}
function loadGroundTruth(): Map<string, GroundTruth> {
const raw = fs.existsSync(LABELS_PATH) ? fs.readFileSync(LABELS_PATH, "utf8") : "";
const arr: GroundTruth[] = raw.trim() ? JSON.parse(raw) : [];
const map = new Map<string, GroundTruth>();
for (const entry of arr) map.set(entry.filename, entry);
return map;
}
function listTestImages(): string[] {
if (!fs.existsSync(TEST_IMAGES_DIR)) return [];
return fs
.readdirSync(TEST_IMAGES_DIR)
.filter(f => [".jpg", ".jpeg", ".png"].includes(path.extname(f).toLowerCase()))
.sort();
}
async function checkServerReachable(baseUrl: string): Promise<void> {
try {
const res = await fetchWithTimeout(`${baseUrl}/api/manual-images`, {}, 10_000);
if (!res.ok) throw new Error(`HTTP ${res.status}`);
} catch (err) {
throw new Error(
`Next dev server not reachable at ${baseUrl} (${(err as Error).message}). ` +
`Start it with "npm run dev" in backend/pfm-web-app.`
);
}
}
async function fetchWithTimeout(url: string, init: RequestInit, timeoutMs: number): Promise<Response> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
try {
return await fetch(url, { ...init, signal: controller.signal });
} finally {
clearTimeout(timer);
}
}
// Store name/address are assigned per-account now, not OCR-detected (see
// v1/documents/upload/route.ts + /api/parse). To score store/alamat fields
// the same way production behaves, simulate "the account assigned to this
// image's ground-truth store uploaded it" by resolving gt.store -> kode_toko
// and passing that through, same as a real account's JWT would.
async function fetchStoreNameToKodeMap(): Promise<Map<string, string>> {
const { Pool } = await import("pg");
const pool = new Pool({
host: process.env.PGHOST || "localhost",
port: parseInt(process.env.PGPORT || "5432", 10),
user: process.env.PGUSER || "postgres",
password: process.env.PGPASSWORD || "postgres",
database: process.env.PGDATABASE || "dopfm",
});
try {
const res = await pool.query("SELECT nama_toko, kode_toko FROM store_master");
return new Map(res.rows.map((r: any) => [r.nama_toko, r.kode_toko]));
} finally {
await pool.end();
}
}
async function fetchParse(baseUrl: string, filename: string, kodeToko?: string): Promise<any> {
const res = await fetchWithTimeout(
`${baseUrl}/api/parse`,
{
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ filename, kodeToko }),
},
FETCH_TIMEOUT_MS
);
const json = await res.json();
if (!res.ok) {
throw new Error(json?.error || `HTTP ${res.status}`);
}
if (!json.postProcessingDetails) {
throw new Error("Response missing postProcessingDetails (parse route may have hit its DB-save fallback)");
}
return json;
}
async function refreshOcrCache(filenames: string[]): Promise<void> {
const { Pool } = await import("pg");
const pool = new Pool({
host: process.env.PGHOST || "localhost",
port: parseInt(process.env.PGPORT || "5432", 10),
user: process.env.PGUSER || "postgres",
password: process.env.PGPASSWORD || "postgres",
database: process.env.PGDATABASE || "dopfm",
});
try {
for (const filename of filenames) {
await pool.query("DELETE FROM documents WHERE filename = $1", [filename]);
}
} finally {
await pool.end();
}
}
function getStageObj(parsed: any, stage: StageName): any {
return parsed?.postProcessingDetails?.[stage] || {};
}
function compareImage(gt: GroundTruth, parsed: any): { checks: Check[]; detail: ImageDetail } {
const checks: Check[] = [];
const headerRows: DetailFieldRow[] = [];
for (const [gtKey, stageKey] of HEADER_FIELD_MAP) {
const gtVal = String(gt[gtKey] ?? "");
const row: DetailFieldRow = { field: gtKey, gt: gtVal, values: {} as any, matches: {} as any };
for (const stage of STAGES) {
const obj = getStageObj(parsed, stage);
const val = obj?.[stageKey] ?? null;
const match = isMatch(gtVal, val);
checks.push({ field: gtKey, stage, match });
row.values[stage] = val;
row.matches[stage] = match;
}
headerRows.push(row);
}
for (const [gtKey, stageKey] of STORE_FIELD_MAP) {
const gtVal = String(gt[gtKey] ?? "");
const row: DetailFieldRow = { field: gtKey, gt: gtVal, values: {} as any, matches: {} as any };
for (const stage of STAGES) {
const obj = getStageObj(parsed, stage);
if (!(stageKey in obj)) {
// Not resolved yet at this stage (store/address lookup only runs once, for layer3Final).
row.values[stage] = null;
row.matches[stage] = null;
continue;
}
const val = obj[stageKey];
const match = isMatch(gtVal, val);
checks.push({ field: gtKey, stage, match });
row.values[stage] = val;
row.matches[stage] = match;
}
headerRows.push(row);
}
const gtCount = gt.items.length;
const countRow: DetailFieldRow = { field: "itemCount", gt: String(gtCount), values: {} as any, matches: {} as any };
for (const stage of STAGES) {
const obj = getStageObj(parsed, stage);
const items: Item[] = obj?.items || [];
const match = items.length === gtCount;
checks.push({ field: "itemCount", stage, match });
countRow.values[stage] = String(items.length);
countRow.matches[stage] = match;
}
headerRows.push(countRow);
const itemRows: DetailItemRow[] = [];
for (let i = 0; i < gtCount; i++) {
const gtItem = gt.items[i];
for (const field of ITEM_FIELDS) {
const rowDetail: DetailItemRow = {
rowIndex: i,
field,
gt: String(gtItem[field] ?? ""),
values: {} as any,
matches: {} as any,
};
for (const stage of STAGES) {
const obj = getStageObj(parsed, stage);
const items: Item[] = obj?.items || [];
const stageItem = items[i];
const val = stageItem ? stageItem[field] : null;
const match = !!stageItem && isMatch(gtItem[field], val);
checks.push({ field, stage, match });
rowDetail.values[stage] = val ?? null;
rowDetail.matches[stage] = match;
}
itemRows.push(rowDetail);
}
}
return { checks, detail: { headerRows, itemRows } };
}
function pct(correct: number, total: number): number {
return total === 0 ? 0 : (correct / total) * 100;
}
function aggregateByField(allChecks: Check[]): Record<string, Record<StageName, { correct: number; total: number }>> {
const agg: Record<string, Record<StageName, { correct: number; total: number }>> = {};
for (const c of allChecks) {
if (!agg[c.field]) {
agg[c.field] = {
layer1RawRegex: { correct: 0, total: 0 },
layer2Sanitized: { correct: 0, total: 0 },
layer3Final: { correct: 0, total: 0 },
};
}
agg[c.field][c.stage].total++;
if (c.match) agg[c.field][c.stage].correct++;
}
return agg;
}
function aggregateOverall(allChecks: Check[]): Record<StageName, { correct: number; total: number }> {
const overall: Record<StageName, { correct: number; total: number }> = {
layer1RawRegex: { correct: 0, total: 0 },
layer2Sanitized: { correct: 0, total: 0 },
layer3Final: { correct: 0, total: 0 },
};
for (const c of allChecks) {
overall[c.stage].total++;
if (c.match) overall[c.stage].correct++;
}
return overall;
}
function getGitCommit(): string {
try {
return execSync("git rev-parse --short HEAD", { cwd: APP_ROOT }).toString().trim();
} catch {
return "unknown";
}
}
function loadLastHistoryEntry(): HistoryEntry | null {
if (!fs.existsSync(HISTORY_PATH)) return null;
const lines = fs.readFileSync(HISTORY_PATH, "utf8").trim().split("\n").filter(Boolean);
if (lines.length === 0) return null;
try {
return JSON.parse(lines[lines.length - 1]);
} catch {
return null;
}
}
function appendHistory(entry: HistoryEntry) {
fs.appendFileSync(HISTORY_PATH, JSON.stringify(entry) + "\n", "utf8");
}
function fmtPct(n: number): string {
return `${n.toFixed(1)}%`;
}
function fmtDelta(curr: number, prev: number | undefined): string {
if (prev === undefined) return "";
const d = curr - prev;
if (Math.abs(d) < 0.05) return " ±0.0";
const sign = d > 0 ? "+" : "";
return `${sign}${d.toFixed(1)}`;
}
function printSummary(
fieldAgg: Record<string, Record<StageName, { correct: number; total: number }>>,
overallAgg: Record<StageName, { correct: number; total: number }>,
perImagePct: Record<string, number>,
prev: HistoryEntry | null,
imageCount: number,
failedImages: string[]
) {
console.log("");
console.log(`OCR accuracy summary (${imageCount} images${failedImages.length ? `, ${failedImages.length} failed` : ""})`);
console.log("");
const fieldCol = 12;
const numCol = 9;
const header =
"Field".padEnd(fieldCol) +
STAGES.map(s => STAGE_LABEL[s].padStart(numCol)).join("") +
" Δ(L3 vs prev)".padStart(16);
console.log(header);
console.log("-".repeat(header.length));
for (const field of FIELD_ORDER) {
const stat = fieldAgg[field];
if (!stat) continue;
const cells = STAGES.map(s => {
const st = stat[s];
return st.total === 0 ? "n/a".padStart(numCol) : fmtPct(pct(st.correct, st.total)).padStart(numCol);
}).join("");
const currL3 = pct(stat.layer3Final.correct, stat.layer3Final.total);
const prevL3 = prev?.fields?.[field]?.layer3Final;
const delta = fmtDelta(currL3, prevL3);
console.log(field.padEnd(fieldCol) + cells + delta.padStart(16));
}
console.log("-".repeat(header.length));
const overallCells = STAGES.map(s => fmtPct(pct(overallAgg[s].correct, overallAgg[s].total)).padStart(numCol)).join("");
const overallL3 = pct(overallAgg.layer3Final.correct, overallAgg.layer3Final.total);
const overallDelta = fmtDelta(overallL3, prev?.overall?.layer3Final);
console.log("OVERALL".padEnd(fieldCol) + overallCells + overallDelta.padStart(16));
if (failedImages.length) {
console.log("");
console.log(`Failed to parse: ${failedImages.join(", ")}`);
}
if (prev) {
const fieldRegressions: string[] = [];
const fieldImprovements: string[] = [];
for (const field of FIELD_ORDER) {
const stat = fieldAgg[field];
if (!stat) continue;
const currL3 = pct(stat.layer3Final.correct, stat.layer3Final.total);
const prevL3 = prev.fields?.[field]?.layer3Final;
if (prevL3 === undefined) continue;
const d = currL3 - prevL3;
if (d <= -0.05) fieldRegressions.push(`${field} ${fmtDelta(currL3, prevL3)}`);
else if (d >= 0.05) fieldImprovements.push(`${field} ${fmtDelta(currL3, prevL3)}`);
}
if (fieldRegressions.length) console.log(`\nField regressions (L3): ${fieldRegressions.join(", ")}`);
if (fieldImprovements.length) console.log(`Field improvements (L3): ${fieldImprovements.join(", ")}`);
const imageRegressions: string[] = [];
const imageImprovements: string[] = [];
for (const [filename, currPct] of Object.entries(perImagePct)) {
const prevPct = prev.perImage?.[filename];
if (prevPct === undefined) continue;
const d = currPct - prevPct;
if (d <= -0.5) imageRegressions.push(`${filename} ${fmtDelta(currPct, prevPct)}`);
else if (d >= 0.5) imageImprovements.push(`${filename} ${fmtDelta(currPct, prevPct)}`);
}
if (imageRegressions.length) console.log(`\nImage regressions: ${imageRegressions.join(", ")}`);
if (imageImprovements.length) console.log(`Image improvements: ${imageImprovements.join(", ")}`);
console.log(`\n(vs run at ${prev.timestamp}${prev.commit !== "unknown" ? `, commit ${prev.commit}` : ""})`);
} else {
console.log("\n(no previous run in accuracy_history.jsonl — this is the baseline)");
}
console.log("");
}
function checkMark(v: boolean | null): string {
if (v === null) return "-";
return v ? "✓" : "✗";
}
function printDetail(filename: string, detail: ImageDetail | undefined) {
console.log(`\n=== Detail: ${filename} ===\n`);
if (!detail) {
console.log("No data for this file (it may not exist in test-images/ or manual_labels.json, or parsing failed this run).\n");
return;
}
const fieldCol = 12;
const valCol = 34;
console.log("Field".padEnd(fieldCol) + STAGES.map(s => STAGE_LABEL[s].padEnd(valCol)).join(""));
console.log(`(ground truth shown per row)`);
for (const row of detail.headerRows) {
console.log(`- ${row.field} = "${row.gt}"`);
for (const stage of STAGES) {
const val = row.values[stage];
const mark = checkMark(row.matches[stage]);
const label = STAGE_LABEL[stage].padEnd(8);
console.log(` ${mark} ${label} ${val === null ? "(n/a)" : `"${val}"`}`);
}
}
console.log("\nItems:");
let currentRow = -1;
for (const row of detail.itemRows) {
if (row.rowIndex !== currentRow) {
currentRow = row.rowIndex;
console.log(` Row ${currentRow + 1}:`);
}
console.log(` - ${row.field} = "${row.gt}"`);
for (const stage of STAGES) {
const val = row.values[stage];
const mark = checkMark(row.matches[stage]);
const label = STAGE_LABEL[stage].padEnd(8);
console.log(` ${mark} ${label} ${val === null ? "(missing row)" : `"${val}"`}`);
}
}
console.log("");
}
async function main() {
const args = parseArgs(process.argv.slice(2));
await checkServerReachable(args.baseUrl);
const gtMap = loadGroundTruth();
const testImages = listTestImages();
const storeNameToKode = await fetchStoreNameToKodeMap();
if (testImages.length === 0) {
console.error(`No test images found in ${TEST_IMAGES_DIR}`);
process.exit(1);
}
const missingGt = testImages.filter(f => !gtMap.has(f));
if (missingGt.length) {
console.warn(`Warning: ${missingGt.length} test image(s) have no ground truth entry and will be skipped: ${missingGt.join(", ")}`);
}
if (args.refreshOcr) {
console.log(`--refresh-ocr: clearing cached OCR for ${testImages.length} images (requires pipeline-api to be reachable)...`);
await refreshOcrCache(testImages);
}
const allChecks: Check[] = [];
const perImagePct: Record<string, number> = {};
const detailsByFile: Record<string, ImageDetail> = {};
const failedImages: string[] = [];
for (const filename of testImages) {
const gt = gtMap.get(filename);
if (!gt) continue;
process.stdout.write(`Parsing ${filename}... `);
try {
const kodeToko = storeNameToKode.get(gt.store);
const parsed = await fetchParse(args.baseUrl, filename, kodeToko);
const { checks, detail } = compareImage(gt, parsed);
allChecks.push(...checks);
detailsByFile[filename] = detail;
const l3Checks = checks.filter(c => c.stage === "layer3Final");
const imgPct = pct(l3Checks.filter(c => c.match).length, l3Checks.length);
perImagePct[filename] = imgPct;
console.log(`done (${imgPct.toFixed(0)}%)`);
} catch (err) {
console.log(`FAILED (${(err as Error).message})`);
failedImages.push(filename);
}
}
const fieldAgg = aggregateByField(allChecks);
const overallAgg = aggregateOverall(allChecks);
const prev = loadLastHistoryEntry();
printSummary(fieldAgg, overallAgg, perImagePct, prev, testImages.length - missingGt.length, failedImages);
if (args.detail) {
printDetail(args.detail, detailsByFile[args.detail]);
}
if (args.dumpJson) {
fs.writeFileSync(
args.dumpJson,
JSON.stringify(
{
timestamp: new Date().toISOString(),
commit: getGitCommit(),
imageCount: testImages.length - missingGt.length,
overallL3: pct(overallAgg.layer3Final.correct, overallAgg.layer3Final.total),
perImagePct,
details: detailsByFile,
},
null,
2
),
"utf8"
);
console.log(`\nWrote full per-image detail to ${args.dumpJson}`);
}
const historyEntry: HistoryEntry = {
timestamp: new Date().toISOString(),
commit: getGitCommit(),
refreshOcr: args.refreshOcr,
imageCount: testImages.length - missingGt.length,
failedImages,
overall: {
layer1RawRegex: pct(overallAgg.layer1RawRegex.correct, overallAgg.layer1RawRegex.total),
layer2Sanitized: pct(overallAgg.layer2Sanitized.correct, overallAgg.layer2Sanitized.total),
layer3Final: pct(overallAgg.layer3Final.correct, overallAgg.layer3Final.total),
},
fields: Object.fromEntries(
Object.entries(fieldAgg).map(([field, stat]) => [
field,
{
layer1RawRegex: pct(stat.layer1RawRegex.correct, stat.layer1RawRegex.total),
layer2Sanitized: pct(stat.layer2Sanitized.correct, stat.layer2Sanitized.total),
layer3Final: pct(stat.layer3Final.correct, stat.layer3Final.total),
},
])
),
perImage: perImagePct,
};
appendHistory(historyEntry);
}
main().catch(err => {
console.error("Fatal error:", err);
process.exit(1);
});