chore: normalize line endings (CRLF -> LF)
No content changes: git diff --ignore-all-space over these files is empty. The churn came from editing on Windows against a repo checked out with LF.
This commit is contained in:
1 parent
15566a6951
commit
caf8e98378
315 files changed
+86950
-86950
No files matched your search
@@ -1,244 +1,244 @@
|
||||
import { query } from "../db";
|
||||
|
||||
// Bounds the classifier call so a wedged GPU container fails fast instead of
|
||||
// hanging indefinitely. Raised from 90s (2026-07-14): hard images now
|
||||
// legitimately take up to ~3 min - a 4-orientation OCR search plus a VL
|
||||
// pipeline fallback when no expiry date is found (see
|
||||
// config/classify_ocr_server.py) - and the old bound was killing exactly
|
||||
// the images those fallbacks exist to save.
|
||||
const PIPELINE_TIMEOUT_MS = 240_000;
|
||||
|
||||
// Thrown when the Python classifier service itself returns a non-2xx response,
|
||||
// so callers can forward its actual status instead of collapsing everything to 500.
|
||||
export class ClassifierError extends Error {
|
||||
status: number;
|
||||
constructor(status: number, message: string) {
|
||||
super(message);
|
||||
this.status = status;
|
||||
}
|
||||
}
|
||||
|
||||
export interface SkuMatch {
|
||||
no_sku: string;
|
||||
nama_item: string;
|
||||
score: number;
|
||||
yoloSimilarity: number;
|
||||
isBestMatch: boolean;
|
||||
}
|
||||
|
||||
export interface ProductScanResult {
|
||||
classification: any;
|
||||
ocr: any;
|
||||
possibleMatches: SkuMatch[];
|
||||
}
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
const len1 = s1.length;
|
||||
const len2 = s2.length;
|
||||
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
|
||||
|
||||
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
|
||||
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
|
||||
|
||||
for (let i = 1; i <= len1; i++) {
|
||||
for (let j = 1; j <= len2; j++) {
|
||||
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
||||
matrix[i][j] = Math.min(
|
||||
matrix[i - 1][j] + 1, // deletion
|
||||
matrix[i][j - 1] + 1, // insertion
|
||||
matrix[i - 1][j - 1] + cost // substitution
|
||||
);
|
||||
}
|
||||
}
|
||||
return matrix[len1][len2];
|
||||
}
|
||||
|
||||
function getStringSimilarity(s1: string, s2: string): number {
|
||||
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
if (!clean1 || !clean2) return 0;
|
||||
const distance = levenshteinDistance(clean1, clean2);
|
||||
const maxLength = Math.max(clean1.length, clean2.length);
|
||||
return (maxLength - distance) / maxLength;
|
||||
}
|
||||
|
||||
// --- OCR-evidence re-ranking of the classifier's top-K candidates ---
|
||||
//
|
||||
// DINOv2's misses are near-twin confusions (same brand line, different
|
||||
// flavor/size) - exactly the cases where the printed variant words differ,
|
||||
// and PaddleOCR usually reads some of them. Within a narrow similarity band
|
||||
// of the top-1 candidate, prefer the one whose distinctive name tokens
|
||||
// actually appear in the OCR'd text. Coverage-normalized so generic
|
||||
// packaging words (e.g. "French Fries", "Ayam") that happen to be unique to
|
||||
// one candidate's *name* can't hijack the ranking. Parameters tuned offline
|
||||
// against the 79-image validation set (scripts/experiment-rerank.mjs,
|
||||
// 2026-07-14: fixes 8 of 18 top-1 misses, breaks 0 of 61 correct).
|
||||
const RERANK_TOP_K = 12;
|
||||
const RERANK_SIM_BAND = 0.12;
|
||||
const RERANK_COVERAGE_MARGIN = 0.25;
|
||||
|
||||
function classNameSku(className: string): string {
|
||||
// foto-kemasan-v2 class names are "<SKU> <NAME...>"
|
||||
return (className || "").trim().split(/\s+/)[0] || "";
|
||||
}
|
||||
|
||||
function tokenizeName(name: string): string[] {
|
||||
return name.toUpperCase().split(/[^A-Z0-9]+/).filter(t => t.length >= 2);
|
||||
}
|
||||
|
||||
function withinEditDistance1(a: string, b: string): boolean {
|
||||
if (a === b) return true;
|
||||
const la = a.length, lb = b.length;
|
||||
if (Math.abs(la - lb) > 1) return false;
|
||||
if (la === lb) {
|
||||
let diff = 0;
|
||||
for (let i = 0; i < la; i++) if (a[i] !== b[i]) diff++;
|
||||
return diff <= 1;
|
||||
}
|
||||
const [s, l] = la < lb ? [a, b] : [b, a];
|
||||
let i = 0, j = 0, skipped = false;
|
||||
while (i < s.length && j < l.length) {
|
||||
if (s[i] === l[j]) { i++; j++; }
|
||||
else if (!skipped) { skipped = true; j++; }
|
||||
else return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
interface OcrTextIndex { squashed: string; tokens: Set<string>; }
|
||||
|
||||
function buildOcrTextIndex(textLines: string[]): OcrTextIndex {
|
||||
const joined = textLines.join(" ").toUpperCase();
|
||||
return {
|
||||
squashed: joined.replace(/[^A-Z0-9]/g, ""),
|
||||
tokens: new Set(tokenizeName(joined))
|
||||
};
|
||||
}
|
||||
|
||||
function tokenFoundInOcr(token: string, ocr: OcrTextIndex): boolean {
|
||||
if (token.length >= 4 && ocr.squashed.includes(token)) return true;
|
||||
if (ocr.tokens.has(token)) return true;
|
||||
if (token.length >= 5) {
|
||||
for (const t of ocr.tokens) {
|
||||
if (Math.abs(t.length - token.length) <= 1 && withinEditDistance1(token, t)) return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Returns the class name of the best candidate after OCR-evidence
|
||||
// re-ranking (the classifier's top-1 unless a close band-mate has clearly
|
||||
// stronger printed-text evidence).
|
||||
function rerankClassCandidates(
|
||||
allProbabilities: Array<{ name: string; confidence: number }>,
|
||||
textLines: string[]
|
||||
): string {
|
||||
if (!allProbabilities.length) return "";
|
||||
const top1Sim = allProbabilities[0].confidence;
|
||||
const band = allProbabilities
|
||||
.slice(0, RERANK_TOP_K)
|
||||
.filter(p => p.confidence >= top1Sim - RERANK_SIM_BAND);
|
||||
if (band.length <= 1 || !textLines.length) return allProbabilities[0].name;
|
||||
|
||||
const ocrIdx = buildOcrTextIndex(textLines);
|
||||
const cands = band.map(p => {
|
||||
const sku = classNameSku(p.name);
|
||||
return { name: p.name, tokens: new Set(tokenizeName(p.name.replace(sku, ""))), coverage: 0 };
|
||||
});
|
||||
const tokenCounts = new Map<string, number>();
|
||||
for (const c of cands) {
|
||||
for (const tok of c.tokens) tokenCounts.set(tok, (tokenCounts.get(tok) || 0) + 1);
|
||||
}
|
||||
for (const c of cands) {
|
||||
let matched = 0, total = 0;
|
||||
for (const tok of c.tokens) {
|
||||
const nWith = tokenCounts.get(tok) || 1;
|
||||
if (nWith >= cands.length) continue; // shared by all band-mates -> no signal
|
||||
const w = 1 / nWith;
|
||||
total += w;
|
||||
if (tokenFoundInOcr(tok, ocrIdx)) matched += w;
|
||||
}
|
||||
c.coverage = total > 0 ? matched / total : 0;
|
||||
}
|
||||
|
||||
let chosen = cands[0];
|
||||
for (const c of cands.slice(1)) {
|
||||
if (c.coverage >= chosen.coverage + RERANK_COVERAGE_MARGIN) chosen = c;
|
||||
}
|
||||
if (chosen !== cands[0]) {
|
||||
console.log(`[Rerank] OCR evidence overrode classifier top-1 "${cands[0].name}" -> "${chosen.name}" (coverage ${cands[0].coverage.toFixed(2)} vs ${chosen.coverage.toFixed(2)})`);
|
||||
}
|
||||
return chosen.name;
|
||||
}
|
||||
|
||||
// Shared by the classic /api/scan-pfm dev route and the authenticated
|
||||
// /api/v1/scan-product route: calls the Python classifier, then matches the
|
||||
// result against sku_master, returning the top-5 candidates.
|
||||
export async function classifyAndMatchProduct(imageBase64: string): Promise<ProductScanResult> {
|
||||
const pyServerUrl = process.env.CLASSIFIER_SERVER_URL || "http://paddleocr-pipeline-api:8120/classify-ocr";
|
||||
|
||||
const response = await fetch(pyServerUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ image_base64: imageBase64 }),
|
||||
signal: AbortSignal.timeout(PIPELINE_TIMEOUT_MS)
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errText = await response.text();
|
||||
throw new ClassifierError(response.status, `Classifier service error: ${errText}`);
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
|
||||
const dbRes = await query("SELECT no_sku, nama_item FROM sku_master");
|
||||
const skuMasterList = dbRes.rows.map(row => ({
|
||||
no_sku: row.no_sku,
|
||||
nama_item: row.nama_item
|
||||
}));
|
||||
|
||||
const extractedSku = data.ocr?.extracted_sku || "";
|
||||
|
||||
// Re-rank the classifier's close candidates using OCR'd package text, then
|
||||
// map the winner straight to its sku_master row by the SKU prefix embedded
|
||||
// in the class name. The old approach (Levenshtein between top-1 class name
|
||||
// and every master nama_item) lost classifier-correct results whenever a
|
||||
// *different* SKU's master name happened to be textually closer.
|
||||
const rerankedName = rerankClassCandidates(
|
||||
data.classification?.all_probabilities || [],
|
||||
data.ocr?.text_lines || []
|
||||
) || data.classification?.top1_name || "";
|
||||
const rerankedSku = classNameSku(rerankedName);
|
||||
|
||||
const matchedList: SkuMatch[] = skuMasterList.map(sku => {
|
||||
const yoloSim = rerankedName ? getStringSimilarity(sku.nama_item, rerankedName) : 0;
|
||||
|
||||
const cleanMasterSku = sku.no_sku.trim();
|
||||
const cleanExtractedSku = extractedSku.trim();
|
||||
const isSkuMatch = cleanExtractedSku && cleanMasterSku === cleanExtractedSku;
|
||||
const isClassifierPick = rerankedSku && cleanMasterSku === rerankedSku;
|
||||
|
||||
const score = isSkuMatch ? 1.0 : isClassifierPick ? 0.995 : yoloSim;
|
||||
|
||||
return {
|
||||
no_sku: sku.no_sku,
|
||||
nama_item: sku.nama_item,
|
||||
score,
|
||||
yoloSimilarity: yoloSim,
|
||||
isBestMatch: false
|
||||
};
|
||||
});
|
||||
|
||||
matchedList.sort((a, b) => b.score - a.score);
|
||||
|
||||
const possibleMatches = matchedList.slice(0, 5).filter(m => m.score > 0.1);
|
||||
if (possibleMatches.length > 0) {
|
||||
possibleMatches[0].isBestMatch = true;
|
||||
}
|
||||
|
||||
return {
|
||||
classification: data.classification,
|
||||
ocr: data.ocr,
|
||||
possibleMatches
|
||||
};
|
||||
}
|
||||
import { query } from "../db";
|
||||
|
||||
// Bounds the classifier call so a wedged GPU container fails fast instead of
|
||||
// hanging indefinitely. Raised from 90s (2026-07-14): hard images now
|
||||
// legitimately take up to ~3 min - a 4-orientation OCR search plus a VL
|
||||
// pipeline fallback when no expiry date is found (see
|
||||
// config/classify_ocr_server.py) - and the old bound was killing exactly
|
||||
// the images those fallbacks exist to save.
|
||||
const PIPELINE_TIMEOUT_MS = 240_000;
|
||||
|
||||
// Thrown when the Python classifier service itself returns a non-2xx response,
|
||||
// so callers can forward its actual status instead of collapsing everything to 500.
|
||||
export class ClassifierError extends Error {
|
||||
status: number;
|
||||
constructor(status: number, message: string) {
|
||||
super(message);
|
||||
this.status = status;
|
||||
}
|
||||
}
|
||||
|
||||
export interface SkuMatch {
|
||||
no_sku: string;
|
||||
nama_item: string;
|
||||
score: number;
|
||||
yoloSimilarity: number;
|
||||
isBestMatch: boolean;
|
||||
}
|
||||
|
||||
export interface ProductScanResult {
|
||||
classification: any;
|
||||
ocr: any;
|
||||
possibleMatches: SkuMatch[];
|
||||
}
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
const len1 = s1.length;
|
||||
const len2 = s2.length;
|
||||
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
|
||||
|
||||
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
|
||||
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
|
||||
|
||||
for (let i = 1; i <= len1; i++) {
|
||||
for (let j = 1; j <= len2; j++) {
|
||||
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
||||
matrix[i][j] = Math.min(
|
||||
matrix[i - 1][j] + 1, // deletion
|
||||
matrix[i][j - 1] + 1, // insertion
|
||||
matrix[i - 1][j - 1] + cost // substitution
|
||||
);
|
||||
}
|
||||
}
|
||||
return matrix[len1][len2];
|
||||
}
|
||||
|
||||
function getStringSimilarity(s1: string, s2: string): number {
|
||||
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
if (!clean1 || !clean2) return 0;
|
||||
const distance = levenshteinDistance(clean1, clean2);
|
||||
const maxLength = Math.max(clean1.length, clean2.length);
|
||||
return (maxLength - distance) / maxLength;
|
||||
}
|
||||
|
||||
// --- OCR-evidence re-ranking of the classifier's top-K candidates ---
|
||||
//
|
||||
// DINOv2's misses are near-twin confusions (same brand line, different
|
||||
// flavor/size) - exactly the cases where the printed variant words differ,
|
||||
// and PaddleOCR usually reads some of them. Within a narrow similarity band
|
||||
// of the top-1 candidate, prefer the one whose distinctive name tokens
|
||||
// actually appear in the OCR'd text. Coverage-normalized so generic
|
||||
// packaging words (e.g. "French Fries", "Ayam") that happen to be unique to
|
||||
// one candidate's *name* can't hijack the ranking. Parameters tuned offline
|
||||
// against the 79-image validation set (scripts/experiment-rerank.mjs,
|
||||
// 2026-07-14: fixes 8 of 18 top-1 misses, breaks 0 of 61 correct).
|
||||
const RERANK_TOP_K = 12;
|
||||
const RERANK_SIM_BAND = 0.12;
|
||||
const RERANK_COVERAGE_MARGIN = 0.25;
|
||||
|
||||
function classNameSku(className: string): string {
|
||||
// foto-kemasan-v2 class names are "<SKU> <NAME...>"
|
||||
return (className || "").trim().split(/\s+/)[0] || "";
|
||||
}
|
||||
|
||||
function tokenizeName(name: string): string[] {
|
||||
return name.toUpperCase().split(/[^A-Z0-9]+/).filter(t => t.length >= 2);
|
||||
}
|
||||
|
||||
function withinEditDistance1(a: string, b: string): boolean {
|
||||
if (a === b) return true;
|
||||
const la = a.length, lb = b.length;
|
||||
if (Math.abs(la - lb) > 1) return false;
|
||||
if (la === lb) {
|
||||
let diff = 0;
|
||||
for (let i = 0; i < la; i++) if (a[i] !== b[i]) diff++;
|
||||
return diff <= 1;
|
||||
}
|
||||
const [s, l] = la < lb ? [a, b] : [b, a];
|
||||
let i = 0, j = 0, skipped = false;
|
||||
while (i < s.length && j < l.length) {
|
||||
if (s[i] === l[j]) { i++; j++; }
|
||||
else if (!skipped) { skipped = true; j++; }
|
||||
else return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
interface OcrTextIndex { squashed: string; tokens: Set<string>; }
|
||||
|
||||
function buildOcrTextIndex(textLines: string[]): OcrTextIndex {
|
||||
const joined = textLines.join(" ").toUpperCase();
|
||||
return {
|
||||
squashed: joined.replace(/[^A-Z0-9]/g, ""),
|
||||
tokens: new Set(tokenizeName(joined))
|
||||
};
|
||||
}
|
||||
|
||||
function tokenFoundInOcr(token: string, ocr: OcrTextIndex): boolean {
|
||||
if (token.length >= 4 && ocr.squashed.includes(token)) return true;
|
||||
if (ocr.tokens.has(token)) return true;
|
||||
if (token.length >= 5) {
|
||||
for (const t of ocr.tokens) {
|
||||
if (Math.abs(t.length - token.length) <= 1 && withinEditDistance1(token, t)) return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Returns the class name of the best candidate after OCR-evidence
|
||||
// re-ranking (the classifier's top-1 unless a close band-mate has clearly
|
||||
// stronger printed-text evidence).
|
||||
function rerankClassCandidates(
|
||||
allProbabilities: Array<{ name: string; confidence: number }>,
|
||||
textLines: string[]
|
||||
): string {
|
||||
if (!allProbabilities.length) return "";
|
||||
const top1Sim = allProbabilities[0].confidence;
|
||||
const band = allProbabilities
|
||||
.slice(0, RERANK_TOP_K)
|
||||
.filter(p => p.confidence >= top1Sim - RERANK_SIM_BAND);
|
||||
if (band.length <= 1 || !textLines.length) return allProbabilities[0].name;
|
||||
|
||||
const ocrIdx = buildOcrTextIndex(textLines);
|
||||
const cands = band.map(p => {
|
||||
const sku = classNameSku(p.name);
|
||||
return { name: p.name, tokens: new Set(tokenizeName(p.name.replace(sku, ""))), coverage: 0 };
|
||||
});
|
||||
const tokenCounts = new Map<string, number>();
|
||||
for (const c of cands) {
|
||||
for (const tok of c.tokens) tokenCounts.set(tok, (tokenCounts.get(tok) || 0) + 1);
|
||||
}
|
||||
for (const c of cands) {
|
||||
let matched = 0, total = 0;
|
||||
for (const tok of c.tokens) {
|
||||
const nWith = tokenCounts.get(tok) || 1;
|
||||
if (nWith >= cands.length) continue; // shared by all band-mates -> no signal
|
||||
const w = 1 / nWith;
|
||||
total += w;
|
||||
if (tokenFoundInOcr(tok, ocrIdx)) matched += w;
|
||||
}
|
||||
c.coverage = total > 0 ? matched / total : 0;
|
||||
}
|
||||
|
||||
let chosen = cands[0];
|
||||
for (const c of cands.slice(1)) {
|
||||
if (c.coverage >= chosen.coverage + RERANK_COVERAGE_MARGIN) chosen = c;
|
||||
}
|
||||
if (chosen !== cands[0]) {
|
||||
console.log(`[Rerank] OCR evidence overrode classifier top-1 "${cands[0].name}" -> "${chosen.name}" (coverage ${cands[0].coverage.toFixed(2)} vs ${chosen.coverage.toFixed(2)})`);
|
||||
}
|
||||
return chosen.name;
|
||||
}
|
||||
|
||||
// Shared by the classic /api/scan-pfm dev route and the authenticated
|
||||
// /api/v1/scan-product route: calls the Python classifier, then matches the
|
||||
// result against sku_master, returning the top-5 candidates.
|
||||
export async function classifyAndMatchProduct(imageBase64: string): Promise<ProductScanResult> {
|
||||
const pyServerUrl = process.env.CLASSIFIER_SERVER_URL || "http://paddleocr-pipeline-api:8120/classify-ocr";
|
||||
|
||||
const response = await fetch(pyServerUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ image_base64: imageBase64 }),
|
||||
signal: AbortSignal.timeout(PIPELINE_TIMEOUT_MS)
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errText = await response.text();
|
||||
throw new ClassifierError(response.status, `Classifier service error: ${errText}`);
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
|
||||
const dbRes = await query("SELECT no_sku, nama_item FROM sku_master");
|
||||
const skuMasterList = dbRes.rows.map(row => ({
|
||||
no_sku: row.no_sku,
|
||||
nama_item: row.nama_item
|
||||
}));
|
||||
|
||||
const extractedSku = data.ocr?.extracted_sku || "";
|
||||
|
||||
// Re-rank the classifier's close candidates using OCR'd package text, then
|
||||
// map the winner straight to its sku_master row by the SKU prefix embedded
|
||||
// in the class name. The old approach (Levenshtein between top-1 class name
|
||||
// and every master nama_item) lost classifier-correct results whenever a
|
||||
// *different* SKU's master name happened to be textually closer.
|
||||
const rerankedName = rerankClassCandidates(
|
||||
data.classification?.all_probabilities || [],
|
||||
data.ocr?.text_lines || []
|
||||
) || data.classification?.top1_name || "";
|
||||
const rerankedSku = classNameSku(rerankedName);
|
||||
|
||||
const matchedList: SkuMatch[] = skuMasterList.map(sku => {
|
||||
const yoloSim = rerankedName ? getStringSimilarity(sku.nama_item, rerankedName) : 0;
|
||||
|
||||
const cleanMasterSku = sku.no_sku.trim();
|
||||
const cleanExtractedSku = extractedSku.trim();
|
||||
const isSkuMatch = cleanExtractedSku && cleanMasterSku === cleanExtractedSku;
|
||||
const isClassifierPick = rerankedSku && cleanMasterSku === rerankedSku;
|
||||
|
||||
const score = isSkuMatch ? 1.0 : isClassifierPick ? 0.995 : yoloSim;
|
||||
|
||||
return {
|
||||
no_sku: sku.no_sku,
|
||||
nama_item: sku.nama_item,
|
||||
score,
|
||||
yoloSimilarity: yoloSim,
|
||||
isBestMatch: false
|
||||
};
|
||||
});
|
||||
|
||||
matchedList.sort((a, b) => b.score - a.score);
|
||||
|
||||
const possibleMatches = matchedList.slice(0, 5).filter(m => m.score > 0.1);
|
||||
if (possibleMatches.length > 0) {
|
||||
possibleMatches[0].isBestMatch = true;
|
||||
}
|
||||
|
||||
return {
|
||||
classification: data.classification,
|
||||
ocr: data.ocr,
|
||||
possibleMatches
|
||||
};
|
||||
}
|
||||
Reference in new issue
Block a user