feat(backend): scan-product accuracy 66.2% -> 79.7% + frozen validation benchmark
Accuracy work on the 79-image product-scan validation set (user goal: 90%): - classify_ocr_server.py: 0/90/180/270-degree expiry-date search (stops at first hit, 0-degree fallback); classification decoupled onto the upright image (rotated frames regressed DINOv2 -6pts until this); cross-line date stitching; tiled full-res OCR pass (defeats the 4000px downscale that killed small inkjet dates); VL-pipeline expiry fallback with keyword-anchored anti-hallucination guard; VL text lines merged into text_lines + VL SKU retry. Visualization endpoints removed entirely (Visual/Spotting grids - unused by frontend, 3x per-scan GPU cost). - product-scan.ts: coverage-normalized OCR-evidence re-ranking of DINOv2 top-K (tuned offline: +8/-0 on top-1 misses), re-ranked class mapped to sku_master by SKU prefix; classifier timeout 90s->240s for fallback paths. - Frozen benchmark: product-test-images-fixed/ (79 renamed images) + freeze/seed/build-undetected/capture/experiment scripts; labels trimmed to the 79 validation entries (training rows kept in .bak-with-training); 5 TRAINED-ON SKUs replaced with fresh held-out photos. - manual-label-scan page: shows last batch-test AI prediction under every field by default (new /api/product-scan-results); serves the fixed folder; fixed total hydration failure via allowedDevOrigins 127.0.0.1. - Measured (all-79, zero failures): sku/name 87.3%, expiry 64.6%, overall 79.7%. Tiles/VL-evidence/VL-SKU deployed but not yet batch-measured. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Gr6HH7JrdsXX8AARejQboM
This commit is contained in:
1 parent
19f1facf9b
commit
e76ccb60a6
156 files changed
+17150
-1405
No files matched your search
@@ -8,7 +8,7 @@ export const dynamic = "force-dynamic";
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const filename = req.nextUrl.searchParams.get("filename");
|
||||
const dirPath = path.join(process.cwd(), "..", "sources", "product-test-images");
|
||||
const dirPath = path.join(process.cwd(), "..", "sources", "product-test-images-fixed");
|
||||
|
||||
// File serving mode
|
||||
if (filename) {
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { errorResponse } from "@/utils/api-error";
|
||||
|
||||
// Serves the most recent accuracy-check-scan.mts detail dump
|
||||
// (sources/product_scan_detail_*.json) so the manual-label-scan page can show
|
||||
// what the AI actually predicted for a given Validation Set image by default,
|
||||
// without re-running the pipeline live for every image browsed. This is the
|
||||
// same predicted value the accuracy harness scores against ground truth -
|
||||
// not a fresh scan, so it reflects the last batch test run.
|
||||
const SOURCES_DIR = path.join(process.cwd(), "..", "sources");
|
||||
|
||||
interface DetailCheck {
|
||||
field: string;
|
||||
match: boolean;
|
||||
expected: string;
|
||||
predicted: string;
|
||||
}
|
||||
|
||||
interface DetailValidationItem {
|
||||
filename: string;
|
||||
method?: string;
|
||||
confidence?: number;
|
||||
checks: DetailCheck[];
|
||||
}
|
||||
|
||||
interface DetailDump {
|
||||
timestamp: string;
|
||||
validation: DetailValidationItem[];
|
||||
}
|
||||
|
||||
function findLatestDump(): { path: string; data: DetailDump } | null {
|
||||
if (!fs.existsSync(SOURCES_DIR)) return null;
|
||||
const candidates = fs
|
||||
.readdirSync(SOURCES_DIR)
|
||||
.filter((f) => /^product_scan_detail_.*\.json$/.test(f))
|
||||
.map((f) => {
|
||||
const p = path.join(SOURCES_DIR, f);
|
||||
return { path: p, mtime: fs.statSync(p).mtimeMs };
|
||||
})
|
||||
.sort((a, b) => b.mtime - a.mtime);
|
||||
|
||||
if (candidates.length === 0) return null;
|
||||
const latest = candidates[0];
|
||||
const data = JSON.parse(fs.readFileSync(latest.path, "utf8"));
|
||||
return { path: latest.path, data };
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
|
||||
const latest = findLatestDump();
|
||||
if (!latest) {
|
||||
return NextResponse.json({ available: false });
|
||||
}
|
||||
|
||||
if (!filename) {
|
||||
return NextResponse.json({ available: true, timestamp: latest.data.timestamp });
|
||||
}
|
||||
|
||||
const item = latest.data.validation.find((v) => v.filename === filename);
|
||||
if (!item) {
|
||||
return NextResponse.json({ available: true, timestamp: latest.data.timestamp, found: false });
|
||||
}
|
||||
|
||||
const byField = Object.fromEntries(item.checks.map((c) => [c.field, c]));
|
||||
|
||||
return NextResponse.json({
|
||||
available: true,
|
||||
found: true,
|
||||
timestamp: latest.data.timestamp,
|
||||
method: item.method,
|
||||
confidence: item.confidence,
|
||||
no_sku: byField.no_sku?.predicted,
|
||||
nama_item: byField.nama_item?.predicted,
|
||||
expiry_date: byField.expiry_date?.predicted
|
||||
});
|
||||
} catch (err: unknown) {
|
||||
console.error("Error in product-scan-results API:", err);
|
||||
const message = err instanceof Error ? err.message : "Internal server error";
|
||||
return errorResponse(500, message);
|
||||
}
|
||||
}
|
||||
@@ -14,41 +14,10 @@ export async function POST(req: NextRequest) {
|
||||
|
||||
const result = await classifyAndMatchProduct(image_base64);
|
||||
|
||||
// Layout-parsing visualization (same pipeline as DO-PFM Visual Grid) - only
|
||||
// used by this desktop test page, not part of the shared classify+match logic.
|
||||
let layoutParsingResult: { layoutParsingResults?: Array<{ outputImages?: Record<string, string> }> } | null = null;
|
||||
const rawB64 = image_base64.includes(",") ? image_base64.split(",")[1] : image_base64;
|
||||
const pipelineUrl = process.env.PIPELINE_URL || "http://localhost:7871/layout-parsing";
|
||||
|
||||
try {
|
||||
const layoutResponse = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
file: rawB64,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
})
|
||||
});
|
||||
|
||||
if (layoutResponse.ok) {
|
||||
const layoutData = await layoutResponse.json();
|
||||
layoutParsingResult = layoutData.result ?? layoutData;
|
||||
} else {
|
||||
console.warn("Layout parsing for visualization failed:", await layoutResponse.text());
|
||||
}
|
||||
} catch (layoutErr) {
|
||||
console.warn("Layout parsing for visualization unavailable:", layoutErr);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
classification: result.classification,
|
||||
ocr: result.ocr,
|
||||
possibleMatches: result.possibleMatches,
|
||||
layoutParsingResult
|
||||
possibleMatches: result.possibleMatches
|
||||
});
|
||||
|
||||
} catch (error: unknown) {
|
||||
|
||||
@@ -18,7 +18,8 @@ export default function ManualLabelScanPage() {
|
||||
notes: ""
|
||||
});
|
||||
const [aiPredicted, setAiPredicted] = useState<AiPredictedData | null>(null);
|
||||
|
||||
const [aiSource, setAiSource] = useState<{ type: "batch" | "live"; timestamp: string; method?: string; confidence?: number } | null>(null);
|
||||
|
||||
const [skuList, setSkuList] = useState<Array<{ no_sku: string; nama_item: string }>>([]);
|
||||
const [isScanning, setIsScanning] = useState(false);
|
||||
const [savingGT, setSavingGT] = useState(false);
|
||||
@@ -39,20 +40,11 @@ export default function ManualLabelScanPage() {
|
||||
setSkuList(skuData.skus || []);
|
||||
}
|
||||
|
||||
// Fetch Training Images
|
||||
const pfmRes = await fetch("/api/produk-pfm");
|
||||
let trainingFiles: { url: string; filename: string }[] = [];
|
||||
if (pfmRes.ok) {
|
||||
const pfmData = await pfmRes.json();
|
||||
trainingFiles = (pfmData.products || []).flatMap((p: any) =>
|
||||
p.images.map((url: string) => ({
|
||||
url,
|
||||
filename: url.replace(/^\/produk-pfm\/foto-kemasan-v2\//, "")
|
||||
}))
|
||||
);
|
||||
}
|
||||
|
||||
// Fetch Test Images
|
||||
// Fetch Test Images — the frozen 79-image Validation Set
|
||||
// (product-test-images-fixed/), the only set the accuracy harness
|
||||
// scores. Gallery/training photos (foto-kemasan-v2/) are not shown
|
||||
// here: they don't need per-photo ground truth, only correct
|
||||
// SKU-folder placement for classifier training.
|
||||
const testRes = await fetch("/api/product-images");
|
||||
let testFiles: { url: string; filename: string }[] = [];
|
||||
if (testRes.ok) {
|
||||
@@ -63,9 +55,8 @@ export default function ManualLabelScanPage() {
|
||||
}));
|
||||
}
|
||||
|
||||
const combined = [...testFiles, ...trainingFiles];
|
||||
setFiles(combined);
|
||||
if (combined.length > 0) setCurrentIndex(0);
|
||||
setFiles(testFiles);
|
||||
if (testFiles.length > 0) setCurrentIndex(0);
|
||||
|
||||
} catch (err) {
|
||||
console.error("Error initializing page", err);
|
||||
@@ -91,11 +82,33 @@ export default function ManualLabelScanPage() {
|
||||
expiry_date: data.expiry_date || "",
|
||||
notes: data.notes || ""
|
||||
});
|
||||
setAiPredicted(null); // Reset AI predictions on new file load
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Error fetching label", err);
|
||||
}
|
||||
|
||||
// Default-load the AI prediction from the last batch accuracy run
|
||||
// (not a live re-scan) so failures are visible immediately while
|
||||
// browsing - "Scan with AI" below can still be used to get a fresh
|
||||
// live result for this exact image.
|
||||
setAiPredicted(null);
|
||||
setAiSource(null);
|
||||
try {
|
||||
const aiRes = await fetch(`/api/product-scan-results?filename=${encodeURIComponent(file.filename)}`);
|
||||
if (aiRes.ok) {
|
||||
const aiData = await aiRes.json();
|
||||
if (aiData.found) {
|
||||
setAiPredicted({
|
||||
no_sku: aiData.no_sku,
|
||||
nama_item: aiData.nama_item,
|
||||
expiry_date: aiData.expiry_date
|
||||
});
|
||||
setAiSource({ type: "batch", timestamp: aiData.timestamp, method: aiData.method, confidence: aiData.confidence });
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Error fetching batch AI result", err);
|
||||
}
|
||||
};
|
||||
loadLabel();
|
||||
}, [currentIndex, files]);
|
||||
@@ -138,13 +151,27 @@ export default function ManualLabelScanPage() {
|
||||
|
||||
if (!scanRes.ok) throw new Error("Pipeline API error");
|
||||
const scanData = await scanRes.json();
|
||||
|
||||
|
||||
// Compare against the sku_master-resolved best match (what the app
|
||||
// actually shows/saves as nama_item, and what the accuracy harness
|
||||
// scores), not classification.top1_name - that's the classifier's raw
|
||||
// internal class label (e.g. the foto-kemasan-v2 folder name), which
|
||||
// structurally never matches a sku_master-style ground truth string
|
||||
// even when the classification itself is correct.
|
||||
const bestMatch = (scanData.possibleMatches || []).find((m: { isBestMatch?: boolean }) => m.isBestMatch);
|
||||
|
||||
setAiPredicted({
|
||||
no_sku: scanData.classification?.top1_name ? skuList.find(s => s.nama_item === scanData.classification.top1_name)?.no_sku : undefined,
|
||||
nama_item: scanData.classification?.top1_name,
|
||||
no_sku: bestMatch?.no_sku,
|
||||
nama_item: bestMatch?.nama_item,
|
||||
expiry_date: scanData.ocr?.extracted_expired_date
|
||||
});
|
||||
|
||||
setAiSource({
|
||||
type: "live",
|
||||
timestamp: new Date().toISOString(),
|
||||
method: scanData.classification?.method,
|
||||
confidence: scanData.classification?.top1_confidence
|
||||
});
|
||||
|
||||
showToast("AI Scan complete!");
|
||||
} catch (err) {
|
||||
showToast(getErrorMessage(err, undefined, "AI Scan failed"), true);
|
||||
@@ -206,6 +233,7 @@ export default function ManualLabelScanPage() {
|
||||
<Editor
|
||||
formData={formData}
|
||||
aiPredicted={aiPredicted}
|
||||
aiSource={aiSource}
|
||||
skuList={skuList}
|
||||
isScanning={isScanning}
|
||||
onScanWithAi={handleScanWithAi}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
"use client";
|
||||
|
||||
import React, { useState, useEffect } from "react";
|
||||
import { extractLayoutVisUrlFromResult } from "@/utils/layoutVisualization";
|
||||
import { getErrorMessage } from "@/utils/client-error";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
@@ -34,16 +33,9 @@ interface ScanResponse {
|
||||
expired_line_index?: number;
|
||||
expired_date_crop_base64?: string;
|
||||
expired_source_line?: string;
|
||||
vis_image_base64?: string;
|
||||
spotting_image_base64?: string;
|
||||
error?: string;
|
||||
};
|
||||
possibleMatches: MatchResult[];
|
||||
layoutParsingResult?: {
|
||||
layoutParsingResults?: Array<{
|
||||
outputImages?: Record<string, string>;
|
||||
}>;
|
||||
};
|
||||
}
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
@@ -115,7 +107,7 @@ export default function ScanPfmPage() {
|
||||
const [savingGT, setSavingGT] = useState<boolean>(false);
|
||||
const [error, setError] = useState<string>("");
|
||||
const [scanResult, setScanResult] = useState<ScanResponse | null>(null);
|
||||
const [activeTab, setActiveTab] = useState<"summary" | "visual" | "spotting" | "json">("summary");
|
||||
const [activeTab, setActiveTab] = useState<"summary" | "json">("summary");
|
||||
|
||||
const [isEditingDate, setIsEditingDate] = useState<boolean>(false);
|
||||
const [editedDate, setEditedDate] = useState<string>("");
|
||||
@@ -385,8 +377,6 @@ export default function ScanPfmPage() {
|
||||
}
|
||||
};
|
||||
|
||||
const visUrl = scanResult ? extractLayoutVisUrlFromResult(scanResult.layoutParsingResult) : null;
|
||||
|
||||
return (
|
||||
<div className="min-h-screen bg-slate-950 text-slate-100 flex flex-col font-sans">
|
||||
|
||||
@@ -637,14 +627,12 @@ export default function ScanPfmPage() {
|
||||
<div className="border-b border-slate-800 flex rounded-xl overflow-hidden bg-slate-950/50">
|
||||
{[
|
||||
{ id: "summary", name: "Scan Summary" },
|
||||
{ id: "visual", name: "Visual Grid" },
|
||||
{ id: "spotting", name: "Spotting Grid" },
|
||||
{ id: "json", name: "Raw Response" }
|
||||
].map((t) => (
|
||||
<button
|
||||
key={t.id}
|
||||
id={`tab-${t.id}`}
|
||||
onClick={() => setActiveTab(t.id as "summary" | "visual" | "spotting" | "json")}
|
||||
onClick={() => setActiveTab(t.id as "summary" | "json")}
|
||||
className={`flex-1 py-3 text-xs font-bold transition-all border-b-2 cursor-pointer select-none ${
|
||||
activeTab === t.id
|
||||
? "border-teal-500 text-teal-400 bg-slate-900/40"
|
||||
@@ -1059,118 +1047,6 @@ export default function ScanPfmPage() {
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Visual Grid Tab */}
|
||||
{activeTab === "visual" && (
|
||||
<div className="space-y-4">
|
||||
<h3 className="text-xs font-bold text-teal-400 uppercase tracking-wider">
|
||||
Layout Visualization Grid
|
||||
</h3>
|
||||
{visUrl ? (
|
||||
<div className="bg-slate-950 rounded-xl border border-slate-800 overflow-hidden shadow-2xl p-2 flex justify-center">
|
||||
{/* eslint-disable-next-line @next/next/no-img-element */}
|
||||
<img
|
||||
src={visUrl}
|
||||
alt="Layout Visualization Grid"
|
||||
className="max-w-full h-auto object-contain rounded"
|
||||
/>
|
||||
</div>
|
||||
) : (
|
||||
<p className="text-xs text-slate-500 italic">
|
||||
No layout visualization image returned. The layout-parsing pipeline may be unavailable.
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Spotting Grid Tab */}
|
||||
{activeTab === "spotting" && (
|
||||
<div className="space-y-6">
|
||||
<div className="space-y-4">
|
||||
<h3 className="text-xs font-bold text-teal-400 uppercase tracking-wider">
|
||||
Spotting Visualization
|
||||
</h3>
|
||||
{scanResult.ocr?.spotting_image_base64 ? (
|
||||
<div className="bg-slate-950 rounded-xl border border-slate-800 overflow-hidden shadow-2xl p-2 flex justify-center">
|
||||
{/* eslint-disable-next-line @next/next/no-img-element */}
|
||||
<img
|
||||
src={scanResult.ocr.spotting_image_base64}
|
||||
alt="Spotting Visualization BBoxes"
|
||||
className="max-w-full h-auto object-contain rounded"
|
||||
/>
|
||||
</div>
|
||||
) : (
|
||||
<p className="text-xs text-slate-500 italic">No spotting visualization image returned by the parser.</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="space-y-3 pt-2 border-t border-slate-800">
|
||||
<h3 className="text-xs font-bold text-amber-400 uppercase tracking-wider">
|
||||
Detected Expired Date (Best Before / BB)
|
||||
</h3>
|
||||
<div className="grid grid-cols-1 sm:grid-cols-2 gap-4">
|
||||
<div className="bg-slate-900/40 border border-slate-800 rounded-xl p-4 flex flex-col gap-2 min-h-[100px]">
|
||||
<span className="text-[9px] font-bold text-slate-500 uppercase tracking-wider">Extracted Date</span>
|
||||
{isEditingDate ? (
|
||||
<div className="flex items-center gap-2 mt-1">
|
||||
<input
|
||||
type="text"
|
||||
value={editedDate}
|
||||
onChange={(e) => setEditedDate(e.target.value)}
|
||||
className="bg-slate-950 border border-slate-700 text-slate-100 rounded-lg px-2 py-1 text-sm font-semibold focus:outline-none focus:ring-1 focus:ring-amber-500 w-full"
|
||||
placeholder="DD/MM/YYYY"
|
||||
autoFocus
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === "Enter") handleSaveDate(editedDate);
|
||||
else if (e.key === "Escape") setIsEditingDate(false);
|
||||
}}
|
||||
/>
|
||||
<button onClick={() => handleSaveDate(editedDate)} className="bg-emerald-600 hover:bg-emerald-500 text-white rounded-lg p-1.5 text-xs font-bold transition-colors">✓</button>
|
||||
<button onClick={() => setIsEditingDate(false)} className="bg-slate-700 hover:bg-slate-600 text-slate-300 rounded-lg p-1.5 text-xs font-bold transition-colors">✗</button>
|
||||
</div>
|
||||
) : (
|
||||
<div className="flex items-center justify-between gap-2">
|
||||
<span className={`text-lg font-bold flex items-center gap-2 ${scanResult.ocr?.extracted_expired_date ? "text-amber-400" : "text-slate-600"}`}>
|
||||
<span className={`h-2 w-2 rounded-full flex-shrink-0 ${scanResult.ocr?.extracted_expired_date ? "bg-amber-400" : "bg-slate-700"}`} />
|
||||
{scanResult.ocr?.extracted_expired_date || "Not detected"}
|
||||
</span>
|
||||
<button
|
||||
onClick={() => { setEditedDate(scanResult.ocr?.extracted_expired_date || ""); setIsEditingDate(true); }}
|
||||
className="text-[10px] bg-slate-800 hover:bg-slate-700 text-slate-400 px-2.5 py-1 rounded-md border border-slate-700 transition-all font-semibold flex items-center gap-1"
|
||||
>
|
||||
✏️ Edit
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
{scanResult.ocr?.expired_source_line && (
|
||||
<p className="text-[10px] text-slate-500 leading-snug">
|
||||
<span className="font-semibold text-slate-400">OCR: </span>
|
||||
{scanResult.ocr.expired_source_line}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="bg-slate-900/40 border border-slate-800 rounded-xl p-4 flex flex-col gap-2 min-h-[100px]">
|
||||
<span className="text-[9px] font-bold text-slate-500 uppercase tracking-wider">Label Crop</span>
|
||||
<div className="flex-1 bg-slate-950 border border-slate-800 rounded-lg p-2 flex items-center justify-center min-h-[72px] overflow-hidden">
|
||||
{scanResult.ocr?.expired_date_crop_base64 ? (
|
||||
// eslint-disable-next-line @next/next/no-img-element
|
||||
<img
|
||||
src={scanResult.ocr.expired_date_crop_base64}
|
||||
alt="Expired date crop"
|
||||
className="max-h-24 max-w-full object-contain brightness-95 contrast-105"
|
||||
/>
|
||||
) : (
|
||||
<span className="text-[10px] text-slate-600 italic text-center px-2">
|
||||
{scanResult.ocr?.extracted_expired_date ? "Crop unavailable" : "No expiry region to crop"}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* JSON Tab */}
|
||||
{activeTab === "json" && (
|
||||
<div className="space-y-4">
|
||||
|
||||
@@ -14,9 +14,17 @@ export interface AiPredictedData {
|
||||
expiry_date?: string;
|
||||
}
|
||||
|
||||
export interface AiSourceInfo {
|
||||
type: "batch" | "live";
|
||||
timestamp: string;
|
||||
method?: string;
|
||||
confidence?: number;
|
||||
}
|
||||
|
||||
interface EditorProps {
|
||||
formData: ScanLabelFormData;
|
||||
aiPredicted: AiPredictedData | null;
|
||||
aiSource: AiSourceInfo | null;
|
||||
skuList: Array<{ no_sku: string; nama_item: string }>;
|
||||
isScanning: boolean;
|
||||
onScanWithAi: () => void;
|
||||
@@ -28,6 +36,7 @@ interface EditorProps {
|
||||
export function Editor({
|
||||
formData,
|
||||
aiPredicted,
|
||||
aiSource,
|
||||
skuList,
|
||||
isScanning,
|
||||
onScanWithAi,
|
||||
@@ -69,8 +78,21 @@ export function Editor({
|
||||
disabled={isScanning || !formData.filename}
|
||||
className="w-full bg-teal-600/20 text-teal-400 hover:bg-teal-600/30 disabled:opacity-50 border border-teal-500/30 rounded-lg py-2 text-xs font-semibold transition flex items-center justify-center gap-2"
|
||||
>
|
||||
{isScanning ? "Scanning with Pipeline..." : "Scan with AI 🤖"}
|
||||
{isScanning ? "Scanning with Pipeline..." : "Scan with AI 🤖 (re-run live)"}
|
||||
</button>
|
||||
{aiSource ? (
|
||||
<p className="text-[10px] text-slate-500 leading-snug">
|
||||
{aiSource.type === "batch" ? (
|
||||
<>Showing result from last batch test ({new Date(aiSource.timestamp).toLocaleString()})</>
|
||||
) : (
|
||||
<>Live scan result ({new Date(aiSource.timestamp).toLocaleTimeString()})</>
|
||||
)}
|
||||
{aiSource.method && <> · {aiSource.method}</>}
|
||||
{typeof aiSource.confidence === "number" && <> · conf {aiSource.confidence.toFixed(3)}</>}
|
||||
</p>
|
||||
) : (
|
||||
<p className="text-[10px] text-slate-600 italic">No AI result yet for this image — click "Scan with AI" or run the accuracy batch test.</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Form Fields */}
|
||||
|
||||
@@ -1,26 +0,0 @@
|
||||
export function normalizeImageSrc(src: string): string {
|
||||
if (!src) return "";
|
||||
if (src.startsWith("http://") || src.startsWith("https://") || src.startsWith("data:")) {
|
||||
return src;
|
||||
}
|
||||
return `data:image/png;base64,${src}`;
|
||||
}
|
||||
|
||||
export interface LayoutPageResult {
|
||||
outputImages?: Record<string, string>;
|
||||
}
|
||||
|
||||
/** Same visualization URL selection as DO-PFM Visual Grid (second image if present, else first). */
|
||||
export function extractLayoutVisUrl(page0: LayoutPageResult | null | undefined): string {
|
||||
const outImgs = page0?.outputImages || {};
|
||||
const sortedUrls = Object.values(outImgs).filter(Boolean) as string[];
|
||||
const visUrl = sortedUrls.length >= 2 ? sortedUrls[1] : sortedUrls[0] || "";
|
||||
return normalizeImageSrc(visUrl);
|
||||
}
|
||||
|
||||
export function extractLayoutVisUrlFromResult(
|
||||
layoutParsingResult: { layoutParsingResults?: LayoutPageResult[] } | null | undefined
|
||||
): string {
|
||||
const page0 = layoutParsingResult?.layoutParsingResults?.[0];
|
||||
return extractLayoutVisUrl(page0);
|
||||
}
|
||||
@@ -1,10 +1,12 @@
|
||||
import { query } from "../db";
|
||||
|
||||
// Bounds the classifier call so a wedged GPU container fails fast instead of
|
||||
// hanging indefinitely - matches the bound `api/parse/route.ts` used to apply
|
||||
// to its own separate inline classify call before it started sharing this
|
||||
// function (see docs/api-contract-map.md G3).
|
||||
const PIPELINE_TIMEOUT_MS = 90_000;
|
||||
// hanging indefinitely. Raised from 90s (2026-07-14): hard images now
|
||||
// legitimately take up to ~3 min - a 4-orientation OCR search plus a VL
|
||||
// pipeline fallback when no expiry date is found (see
|
||||
// config/classify_ocr_server.py) - and the old bound was killing exactly
|
||||
// the images those fallbacks exist to save.
|
||||
const PIPELINE_TIMEOUT_MS = 240_000;
|
||||
|
||||
// Thrown when the Python classifier service itself returns a non-2xx response,
|
||||
// so callers can forward its actual status instead of collapsing everything to 500.
|
||||
@@ -60,6 +62,115 @@ function getStringSimilarity(s1: string, s2: string): number {
|
||||
return (maxLength - distance) / maxLength;
|
||||
}
|
||||
|
||||
// --- OCR-evidence re-ranking of the classifier's top-K candidates ---
|
||||
//
|
||||
// DINOv2's misses are near-twin confusions (same brand line, different
|
||||
// flavor/size) - exactly the cases where the printed variant words differ,
|
||||
// and PaddleOCR usually reads some of them. Within a narrow similarity band
|
||||
// of the top-1 candidate, prefer the one whose distinctive name tokens
|
||||
// actually appear in the OCR'd text. Coverage-normalized so generic
|
||||
// packaging words (e.g. "French Fries", "Ayam") that happen to be unique to
|
||||
// one candidate's *name* can't hijack the ranking. Parameters tuned offline
|
||||
// against the 79-image validation set (scripts/experiment-rerank.mjs,
|
||||
// 2026-07-14: fixes 8 of 18 top-1 misses, breaks 0 of 61 correct).
|
||||
const RERANK_TOP_K = 12;
|
||||
const RERANK_SIM_BAND = 0.12;
|
||||
const RERANK_COVERAGE_MARGIN = 0.25;
|
||||
|
||||
function classNameSku(className: string): string {
|
||||
// foto-kemasan-v2 class names are "<SKU> <NAME...>"
|
||||
return (className || "").trim().split(/\s+/)[0] || "";
|
||||
}
|
||||
|
||||
function tokenizeName(name: string): string[] {
|
||||
return name.toUpperCase().split(/[^A-Z0-9]+/).filter(t => t.length >= 2);
|
||||
}
|
||||
|
||||
function withinEditDistance1(a: string, b: string): boolean {
|
||||
if (a === b) return true;
|
||||
const la = a.length, lb = b.length;
|
||||
if (Math.abs(la - lb) > 1) return false;
|
||||
if (la === lb) {
|
||||
let diff = 0;
|
||||
for (let i = 0; i < la; i++) if (a[i] !== b[i]) diff++;
|
||||
return diff <= 1;
|
||||
}
|
||||
const [s, l] = la < lb ? [a, b] : [b, a];
|
||||
let i = 0, j = 0, skipped = false;
|
||||
while (i < s.length && j < l.length) {
|
||||
if (s[i] === l[j]) { i++; j++; }
|
||||
else if (!skipped) { skipped = true; j++; }
|
||||
else return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
interface OcrTextIndex { squashed: string; tokens: Set<string>; }
|
||||
|
||||
function buildOcrTextIndex(textLines: string[]): OcrTextIndex {
|
||||
const joined = textLines.join(" ").toUpperCase();
|
||||
return {
|
||||
squashed: joined.replace(/[^A-Z0-9]/g, ""),
|
||||
tokens: new Set(tokenizeName(joined))
|
||||
};
|
||||
}
|
||||
|
||||
function tokenFoundInOcr(token: string, ocr: OcrTextIndex): boolean {
|
||||
if (token.length >= 4 && ocr.squashed.includes(token)) return true;
|
||||
if (ocr.tokens.has(token)) return true;
|
||||
if (token.length >= 5) {
|
||||
for (const t of ocr.tokens) {
|
||||
if (Math.abs(t.length - token.length) <= 1 && withinEditDistance1(token, t)) return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Returns the class name of the best candidate after OCR-evidence
|
||||
// re-ranking (the classifier's top-1 unless a close band-mate has clearly
|
||||
// stronger printed-text evidence).
|
||||
function rerankClassCandidates(
|
||||
allProbabilities: Array<{ name: string; confidence: number }>,
|
||||
textLines: string[]
|
||||
): string {
|
||||
if (!allProbabilities.length) return "";
|
||||
const top1Sim = allProbabilities[0].confidence;
|
||||
const band = allProbabilities
|
||||
.slice(0, RERANK_TOP_K)
|
||||
.filter(p => p.confidence >= top1Sim - RERANK_SIM_BAND);
|
||||
if (band.length <= 1 || !textLines.length) return allProbabilities[0].name;
|
||||
|
||||
const ocrIdx = buildOcrTextIndex(textLines);
|
||||
const cands = band.map(p => {
|
||||
const sku = classNameSku(p.name);
|
||||
return { name: p.name, tokens: new Set(tokenizeName(p.name.replace(sku, ""))), coverage: 0 };
|
||||
});
|
||||
const tokenCounts = new Map<string, number>();
|
||||
for (const c of cands) {
|
||||
for (const tok of c.tokens) tokenCounts.set(tok, (tokenCounts.get(tok) || 0) + 1);
|
||||
}
|
||||
for (const c of cands) {
|
||||
let matched = 0, total = 0;
|
||||
for (const tok of c.tokens) {
|
||||
const nWith = tokenCounts.get(tok) || 1;
|
||||
if (nWith >= cands.length) continue; // shared by all band-mates -> no signal
|
||||
const w = 1 / nWith;
|
||||
total += w;
|
||||
if (tokenFoundInOcr(tok, ocrIdx)) matched += w;
|
||||
}
|
||||
c.coverage = total > 0 ? matched / total : 0;
|
||||
}
|
||||
|
||||
let chosen = cands[0];
|
||||
for (const c of cands.slice(1)) {
|
||||
if (c.coverage >= chosen.coverage + RERANK_COVERAGE_MARGIN) chosen = c;
|
||||
}
|
||||
if (chosen !== cands[0]) {
|
||||
console.log(`[Rerank] OCR evidence overrode classifier top-1 "${cands[0].name}" -> "${chosen.name}" (coverage ${cands[0].coverage.toFixed(2)} vs ${chosen.coverage.toFixed(2)})`);
|
||||
}
|
||||
return chosen.name;
|
||||
}
|
||||
|
||||
// Shared by the classic /api/scan-pfm dev route and the authenticated
|
||||
// /api/v1/scan-product route: calls the Python classifier, then matches the
|
||||
// result against sku_master, returning the top-5 candidates.
|
||||
@@ -86,17 +197,28 @@ export async function classifyAndMatchProduct(imageBase64: string): Promise<Prod
|
||||
nama_item: row.nama_item
|
||||
}));
|
||||
|
||||
const top1Name = data.classification?.top1_name || "";
|
||||
const extractedSku = data.ocr?.extracted_sku || "";
|
||||
|
||||
// Re-rank the classifier's close candidates using OCR'd package text, then
|
||||
// map the winner straight to its sku_master row by the SKU prefix embedded
|
||||
// in the class name. The old approach (Levenshtein between top-1 class name
|
||||
// and every master nama_item) lost classifier-correct results whenever a
|
||||
// *different* SKU's master name happened to be textually closer.
|
||||
const rerankedName = rerankClassCandidates(
|
||||
data.classification?.all_probabilities || [],
|
||||
data.ocr?.text_lines || []
|
||||
) || data.classification?.top1_name || "";
|
||||
const rerankedSku = classNameSku(rerankedName);
|
||||
|
||||
const matchedList: SkuMatch[] = skuMasterList.map(sku => {
|
||||
const yoloSim = top1Name ? getStringSimilarity(sku.nama_item, top1Name) : 0;
|
||||
const yoloSim = rerankedName ? getStringSimilarity(sku.nama_item, rerankedName) : 0;
|
||||
|
||||
const cleanMasterSku = sku.no_sku.trim();
|
||||
const cleanExtractedSku = extractedSku.trim();
|
||||
const isSkuMatch = cleanExtractedSku && cleanMasterSku === cleanExtractedSku;
|
||||
const isClassifierPick = rerankedSku && cleanMasterSku === rerankedSku;
|
||||
|
||||
const score = isSkuMatch ? 1.0 : yoloSim;
|
||||
const score = isSkuMatch ? 1.0 : isClassifierPick ? 0.995 : yoloSim;
|
||||
|
||||
return {
|
||||
no_sku: sku.no_sku,
|
||||
|
||||
Reference in new issue
Block a user