feat: update backend OCR parser, web app, mobile app camera/preview UI, tests, and documentation with sample images

This commit is contained in:
Rafhan Mazaya Fathurrahman committed 2026-07-02 13:33:57 +07:00
1 parent aa3233e411
commit bdb3a49742
71 files changed
+3360 -764

No files matched your search

+201 -60
View File
@@ -5,6 +5,36 @@ import crypto from "crypto";
import { query, cleanupAndReindexItems, resolveStoreFromText } from "../../../db";
import { parseDOMetadata, sanitizeParsedMetadata } from "../../../utils/parser";
function levenshteinDistance(s1: string, s2: string): number {
const len1 = s1.length;
const len2 = s2.length;
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
for (let i = 1; i <= len1; i++) {
for (let j = 1; j <= len2; j++) {
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
matrix[i][j] = Math.min(
matrix[i - 1][j] + 1, // deletion
matrix[i][j - 1] + 1, // insertion
matrix[i - 1][j - 1] + cost // substitution
);
}
}
return matrix[len1][len2];
}
function getStringSimilarity(s1: string, s2: string): number {
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
if (!clean1 || !clean2) return 0;
const distance = levenshteinDistance(clean1, clean2);
const maxLength = Math.max(clean1.length, clean2.length);
return (maxLength - distance) / maxLength;
}
export async function POST(req: NextRequest) {
let safeFile = "";
try {
@@ -33,60 +63,6 @@ export async function POST(req: NextRequest) {
const fileBuffer = fs.readFileSync(filePath);
const fileHash = crypto.createHash("sha256").update(fileBuffer).digest("hex");
// 2. Check if document exists in database (by filename OR hash) and is already parsed
const checkRes = await query(
"SELECT id, layout_parsing_result FROM documents WHERE (filename = $1 OR file_hash = $2) AND parsed = true AND layout_parsing_result IS NOT NULL",
[safeFile, fileHash]
);
if (checkRes.rowCount && checkRes.rowCount > 0) {
const doc = checkRes.rows[0];
const docId = doc.id;
const pipelineResult = doc.layout_parsing_result;
// Clean up and re-index invalid items first
await cleanupAndReindexItems(docId);
// Fetch items
const itemsRes = await query(
`SELECT row_index,
kode_barang, nama_barang, banyak, jumlah,
is_flagged, remark
FROM ocr_items
WHERE document_id = $1
ORDER BY row_index`,
[docId]
);
const items = itemsRes.rows.map(row => ({
kodeBarang: row.kode_barang,
namaBarang: row.nama_barang,
banyak: row.banyak,
jumlah: row.jumlah
}));
const flagged: Record<number, boolean> = {};
const remarks: Record<number, string> = {};
itemsRes.rows.forEach(row => {
if (row.is_flagged) {
flagged[row.row_index] = true;
}
if (row.remark && row.remark.trim()) {
remarks[row.row_index] = row.remark;
}
});
return NextResponse.json({
errorCode: 0,
errorMsg: "Success",
result: pipelineResult,
items,
flagged,
remarks
});
}
const b64 = fileBuffer.toString("base64");
// Form payload
@@ -96,7 +72,7 @@ export async function POST(req: NextRequest) {
useLayoutDetection: true,
fileType: 1,
useDocUnwarping: false,
useDocOrientationClassify: false
useDocOrientationClassify: true
};
// Post to Pipeline API
@@ -137,6 +113,7 @@ export async function POST(req: NextRequest) {
// Check if the image is not straight (tilt > 1.0 degree)
const tilt = calculateAverageTilt(data);
let unwarped = false;
if (tilt > 1.0) {
console.log(`Parsed document ${safeFile} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
const unwarpPayload = {
@@ -154,6 +131,7 @@ export async function POST(req: NextRequest) {
if (unwarpResponse.ok) {
data = await unwarpResponse.json();
console.log(`Document unwarped successfully.`);
unwarped = true;
} else {
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
}
@@ -167,12 +145,151 @@ export async function POST(req: NextRequest) {
// 3. Save to database
try {
const pipelineResult = data.result || data;
pipelineResult.pipeline_info = {
tilt,
unwarped,
original_tilt: tilt
};
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
const markdownText = page0?.markdown?.text || "";
const rawMetadata = parseDOMetadata(markdownText);
// Second-layer sanity check: enforces strict field formats and auto-corrects anomalies
const docMetadata = sanitizeParsedMetadata(rawMetadata as any);
// Load SKU Master list from DB (including new packaging details for triple check validation)
const skuDbRes = await query("SELECT no_sku, nama_item, standar_jumlah, jenis_outer FROM sku_master");
const skuMasterList = skuDbRes.rows.map(row => ({
no_sku: row.no_sku.toString().trim(),
nama_item: row.nama_item.toString().trim(),
standar_jumlah: row.standar_jumlah ? row.standar_jumlah.toString().trim() : null,
jenis_outer: row.jenis_outer ? row.jenis_outer.toString().trim() : null
}));
// Cross-check items using intelligent Triple-Check (SKU, Name, and Unit)
const checkedItems: typeof docMetadata.items = [];
for (const item of docMetadata.items) {
const ocrSku = item.kodeBarang ? item.kodeBarang.trim() : "";
const ocrName = item.namaBarang ? item.namaBarang.trim() : "";
const ocrQty = item.banyak ? item.banyak.trim() : "";
const ocrPrice = item.jumlah ? item.jumlah.trim() : "";
// Extract alphabetic characters as unit symbols
const qtyUnitMatch = ocrQty.match(/[a-zA-Z]+/g);
const qtyUnit = qtyUnitMatch ? qtyUnitMatch.join(" ") : "";
const priceUnitMatch = ocrPrice.match(/[a-zA-Z]+/g);
const priceUnit = priceUnitMatch ? priceUnitMatch.join(" ") : "";
// Remember the original parsed SKU for traceability
(item as any).kodeBarangOriginal = ocrSku;
// Find the best master candidate using a weighted combined score
let bestMatch: typeof skuMasterList[0] | null = null;
let bestScore = 0;
// First, check if there is an exact SKU match. If yes, trust the SKU completely and bypass fuzzy matching.
const exactSkuMatch = skuMasterList.find(c => ocrSku === c.no_sku);
if (exactSkuMatch) {
bestMatch = exactSkuMatch;
bestScore = 1.0;
} else {
for (const candidate of skuMasterList) {
// 1. SKU Similarity (weight = 0.5)
let skuSim = 0;
if (ocrSku === candidate.no_sku) {
skuSim = 1.0;
} else if (ocrSku && /^\d+$/.test(ocrSku.replace(/[^0-9]/g, ''))) {
const cleanOcrSku = ocrSku.replace(/[^0-9]/g, '');
skuSim = getStringSimilarity(cleanOcrSku, candidate.no_sku);
}
// 2. Name Similarity (weight = 0.4)
const nameSim = getStringSimilarity(ocrName, candidate.nama_item);
// 3. Packaging Unit Match Bonus (weight = 0.1)
let unitBonus = 0;
let matchesOuter = false;
let matchesInner = false;
const cleanQtyUnit = qtyUnit.toLowerCase();
const cleanPriceUnit = priceUnit.toLowerCase();
if (candidate.jenis_outer) {
const cOuter = candidate.jenis_outer.toLowerCase();
if (cleanQtyUnit.includes(cOuter) ||
(cOuter === "karung" && cleanQtyUnit.includes("krg")) ||
(cOuter === "box" && cleanQtyUnit.includes("box")) ||
(cOuter === "bag" && cleanQtyUnit.includes("bag"))) {
matchesOuter = true;
}
}
if (candidate.standar_jumlah) {
const cInner = candidate.standar_jumlah.toLowerCase();
if (cleanPriceUnit.includes(cInner) ||
(cInner === "pac" && cleanPriceUnit.includes("pai")) ||
(cInner === "pc" && cleanPriceUnit.includes("pc")) ||
(cInner === "kg" && cleanPriceUnit.includes("kg"))) {
matchesInner = true;
}
}
if (matchesOuter) unitBonus += 0.5;
if (matchesInner) unitBonus += 0.5;
const score = (0.5 * skuSim) + (0.4 * nameSim) + (0.1 * unitBonus);
if (score > bestScore) {
bestScore = score;
bestMatch = candidate;
}
}
}
if (bestMatch && bestScore >= 0.6) {
// Triple-check matched! Correct the SKU and description to master values,
// and auto-standardize the quantities units!
item.kodeBarang = bestMatch.no_sku;
item.namaBarang = bestMatch.nama_item;
// Auto-standardize Qty (banyak) units
const numericQtyMatch = ocrQty.match(/[\d.,]+/);
if (numericQtyMatch) {
const numericQty = numericQtyMatch[0];
const standardOuter = bestMatch.jenis_outer || "";
if (standardOuter.toLowerCase() === "karung") {
item.banyak = `${numericQty} KRG`;
} else if (standardOuter.toLowerCase() === "box") {
item.banyak = `${numericQty} BOX`;
} else if (standardOuter.toLowerCase() === "bag") {
item.banyak = `${numericQty} BAG`;
} else {
item.banyak = `${numericQty} ${standardOuter.toUpperCase()}`;
}
}
// Auto-standardize Price/Total (jumlah) units
const numericPriceMatch = ocrPrice.match(/[\d.,]+/);
if (numericPriceMatch) {
const numericPrice = numericPriceMatch[0];
const standardInner = (bestMatch.standar_jumlah || "").toUpperCase();
item.jumlah = `${numericPrice} ${standardInner}`;
}
checkedItems.push(item);
} else {
// Case 3: No match by SKU/Name/Unit triple check.
// If the original SKU is a valid 8-digit code, keep it (to allow new SKUs not yet in master list).
// Otherwise, discard the row entirely to filter out layout parsing table noise (like signatures).
if (/^\d{8}$/.test(ocrSku)) {
checkedItems.push(item);
} else {
console.log(`Filtering out table noise row: SKU="${ocrSku}", Name="${ocrName}"`);
}
}
}
docMetadata.items = checkedItems;
// Resolve store information using master database
const resolvedStore = await resolveStoreFromText(markdownText);
(docMetadata as any).orderUntuk = resolvedStore.orderUntuk;
@@ -180,9 +297,30 @@ export async function POST(req: NextRequest) {
const stats = fs.statSync(filePath);
const logsPayload = {
filename: safeFile,
vllm_calls: [] as any[],
ocr_raw: pipelineResult,
stage_1_output: docMetadata,
stage_2_output: docMetadata,
frontend_response: {
filename: safeFile,
result: {
errorCode: 0,
errorMsg: "Success",
result: pipelineResult
}
},
pipeline_info: {
tilt,
unwarped,
original_tilt: tilt
}
};
const insertDocRes = await query(`
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash, processing_logs)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
ON CONFLICT (filename) DO UPDATE
SET upload_time = EXCLUDED.upload_time,
size = EXCLUDED.size,
@@ -190,7 +328,8 @@ export async function POST(req: NextRequest) {
metadata = EXCLUDED.metadata,
layout_parsing_result = EXCLUDED.layout_parsing_result,
is_sample = EXCLUDED.is_sample,
file_hash = EXCLUDED.file_hash
file_hash = EXCLUDED.file_hash,
processing_logs = EXCLUDED.processing_logs
RETURNING id
`, [
safeFile,
@@ -200,7 +339,8 @@ export async function POST(req: NextRequest) {
JSON.stringify(docMetadata),
JSON.stringify(pipelineResult),
isSample,
fileHash
fileHash,
JSON.stringify(logsPayload)
]);
const docId = insertDocRes.rows[0].id;
@@ -219,10 +359,11 @@ export async function POST(req: NextRequest) {
jumlah_original, jumlah,
is_flagged, remark
)
VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
VALUES ($1, $2, $3, $4, $5, $6, $6, $7, $7, false, '')
`, [
docId,
i,
(item as any).kodeBarangOriginal || item.kodeBarang,
item.kodeBarang,
item.namaBarang,
item.banyak,
@@ -0,0 +1,233 @@
import { NextRequest, NextResponse } from "next/server";
import { query } from "../../../db";
import { correctVisualDigits } from "../../../utils/parser";
export const dynamic = "force-dynamic";
function levenshteinDistance(s1: string, s2: string): number {
const len1 = s1.length;
const len2 = s2.length;
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
for (let i = 1; i <= len1; i++) {
for (let j = 1; j <= len2; j++) {
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
matrix[i][j] = Math.min(
matrix[i - 1][j] + 1, // deletion
matrix[i][j - 1] + 1, // insertion
matrix[i - 1][j - 1] + cost // substitution
);
}
}
return matrix[len1][len2];
}
function getStringSimilarity(s1: string, s2: string): number {
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
if (!clean1 || !clean2) return 0;
const distance = levenshteinDistance(clean1, clean2);
const maxLength = Math.max(clean1.length, clean2.length);
return (maxLength - distance) / maxLength;
}
// Re-implement cleanDateValue directly so we don't have to deal with exports issues if any
const MONTHS_MAP: Record<string, string> = {
january: "January", januari: "January", janov: "January", jan: "January",
february: "February", februari: "February", feb: "February",
march: "March", maret: "March", mar: "March",
april: "April", apr: "April",
may: "May", mei: "May",
june: "June", juni: "June", jun: "June",
july: "July", juli: "July", jul: "July",
august: "August", agustus: "August", agt: "August", ags: "August", aug: "August",
september: "September", sept: "September", sep: "September",
oktober: "October", october: "October", okt: "October", oct: "October",
november: "November", nopember: "November", nov: "November",
desember: "December", december: "December", des: "December", dec: "December"
};
function cleanDateValue(raw: string): string {
if (!raw) return "Not Found";
const cleaned = raw.trim();
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
const today = new Date();
let day: number | null = null;
let monthStr: string | null = null;
let year: number | null = null;
const yearMatch = cleaned.match(/\b(20\d{2})\b/);
if (yearMatch) {
const parsedYear = parseInt(yearMatch[1], 10);
if (parsedYear >= 2010 && parsedYear <= 2035) {
year = parsedYear;
}
}
const lowerRaw = cleaned.toLowerCase();
const monthsKeys = Object.keys(MONTHS_MAP);
monthsKeys.sort((a, b) => b.length - a.length);
for (const key of monthsKeys) {
if (lowerRaw.includes(key)) {
monthStr = MONTHS_MAP[key] || null;
break;
}
}
let textForDay = cleaned;
if (year) {
textForDay = textForDay.replace(year.toString(), "");
}
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
if (dayMatches) {
for (const matchStr of dayMatches) {
const parsedDay = parseInt(matchStr, 10);
if (parsedDay >= 1 && parsedDay <= 31) {
day = parsedDay;
break;
}
}
}
const currentYear = today.getFullYear();
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
const currentMonth = currentMonthNames[today.getMonth()];
const currentDay = today.getDate();
const finalDay = day !== null ? day : currentDay;
const finalMonth = monthStr !== null ? monthStr : currentMonth;
const finalYear = year !== null ? year : currentYear;
return `${finalDay} ${finalMonth} ${finalYear}`;
}
export async function GET(req: NextRequest) {
const results: string[] = [];
let passed = true;
const assert = (condition: boolean, desc: string) => {
if (condition) {
results.push(`[PASS] ${desc}`);
} else {
results.push(`[FAIL] ${desc}`);
passed = false;
}
};
// 1. Test Visual Digit Correction
const so1 = correctVisualDigits("16O29B7162");
assert(so1 === "1602987162", `correctVisualDigits("16O29B7162") -> got "${so1}", expected "1602987162"`);
const do1 = correctVisualDigits("1602l87");
assert(do1 === "1602187", `correctVisualDigits("1602l87") -> got "${do1}", expected "1602187"`);
const so2 = correctVisualDigits("16O29B7162-OK");
assert(so2 === "1602987162", `correctVisualDigits("16O29B7162-OK") -> got "${so2}", expected "1602987162"`);
// 2. Test Date Lenient Parsing & Fallback Auto-Fill
const today = new Date();
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
const currentMonth = currentMonthNames[today.getMonth()];
const currentDay = today.getDate();
const currentYear = today.getFullYear();
const d1 = cleanDateValue("30-Hv-2026");
assert(d1 === `30 ${currentMonth} 2026`, `cleanDateValue("30-Hv-2026") -> got "${d1}", expected "30 ${currentMonth} 2026"`);
const d2 = cleanDateValue("Hv-Jan-2026");
assert(d2 === `${currentDay} January 2026`, `cleanDateValue("Hv-Jan-2026") -> got "${d2}", expected "${currentDay} January 2026"`);
const d3 = cleanDateValue("30-Jan");
assert(d3 === `30 January ${currentYear}`, `cleanDateValue("30-Jan") -> got "${d3}", expected "30 January ${currentYear}"`);
// 3. Test Two-Way Database SKU Cross-Check
try {
const skuDbRes = await query("SELECT no_sku, nama_item FROM sku_master");
const skuMasterList = skuDbRes.rows.map(row => ({
no_sku: row.no_sku.toString().trim(),
nama_item: row.nama_item.toString().trim()
}));
// Mock an OCR parsed items list
const items = [
{
kodeBarang: "11048006",
namaBarang: "BEBEK PARTING wrong ocr text",
banyak: "10 BAG",
jumlah: "100000"
},
{
kodeBarang: "Not Found",
namaBarang: "CEKER BERKUKU FROZEN PACK",
banyak: "20 KRG",
jumlah: "200000"
},
{
kodeBarang: "Not Found",
namaBarang: "Tanda Tangan Supit",
banyak: "Bag. Pengeluaran Barang",
jumlah: "Bagian Penjualan"
}
];
const checkedItems: typeof items = [];
for (const item of items) {
const ocrSku = item.kodeBarang ? item.kodeBarang.trim() : "";
const ocrName = item.namaBarang ? item.namaBarang.trim() : "";
const matchedBySku = /^\d{8}$/.test(ocrSku) ? skuMasterList.find(sku => sku.no_sku === ocrSku) : null;
if (matchedBySku) {
item.kodeBarang = matchedBySku.no_sku;
item.namaBarang = matchedBySku.nama_item;
checkedItems.push(item);
} else {
let bestMatch: typeof skuMasterList[0] | null = null;
let bestScore = 0;
for (const sku of skuMasterList) {
const score = getStringSimilarity(sku.nama_item, ocrName);
if (score > bestScore) {
bestScore = score;
bestMatch = sku;
}
}
if (bestMatch && bestScore >= 0.6) {
item.kodeBarang = bestMatch.no_sku;
item.namaBarang = bestMatch.nama_item;
checkedItems.push(item);
} else {
if (/^\d{8}$/.test(ocrSku)) {
checkedItems.push(item);
}
}
}
}
// Verify checkedItems length (noise item discarded)
assert(checkedItems.length === 2, `checkedItems length should be 2, got ${checkedItems.length} (noise footer row successfully discarded)`);
// Verify item 1 description correction
assert(checkedItems[0].kodeBarang === "11048006", "Item 1 SKU should remain 11048006");
assert(checkedItems[0].namaBarang === "BEBEK PARTING-NEW(*)", `Item 1 name corrected from DB -> got "${checkedItems[0].namaBarang}"`);
// Verify item 2 SKU fuzzy autocomplete from description
assert(checkedItems[1].kodeBarang === "11110059", `Item 2 SKU autocompleted from DB -> got "${checkedItems[1].kodeBarang}"`);
assert(checkedItems[1].namaBarang === "CEKER BERKUKU FROZEN PACK 1 KG(*)", `Item 2 name corrected from DB -> got "${checkedItems[1].namaBarang}"`);
} catch (err: any) {
passed = false;
results.push(`[ERROR] Database SKU check failed: ${err.message}`);
}
return NextResponse.json({
status: passed ? "success" : "failed",
results
});
}
+18 -27
View File
@@ -35,29 +35,7 @@ export async function POST(req: NextRequest) {
// Compute hash to check for duplicate content
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
// Check if same content already exists in database
const dupRes = await query(
"SELECT filename, layout_parsing_result FROM documents WHERE file_hash = $1",
[fileHash]
);
if (dupRes.rowCount && dupRes.rowCount > 0) {
const existingDoc = dupRes.rows[0];
console.log(`Uploaded file matches existing database record (file_hash: ${fileHash}). Reusing existing file: ${existingDoc.filename}`);
// Reconstruct full response wrapping to match fresh API response
const wrappedResult = {
errorCode: 0,
errorMsg: "Success",
result: existingDoc.layout_parsing_result
};
return NextResponse.json({
filename: existingDoc.filename,
result: wrappedResult,
alreadyExists: true
});
}
fs.writeFileSync(filePath, buffer);
@@ -71,7 +49,7 @@ export async function POST(req: NextRequest) {
useLayoutDetection: true,
fileType: 1,
useDocUnwarping: false,
useDocOrientationClassify: false
useDocOrientationClassify: true
};
const pipelineUrl = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
@@ -87,7 +65,7 @@ export async function POST(req: NextRequest) {
});
if (!response.ok) {
clearActiveLog();
clearActiveLog(filename);
const errText = await response.text();
return NextResponse.json({ error: `Pipeline API error: ${errText}` }, { status: response.status });
}
@@ -96,6 +74,7 @@ export async function POST(req: NextRequest) {
// Check if the image is not straight (tilt > 1.0 degree)
const tilt = calculateAverageTilt(data);
let unwarped = false;
if (tilt > 1.0) {
console.log(`Uploaded document ${filename} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
const unwarpPayload = {
@@ -113,6 +92,7 @@ export async function POST(req: NextRequest) {
if (unwarpResponse.ok) {
data = await unwarpResponse.json();
console.log(`Document unwarped successfully.`);
unwarped = true;
} else {
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
}
@@ -125,6 +105,12 @@ export async function POST(req: NextRequest) {
// Save to PostgreSQL database
try {
const pipelineResult = data.result || data;
pipelineResult.pipeline_info = {
tilt,
unwarped,
original_tilt: tilt
};
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
const markdownText = page0?.markdown?.text || "";
const docMetadata = parseDOMetadata(markdownText);
@@ -149,16 +135,21 @@ export async function POST(req: NextRequest) {
};
// Retrieve and finalize active log data
const activeLog = getActiveLog();
const activeLog = getActiveLog(filename);
let logsPayload: any = null;
if (activeLog && activeLog.filename === filename) {
activeLog.ocr_raw = pipelineResult;
activeLog.stage_1_output = docMetadata;
activeLog.stage_2_output = sanitizedMetadata;
activeLog.frontend_response = clientResponse;
logsPayload = activeLog;
activeLog.pipeline_info = {
tilt,
unwarped,
original_tilt: tilt
};
logsPayload = { ...activeLog };
}
clearActiveLog();
clearActiveLog(filename);
const insertDocRes = await query(`
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash, processing_logs)
@@ -49,42 +49,24 @@ export async function POST(req: NextRequest) {
const latitude = latVal ? parseFloat(latVal.toString()) : null;
const longitude = lngVal ? parseFloat(lngVal.toString()) : null;
// Check if the exact file content already exists in DB
const dupRes = await query(
"SELECT id, filename, latitude, longitude, metadata FROM documents WHERE file_hash = $1 AND is_sample = false",
[fileHash]
);
let docId: number;
let finalFilename = filename;
if (dupRes.rowCount && dupRes.rowCount > 0) {
const existingDoc = dupRes.rows[0];
docId = existingDoc.id;
finalFilename = existingDoc.filename;
console.log(`Reusing existing document record (id: ${docId}) for hash match.`);
// Clean up the newly written file since we are reusing the existing one
if (fs.existsSync(filePath) && finalFilename !== filename) {
fs.unlinkSync(filePath);
}
} else {
const insertRes = await query(`
INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
RETURNING id
`, [
filename,
new Date(),
buffer.length,
false,
false,
fileHash,
latitude,
longitude
]);
docId = insertRes.rows[0].id;
}
const insertRes = await query(`
INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude)
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
RETURNING id
`, [
filename,
new Date(),
buffer.length,
false,
false,
fileHash,
latitude,
longitude
]);
docId = insertRes.rows[0].id;
// Trigger parsing synchronously to ensure it is processed immediately on receiving the image
try {
@@ -1,5 +1,5 @@
import { NextRequest, NextResponse } from "next/server";
import { logVllmCall } from "../../../../utils/active-log";
import { logVllmCallToAll } from "../../../../utils/active-log";
export const dynamic = "force-dynamic";
@@ -87,7 +87,7 @@ async function handleProxy(req: NextRequest) {
if (pathname.includes("/chat/completions") || pathname.includes("/completions")) {
// Make a clean copy of the request to log (hiding huge base64 images if they clutter logs)
const cleanReq = sanitizeLogPayload(reqBody);
logVllmCall(cleanReq, resBody);
logVllmCallToAll(cleanReq, resBody);
}
// Return the response back to pipeline-api
+38 -7
View File
@@ -218,7 +218,38 @@ export default function Home() {
)
},
{
title: `2. vLLM Prompt ("Before")`,
title: "2. Pipeline OCR (Auto-Deskew & Raw JSON)",
description: "Inspect Pipeline API response & deskew actions",
content: (
<div className="space-y-4">
<h4 className="text-sm font-semibold text-teal-400">Pipeline API Processing Info (B)</h4>
{(() => {
const res = selectedDetails?.result || selectedLogs?.ocr_raw;
const info = res?.pipeline_info || selectedLogs?.pipeline_info;
return (
<div className="bg-slate-900/60 border border-slate-700/50 rounded-lg p-4 font-mono text-xs space-y-2 text-slate-300">
<p><span className="text-slate-500">Average Tilt Detected:</span> {info?.tilt !== undefined ? `${parseFloat(info.tilt).toFixed(2)}°` : "N/A"}</p>
<p><span className="text-slate-500">Unwarping / Deskew Triggered:</span> {info?.unwarped !== undefined ? (info.unwarped ? "Yes (Tilt > 1.0°)" : "No (Tilt <= 1.0°)") : "N/A"}</p>
<p><span className="text-slate-500">Original Tilt:</span> {info?.original_tilt !== undefined ? `${parseFloat(info.original_tilt).toFixed(2)}°` : "N/A"}</p>
</div>
);
})()}
<h4 className="text-sm font-semibold text-teal-400">Raw Pipeline Response JSON</h4>
{selectedDetails?.result || selectedLogs?.ocr_raw ? (
<div className="bg-slate-900/60 border border-slate-700/50 rounded-lg p-4 font-mono text-xs overflow-auto max-h-96 text-slate-300">
<pre>{JSON.stringify(selectedDetails?.result || selectedLogs?.ocr_raw, null, 2)}</pre>
</div>
) : (
<div className="bg-slate-900/60 border border-slate-700/50 rounded-lg p-4 text-xs text-slate-400 italic">
No raw pipeline results parsed.
</div>
)}
</div>
)
},
{
title: `3. vLLM Prompt ("Before")`,
description: `Inspect prompt parameters & system role (${vllmCallsCount} call${vllmCallsCount > 1 ? "s" : ""})`,
content: (
<div className="space-y-4">
@@ -244,7 +275,7 @@ export default function Home() {
)
},
{
title: `3. vLLM Response ("After")`,
title: `4. vLLM Response ("After")`,
description: "Inspect raw output returned by VLM",
content: (
<div className="space-y-4">
@@ -270,7 +301,7 @@ export default function Home() {
)
},
{
title: "4. Raw OCR Layout Markdown",
title: "5. Raw OCR Layout Markdown",
description: "View parsed layout markdown structure",
content: (
<div className="space-y-4">
@@ -296,7 +327,7 @@ export default function Home() {
)
},
{
title: "5. Stage 1 Extracted Fields",
title: "6. Stage 1 Extracted Fields",
description: "View output of regex schema matching",
content: (
<div className="space-y-4">
@@ -314,7 +345,7 @@ export default function Home() {
)
},
{
title: "6. Stage 2 Sanitized Output",
title: "7. Stage 2 Sanitized Output",
description: "View results after validation checks",
content: (
<div className="space-y-4">
@@ -332,7 +363,7 @@ export default function Home() {
)
},
{
title: "7. Client Gateway Response",
title: "8. Client Gateway Response",
description: "Inspect JSON payload sent to frontends",
content: (
<div className="space-y-4">
@@ -674,7 +705,7 @@ export default function Home() {
Layer Detail view
</h4>
<span className="text-[10px] bg-slate-800 text-slate-400 px-2 py-0.5 rounded-full font-mono font-semibold">
Layer {activeStep + 1} of 7
Layer {activeStep + 1} of {steps.length}
</span>
</div>
+42 -14
View File
@@ -15,37 +15,65 @@ export interface ActiveUploadLog {
frontend_response?: any;
}
// Store active log in global context to persist across Next.js hot-reloads
// Store active logs in global context as a Map keyed by filename
// This supports concurrent uploads without race conditions
const globalForActiveLog = global as unknown as {
activeLog: ActiveUploadLog | null;
activeLogs: Map<string, ActiveUploadLog>;
};
// Initialise the map once (survives Next.js hot-reloads on the same process)
if (!globalForActiveLog.activeLogs) {
globalForActiveLog.activeLogs = new Map();
}
export function startActiveLog(filename: string) {
globalForActiveLog.activeLog = {
globalForActiveLog.activeLogs.set(filename, {
filename,
vllm_calls: []
};
});
console.log(`[ActiveLog] Started tracking log for ${filename}`);
}
export function logVllmCall(request: any, response: any) {
if (globalForActiveLog.activeLog) {
globalForActiveLog.activeLog.vllm_calls.push({
export function logVllmCall(filename: string, request: any, response: any) {
const log = globalForActiveLog.activeLogs.get(filename);
if (log) {
log.vllm_calls.push({
request,
response,
timestamp: new Date().toISOString()
});
console.log(`[ActiveLog] Logged vLLM call for ${globalForActiveLog.activeLog.filename} (total calls: ${globalForActiveLog.activeLog.vllm_calls.length})`);
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
} else {
console.log("[ActiveLog] Warning: Attempted to log vLLM call but no active log session is running.");
console.log(`[ActiveLog] Warning: Attempted to log vLLM call for "${filename}" but no active log session is running.`);
}
}
export function getActiveLog(): ActiveUploadLog | null {
return globalForActiveLog.activeLog;
export function getActiveLog(filename: string): ActiveUploadLog | null {
return globalForActiveLog.activeLogs.get(filename) || null;
}
export function clearActiveLog() {
globalForActiveLog.activeLog = null;
console.log("[ActiveLog] Cleared active log tracking context");
export function clearActiveLog(filename: string) {
globalForActiveLog.activeLogs.delete(filename);
console.log(`[ActiveLog] Cleared active log tracking context for ${filename}`);
}
/**
* Log a vLLM call to ALL currently active upload sessions.
* Used by the vllm-proxy, which doesn't have per-upload filename context,
* since the pipeline-api processes exactly one upload at a time.
*/
export function logVllmCallToAll(request: any, response: any) {
const sessions = globalForActiveLog.activeLogs;
if (sessions.size === 0) {
console.log("[ActiveLog] Warning: Attempted to log vLLM call but no active log session is running.");
return;
}
for (const [filename, log] of sessions) {
log.vllm_calls.push({
request,
response,
timestamp: new Date().toISOString()
});
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
}
}
+343 -197
View File
@@ -18,31 +18,30 @@ function cleanFinalValue(val: string, preserveNewlines = false): string {
function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
if (!raw || raw === "Not Found") return "Not Found";
// Strip leading label noise like "No. PO : " before matching
// Strip leading label noise like "No. PO : " before matching, allowing common visual confusions
const stripped = raw
.replace(/^No\.?\s*PO\s*[:\-]?\s*/i, "")
.replace(/^No\.?\s*(?:PO|P0|07|70|F0|O0)\s*[:\-]?\s*/i, "")
.trim();
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn
// ALWAYS use currentYearLastTwo — never trust OCR year (can be corrupted)
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn, 07/26/nnn
// Structure: [PREFIX][optional_noise_digits][/][optional_year_segment][/]?[NUMBER]
// We find the LAST slash and take everything after it as the real number
const withSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)\d*[ \t]*[\/\-][ \t]*(?:\d{0,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
const withSlash = /^(?:[A-Z0-9]{1,4})[ \t]*[\/\-][ \t]*(?:\d{1,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
const m1 = stripped.match(withSlash);
if (m1) {
return `PO/${currentYearLastTwo}/${m1[1]}`;
}
// Pattern 2: No slashes — OCR fused: PO12070000190729 or F012070000170727
// Pattern 2: No slashes — OCR fused: PO12070000190729 or F012070000170727, 7012010000100029
// Structure: [PREFIX][digits_with_noise][real_number_starting_0000]
const noSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)(\d+)$/i;
const noSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0|07|70|11|17|76|0|7)(\d+)$/i;
const m2 = stripped.match(noSlash);
if (m2) {
const digits = m2[1];
// Real PO number starts with 0000 in observed patterns
const numberPart = digits.replace(/^\d{2,4}(0{4}\d+)$/, "$1");
if (numberPart && numberPart !== digits) {
return `PO/${currentYearLastTwo}/${numberPart}`;
// Real PO number starts with 0000 (or 000, 00)
const matchNum = digits.match(/(0{2,}\d+)$/);
if (matchNum) {
return `PO/${currentYearLastTwo}/${matchNum[1]}`;
}
// Fallback: strip up to 4 leading noise digits
const fallbackDigits = digits.replace(/^\d{2,4}/, "");
@@ -52,8 +51,12 @@ function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
return `PO/${currentYearLastTwo}/${digits}`;
}
// Pattern 3: Just a raw number (8+ digits) — not a valid PO format
// Pattern 3: Just a raw number (8+ digits) — not a valid PO format unless it has 0000
if (/^\d{8,}$/.test(stripped)) {
const matchNum = stripped.match(/(0{2,}\d+)$/);
if (matchNum) {
return `PO/${currentYearLastTwo}/${matchNum[1]}`;
}
return "Not Found";
}
@@ -70,26 +73,79 @@ function getYearFromDate(dateStr: string): string {
return new Date().getFullYear().toString().slice(-2);
}
const MONTHS_MAP: Record<string, string> = {
january: "January", januari: "January", janov: "January", jan: "January",
february: "February", februari: "February", feb: "February",
march: "March", maret: "March", mar: "March",
april: "April", apr: "April",
may: "May", mei: "May",
june: "June", juni: "June", jun: "June",
july: "July", juli: "July", jul: "July",
august: "August", agustus: "August", agt: "August", ags: "August", aug: "August",
september: "September", sept: "September", sep: "September",
oktober: "October", october: "October", okt: "October", oct: "October",
november: "November", nopember: "November", nov: "November",
desember: "December", december: "December", des: "December", dec: "December"
};
function cleanDateValue(raw: string): string {
if (!raw || raw === "Not Found") return "Not Found";
// Enforce dd Month yyyy pattern (digits, month letters, year digits)
// Permissive of various spacing/dashes/slashes
const pattern = /\b(\d{1,2})[ \t\-\/]*(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)([a-zA-Z]*)[ \t\-\/]*(\d{4})\b/i;
const match = raw.match(pattern);
if (match) {
const day = match[1];
const month = match[2] + match[3];
const year = match[4];
// Capitalize month first letter, keep rest lowercase (e.g. May, June)
const formattedMonth = month.charAt(0).toUpperCase() + month.slice(1).toLowerCase();
return `${day} ${formattedMonth} ${year}`;
if (!raw) return "Not Found";
const cleaned = raw.trim();
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
const today = new Date();
let day: number | null = null;
let monthStr: string | null = null;
let year: number | null = null;
// 1. Try to find 4-digit year (2010 to 2035)
const yearMatch = cleaned.match(/\b(20\d{2})\b/);
if (yearMatch) {
const parsedYear = parseInt(yearMatch[1], 10);
if (parsedYear >= 2010 && parsedYear <= 2035) {
year = parsedYear;
}
}
// Fallback: If no standard date pattern is found, return Not Found
return "Not Found";
// 2. Try to find month using keywords
const lowerRaw = cleaned.toLowerCase();
const monthsKeys = Object.keys(MONTHS_MAP);
monthsKeys.sort((a, b) => b.length - a.length);
for (const key of monthsKeys) {
if (lowerRaw.includes(key)) {
monthStr = MONTHS_MAP[key] || null;
break;
}
}
// 3. Try to find 1 or 2 digit day (not part of the year)
let textForDay = cleaned;
if (year) {
textForDay = textForDay.replace(year.toString(), "");
}
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
if (dayMatches) {
for (const matchStr of dayMatches) {
const parsedDay = parseInt(matchStr, 10);
if (parsedDay >= 1 && parsedDay <= 31) {
day = parsedDay;
break;
}
}
}
// 4. Fallback fill-in from current date
const currentYear = today.getFullYear();
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
const currentMonth = currentMonthNames[today.getMonth()];
const currentDay = today.getDate();
const finalDay = day !== null ? day : currentDay;
const finalMonth = monthStr !== null ? monthStr : currentMonth;
const finalYear = year !== null ? year : currentYear;
return `${finalDay} ${finalMonth} ${finalYear}`;
}
export function parseDOMetadata(markdown: string) {
@@ -363,185 +419,232 @@ export function parseDOMetadata(markdown: string) {
metadata.noSO = cleanFinalValue(metadata.noSO);
metadata.noDO = cleanFinalValue(metadata.noDO);
// Always use current year for fused (no-slash) PO patterns — OCR corrupts year digits
const currentYearLastTwo = new Date().getFullYear().toString().slice(-2);
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), currentYearLastTwo);
// Extract PO year dynamically from parsed document date (or default to current year if date not found)
const docYearLastTwo = getYearFromDate(metadata.tanggal);
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), docYearLastTwo);
// Parse HTML tables for items
const tableRegex = /<table[^>]*>([\s\S]*?)<\/table>/g;
let match;
while ((match = tableRegex.exec(markdown)) !== null) {
const tableHtml = match[1];
// Parse all raw rows and cells first to get td details, including rowspan and colspan
const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/g;
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
interface CellInfo {
text: string;
rowspan: number;
colspan: number;
}
const rawRows: CellInfo[][] = [];
let trMatch;
let rowIndex = 0;
while ((trMatch = trRegex.exec(tableHtml)) !== null) {
const rowHtml = trMatch[1];
const rowCells: CellInfo[] = [];
let tdMatch;
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
const tdHtml = tdMatch[0];
const cellContent = tdMatch[1];
const rsMatch = tdHtml.match(/rowspan=["']?(\d+)["']?/i);
const rowspan = rsMatch ? parseInt(rsMatch[1], 10) : 1;
const csMatch = tdHtml.match(/colspan=["']?(\d+)["']?/i);
const colspan = csMatch ? parseInt(csMatch[1], 10) : 1;
const cellText = cellContent.replace(/<[^>]*>/g, "").trim().replace(/\\n/g, "\n");
rowCells.push({
text: cellText,
rowspan,
colspan
});
}
if (rowCells.length > 0) {
rawRows.push(rowCells);
}
}
if (rawRows.length === 0) continue;
// Determine the maximum columns in the grid
let maxCols = 0;
for (const cell of rawRows[0]) {
maxCols += cell.colspan;
}
const numRows = rawRows.length;
const grid: string[][] = Array.from({ length: numRows }, () => new Array(maxCols).fill(""));
// Fill the grid, respecting rowspan and colspan
for (let r = 0; r < numRows; r++) {
const rowCells = rawRows[r];
let cellIndex = 0;
for (let c = 0; c < maxCols; c++) {
// Skip if already filled
if (grid[r][c] !== "") {
continue;
}
if (cellIndex >= rowCells.length) {
break;
}
const cell = rowCells[cellIndex++];
const lines = cell.text.split("\n").map(l => l.trim()).filter(Boolean);
for (let dr = 0; dr < cell.rowspan; dr++) {
if (r + dr >= numRows) break;
for (let dc = 0; dc < cell.colspan; dc++) {
if (c + dc >= maxCols) break;
let cellValue = cell.text;
if (cell.rowspan > 1 && lines.length > 0) {
cellValue = lines[dr] ?? lines[lines.length - 1] ?? "";
}
grid[r + dr][c + dc] = cellValue;
}
}
}
}
// Process the grid
let kIdx = 0;
let nIdx = 1;
let bIdx = 2;
let jIdx = 3;
let isItemsTable = false;
while ((trMatch = trRegex.exec(tableHtml)) !== null) {
const rowHtml = trMatch[1];
if (rowIndex === 0) {
// Parse header row
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
let tdMatch;
const headerCells: string[] = [];
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
headerCells.push(tdMatch[1].replace(/<[^>]*>/g, "").trim().toLowerCase());
// Check headers in grid[0]
const headerCells = grid[0].map(h => h.toLowerCase());
const foundKode = headerCells.findIndex(h => h.includes("kode") || h.includes("item code"));
const foundNama = headerCells.findIndex(h => h.includes("nama") || h.includes("item name") || h.includes("description"));
const foundBanyak = headerCells.findIndex(h => h.includes("banyak") || h.includes("qty") || h.includes("quantity"));
const foundJumlah = headerCells.findIndex(h => h.includes("jumlah") || h.includes("total"));
if (foundKode !== -1 || foundNama !== -1) {
isItemsTable = true;
kIdx = foundKode !== -1 ? foundKode : 0;
nIdx = foundNama !== -1 ? foundNama : 1;
bIdx = foundBanyak !== -1 ? foundBanyak : 2;
jIdx = foundJumlah !== -1 ? foundJumlah : 3;
}
if (isItemsTable) {
for (let r = 1; r < numRows; r++) {
const cells = grid[r];
const kodeCell = cells[kIdx] || "";
const namaCell = cells[nIdx] || "";
let banyakCell = "";
let jumlahCell = "";
// Check if there is an extra column before banyak that we should merge with banyak
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
const qtyCell = cells[bIdx - 1] || "";
const unitCell = cells[bIdx] || "";
banyakCell = `${qtyCell} ${unitCell}`.trim();
} else {
banyakCell = cells[bIdx] || "";
}
const foundKode = headerCells.findIndex(h => h.includes("kode") || h.includes("item code"));
const foundNama = headerCells.findIndex(h => h.includes("nama") || h.includes("item name") || h.includes("description"));
const foundBanyak = headerCells.findIndex(h => h.includes("banyak") || h.includes("qty") || h.includes("quantity"));
const foundJumlah = headerCells.findIndex(h => h.includes("jumlah") || h.includes("total"));
if (foundKode !== -1 || foundNama !== -1) {
isItemsTable = true;
kIdx = foundKode !== -1 ? foundKode : 0;
nIdx = foundNama !== -1 ? foundNama : 1;
bIdx = foundBanyak !== -1 ? foundBanyak : 2;
jIdx = foundJumlah !== -1 ? foundJumlah : 3;
}
} else {
if (isItemsTable) {
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
let tdMatch;
const cells: string[] = [];
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
// Normalize literal \n text if returned as literal string "\n"
const cellText = tdMatch[1].replace(/<[^>]*>/g, "").trim().replace(/\\n/g, "\n");
cells.push(cellText);
if (jIdx !== -1) {
jumlahCell = cells[jIdx] || "";
} else {
if (cells.length === 5 && bIdx === 3) {
jumlahCell = cells[4] || "";
} else {
jumlahCell = cells[3] || "";
}
if (cells.length >= 3) {
const kodeCell = cells[kIdx] || "";
const namaCell = cells[nIdx] || "";
let banyakCell = "";
let jumlahCell = "";
// Check if there is an extra column before banyak that we should merge with banyak
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
const qtyCell = cells[bIdx - 1] || "";
const unitCell = cells[bIdx] || "";
const qtyLines = qtyCell.split("\n").map(l => l.trim());
const unitLines = unitCell.split("\n").map(l => l.trim());
const combinedLines: string[] = [];
const maxQLen = Math.max(qtyLines.length, unitLines.length);
for (let idx = 0; idx < maxQLen; idx++) {
let q = qtyLines[idx] || "";
const u = unitLines[idx] || "";
// Default to "1" if quantity is missing for a valid item row
const numItems = kodeCell.split("\n").map(p => p.trim()).filter(Boolean).length;
if (!q && idx < numItems) {
q = "1";
}
combinedLines.push(`${q} ${u}`.trim());
}
banyakCell = combinedLines.join("\n");
} else {
banyakCell = cells[bIdx] || "";
}
if (jIdx !== -1) {
jumlahCell = cells[jIdx] || "";
} else {
if (cells.length === 5 && bIdx === 3) {
jumlahCell = cells[4] || "";
} else {
jumlahCell = cells[3] || "";
}
}
// Split cell contents by newlines to support combined rows
const kodeParts = kodeCell.split("\n").map(p => p.trim()).filter(Boolean);
const namaParts = namaCell.split("\n").map(p => p.trim()).filter(Boolean);
const banyakParts = banyakCell.split("\n").map(p => p.trim()).filter(Boolean);
const jumlahParts = jumlahCell.split("\n").map(p => p.trim()).filter(Boolean);
const isWatermark = (s: string) => {
const sl = s.toLowerCase();
return (
sl === "asli" ||
sl === "copy" ||
sl === "nama barang" ||
sl === "tanda tangan supir" ||
sl === "penerima barang" ||
sl === "barang dikirim dalam keadaan baik" ||
sl === "jumlah"
);
};
// Filter watermark keywords from each parts array
const cleanKodes = kodeParts.filter(p => !isWatermark(p));
const cleanNamas = namaParts.filter(p => !isWatermark(p));
let cleanBanyaks = banyakParts.filter(p => !isWatermark(p));
const cleanJumlahs = jumlahParts.filter(p => !isWatermark(p));
if (cleanBanyaks.length === 2 * cleanKodes.length) {
const halved: string[] = [];
const half = cleanKodes.length;
for (let i = 0; i < half; i++) {
const qty = cleanBanyaks[i] || "";
const unit = cleanBanyaks[i + half] || "";
halved.push(`${qty} ${unit}`.trim());
}
cleanBanyaks = halved;
}
const maxLen = Math.max(cleanKodes.length, cleanNamas.length, cleanBanyaks.length, cleanJumlahs.length);
for (let i = 0; i < maxLen; i++) {
const k = cleanKodes[i] || "";
const n = cleanNamas[i] || "";
let b = cleanBanyaks[i] || "";
const j = cleanJumlahs[i] || "";
if (isWatermark(k) || isWatermark(n)) {
continue;
}
// Clean checkmarks and extra spaces from banyak
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
// Fallback for Banyak if empty or purely alphabetical unit
if (!b) {
b = "1";
} else if (/^[a-zA-Z]+$/.test(b)) {
b = `1 ${b}`;
}
// Autocomplete packaging units if Banyak is purely numeric
if (b && /^\d+$/.test(b)) {
const code = k.trim();
const name = n.toLowerCase();
if (code === "11310024" || name.includes("griller")) {
b = `${b} KRG`;
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
b = `${b} BAG`;
}
}
// Validate kodeBarang: must not be blank and must match exactly 8 digits
const cleanKode = k.trim();
if (cleanKode === "" || !/^\d{8}$/.test(cleanKode)) {
continue;
}
metadata.items.push({
kodeBarang: k,
namaBarang: n,
banyak: b,
jumlah: j
});
}
// Split cell contents by newlines to support combined rows
const kodeParts = kodeCell.split("\n").map(p => p.trim()).filter(Boolean);
const namaParts = namaCell.split("\n").map(p => p.trim()).filter(Boolean);
const banyakParts = banyakCell.split("\n").map(p => p.trim()).filter(Boolean);
const jumlahParts = jumlahCell.split("\n").map(p => p.trim()).filter(Boolean);
const isWatermark = (s: string) => {
const sl = s.toLowerCase();
return (
sl === "asli" ||
sl === "copy" ||
sl === "nama barang" ||
sl === "tanda tangan supir" ||
sl === "penerima barang" ||
sl === "barang dikirim dalam keadaan baik" ||
sl === "jumlah"
);
};
// Filter watermark keywords from each parts array
const cleanKodes = kodeParts.filter(p => !isWatermark(p));
const cleanNamas = namaParts.filter(p => !isWatermark(p));
let cleanBanyaks = banyakParts.filter(p => !isWatermark(p));
const cleanJumlahs = jumlahParts.filter(p => !isWatermark(p));
if (cleanBanyaks.length === 2 * cleanKodes.length) {
const halved: string[] = [];
const half = cleanKodes.length;
for (let i = 0; i < half; i++) {
const qty = cleanBanyaks[i] || "";
const unit = cleanBanyaks[i + half] || "";
halved.push(`${qty} ${unit}`.trim());
}
cleanBanyaks = halved;
}
const maxLen = Math.max(cleanKodes.length, cleanNamas.length, cleanBanyaks.length, cleanJumlahs.length);
for (let i = 0; i < maxLen; i++) {
const k = cleanKodes[i] || "";
const n = cleanNamas[i] || "";
let b = cleanBanyaks[i] || "";
const j = cleanJumlahs[i] || "";
if (isWatermark(k) || isWatermark(n)) {
continue;
}
// Clean checkmarks and extra spaces from banyak
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
// Fallback for Banyak if empty or purely alphabetical unit
if (!b) {
b = "1";
} else if (/^[a-zA-Z]+$/.test(b)) {
b = `1 ${b}`;
}
// Autocomplete packaging units if Banyak is purely numeric
if (b && /^\d+$/.test(b)) {
const code = k.trim();
const name = n.toLowerCase();
if (code === "11310024" || name.includes("griller")) {
b = `${b} KRG`;
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
b = `${b} BAG`;
}
}
const cleanKode = k.trim();
const cleanNama = n.trim();
if (cleanKode === "" && cleanNama === "") {
continue;
}
metadata.items.push({
kodeBarang: k,
namaBarang: n,
banyak: b,
jumlah: j
});
}
}
rowIndex++;
}
}
@@ -615,6 +718,44 @@ function formatPlatNumber(raw: string): string {
const VALID_MONTHS = ["January","February","March","April","May","June","July","August","September","October","November","December"];
const MONTH_SHORT = ["Jan","Feb","Mar","Apr","May","Jun","Jul","Aug","Sep","Oct","Nov","Dec"];
export function correctVisualDigits(val: string): string {
if (!val || val === "Not Found") return "";
let cleaned = val.trim();
const parts = cleaned.split(/[^a-zA-Z0-9]+/);
let bestPart = parts[0] || "";
let maxDigitsCount = 0;
for (const part of parts) {
const digitsCount = (part.match(/[0-9]/g) || []).length;
if (digitsCount > maxDigitsCount) {
maxDigitsCount = digitsCount;
bestPart = part;
}
}
if (maxDigitsCount === 0) {
let maxLen = 0;
for (const part of parts) {
if (part.length > maxLen) {
maxLen = part.length;
bestPart = part;
}
}
}
bestPart = bestPart
.replace(/[Oo]/g, "0")
.replace(/[Ii|l]/g, "1")
.replace(/[Bb]/g, "8")
.replace(/[Ss]/g, "5")
.replace(/[Zz]/g, "2")
.replace(/[Gg]/g, "9")
.replace(/[^0-9]/g, "");
return bestPart;
}
export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata> & Record<string, any>): typeof meta {
const currentYY = new Date().getFullYear().toString().slice(-2);
const currentFullYear = new Date().getFullYear();
@@ -641,22 +782,26 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
}
// --- noPO ---
// Must match PO/YY/NNNN+ where YY = current year, NNNN = 4+ digits
// If year segment doesn't match current year, auto-correct it (parser already forces current year,
// but this is a safety net in case anything slipped through)
// Get year from parsed date (or default to current year if date not found)
const docYY = result.tanggal && result.tanggal !== "Not Found" ? getYearFromDate(result.tanggal) : currentYY;
// --- noPO ---
// Must match PO/YY/NNNN+ where YY is the document-specific year segment, NNNN = 4+ digits
const noPO = (result.noPO || "").trim();
const poPattern = /^PO\/(\d{2})\/(\d{4,})$/i;
const pm = noPO.match(poPattern);
if (pm) {
// Auto-correct year to current year regardless of what was parsed
result.noPO = `PO/${currentYY}/${pm[2]}`;
// Keep the parsed year segment if it matches docYY, or fall back to docYY
const yearSegment = pm[1] === docYY ? pm[1] : docYY;
result.noPO = `PO/${yearSegment}/${pm[2]}`;
} else {
result.noPO = "Not Found";
}
// --- noSO ---
// Must be numeric string, 7-12 digits
const noSO = (result.noSO || "").trim();
let noSO = (result.noSO || "").trim();
noSO = correctVisualDigits(noSO);
if (/^\d{7,12}$/.test(noSO)) {
result.noSO = noSO;
} else {
@@ -665,7 +810,8 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
// --- noDO ---
// Must be numeric string, 7-12 digits
const noDO = (result.noDO || "").trim();
let noDO = (result.noDO || "").trim();
noDO = correctVisualDigits(noDO);
if (/^\d{7,12}$/.test(noDO)) {
result.noDO = noDO;
} else {