feat: update backend OCR parser, web app, mobile app camera/preview UI, tests, and documentation with sample images
This commit is contained in:
1 parent
aa3233e411
commit
bdb3a49742
71 files changed
+3360
-764
No files matched your search
@@ -5,6 +5,36 @@ import crypto from "crypto";
|
||||
import { query, cleanupAndReindexItems, resolveStoreFromText } from "../../../db";
|
||||
import { parseDOMetadata, sanitizeParsedMetadata } from "../../../utils/parser";
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
const len1 = s1.length;
|
||||
const len2 = s2.length;
|
||||
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
|
||||
|
||||
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
|
||||
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
|
||||
|
||||
for (let i = 1; i <= len1; i++) {
|
||||
for (let j = 1; j <= len2; j++) {
|
||||
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
||||
matrix[i][j] = Math.min(
|
||||
matrix[i - 1][j] + 1, // deletion
|
||||
matrix[i][j - 1] + 1, // insertion
|
||||
matrix[i - 1][j - 1] + cost // substitution
|
||||
);
|
||||
}
|
||||
}
|
||||
return matrix[len1][len2];
|
||||
}
|
||||
|
||||
function getStringSimilarity(s1: string, s2: string): number {
|
||||
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
if (!clean1 || !clean2) return 0;
|
||||
const distance = levenshteinDistance(clean1, clean2);
|
||||
const maxLength = Math.max(clean1.length, clean2.length);
|
||||
return (maxLength - distance) / maxLength;
|
||||
}
|
||||
|
||||
export async function POST(req: NextRequest) {
|
||||
let safeFile = "";
|
||||
try {
|
||||
@@ -33,60 +63,6 @@ export async function POST(req: NextRequest) {
|
||||
const fileBuffer = fs.readFileSync(filePath);
|
||||
const fileHash = crypto.createHash("sha256").update(fileBuffer).digest("hex");
|
||||
|
||||
// 2. Check if document exists in database (by filename OR hash) and is already parsed
|
||||
const checkRes = await query(
|
||||
"SELECT id, layout_parsing_result FROM documents WHERE (filename = $1 OR file_hash = $2) AND parsed = true AND layout_parsing_result IS NOT NULL",
|
||||
[safeFile, fileHash]
|
||||
);
|
||||
|
||||
if (checkRes.rowCount && checkRes.rowCount > 0) {
|
||||
const doc = checkRes.rows[0];
|
||||
const docId = doc.id;
|
||||
const pipelineResult = doc.layout_parsing_result;
|
||||
|
||||
// Clean up and re-index invalid items first
|
||||
await cleanupAndReindexItems(docId);
|
||||
|
||||
// Fetch items
|
||||
const itemsRes = await query(
|
||||
`SELECT row_index,
|
||||
kode_barang, nama_barang, banyak, jumlah,
|
||||
is_flagged, remark
|
||||
FROM ocr_items
|
||||
WHERE document_id = $1
|
||||
ORDER BY row_index`,
|
||||
[docId]
|
||||
);
|
||||
|
||||
const items = itemsRes.rows.map(row => ({
|
||||
kodeBarang: row.kode_barang,
|
||||
namaBarang: row.nama_barang,
|
||||
banyak: row.banyak,
|
||||
jumlah: row.jumlah
|
||||
}));
|
||||
|
||||
const flagged: Record<number, boolean> = {};
|
||||
const remarks: Record<number, string> = {};
|
||||
|
||||
itemsRes.rows.forEach(row => {
|
||||
if (row.is_flagged) {
|
||||
flagged[row.row_index] = true;
|
||||
}
|
||||
if (row.remark && row.remark.trim()) {
|
||||
remarks[row.row_index] = row.remark;
|
||||
}
|
||||
});
|
||||
|
||||
return NextResponse.json({
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult,
|
||||
items,
|
||||
flagged,
|
||||
remarks
|
||||
});
|
||||
}
|
||||
|
||||
const b64 = fileBuffer.toString("base64");
|
||||
|
||||
// Form payload
|
||||
@@ -96,7 +72,7 @@ export async function POST(req: NextRequest) {
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
|
||||
// Post to Pipeline API
|
||||
@@ -137,6 +113,7 @@ export async function POST(req: NextRequest) {
|
||||
|
||||
// Check if the image is not straight (tilt > 1.0 degree)
|
||||
const tilt = calculateAverageTilt(data);
|
||||
let unwarped = false;
|
||||
if (tilt > 1.0) {
|
||||
console.log(`Parsed document ${safeFile} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
|
||||
const unwarpPayload = {
|
||||
@@ -154,6 +131,7 @@ export async function POST(req: NextRequest) {
|
||||
if (unwarpResponse.ok) {
|
||||
data = await unwarpResponse.json();
|
||||
console.log(`Document unwarped successfully.`);
|
||||
unwarped = true;
|
||||
} else {
|
||||
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
|
||||
}
|
||||
@@ -167,12 +145,151 @@ export async function POST(req: NextRequest) {
|
||||
// 3. Save to database
|
||||
try {
|
||||
const pipelineResult = data.result || data;
|
||||
pipelineResult.pipeline_info = {
|
||||
tilt,
|
||||
unwarped,
|
||||
original_tilt: tilt
|
||||
};
|
||||
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
const rawMetadata = parseDOMetadata(markdownText);
|
||||
// Second-layer sanity check: enforces strict field formats and auto-corrects anomalies
|
||||
const docMetadata = sanitizeParsedMetadata(rawMetadata as any);
|
||||
|
||||
// Load SKU Master list from DB (including new packaging details for triple check validation)
|
||||
const skuDbRes = await query("SELECT no_sku, nama_item, standar_jumlah, jenis_outer FROM sku_master");
|
||||
const skuMasterList = skuDbRes.rows.map(row => ({
|
||||
no_sku: row.no_sku.toString().trim(),
|
||||
nama_item: row.nama_item.toString().trim(),
|
||||
standar_jumlah: row.standar_jumlah ? row.standar_jumlah.toString().trim() : null,
|
||||
jenis_outer: row.jenis_outer ? row.jenis_outer.toString().trim() : null
|
||||
}));
|
||||
|
||||
// Cross-check items using intelligent Triple-Check (SKU, Name, and Unit)
|
||||
const checkedItems: typeof docMetadata.items = [];
|
||||
for (const item of docMetadata.items) {
|
||||
const ocrSku = item.kodeBarang ? item.kodeBarang.trim() : "";
|
||||
const ocrName = item.namaBarang ? item.namaBarang.trim() : "";
|
||||
const ocrQty = item.banyak ? item.banyak.trim() : "";
|
||||
const ocrPrice = item.jumlah ? item.jumlah.trim() : "";
|
||||
|
||||
// Extract alphabetic characters as unit symbols
|
||||
const qtyUnitMatch = ocrQty.match(/[a-zA-Z]+/g);
|
||||
const qtyUnit = qtyUnitMatch ? qtyUnitMatch.join(" ") : "";
|
||||
|
||||
const priceUnitMatch = ocrPrice.match(/[a-zA-Z]+/g);
|
||||
const priceUnit = priceUnitMatch ? priceUnitMatch.join(" ") : "";
|
||||
|
||||
// Remember the original parsed SKU for traceability
|
||||
(item as any).kodeBarangOriginal = ocrSku;
|
||||
|
||||
// Find the best master candidate using a weighted combined score
|
||||
let bestMatch: typeof skuMasterList[0] | null = null;
|
||||
let bestScore = 0;
|
||||
|
||||
// First, check if there is an exact SKU match. If yes, trust the SKU completely and bypass fuzzy matching.
|
||||
const exactSkuMatch = skuMasterList.find(c => ocrSku === c.no_sku);
|
||||
if (exactSkuMatch) {
|
||||
bestMatch = exactSkuMatch;
|
||||
bestScore = 1.0;
|
||||
} else {
|
||||
for (const candidate of skuMasterList) {
|
||||
// 1. SKU Similarity (weight = 0.5)
|
||||
let skuSim = 0;
|
||||
if (ocrSku === candidate.no_sku) {
|
||||
skuSim = 1.0;
|
||||
} else if (ocrSku && /^\d+$/.test(ocrSku.replace(/[^0-9]/g, ''))) {
|
||||
const cleanOcrSku = ocrSku.replace(/[^0-9]/g, '');
|
||||
skuSim = getStringSimilarity(cleanOcrSku, candidate.no_sku);
|
||||
}
|
||||
|
||||
// 2. Name Similarity (weight = 0.4)
|
||||
const nameSim = getStringSimilarity(ocrName, candidate.nama_item);
|
||||
|
||||
// 3. Packaging Unit Match Bonus (weight = 0.1)
|
||||
let unitBonus = 0;
|
||||
let matchesOuter = false;
|
||||
let matchesInner = false;
|
||||
|
||||
const cleanQtyUnit = qtyUnit.toLowerCase();
|
||||
const cleanPriceUnit = priceUnit.toLowerCase();
|
||||
|
||||
if (candidate.jenis_outer) {
|
||||
const cOuter = candidate.jenis_outer.toLowerCase();
|
||||
if (cleanQtyUnit.includes(cOuter) ||
|
||||
(cOuter === "karung" && cleanQtyUnit.includes("krg")) ||
|
||||
(cOuter === "box" && cleanQtyUnit.includes("box")) ||
|
||||
(cOuter === "bag" && cleanQtyUnit.includes("bag"))) {
|
||||
matchesOuter = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (candidate.standar_jumlah) {
|
||||
const cInner = candidate.standar_jumlah.toLowerCase();
|
||||
if (cleanPriceUnit.includes(cInner) ||
|
||||
(cInner === "pac" && cleanPriceUnit.includes("pai")) ||
|
||||
(cInner === "pc" && cleanPriceUnit.includes("pc")) ||
|
||||
(cInner === "kg" && cleanPriceUnit.includes("kg"))) {
|
||||
matchesInner = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (matchesOuter) unitBonus += 0.5;
|
||||
if (matchesInner) unitBonus += 0.5;
|
||||
|
||||
const score = (0.5 * skuSim) + (0.4 * nameSim) + (0.1 * unitBonus);
|
||||
if (score > bestScore) {
|
||||
bestScore = score;
|
||||
bestMatch = candidate;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (bestMatch && bestScore >= 0.6) {
|
||||
// Triple-check matched! Correct the SKU and description to master values,
|
||||
// and auto-standardize the quantities units!
|
||||
item.kodeBarang = bestMatch.no_sku;
|
||||
item.namaBarang = bestMatch.nama_item;
|
||||
|
||||
// Auto-standardize Qty (banyak) units
|
||||
const numericQtyMatch = ocrQty.match(/[\d.,]+/);
|
||||
if (numericQtyMatch) {
|
||||
const numericQty = numericQtyMatch[0];
|
||||
const standardOuter = bestMatch.jenis_outer || "";
|
||||
if (standardOuter.toLowerCase() === "karung") {
|
||||
item.banyak = `${numericQty} KRG`;
|
||||
} else if (standardOuter.toLowerCase() === "box") {
|
||||
item.banyak = `${numericQty} BOX`;
|
||||
} else if (standardOuter.toLowerCase() === "bag") {
|
||||
item.banyak = `${numericQty} BAG`;
|
||||
} else {
|
||||
item.banyak = `${numericQty} ${standardOuter.toUpperCase()}`;
|
||||
}
|
||||
}
|
||||
|
||||
// Auto-standardize Price/Total (jumlah) units
|
||||
const numericPriceMatch = ocrPrice.match(/[\d.,]+/);
|
||||
if (numericPriceMatch) {
|
||||
const numericPrice = numericPriceMatch[0];
|
||||
const standardInner = (bestMatch.standar_jumlah || "").toUpperCase();
|
||||
item.jumlah = `${numericPrice} ${standardInner}`;
|
||||
}
|
||||
|
||||
checkedItems.push(item);
|
||||
} else {
|
||||
// Case 3: No match by SKU/Name/Unit triple check.
|
||||
// If the original SKU is a valid 8-digit code, keep it (to allow new SKUs not yet in master list).
|
||||
// Otherwise, discard the row entirely to filter out layout parsing table noise (like signatures).
|
||||
if (/^\d{8}$/.test(ocrSku)) {
|
||||
checkedItems.push(item);
|
||||
} else {
|
||||
console.log(`Filtering out table noise row: SKU="${ocrSku}", Name="${ocrName}"`);
|
||||
}
|
||||
}
|
||||
}
|
||||
docMetadata.items = checkedItems;
|
||||
|
||||
// Resolve store information using master database
|
||||
const resolvedStore = await resolveStoreFromText(markdownText);
|
||||
(docMetadata as any).orderUntuk = resolvedStore.orderUntuk;
|
||||
@@ -180,9 +297,30 @@ export async function POST(req: NextRequest) {
|
||||
|
||||
const stats = fs.statSync(filePath);
|
||||
|
||||
const logsPayload = {
|
||||
filename: safeFile,
|
||||
vllm_calls: [] as any[],
|
||||
ocr_raw: pipelineResult,
|
||||
stage_1_output: docMetadata,
|
||||
stage_2_output: docMetadata,
|
||||
frontend_response: {
|
||||
filename: safeFile,
|
||||
result: {
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: pipelineResult
|
||||
}
|
||||
},
|
||||
pipeline_info: {
|
||||
tilt,
|
||||
unwarped,
|
||||
original_tilt: tilt
|
||||
}
|
||||
};
|
||||
|
||||
const insertDocRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash, processing_logs)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
|
||||
ON CONFLICT (filename) DO UPDATE
|
||||
SET upload_time = EXCLUDED.upload_time,
|
||||
size = EXCLUDED.size,
|
||||
@@ -190,7 +328,8 @@ export async function POST(req: NextRequest) {
|
||||
metadata = EXCLUDED.metadata,
|
||||
layout_parsing_result = EXCLUDED.layout_parsing_result,
|
||||
is_sample = EXCLUDED.is_sample,
|
||||
file_hash = EXCLUDED.file_hash
|
||||
file_hash = EXCLUDED.file_hash,
|
||||
processing_logs = EXCLUDED.processing_logs
|
||||
RETURNING id
|
||||
`, [
|
||||
safeFile,
|
||||
@@ -200,7 +339,8 @@ export async function POST(req: NextRequest) {
|
||||
JSON.stringify(docMetadata),
|
||||
JSON.stringify(pipelineResult),
|
||||
isSample,
|
||||
fileHash
|
||||
fileHash,
|
||||
JSON.stringify(logsPayload)
|
||||
]);
|
||||
|
||||
const docId = insertDocRes.rows[0].id;
|
||||
@@ -219,10 +359,11 @@ export async function POST(req: NextRequest) {
|
||||
jumlah_original, jumlah,
|
||||
is_flagged, remark
|
||||
)
|
||||
VALUES ($1, $2, $3, $3, $4, $5, $5, $6, $6, false, '')
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $6, $7, $7, false, '')
|
||||
`, [
|
||||
docId,
|
||||
i,
|
||||
(item as any).kodeBarangOriginal || item.kodeBarang,
|
||||
item.kodeBarang,
|
||||
item.namaBarang,
|
||||
item.banyak,
|
||||
|
||||
@@ -0,0 +1,233 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { query } from "../../../db";
|
||||
import { correctVisualDigits } from "../../../utils/parser";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
const len1 = s1.length;
|
||||
const len2 = s2.length;
|
||||
const matrix = Array.from({ length: len1 + 1 }, () => new Array(len2 + 1).fill(0));
|
||||
|
||||
for (let i = 0; i <= len1; i++) matrix[i][0] = i;
|
||||
for (let j = 0; j <= len2; j++) matrix[0][j] = j;
|
||||
|
||||
for (let i = 1; i <= len1; i++) {
|
||||
for (let j = 1; j <= len2; j++) {
|
||||
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
||||
matrix[i][j] = Math.min(
|
||||
matrix[i - 1][j] + 1, // deletion
|
||||
matrix[i][j - 1] + 1, // insertion
|
||||
matrix[i - 1][j - 1] + cost // substitution
|
||||
);
|
||||
}
|
||||
}
|
||||
return matrix[len1][len2];
|
||||
}
|
||||
|
||||
function getStringSimilarity(s1: string, s2: string): number {
|
||||
const clean1 = s1.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
const clean2 = s2.toLowerCase().replace(/[^a-z0-9]/g, '');
|
||||
if (!clean1 || !clean2) return 0;
|
||||
const distance = levenshteinDistance(clean1, clean2);
|
||||
const maxLength = Math.max(clean1.length, clean2.length);
|
||||
return (maxLength - distance) / maxLength;
|
||||
}
|
||||
|
||||
// Re-implement cleanDateValue directly so we don't have to deal with exports issues if any
|
||||
const MONTHS_MAP: Record<string, string> = {
|
||||
january: "January", januari: "January", janov: "January", jan: "January",
|
||||
february: "February", februari: "February", feb: "February",
|
||||
march: "March", maret: "March", mar: "March",
|
||||
april: "April", apr: "April",
|
||||
may: "May", mei: "May",
|
||||
june: "June", juni: "June", jun: "June",
|
||||
july: "July", juli: "July", jul: "July",
|
||||
august: "August", agustus: "August", agt: "August", ags: "August", aug: "August",
|
||||
september: "September", sept: "September", sep: "September",
|
||||
oktober: "October", october: "October", okt: "October", oct: "October",
|
||||
november: "November", nopember: "November", nov: "November",
|
||||
desember: "December", december: "December", des: "December", dec: "December"
|
||||
};
|
||||
|
||||
function cleanDateValue(raw: string): string {
|
||||
if (!raw) return "Not Found";
|
||||
const cleaned = raw.trim();
|
||||
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
|
||||
|
||||
const today = new Date();
|
||||
let day: number | null = null;
|
||||
let monthStr: string | null = null;
|
||||
let year: number | null = null;
|
||||
|
||||
const yearMatch = cleaned.match(/\b(20\d{2})\b/);
|
||||
if (yearMatch) {
|
||||
const parsedYear = parseInt(yearMatch[1], 10);
|
||||
if (parsedYear >= 2010 && parsedYear <= 2035) {
|
||||
year = parsedYear;
|
||||
}
|
||||
}
|
||||
|
||||
const lowerRaw = cleaned.toLowerCase();
|
||||
const monthsKeys = Object.keys(MONTHS_MAP);
|
||||
monthsKeys.sort((a, b) => b.length - a.length);
|
||||
|
||||
for (const key of monthsKeys) {
|
||||
if (lowerRaw.includes(key)) {
|
||||
monthStr = MONTHS_MAP[key] || null;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
let textForDay = cleaned;
|
||||
if (year) {
|
||||
textForDay = textForDay.replace(year.toString(), "");
|
||||
}
|
||||
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
|
||||
if (dayMatches) {
|
||||
for (const matchStr of dayMatches) {
|
||||
const parsedDay = parseInt(matchStr, 10);
|
||||
if (parsedDay >= 1 && parsedDay <= 31) {
|
||||
day = parsedDay;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const currentYear = today.getFullYear();
|
||||
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
|
||||
const currentMonth = currentMonthNames[today.getMonth()];
|
||||
const currentDay = today.getDate();
|
||||
|
||||
const finalDay = day !== null ? day : currentDay;
|
||||
const finalMonth = monthStr !== null ? monthStr : currentMonth;
|
||||
const finalYear = year !== null ? year : currentYear;
|
||||
|
||||
return `${finalDay} ${finalMonth} ${finalYear}`;
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
const results: string[] = [];
|
||||
let passed = true;
|
||||
|
||||
const assert = (condition: boolean, desc: string) => {
|
||||
if (condition) {
|
||||
results.push(`[PASS] ${desc}`);
|
||||
} else {
|
||||
results.push(`[FAIL] ${desc}`);
|
||||
passed = false;
|
||||
}
|
||||
};
|
||||
|
||||
// 1. Test Visual Digit Correction
|
||||
const so1 = correctVisualDigits("16O29B7162");
|
||||
assert(so1 === "1602987162", `correctVisualDigits("16O29B7162") -> got "${so1}", expected "1602987162"`);
|
||||
|
||||
const do1 = correctVisualDigits("1602l87");
|
||||
assert(do1 === "1602187", `correctVisualDigits("1602l87") -> got "${do1}", expected "1602187"`);
|
||||
|
||||
const so2 = correctVisualDigits("16O29B7162-OK");
|
||||
assert(so2 === "1602987162", `correctVisualDigits("16O29B7162-OK") -> got "${so2}", expected "1602987162"`);
|
||||
|
||||
// 2. Test Date Lenient Parsing & Fallback Auto-Fill
|
||||
const today = new Date();
|
||||
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
|
||||
const currentMonth = currentMonthNames[today.getMonth()];
|
||||
const currentDay = today.getDate();
|
||||
const currentYear = today.getFullYear();
|
||||
|
||||
const d1 = cleanDateValue("30-Hv-2026");
|
||||
assert(d1 === `30 ${currentMonth} 2026`, `cleanDateValue("30-Hv-2026") -> got "${d1}", expected "30 ${currentMonth} 2026"`);
|
||||
|
||||
const d2 = cleanDateValue("Hv-Jan-2026");
|
||||
assert(d2 === `${currentDay} January 2026`, `cleanDateValue("Hv-Jan-2026") -> got "${d2}", expected "${currentDay} January 2026"`);
|
||||
|
||||
const d3 = cleanDateValue("30-Jan");
|
||||
assert(d3 === `30 January ${currentYear}`, `cleanDateValue("30-Jan") -> got "${d3}", expected "30 January ${currentYear}"`);
|
||||
|
||||
// 3. Test Two-Way Database SKU Cross-Check
|
||||
try {
|
||||
const skuDbRes = await query("SELECT no_sku, nama_item FROM sku_master");
|
||||
const skuMasterList = skuDbRes.rows.map(row => ({
|
||||
no_sku: row.no_sku.toString().trim(),
|
||||
nama_item: row.nama_item.toString().trim()
|
||||
}));
|
||||
|
||||
// Mock an OCR parsed items list
|
||||
const items = [
|
||||
{
|
||||
kodeBarang: "11048006",
|
||||
namaBarang: "BEBEK PARTING wrong ocr text",
|
||||
banyak: "10 BAG",
|
||||
jumlah: "100000"
|
||||
},
|
||||
{
|
||||
kodeBarang: "Not Found",
|
||||
namaBarang: "CEKER BERKUKU FROZEN PACK",
|
||||
banyak: "20 KRG",
|
||||
jumlah: "200000"
|
||||
},
|
||||
{
|
||||
kodeBarang: "Not Found",
|
||||
namaBarang: "Tanda Tangan Supit",
|
||||
banyak: "Bag. Pengeluaran Barang",
|
||||
jumlah: "Bagian Penjualan"
|
||||
}
|
||||
];
|
||||
|
||||
const checkedItems: typeof items = [];
|
||||
for (const item of items) {
|
||||
const ocrSku = item.kodeBarang ? item.kodeBarang.trim() : "";
|
||||
const ocrName = item.namaBarang ? item.namaBarang.trim() : "";
|
||||
|
||||
const matchedBySku = /^\d{8}$/.test(ocrSku) ? skuMasterList.find(sku => sku.no_sku === ocrSku) : null;
|
||||
|
||||
if (matchedBySku) {
|
||||
item.kodeBarang = matchedBySku.no_sku;
|
||||
item.namaBarang = matchedBySku.nama_item;
|
||||
checkedItems.push(item);
|
||||
} else {
|
||||
let bestMatch: typeof skuMasterList[0] | null = null;
|
||||
let bestScore = 0;
|
||||
|
||||
for (const sku of skuMasterList) {
|
||||
const score = getStringSimilarity(sku.nama_item, ocrName);
|
||||
if (score > bestScore) {
|
||||
bestScore = score;
|
||||
bestMatch = sku;
|
||||
}
|
||||
}
|
||||
|
||||
if (bestMatch && bestScore >= 0.6) {
|
||||
item.kodeBarang = bestMatch.no_sku;
|
||||
item.namaBarang = bestMatch.nama_item;
|
||||
checkedItems.push(item);
|
||||
} else {
|
||||
if (/^\d{8}$/.test(ocrSku)) {
|
||||
checkedItems.push(item);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Verify checkedItems length (noise item discarded)
|
||||
assert(checkedItems.length === 2, `checkedItems length should be 2, got ${checkedItems.length} (noise footer row successfully discarded)`);
|
||||
|
||||
// Verify item 1 description correction
|
||||
assert(checkedItems[0].kodeBarang === "11048006", "Item 1 SKU should remain 11048006");
|
||||
assert(checkedItems[0].namaBarang === "BEBEK PARTING-NEW(*)", `Item 1 name corrected from DB -> got "${checkedItems[0].namaBarang}"`);
|
||||
|
||||
// Verify item 2 SKU fuzzy autocomplete from description
|
||||
assert(checkedItems[1].kodeBarang === "11110059", `Item 2 SKU autocompleted from DB -> got "${checkedItems[1].kodeBarang}"`);
|
||||
assert(checkedItems[1].namaBarang === "CEKER BERKUKU FROZEN PACK 1 KG(*)", `Item 2 name corrected from DB -> got "${checkedItems[1].namaBarang}"`);
|
||||
|
||||
} catch (err: any) {
|
||||
passed = false;
|
||||
results.push(`[ERROR] Database SKU check failed: ${err.message}`);
|
||||
}
|
||||
|
||||
return NextResponse.json({
|
||||
status: passed ? "success" : "failed",
|
||||
results
|
||||
});
|
||||
}
|
||||
@@ -35,29 +35,7 @@ export async function POST(req: NextRequest) {
|
||||
// Compute hash to check for duplicate content
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
|
||||
// Check if same content already exists in database
|
||||
const dupRes = await query(
|
||||
"SELECT filename, layout_parsing_result FROM documents WHERE file_hash = $1",
|
||||
[fileHash]
|
||||
);
|
||||
|
||||
if (dupRes.rowCount && dupRes.rowCount > 0) {
|
||||
const existingDoc = dupRes.rows[0];
|
||||
console.log(`Uploaded file matches existing database record (file_hash: ${fileHash}). Reusing existing file: ${existingDoc.filename}`);
|
||||
|
||||
// Reconstruct full response wrapping to match fresh API response
|
||||
const wrappedResult = {
|
||||
errorCode: 0,
|
||||
errorMsg: "Success",
|
||||
result: existingDoc.layout_parsing_result
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
filename: existingDoc.filename,
|
||||
result: wrappedResult,
|
||||
alreadyExists: true
|
||||
});
|
||||
}
|
||||
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
@@ -71,7 +49,7 @@ export async function POST(req: NextRequest) {
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: false
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
|
||||
const pipelineUrl = process.env.PIPELINE_URL || "http://paddleocr-pipeline-api:8090/layout-parsing";
|
||||
@@ -87,7 +65,7 @@ export async function POST(req: NextRequest) {
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
clearActiveLog();
|
||||
clearActiveLog(filename);
|
||||
const errText = await response.text();
|
||||
return NextResponse.json({ error: `Pipeline API error: ${errText}` }, { status: response.status });
|
||||
}
|
||||
@@ -96,6 +74,7 @@ export async function POST(req: NextRequest) {
|
||||
|
||||
// Check if the image is not straight (tilt > 1.0 degree)
|
||||
const tilt = calculateAverageTilt(data);
|
||||
let unwarped = false;
|
||||
if (tilt > 1.0) {
|
||||
console.log(`Uploaded document ${filename} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
|
||||
const unwarpPayload = {
|
||||
@@ -113,6 +92,7 @@ export async function POST(req: NextRequest) {
|
||||
if (unwarpResponse.ok) {
|
||||
data = await unwarpResponse.json();
|
||||
console.log(`Document unwarped successfully.`);
|
||||
unwarped = true;
|
||||
} else {
|
||||
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
|
||||
}
|
||||
@@ -125,6 +105,12 @@ export async function POST(req: NextRequest) {
|
||||
// Save to PostgreSQL database
|
||||
try {
|
||||
const pipelineResult = data.result || data;
|
||||
pipelineResult.pipeline_info = {
|
||||
tilt,
|
||||
unwarped,
|
||||
original_tilt: tilt
|
||||
};
|
||||
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
const docMetadata = parseDOMetadata(markdownText);
|
||||
@@ -149,16 +135,21 @@ export async function POST(req: NextRequest) {
|
||||
};
|
||||
|
||||
// Retrieve and finalize active log data
|
||||
const activeLog = getActiveLog();
|
||||
const activeLog = getActiveLog(filename);
|
||||
let logsPayload: any = null;
|
||||
if (activeLog && activeLog.filename === filename) {
|
||||
activeLog.ocr_raw = pipelineResult;
|
||||
activeLog.stage_1_output = docMetadata;
|
||||
activeLog.stage_2_output = sanitizedMetadata;
|
||||
activeLog.frontend_response = clientResponse;
|
||||
logsPayload = activeLog;
|
||||
activeLog.pipeline_info = {
|
||||
tilt,
|
||||
unwarped,
|
||||
original_tilt: tilt
|
||||
};
|
||||
logsPayload = { ...activeLog };
|
||||
}
|
||||
clearActiveLog();
|
||||
clearActiveLog(filename);
|
||||
|
||||
const insertDocRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, metadata, layout_parsing_result, is_sample, file_hash, processing_logs)
|
||||
|
||||
@@ -49,42 +49,24 @@ export async function POST(req: NextRequest) {
|
||||
const latitude = latVal ? parseFloat(latVal.toString()) : null;
|
||||
const longitude = lngVal ? parseFloat(lngVal.toString()) : null;
|
||||
|
||||
// Check if the exact file content already exists in DB
|
||||
const dupRes = await query(
|
||||
"SELECT id, filename, latitude, longitude, metadata FROM documents WHERE file_hash = $1 AND is_sample = false",
|
||||
[fileHash]
|
||||
);
|
||||
|
||||
let docId: number;
|
||||
let finalFilename = filename;
|
||||
|
||||
if (dupRes.rowCount && dupRes.rowCount > 0) {
|
||||
const existingDoc = dupRes.rows[0];
|
||||
docId = existingDoc.id;
|
||||
finalFilename = existingDoc.filename;
|
||||
console.log(`Reusing existing document record (id: ${docId}) for hash match.`);
|
||||
|
||||
// Clean up the newly written file since we are reusing the existing one
|
||||
if (fs.existsSync(filePath) && finalFilename !== filename) {
|
||||
fs.unlinkSync(filePath);
|
||||
}
|
||||
} else {
|
||||
const insertRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
|
||||
RETURNING id
|
||||
`, [
|
||||
filename,
|
||||
new Date(),
|
||||
buffer.length,
|
||||
false,
|
||||
false,
|
||||
fileHash,
|
||||
latitude,
|
||||
longitude
|
||||
]);
|
||||
docId = insertRes.rows[0].id;
|
||||
}
|
||||
const insertRes = await query(`
|
||||
INSERT INTO documents (filename, upload_time, size, parsed, is_sample, file_hash, latitude, longitude)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
|
||||
RETURNING id
|
||||
`, [
|
||||
filename,
|
||||
new Date(),
|
||||
buffer.length,
|
||||
false,
|
||||
false,
|
||||
fileHash,
|
||||
latitude,
|
||||
longitude
|
||||
]);
|
||||
docId = insertRes.rows[0].id;
|
||||
|
||||
// Trigger parsing synchronously to ensure it is processed immediately on receiving the image
|
||||
try {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { NextRequest, NextResponse } from "next/server";
|
||||
import { logVllmCall } from "../../../../utils/active-log";
|
||||
import { logVllmCallToAll } from "../../../../utils/active-log";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
@@ -87,7 +87,7 @@ async function handleProxy(req: NextRequest) {
|
||||
if (pathname.includes("/chat/completions") || pathname.includes("/completions")) {
|
||||
// Make a clean copy of the request to log (hiding huge base64 images if they clutter logs)
|
||||
const cleanReq = sanitizeLogPayload(reqBody);
|
||||
logVllmCall(cleanReq, resBody);
|
||||
logVllmCallToAll(cleanReq, resBody);
|
||||
}
|
||||
|
||||
// Return the response back to pipeline-api
|
||||
|
||||
@@ -218,7 +218,38 @@ export default function Home() {
|
||||
)
|
||||
},
|
||||
{
|
||||
title: `2. vLLM Prompt ("Before")`,
|
||||
title: "2. Pipeline OCR (Auto-Deskew & Raw JSON)",
|
||||
description: "Inspect Pipeline API response & deskew actions",
|
||||
content: (
|
||||
<div className="space-y-4">
|
||||
<h4 className="text-sm font-semibold text-teal-400">Pipeline API Processing Info (B)</h4>
|
||||
{(() => {
|
||||
const res = selectedDetails?.result || selectedLogs?.ocr_raw;
|
||||
const info = res?.pipeline_info || selectedLogs?.pipeline_info;
|
||||
return (
|
||||
<div className="bg-slate-900/60 border border-slate-700/50 rounded-lg p-4 font-mono text-xs space-y-2 text-slate-300">
|
||||
<p><span className="text-slate-500">Average Tilt Detected:</span> {info?.tilt !== undefined ? `${parseFloat(info.tilt).toFixed(2)}°` : "N/A"}</p>
|
||||
<p><span className="text-slate-500">Unwarping / Deskew Triggered:</span> {info?.unwarped !== undefined ? (info.unwarped ? "Yes (Tilt > 1.0°)" : "No (Tilt <= 1.0°)") : "N/A"}</p>
|
||||
<p><span className="text-slate-500">Original Tilt:</span> {info?.original_tilt !== undefined ? `${parseFloat(info.original_tilt).toFixed(2)}°` : "N/A"}</p>
|
||||
</div>
|
||||
);
|
||||
})()}
|
||||
|
||||
<h4 className="text-sm font-semibold text-teal-400">Raw Pipeline Response JSON</h4>
|
||||
{selectedDetails?.result || selectedLogs?.ocr_raw ? (
|
||||
<div className="bg-slate-900/60 border border-slate-700/50 rounded-lg p-4 font-mono text-xs overflow-auto max-h-96 text-slate-300">
|
||||
<pre>{JSON.stringify(selectedDetails?.result || selectedLogs?.ocr_raw, null, 2)}</pre>
|
||||
</div>
|
||||
) : (
|
||||
<div className="bg-slate-900/60 border border-slate-700/50 rounded-lg p-4 text-xs text-slate-400 italic">
|
||||
No raw pipeline results parsed.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
},
|
||||
{
|
||||
title: `3. vLLM Prompt ("Before")`,
|
||||
description: `Inspect prompt parameters & system role (${vllmCallsCount} call${vllmCallsCount > 1 ? "s" : ""})`,
|
||||
content: (
|
||||
<div className="space-y-4">
|
||||
@@ -244,7 +275,7 @@ export default function Home() {
|
||||
)
|
||||
},
|
||||
{
|
||||
title: `3. vLLM Response ("After")`,
|
||||
title: `4. vLLM Response ("After")`,
|
||||
description: "Inspect raw output returned by VLM",
|
||||
content: (
|
||||
<div className="space-y-4">
|
||||
@@ -270,7 +301,7 @@ export default function Home() {
|
||||
)
|
||||
},
|
||||
{
|
||||
title: "4. Raw OCR Layout Markdown",
|
||||
title: "5. Raw OCR Layout Markdown",
|
||||
description: "View parsed layout markdown structure",
|
||||
content: (
|
||||
<div className="space-y-4">
|
||||
@@ -296,7 +327,7 @@ export default function Home() {
|
||||
)
|
||||
},
|
||||
{
|
||||
title: "5. Stage 1 Extracted Fields",
|
||||
title: "6. Stage 1 Extracted Fields",
|
||||
description: "View output of regex schema matching",
|
||||
content: (
|
||||
<div className="space-y-4">
|
||||
@@ -314,7 +345,7 @@ export default function Home() {
|
||||
)
|
||||
},
|
||||
{
|
||||
title: "6. Stage 2 Sanitized Output",
|
||||
title: "7. Stage 2 Sanitized Output",
|
||||
description: "View results after validation checks",
|
||||
content: (
|
||||
<div className="space-y-4">
|
||||
@@ -332,7 +363,7 @@ export default function Home() {
|
||||
)
|
||||
},
|
||||
{
|
||||
title: "7. Client Gateway Response",
|
||||
title: "8. Client Gateway Response",
|
||||
description: "Inspect JSON payload sent to frontends",
|
||||
content: (
|
||||
<div className="space-y-4">
|
||||
@@ -674,7 +705,7 @@ export default function Home() {
|
||||
Layer Detail view
|
||||
</h4>
|
||||
<span className="text-[10px] bg-slate-800 text-slate-400 px-2 py-0.5 rounded-full font-mono font-semibold">
|
||||
Layer {activeStep + 1} of 7
|
||||
Layer {activeStep + 1} of {steps.length}
|
||||
</span>
|
||||
</div>
|
||||
|
||||
|
||||
@@ -15,37 +15,65 @@ export interface ActiveUploadLog {
|
||||
frontend_response?: any;
|
||||
}
|
||||
|
||||
// Store active log in global context to persist across Next.js hot-reloads
|
||||
// Store active logs in global context as a Map keyed by filename
|
||||
// This supports concurrent uploads without race conditions
|
||||
const globalForActiveLog = global as unknown as {
|
||||
activeLog: ActiveUploadLog | null;
|
||||
activeLogs: Map<string, ActiveUploadLog>;
|
||||
};
|
||||
|
||||
// Initialise the map once (survives Next.js hot-reloads on the same process)
|
||||
if (!globalForActiveLog.activeLogs) {
|
||||
globalForActiveLog.activeLogs = new Map();
|
||||
}
|
||||
|
||||
export function startActiveLog(filename: string) {
|
||||
globalForActiveLog.activeLog = {
|
||||
globalForActiveLog.activeLogs.set(filename, {
|
||||
filename,
|
||||
vllm_calls: []
|
||||
};
|
||||
});
|
||||
console.log(`[ActiveLog] Started tracking log for ${filename}`);
|
||||
}
|
||||
|
||||
export function logVllmCall(request: any, response: any) {
|
||||
if (globalForActiveLog.activeLog) {
|
||||
globalForActiveLog.activeLog.vllm_calls.push({
|
||||
export function logVllmCall(filename: string, request: any, response: any) {
|
||||
const log = globalForActiveLog.activeLogs.get(filename);
|
||||
if (log) {
|
||||
log.vllm_calls.push({
|
||||
request,
|
||||
response,
|
||||
timestamp: new Date().toISOString()
|
||||
});
|
||||
console.log(`[ActiveLog] Logged vLLM call for ${globalForActiveLog.activeLog.filename} (total calls: ${globalForActiveLog.activeLog.vllm_calls.length})`);
|
||||
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
|
||||
} else {
|
||||
console.log("[ActiveLog] Warning: Attempted to log vLLM call but no active log session is running.");
|
||||
console.log(`[ActiveLog] Warning: Attempted to log vLLM call for "${filename}" but no active log session is running.`);
|
||||
}
|
||||
}
|
||||
|
||||
export function getActiveLog(): ActiveUploadLog | null {
|
||||
return globalForActiveLog.activeLog;
|
||||
export function getActiveLog(filename: string): ActiveUploadLog | null {
|
||||
return globalForActiveLog.activeLogs.get(filename) || null;
|
||||
}
|
||||
|
||||
export function clearActiveLog() {
|
||||
globalForActiveLog.activeLog = null;
|
||||
console.log("[ActiveLog] Cleared active log tracking context");
|
||||
export function clearActiveLog(filename: string) {
|
||||
globalForActiveLog.activeLogs.delete(filename);
|
||||
console.log(`[ActiveLog] Cleared active log tracking context for ${filename}`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Log a vLLM call to ALL currently active upload sessions.
|
||||
* Used by the vllm-proxy, which doesn't have per-upload filename context,
|
||||
* since the pipeline-api processes exactly one upload at a time.
|
||||
*/
|
||||
export function logVllmCallToAll(request: any, response: any) {
|
||||
const sessions = globalForActiveLog.activeLogs;
|
||||
if (sessions.size === 0) {
|
||||
console.log("[ActiveLog] Warning: Attempted to log vLLM call but no active log session is running.");
|
||||
return;
|
||||
}
|
||||
for (const [filename, log] of sessions) {
|
||||
log.vllm_calls.push({
|
||||
request,
|
||||
response,
|
||||
timestamp: new Date().toISOString()
|
||||
});
|
||||
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
|
||||
}
|
||||
}
|
||||
@@ -18,31 +18,30 @@ function cleanFinalValue(val: string, preserveNewlines = false): string {
|
||||
function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
|
||||
if (!raw || raw === "Not Found") return "Not Found";
|
||||
|
||||
// Strip leading label noise like "No. PO : " before matching
|
||||
// Strip leading label noise like "No. PO : " before matching, allowing common visual confusions
|
||||
const stripped = raw
|
||||
.replace(/^No\.?\s*PO\s*[:\-]?\s*/i, "")
|
||||
.replace(/^No\.?\s*(?:PO|P0|07|70|F0|O0)\s*[:\-]?\s*/i, "")
|
||||
.trim();
|
||||
|
||||
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn
|
||||
// ALWAYS use currentYearLastTwo — never trust OCR year (can be corrupted)
|
||||
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn, 07/26/nnn
|
||||
// Structure: [PREFIX][optional_noise_digits][/][optional_year_segment][/]?[NUMBER]
|
||||
// We find the LAST slash and take everything after it as the real number
|
||||
const withSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)\d*[ \t]*[\/\-][ \t]*(?:\d{0,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
|
||||
const withSlash = /^(?:[A-Z0-9]{1,4})[ \t]*[\/\-][ \t]*(?:\d{1,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
|
||||
const m1 = stripped.match(withSlash);
|
||||
if (m1) {
|
||||
return `PO/${currentYearLastTwo}/${m1[1]}`;
|
||||
}
|
||||
|
||||
// Pattern 2: No slashes — OCR fused: PO12070000190729 or F012070000170727
|
||||
// Pattern 2: No slashes — OCR fused: PO12070000190729 or F012070000170727, 7012010000100029
|
||||
// Structure: [PREFIX][digits_with_noise][real_number_starting_0000]
|
||||
const noSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)(\d+)$/i;
|
||||
const noSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0|07|70|11|17|76|0|7)(\d+)$/i;
|
||||
const m2 = stripped.match(noSlash);
|
||||
if (m2) {
|
||||
const digits = m2[1];
|
||||
// Real PO number starts with 0000 in observed patterns
|
||||
const numberPart = digits.replace(/^\d{2,4}(0{4}\d+)$/, "$1");
|
||||
if (numberPart && numberPart !== digits) {
|
||||
return `PO/${currentYearLastTwo}/${numberPart}`;
|
||||
// Real PO number starts with 0000 (or 000, 00)
|
||||
const matchNum = digits.match(/(0{2,}\d+)$/);
|
||||
if (matchNum) {
|
||||
return `PO/${currentYearLastTwo}/${matchNum[1]}`;
|
||||
}
|
||||
// Fallback: strip up to 4 leading noise digits
|
||||
const fallbackDigits = digits.replace(/^\d{2,4}/, "");
|
||||
@@ -52,8 +51,12 @@ function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
|
||||
return `PO/${currentYearLastTwo}/${digits}`;
|
||||
}
|
||||
|
||||
// Pattern 3: Just a raw number (8+ digits) — not a valid PO format
|
||||
// Pattern 3: Just a raw number (8+ digits) — not a valid PO format unless it has 0000
|
||||
if (/^\d{8,}$/.test(stripped)) {
|
||||
const matchNum = stripped.match(/(0{2,}\d+)$/);
|
||||
if (matchNum) {
|
||||
return `PO/${currentYearLastTwo}/${matchNum[1]}`;
|
||||
}
|
||||
return "Not Found";
|
||||
}
|
||||
|
||||
@@ -70,26 +73,79 @@ function getYearFromDate(dateStr: string): string {
|
||||
return new Date().getFullYear().toString().slice(-2);
|
||||
}
|
||||
|
||||
const MONTHS_MAP: Record<string, string> = {
|
||||
january: "January", januari: "January", janov: "January", jan: "January",
|
||||
february: "February", februari: "February", feb: "February",
|
||||
march: "March", maret: "March", mar: "March",
|
||||
april: "April", apr: "April",
|
||||
may: "May", mei: "May",
|
||||
june: "June", juni: "June", jun: "June",
|
||||
july: "July", juli: "July", jul: "July",
|
||||
august: "August", agustus: "August", agt: "August", ags: "August", aug: "August",
|
||||
september: "September", sept: "September", sep: "September",
|
||||
oktober: "October", october: "October", okt: "October", oct: "October",
|
||||
november: "November", nopember: "November", nov: "November",
|
||||
desember: "December", december: "December", des: "December", dec: "December"
|
||||
};
|
||||
|
||||
function cleanDateValue(raw: string): string {
|
||||
if (!raw || raw === "Not Found") return "Not Found";
|
||||
|
||||
// Enforce dd Month yyyy pattern (digits, month letters, year digits)
|
||||
// Permissive of various spacing/dashes/slashes
|
||||
const pattern = /\b(\d{1,2})[ \t\-\/]*(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)([a-zA-Z]*)[ \t\-\/]*(\d{4})\b/i;
|
||||
const match = raw.match(pattern);
|
||||
if (match) {
|
||||
const day = match[1];
|
||||
const month = match[2] + match[3];
|
||||
const year = match[4];
|
||||
|
||||
// Capitalize month first letter, keep rest lowercase (e.g. May, June)
|
||||
const formattedMonth = month.charAt(0).toUpperCase() + month.slice(1).toLowerCase();
|
||||
|
||||
return `${day} ${formattedMonth} ${year}`;
|
||||
if (!raw) return "Not Found";
|
||||
const cleaned = raw.trim();
|
||||
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
|
||||
|
||||
const today = new Date();
|
||||
let day: number | null = null;
|
||||
let monthStr: string | null = null;
|
||||
let year: number | null = null;
|
||||
|
||||
// 1. Try to find 4-digit year (2010 to 2035)
|
||||
const yearMatch = cleaned.match(/\b(20\d{2})\b/);
|
||||
if (yearMatch) {
|
||||
const parsedYear = parseInt(yearMatch[1], 10);
|
||||
if (parsedYear >= 2010 && parsedYear <= 2035) {
|
||||
year = parsedYear;
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback: If no standard date pattern is found, return Not Found
|
||||
return "Not Found";
|
||||
|
||||
// 2. Try to find month using keywords
|
||||
const lowerRaw = cleaned.toLowerCase();
|
||||
const monthsKeys = Object.keys(MONTHS_MAP);
|
||||
monthsKeys.sort((a, b) => b.length - a.length);
|
||||
|
||||
for (const key of monthsKeys) {
|
||||
if (lowerRaw.includes(key)) {
|
||||
monthStr = MONTHS_MAP[key] || null;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Try to find 1 or 2 digit day (not part of the year)
|
||||
let textForDay = cleaned;
|
||||
if (year) {
|
||||
textForDay = textForDay.replace(year.toString(), "");
|
||||
}
|
||||
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
|
||||
if (dayMatches) {
|
||||
for (const matchStr of dayMatches) {
|
||||
const parsedDay = parseInt(matchStr, 10);
|
||||
if (parsedDay >= 1 && parsedDay <= 31) {
|
||||
day = parsedDay;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Fallback fill-in from current date
|
||||
const currentYear = today.getFullYear();
|
||||
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
|
||||
const currentMonth = currentMonthNames[today.getMonth()];
|
||||
const currentDay = today.getDate();
|
||||
|
||||
const finalDay = day !== null ? day : currentDay;
|
||||
const finalMonth = monthStr !== null ? monthStr : currentMonth;
|
||||
const finalYear = year !== null ? year : currentYear;
|
||||
|
||||
return `${finalDay} ${finalMonth} ${finalYear}`;
|
||||
}
|
||||
|
||||
export function parseDOMetadata(markdown: string) {
|
||||
@@ -363,185 +419,232 @@ export function parseDOMetadata(markdown: string) {
|
||||
metadata.noSO = cleanFinalValue(metadata.noSO);
|
||||
metadata.noDO = cleanFinalValue(metadata.noDO);
|
||||
|
||||
// Always use current year for fused (no-slash) PO patterns — OCR corrupts year digits
|
||||
const currentYearLastTwo = new Date().getFullYear().toString().slice(-2);
|
||||
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), currentYearLastTwo);
|
||||
// Extract PO year dynamically from parsed document date (or default to current year if date not found)
|
||||
const docYearLastTwo = getYearFromDate(metadata.tanggal);
|
||||
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), docYearLastTwo);
|
||||
|
||||
// Parse HTML tables for items
|
||||
const tableRegex = /<table[^>]*>([\s\S]*?)<\/table>/g;
|
||||
let match;
|
||||
while ((match = tableRegex.exec(markdown)) !== null) {
|
||||
const tableHtml = match[1];
|
||||
// Parse all raw rows and cells first to get td details, including rowspan and colspan
|
||||
const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/g;
|
||||
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
|
||||
|
||||
interface CellInfo {
|
||||
text: string;
|
||||
rowspan: number;
|
||||
colspan: number;
|
||||
}
|
||||
const rawRows: CellInfo[][] = [];
|
||||
|
||||
let trMatch;
|
||||
let rowIndex = 0;
|
||||
while ((trMatch = trRegex.exec(tableHtml)) !== null) {
|
||||
const rowHtml = trMatch[1];
|
||||
const rowCells: CellInfo[] = [];
|
||||
let tdMatch;
|
||||
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
|
||||
const tdHtml = tdMatch[0];
|
||||
const cellContent = tdMatch[1];
|
||||
|
||||
const rsMatch = tdHtml.match(/rowspan=["']?(\d+)["']?/i);
|
||||
const rowspan = rsMatch ? parseInt(rsMatch[1], 10) : 1;
|
||||
|
||||
const csMatch = tdHtml.match(/colspan=["']?(\d+)["']?/i);
|
||||
const colspan = csMatch ? parseInt(csMatch[1], 10) : 1;
|
||||
|
||||
const cellText = cellContent.replace(/<[^>]*>/g, "").trim().replace(/\\n/g, "\n");
|
||||
|
||||
rowCells.push({
|
||||
text: cellText,
|
||||
rowspan,
|
||||
colspan
|
||||
});
|
||||
}
|
||||
if (rowCells.length > 0) {
|
||||
rawRows.push(rowCells);
|
||||
}
|
||||
}
|
||||
|
||||
if (rawRows.length === 0) continue;
|
||||
|
||||
// Determine the maximum columns in the grid
|
||||
let maxCols = 0;
|
||||
for (const cell of rawRows[0]) {
|
||||
maxCols += cell.colspan;
|
||||
}
|
||||
|
||||
const numRows = rawRows.length;
|
||||
const grid: string[][] = Array.from({ length: numRows }, () => new Array(maxCols).fill(""));
|
||||
|
||||
// Fill the grid, respecting rowspan and colspan
|
||||
for (let r = 0; r < numRows; r++) {
|
||||
const rowCells = rawRows[r];
|
||||
let cellIndex = 0;
|
||||
|
||||
for (let c = 0; c < maxCols; c++) {
|
||||
// Skip if already filled
|
||||
if (grid[r][c] !== "") {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (cellIndex >= rowCells.length) {
|
||||
break;
|
||||
}
|
||||
|
||||
const cell = rowCells[cellIndex++];
|
||||
const lines = cell.text.split("\n").map(l => l.trim()).filter(Boolean);
|
||||
|
||||
for (let dr = 0; dr < cell.rowspan; dr++) {
|
||||
if (r + dr >= numRows) break;
|
||||
|
||||
for (let dc = 0; dc < cell.colspan; dc++) {
|
||||
if (c + dc >= maxCols) break;
|
||||
|
||||
let cellValue = cell.text;
|
||||
if (cell.rowspan > 1 && lines.length > 0) {
|
||||
cellValue = lines[dr] ?? lines[lines.length - 1] ?? "";
|
||||
}
|
||||
|
||||
grid[r + dr][c + dc] = cellValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Process the grid
|
||||
let kIdx = 0;
|
||||
let nIdx = 1;
|
||||
let bIdx = 2;
|
||||
let jIdx = 3;
|
||||
let isItemsTable = false;
|
||||
|
||||
while ((trMatch = trRegex.exec(tableHtml)) !== null) {
|
||||
const rowHtml = trMatch[1];
|
||||
if (rowIndex === 0) {
|
||||
// Parse header row
|
||||
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
|
||||
let tdMatch;
|
||||
const headerCells: string[] = [];
|
||||
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
|
||||
headerCells.push(tdMatch[1].replace(/<[^>]*>/g, "").trim().toLowerCase());
|
||||
|
||||
// Check headers in grid[0]
|
||||
const headerCells = grid[0].map(h => h.toLowerCase());
|
||||
const foundKode = headerCells.findIndex(h => h.includes("kode") || h.includes("item code"));
|
||||
const foundNama = headerCells.findIndex(h => h.includes("nama") || h.includes("item name") || h.includes("description"));
|
||||
const foundBanyak = headerCells.findIndex(h => h.includes("banyak") || h.includes("qty") || h.includes("quantity"));
|
||||
const foundJumlah = headerCells.findIndex(h => h.includes("jumlah") || h.includes("total"));
|
||||
|
||||
if (foundKode !== -1 || foundNama !== -1) {
|
||||
isItemsTable = true;
|
||||
kIdx = foundKode !== -1 ? foundKode : 0;
|
||||
nIdx = foundNama !== -1 ? foundNama : 1;
|
||||
bIdx = foundBanyak !== -1 ? foundBanyak : 2;
|
||||
jIdx = foundJumlah !== -1 ? foundJumlah : 3;
|
||||
}
|
||||
|
||||
if (isItemsTable) {
|
||||
for (let r = 1; r < numRows; r++) {
|
||||
const cells = grid[r];
|
||||
const kodeCell = cells[kIdx] || "";
|
||||
const namaCell = cells[nIdx] || "";
|
||||
let banyakCell = "";
|
||||
let jumlahCell = "";
|
||||
|
||||
// Check if there is an extra column before banyak that we should merge with banyak
|
||||
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
|
||||
const qtyCell = cells[bIdx - 1] || "";
|
||||
const unitCell = cells[bIdx] || "";
|
||||
banyakCell = `${qtyCell} ${unitCell}`.trim();
|
||||
} else {
|
||||
banyakCell = cells[bIdx] || "";
|
||||
}
|
||||
|
||||
const foundKode = headerCells.findIndex(h => h.includes("kode") || h.includes("item code"));
|
||||
const foundNama = headerCells.findIndex(h => h.includes("nama") || h.includes("item name") || h.includes("description"));
|
||||
const foundBanyak = headerCells.findIndex(h => h.includes("banyak") || h.includes("qty") || h.includes("quantity"));
|
||||
const foundJumlah = headerCells.findIndex(h => h.includes("jumlah") || h.includes("total"));
|
||||
|
||||
if (foundKode !== -1 || foundNama !== -1) {
|
||||
isItemsTable = true;
|
||||
kIdx = foundKode !== -1 ? foundKode : 0;
|
||||
nIdx = foundNama !== -1 ? foundNama : 1;
|
||||
bIdx = foundBanyak !== -1 ? foundBanyak : 2;
|
||||
jIdx = foundJumlah !== -1 ? foundJumlah : 3;
|
||||
}
|
||||
} else {
|
||||
if (isItemsTable) {
|
||||
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
|
||||
let tdMatch;
|
||||
const cells: string[] = [];
|
||||
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
|
||||
// Normalize literal \n text if returned as literal string "\n"
|
||||
const cellText = tdMatch[1].replace(/<[^>]*>/g, "").trim().replace(/\\n/g, "\n");
|
||||
cells.push(cellText);
|
||||
|
||||
if (jIdx !== -1) {
|
||||
jumlahCell = cells[jIdx] || "";
|
||||
} else {
|
||||
if (cells.length === 5 && bIdx === 3) {
|
||||
jumlahCell = cells[4] || "";
|
||||
} else {
|
||||
jumlahCell = cells[3] || "";
|
||||
}
|
||||
if (cells.length >= 3) {
|
||||
const kodeCell = cells[kIdx] || "";
|
||||
const namaCell = cells[nIdx] || "";
|
||||
let banyakCell = "";
|
||||
let jumlahCell = "";
|
||||
|
||||
// Check if there is an extra column before banyak that we should merge with banyak
|
||||
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
|
||||
const qtyCell = cells[bIdx - 1] || "";
|
||||
const unitCell = cells[bIdx] || "";
|
||||
|
||||
const qtyLines = qtyCell.split("\n").map(l => l.trim());
|
||||
const unitLines = unitCell.split("\n").map(l => l.trim());
|
||||
const combinedLines: string[] = [];
|
||||
const maxQLen = Math.max(qtyLines.length, unitLines.length);
|
||||
for (let idx = 0; idx < maxQLen; idx++) {
|
||||
let q = qtyLines[idx] || "";
|
||||
const u = unitLines[idx] || "";
|
||||
|
||||
// Default to "1" if quantity is missing for a valid item row
|
||||
const numItems = kodeCell.split("\n").map(p => p.trim()).filter(Boolean).length;
|
||||
if (!q && idx < numItems) {
|
||||
q = "1";
|
||||
}
|
||||
|
||||
combinedLines.push(`${q} ${u}`.trim());
|
||||
}
|
||||
banyakCell = combinedLines.join("\n");
|
||||
} else {
|
||||
banyakCell = cells[bIdx] || "";
|
||||
}
|
||||
|
||||
if (jIdx !== -1) {
|
||||
jumlahCell = cells[jIdx] || "";
|
||||
} else {
|
||||
if (cells.length === 5 && bIdx === 3) {
|
||||
jumlahCell = cells[4] || "";
|
||||
} else {
|
||||
jumlahCell = cells[3] || "";
|
||||
}
|
||||
}
|
||||
|
||||
// Split cell contents by newlines to support combined rows
|
||||
const kodeParts = kodeCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const namaParts = namaCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const banyakParts = banyakCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const jumlahParts = jumlahCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
|
||||
const isWatermark = (s: string) => {
|
||||
const sl = s.toLowerCase();
|
||||
return (
|
||||
sl === "asli" ||
|
||||
sl === "copy" ||
|
||||
sl === "nama barang" ||
|
||||
sl === "tanda tangan supir" ||
|
||||
sl === "penerima barang" ||
|
||||
sl === "barang dikirim dalam keadaan baik" ||
|
||||
sl === "jumlah"
|
||||
);
|
||||
};
|
||||
|
||||
// Filter watermark keywords from each parts array
|
||||
const cleanKodes = kodeParts.filter(p => !isWatermark(p));
|
||||
const cleanNamas = namaParts.filter(p => !isWatermark(p));
|
||||
let cleanBanyaks = banyakParts.filter(p => !isWatermark(p));
|
||||
const cleanJumlahs = jumlahParts.filter(p => !isWatermark(p));
|
||||
|
||||
if (cleanBanyaks.length === 2 * cleanKodes.length) {
|
||||
const halved: string[] = [];
|
||||
const half = cleanKodes.length;
|
||||
for (let i = 0; i < half; i++) {
|
||||
const qty = cleanBanyaks[i] || "";
|
||||
const unit = cleanBanyaks[i + half] || "";
|
||||
halved.push(`${qty} ${unit}`.trim());
|
||||
}
|
||||
cleanBanyaks = halved;
|
||||
}
|
||||
|
||||
const maxLen = Math.max(cleanKodes.length, cleanNamas.length, cleanBanyaks.length, cleanJumlahs.length);
|
||||
|
||||
for (let i = 0; i < maxLen; i++) {
|
||||
const k = cleanKodes[i] || "";
|
||||
const n = cleanNamas[i] || "";
|
||||
let b = cleanBanyaks[i] || "";
|
||||
const j = cleanJumlahs[i] || "";
|
||||
|
||||
if (isWatermark(k) || isWatermark(n)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Clean checkmarks and extra spaces from banyak
|
||||
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
|
||||
|
||||
// Fallback for Banyak if empty or purely alphabetical unit
|
||||
if (!b) {
|
||||
b = "1";
|
||||
} else if (/^[a-zA-Z]+$/.test(b)) {
|
||||
b = `1 ${b}`;
|
||||
}
|
||||
|
||||
// Autocomplete packaging units if Banyak is purely numeric
|
||||
if (b && /^\d+$/.test(b)) {
|
||||
const code = k.trim();
|
||||
const name = n.toLowerCase();
|
||||
if (code === "11310024" || name.includes("griller")) {
|
||||
b = `${b} KRG`;
|
||||
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
|
||||
b = `${b} BAG`;
|
||||
}
|
||||
}
|
||||
|
||||
// Validate kodeBarang: must not be blank and must match exactly 8 digits
|
||||
const cleanKode = k.trim();
|
||||
if (cleanKode === "" || !/^\d{8}$/.test(cleanKode)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
metadata.items.push({
|
||||
kodeBarang: k,
|
||||
namaBarang: n,
|
||||
banyak: b,
|
||||
jumlah: j
|
||||
});
|
||||
}
|
||||
|
||||
// Split cell contents by newlines to support combined rows
|
||||
const kodeParts = kodeCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const namaParts = namaCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const banyakParts = banyakCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const jumlahParts = jumlahCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
|
||||
const isWatermark = (s: string) => {
|
||||
const sl = s.toLowerCase();
|
||||
return (
|
||||
sl === "asli" ||
|
||||
sl === "copy" ||
|
||||
sl === "nama barang" ||
|
||||
sl === "tanda tangan supir" ||
|
||||
sl === "penerima barang" ||
|
||||
sl === "barang dikirim dalam keadaan baik" ||
|
||||
sl === "jumlah"
|
||||
);
|
||||
};
|
||||
|
||||
// Filter watermark keywords from each parts array
|
||||
const cleanKodes = kodeParts.filter(p => !isWatermark(p));
|
||||
const cleanNamas = namaParts.filter(p => !isWatermark(p));
|
||||
let cleanBanyaks = banyakParts.filter(p => !isWatermark(p));
|
||||
const cleanJumlahs = jumlahParts.filter(p => !isWatermark(p));
|
||||
|
||||
if (cleanBanyaks.length === 2 * cleanKodes.length) {
|
||||
const halved: string[] = [];
|
||||
const half = cleanKodes.length;
|
||||
for (let i = 0; i < half; i++) {
|
||||
const qty = cleanBanyaks[i] || "";
|
||||
const unit = cleanBanyaks[i + half] || "";
|
||||
halved.push(`${qty} ${unit}`.trim());
|
||||
}
|
||||
cleanBanyaks = halved;
|
||||
}
|
||||
|
||||
const maxLen = Math.max(cleanKodes.length, cleanNamas.length, cleanBanyaks.length, cleanJumlahs.length);
|
||||
|
||||
for (let i = 0; i < maxLen; i++) {
|
||||
const k = cleanKodes[i] || "";
|
||||
const n = cleanNamas[i] || "";
|
||||
let b = cleanBanyaks[i] || "";
|
||||
const j = cleanJumlahs[i] || "";
|
||||
|
||||
if (isWatermark(k) || isWatermark(n)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Clean checkmarks and extra spaces from banyak
|
||||
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
|
||||
|
||||
// Fallback for Banyak if empty or purely alphabetical unit
|
||||
if (!b) {
|
||||
b = "1";
|
||||
} else if (/^[a-zA-Z]+$/.test(b)) {
|
||||
b = `1 ${b}`;
|
||||
}
|
||||
|
||||
// Autocomplete packaging units if Banyak is purely numeric
|
||||
if (b && /^\d+$/.test(b)) {
|
||||
const code = k.trim();
|
||||
const name = n.toLowerCase();
|
||||
if (code === "11310024" || name.includes("griller")) {
|
||||
b = `${b} KRG`;
|
||||
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
|
||||
b = `${b} BAG`;
|
||||
}
|
||||
}
|
||||
|
||||
const cleanKode = k.trim();
|
||||
const cleanNama = n.trim();
|
||||
if (cleanKode === "" && cleanNama === "") {
|
||||
continue;
|
||||
}
|
||||
|
||||
metadata.items.push({
|
||||
kodeBarang: k,
|
||||
namaBarang: n,
|
||||
banyak: b,
|
||||
jumlah: j
|
||||
});
|
||||
}
|
||||
}
|
||||
rowIndex++;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -615,6 +718,44 @@ function formatPlatNumber(raw: string): string {
|
||||
const VALID_MONTHS = ["January","February","March","April","May","June","July","August","September","October","November","December"];
|
||||
const MONTH_SHORT = ["Jan","Feb","Mar","Apr","May","Jun","Jul","Aug","Sep","Oct","Nov","Dec"];
|
||||
|
||||
export function correctVisualDigits(val: string): string {
|
||||
if (!val || val === "Not Found") return "";
|
||||
let cleaned = val.trim();
|
||||
|
||||
const parts = cleaned.split(/[^a-zA-Z0-9]+/);
|
||||
let bestPart = parts[0] || "";
|
||||
let maxDigitsCount = 0;
|
||||
|
||||
for (const part of parts) {
|
||||
const digitsCount = (part.match(/[0-9]/g) || []).length;
|
||||
if (digitsCount > maxDigitsCount) {
|
||||
maxDigitsCount = digitsCount;
|
||||
bestPart = part;
|
||||
}
|
||||
}
|
||||
|
||||
if (maxDigitsCount === 0) {
|
||||
let maxLen = 0;
|
||||
for (const part of parts) {
|
||||
if (part.length > maxLen) {
|
||||
maxLen = part.length;
|
||||
bestPart = part;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bestPart = bestPart
|
||||
.replace(/[Oo]/g, "0")
|
||||
.replace(/[Ii|l]/g, "1")
|
||||
.replace(/[Bb]/g, "8")
|
||||
.replace(/[Ss]/g, "5")
|
||||
.replace(/[Zz]/g, "2")
|
||||
.replace(/[Gg]/g, "9")
|
||||
.replace(/[^0-9]/g, "");
|
||||
|
||||
return bestPart;
|
||||
}
|
||||
|
||||
export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata> & Record<string, any>): typeof meta {
|
||||
const currentYY = new Date().getFullYear().toString().slice(-2);
|
||||
const currentFullYear = new Date().getFullYear();
|
||||
@@ -641,22 +782,26 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
|
||||
}
|
||||
|
||||
// --- noPO ---
|
||||
// Must match PO/YY/NNNN+ where YY = current year, NNNN = 4+ digits
|
||||
// If year segment doesn't match current year, auto-correct it (parser already forces current year,
|
||||
// but this is a safety net in case anything slipped through)
|
||||
// Get year from parsed date (or default to current year if date not found)
|
||||
const docYY = result.tanggal && result.tanggal !== "Not Found" ? getYearFromDate(result.tanggal) : currentYY;
|
||||
|
||||
// --- noPO ---
|
||||
// Must match PO/YY/NNNN+ where YY is the document-specific year segment, NNNN = 4+ digits
|
||||
const noPO = (result.noPO || "").trim();
|
||||
const poPattern = /^PO\/(\d{2})\/(\d{4,})$/i;
|
||||
const pm = noPO.match(poPattern);
|
||||
if (pm) {
|
||||
// Auto-correct year to current year regardless of what was parsed
|
||||
result.noPO = `PO/${currentYY}/${pm[2]}`;
|
||||
// Keep the parsed year segment if it matches docYY, or fall back to docYY
|
||||
const yearSegment = pm[1] === docYY ? pm[1] : docYY;
|
||||
result.noPO = `PO/${yearSegment}/${pm[2]}`;
|
||||
} else {
|
||||
result.noPO = "Not Found";
|
||||
}
|
||||
|
||||
// --- noSO ---
|
||||
// Must be numeric string, 7-12 digits
|
||||
const noSO = (result.noSO || "").trim();
|
||||
let noSO = (result.noSO || "").trim();
|
||||
noSO = correctVisualDigits(noSO);
|
||||
if (/^\d{7,12}$/.test(noSO)) {
|
||||
result.noSO = noSO;
|
||||
} else {
|
||||
@@ -665,7 +810,8 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
|
||||
|
||||
// --- noDO ---
|
||||
// Must be numeric string, 7-12 digits
|
||||
const noDO = (result.noDO || "").trim();
|
||||
let noDO = (result.noDO || "").trim();
|
||||
noDO = correctVisualDigits(noDO);
|
||||
if (/^\d{7,12}$/.test(noDO)) {
|
||||
result.noDO = noDO;
|
||||
} else {
|
||||
|
||||
Reference in new issue
Block a user