fix: resolve table column shift, normalize units and standardize dates

This commit is contained in:
Rafhan Mazaya Fathurrahman committed 2026-07-03 16:04:57 +07:00
1 parent 095d4b799a
commit 2900eb670b
47 files changed
+11881 -2724

No files matched your search

+46 -14
View File
@@ -51,12 +51,15 @@ export async function POST(req: NextRequest) {
let isSample = true;
if (!fs.existsSync(filePath)) {
filePath = path.join("/uploads", safeFile);
filePath = path.join(process.cwd(), "public", "test-images", safeFile);
if (!fs.existsSync(filePath)) {
return NextResponse.json({ error: "File not found" }, { status: 404 });
filePath = path.join("/uploads", safeFile);
if (!fs.existsSync(filePath)) {
return NextResponse.json({ error: "File not found" }, { status: 404 });
}
isUpload = true;
isSample = false;
}
isUpload = true;
isSample = false;
}
// Read file and compute content hash
@@ -156,6 +159,7 @@ export async function POST(req: NextRequest) {
const rawMetadata = parseDOMetadata(markdownText);
// Second-layer sanity check: enforces strict field formats and auto-corrects anomalies
const docMetadata = sanitizeParsedMetadata(rawMetadata as any);
const docMetadataSanitized = JSON.parse(JSON.stringify(docMetadata));
// Load SKU Master list from DB (including new packaging details for triple check validation)
const skuDbRes = await query("SELECT no_sku, nama_item, standar_jumlah, jenis_outer FROM sku_master");
@@ -256,15 +260,25 @@ export async function POST(req: NextRequest) {
const numericQtyMatch = ocrQty.match(/[\d.,]+/);
if (numericQtyMatch) {
const numericQty = numericQtyMatch[0];
const qtyUnitMatch = ocrQty.match(/[a-zA-Z]+/);
const parsedQtyUnit = qtyUnitMatch ? qtyUnitMatch[0].toUpperCase() : "";
const isQtyUnitValid = ["KRG", "BOX", "BAG", "PAC", "PC", "KG", "PCS"].includes(parsedQtyUnit);
const standardOuter = bestMatch.jenis_outer || "";
if (standardOuter.toLowerCase() === "karung") {
item.banyak = `${numericQty} KRG`;
} else if (standardOuter.toLowerCase() === "box") {
item.banyak = `${numericQty} BOX`;
} else if (standardOuter.toLowerCase() === "bag") {
item.banyak = `${numericQty} BAG`;
if (isQtyUnitValid) {
item.banyak = `${numericQty} ${parsedQtyUnit}`;
} else if (standardOuter) {
if (standardOuter.toLowerCase() === "karung") {
item.banyak = `${numericQty} KRG`;
} else if (standardOuter.toLowerCase() === "box") {
item.banyak = `${numericQty} BOX`;
} else if (standardOuter.toLowerCase() === "bag") {
item.banyak = `${numericQty} BAG`;
} else {
item.banyak = `${numericQty} ${standardOuter.toUpperCase()}`;
}
} else {
item.banyak = `${numericQty} ${standardOuter.toUpperCase()}`;
item.banyak = ocrQty;
}
}
@@ -272,8 +286,20 @@ export async function POST(req: NextRequest) {
const numericPriceMatch = ocrPrice.match(/[\d.,]+/);
if (numericPriceMatch) {
const numericPrice = numericPriceMatch[0];
const standardInner = (bestMatch.standar_jumlah || "").toUpperCase();
item.jumlah = `${numericPrice} ${standardInner}`;
const priceUnitMatch = ocrPrice.match(/[a-zA-Z]+/);
const parsedPriceUnit = priceUnitMatch ? priceUnitMatch[0].toUpperCase() : "";
const isPriceUnitValid = ["KRG", "BOX", "BAG", "PAC", "PC", "KG", "PCS"].includes(parsedPriceUnit);
const standardInner = bestMatch.standar_jumlah || "";
if (standardInner.toLowerCase() === "pc" || standardInner.toLowerCase() === "pcs") {
item.jumlah = `${numericPrice} ${standardInner.toUpperCase()}`;
} else if (isPriceUnitValid) {
item.jumlah = `${numericPrice} ${parsedPriceUnit}`;
} else if (standardInner) {
item.jumlah = `${numericPrice} ${standardInner.toUpperCase()}`;
} else {
item.jumlah = ocrPrice;
}
}
checkedItems.push(item);
@@ -371,12 +397,18 @@ export async function POST(req: NextRequest) {
]);
}
// Return wrapped success response with items, flagged, remarks
// Return wrapped success response with items, flagged, remarks, and intermediate post-processing details
return NextResponse.json({
errorCode: 0,
errorMsg: "Success",
result: pipelineResult,
items: docMetadata.items,
postProcessingDetails: {
rawMarkdown: markdownText,
layer1RawRegex: rawMetadata,
layer2Sanitized: docMetadataSanitized,
layer3Final: docMetadata
},
flagged: {},
remarks: {}
});
+26 -1
View File
@@ -29,19 +29,43 @@ function runTests() {
{ name: "Date (Asli/Copy) prefix noise", markdown: "Tanggal: (Asli/Copy) 15 May 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "15 May 2026" } },
{ name: "Date junk suffix cut", markdown: "Tanggal: 25 May 2020 (Printed by system)", expected: { tanggal: "25 May 2020" } },
{ name: "Date bad OCR month Hv -> Not Found", markdown: "Tanggal:25 Hv 2024\nNo.SO : 1601001206", expected: { tanggal: "Not Found" } },
{ name: "Date single digit 7 May 2026 -> 07 May 2026", markdown: "Tanggal: 7 May 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "07 May 2026" } },
{ name: "Date single digit 4 Apr 2026 -> 04 April 2026", markdown: "Tanggal: 4 Apr 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "04 April 2026" } },
{ name: "00117709 before Tanggal must not pollute date", markdown: "00117709\nTanggal:25May2020\nNo.SO : 1091721200\nNo. DO : 1657943004\nNo. PO : F0/26/0000190929", expected: { tanggal: "25 May 2020" } },
{ name: "Plate B 9427 UXT", markdown: "Truck No. B 9427 UXT\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9427 UXT" } },
{ name: "Plate B-9999-XYZ dash", markdown: "No. Polisi: B-9999-XYZ\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9999 XYZ" } },
{ name: "Plate ignore PO/SO prefix", markdown: "Plate is PO 1234 SO but real truck is A 123 B\nNo. PO : PO/26/0000178435", expected: { platTruk: "A 123 B" } },
{ name: "Plate B9427UXT adjacent", markdown: "No Polisi B9427UXT\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9427 UXT" } },
{ name: "Plate real doc B 9723 CXS", markdown: "Truck No.\nB 9723 CXS\nWH 01 / 01\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9723 CXS" } },
{
name: "Table column shift alignment correction",
markdown: "No. PO : PO/26/0000178435\n" +
"<table>" +
"<tr><td>Kode Barang</td><td>Nama Barang</td><td>Banyak</td><td>Jumlah</td></tr>" +
"<tr><td></td><td>Item A</td><td>2 KRG</td><td>40 PC</td></tr>" +
"<tr><td>11310014</td><td>Item B</td><td>2 KRG</td><td>40 PC</td></tr>" +
"<tr><td>11310024</td><td>Item C</td><td>1 BOX</td><td>10 KG</td></tr>" +
"<tr><td>11720055</td><td></td><td></td><td></td></tr>" +
"</table>",
expected: {
items: [
{ kodeBarang: "11310014", namaBarang: "Item A", banyak: "2 KRG", jumlah: "40 PC" },
{ kodeBarang: "11310024", namaBarang: "Item B", banyak: "2 KRG", jumlah: "40 PC" },
{ kodeBarang: "11720055", namaBarang: "Item C", banyak: "1 BOX", jumlah: "10 KG" }
]
} as any
}
];
for (const t of parseTests) {
try {
const result = parseDOMetadata(t.markdown) as any;
for (const [key, val] of Object.entries(t.expected)) {
assert.strictEqual(result[key], val, `field [${key}] expected "${val}" got "${result[key]}"`);
if (key === "items") {
assert.deepStrictEqual(result.items, val);
} else {
assert.strictEqual(result[key], val, `field [${key}] expected "${val}" got "${result[key]}"`);
}
}
console.log(`[PASS] ${t.name}`);
} catch (err: any) {
@@ -56,6 +80,7 @@ function runTests() {
// tanggal valid
{ name: "sanitize: valid tanggal 30 June 2026 passes", input: { tanggal: "30 June 2026" }, expected: { tanggal: "30 June 2026" } },
{ name: "sanitize: valid tanggal 25 May 2020 passes", input: { tanggal: "25 May 2020" }, expected: { tanggal: "25 May 2020" } },
{ name: "sanitize: single digit tanggal 4 April 2026 -> 04 April 2026", input: { tanggal: "4 April 2026" }, expected: { tanggal: "04 April 2026" } },
// tanggal invalid
{ name: "sanitize: tanggal bad month Hv -> Not Found", input: { tanggal: "25 Hv 2024" }, expected: { tanggal: "Not Found" } },
{ name: "sanitize: tanggal as number 0011770 -> Not Found", input: { tanggal: "0011770" }, expected: { tanggal: "Not Found" } },
+130 -40
View File
@@ -26,7 +26,7 @@ function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn, 07/26/nnn
// Structure: [PREFIX][optional_noise_digits][/][optional_year_segment][/]?[NUMBER]
// We find the LAST slash and take everything after it as the real number
const withSlash = /^(?:[A-Z0-9]{1,4})[ \t]*[\/\-][ \t]*(?:\d{1,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
const withSlash = /^(?:[A-Z0-9]+)[ \t]*[\/\-][ \t]*(?:\d{1,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
const m1 = stripped.match(withSlash);
if (m1) {
return `PO/${currentYearLastTwo}/${m1[1]}`;
@@ -93,7 +93,20 @@ function cleanDateValue(raw: string): string {
const cleaned = raw.trim();
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
const today = new Date();
// Try unified regex first
const unifiedPattern = /(\d{1,2})[ \t\-\/]*(jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|maret|apr(?:il)?|may|mei|jun(?:[ei])?|jul(?:[ii])?|aug(?:ustus)?|agt|ags|sep(?:tember)?|oct(?:ober)?|okt|nov(?:ember)?|nop(?:ember)?|dec(?:ember)?|des(?:ember)?)[ \t\-\/]*(20\d{2})/i;
const match = cleaned.match(unifiedPattern);
if (match) {
const day = parseInt(match[1], 10);
const monthKey = match[2].toLowerCase();
const year = parseInt(match[3], 10);
const month = MONTHS_MAP[monthKey] || monthKey;
if (day >= 1 && day <= 31 && year >= 2010 && year <= 2035) {
const paddedDay = day.toString().padStart(2, '0');
return `${paddedDay} ${month} ${year}`;
}
}
let day: number | null = null;
let monthStr: string | null = null;
let year: number | null = null;
@@ -119,12 +132,14 @@ function cleanDateValue(raw: string): string {
}
}
if (!monthStr) return "Not Found";
// 3. Try to find 1 or 2 digit day (not part of the year)
let textForDay = cleaned;
if (year) {
textForDay = textForDay.replace(year.toString(), "");
}
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g) || textForDay.match(/(\d{1,2})/g);
if (dayMatches) {
for (const matchStr of dayMatches) {
const parsedDay = parseInt(matchStr, 10);
@@ -135,17 +150,12 @@ function cleanDateValue(raw: string): string {
}
}
// 4. Fallback fill-in from current date
const currentYear = today.getFullYear();
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
const currentMonth = currentMonthNames[today.getMonth()];
const currentDay = today.getDate();
if (day !== null && year !== null) {
const paddedDay = day.toString().padStart(2, '0');
return `${paddedDay} ${monthStr} ${year}`;
}
const finalDay = day !== null ? day : currentDay;
const finalMonth = monthStr !== null ? monthStr : currentMonth;
const finalYear = year !== null ? year : currentYear;
return `${finalDay} ${finalMonth} ${finalYear}`;
return "Not Found";
}
export function parseDOMetadata(markdown: string) {
@@ -317,25 +327,38 @@ export function parseDOMetadata(markdown: string) {
metadata.tanggal = dateMatch[0];
}
// 2. Real SO is the value that was matched under No. DO
if (/^\d{10}$/.test(originalDO)) {
metadata.noSO = originalDO;
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex);
if (m && m.length > 0) {
metadata.noSO = m[0];
}
}
// Only shift values if the PO field is shifted (i.e. PO field is a 10-digit number instead of PO format).
// If the PO field has a valid PO format (e.g. PO/26/...), PO and DO are in correct places and not shifted.
const isPoValid = /^PO\/[A-Z0-9\-\/]+\b/i.test(originalPO);
// 3. Real DO is the value that was matched under No. PO
if (/^\d{10}$/.test(originalPO)) {
metadata.noDO = originalPO;
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex);
if (m && m.length > 1) {
metadata.noDO = m[1];
if (!isPoValid) {
// 2. Real SO is the value that was matched under No. DO
if (/^\d{10}$/.test(originalDO)) {
metadata.noSO = originalDO;
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex);
const cleanMatches = m ? m.filter(val => !originalPO.includes(val)) : [];
if (cleanMatches.length > 0) {
metadata.noSO = cleanMatches[0];
}
}
// 3. Real DO is the value that was matched under No. PO
if (/^\d{10}$/.test(originalPO)) {
metadata.noDO = originalPO;
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex);
const cleanMatches = m ? m.filter(val => !originalPO.includes(val)) : [];
if (cleanMatches.length > 1) {
metadata.noDO = cleanMatches[1];
}
}
} else {
// PO is valid and NOT shifted. SO was just empty or noise.
if (isShortSO || isDateInSO) {
metadata.noSO = "Not Found";
}
}
@@ -420,7 +443,11 @@ export function parseDOMetadata(markdown: string) {
metadata.noDO = cleanFinalValue(metadata.noDO);
// Extract PO year dynamically from parsed document date (or default to current year if date not found)
const docYearLastTwo = getYearFromDate(metadata.tanggal);
const currentYY = new Date().getFullYear().toString().slice(-2);
let docYearLastTwo = getYearFromDate(metadata.tanggal);
if (docYearLastTwo !== currentYY) {
docYearLastTwo = currentYY;
}
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), docYearLastTwo);
// Parse HTML tables for items
@@ -536,6 +563,45 @@ export function parseDOMetadata(markdown: string) {
}
if (isItemsTable) {
// Check for SKU column downward shift of 1 row:
// - First data row (r=1) has empty SKU, but has description
// - A downstream row (shiftBoundaryRow) has SKU, but description, banyak, jumlah are empty
if (numRows > 2 && kIdx !== -1 && nIdx !== -1) {
const firstRowSku = (grid[1][kIdx] || "").trim();
const firstRowDesc = (grid[1][nIdx] || "").trim();
if (firstRowSku === "" && firstRowDesc !== "") {
let shiftBoundaryRow = -1;
for (let r = 2; r < numRows; r++) {
const skuVal = (grid[r][kIdx] || "").trim();
let othersEmpty = true;
for (let c = 0; c < maxCols; c++) {
if (c !== kIdx) {
const val = (grid[r][c] || "").trim();
if (val !== "") {
othersEmpty = false;
break;
}
}
}
if (skuVal !== "" && /^\d{8}$/.test(skuVal) && othersEmpty) {
shiftBoundaryRow = r;
break;
}
}
if (shiftBoundaryRow !== -1) {
console.log(`Detected shifted SKU column in table. Shift boundary row: ${shiftBoundaryRow}. Correcting alignment...`);
for (let r = 1; r < shiftBoundaryRow; r++) {
grid[r][kIdx] = grid[r + 1][kIdx];
}
grid[shiftBoundaryRow][kIdx] = "";
}
}
}
for (let r = 1; r < numRows; r++) {
const cells = grid[r];
const kodeCell = cells[kIdx] || "";
@@ -544,7 +610,7 @@ export function parseDOMetadata(markdown: string) {
let jumlahCell = "";
// Check if there is an extra column before banyak that we should merge with banyak
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx && headerCells[bIdx - 1] !== headerCells[nIdx]) {
const qtyCell = cells[bIdx - 1] || "";
const unitCell = cells[bIdx] || "";
banyakCell = `${qtyCell} ${unitCell}`.trim();
@@ -612,6 +678,11 @@ export function parseDOMetadata(markdown: string) {
// Clean checkmarks and extra spaces from banyak
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
b = b.replace(/\b80[xX]\b/g, "BOX")
.replace(/\bB0[xX]\b/g, "BOX")
.replace(/\b8[aA][gG]\b/g, "BAG")
.replace(/\b[pP][aA][iI]\b/g, "PAC")
.trim();
// Fallback for Banyak if empty or purely alphabetical unit
if (!b) {
@@ -626,11 +697,21 @@ export function parseDOMetadata(markdown: string) {
const name = n.toLowerCase();
if (code === "11310024" || name.includes("griller")) {
b = `${b} KRG`;
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
} else if (code === "11640053" || name.includes("bone in leg")) {
b = `${b} BAG`;
} else if (code.startsWith("21") || code.startsWith("12") || name.includes("nasi") || name.includes("rice") || name.includes("nugget") || name.includes("sosis") || name.includes("sausage") || name.includes("fiesta") || name.includes("champ")) {
b = `${b} BOX`;
}
}
// Clean checkmarks and extra spaces from jumlah
let cleanJ = j.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
cleanJ = cleanJ.replace(/\b80[xX]\b/g, "BOX")
.replace(/\bB0[xX]\b/g, "BOX")
.replace(/\b8[aA][gG]\b/g, "BAG")
.replace(/\b[pP][aA][iI]\b/g, "PAC")
.trim();
const cleanKode = k.trim();
const cleanNama = n.trim();
if (cleanKode === "" && cleanNama === "") {
@@ -641,7 +722,7 @@ export function parseDOMetadata(markdown: string) {
kodeBarang: k,
namaBarang: n,
banyak: b,
jumlah: j
jumlah: cleanJ
});
}
}
@@ -671,6 +752,11 @@ export function extractPlatTruk(text: string): string {
];
const isValidPrefix = (p: string) => arrayPlat.includes(p.toUpperCase());
const FORBIDDEN_SUFFIXES = [
"BOX", "KRG", "BAG", "PAC", "KG", "PC", "PCS", "LGT", "K", "EKR", "LTR", "BKS",
"IDN", "WH", "PM", "TTD", "CPI", "HO", "NO", "GR", "G"
];
const isForbiddenSuffix = (s: string) => FORBIDDEN_SUFFIXES.includes(s.toUpperCase());
// 1. Look for explicit labels: No. Polisi, No. Pol, No. Polisi:, No. Kendaraan, Plat No, Plat, Truck No., etc.
const labelRegex = /(?:No\.?\s*(?:Polisi|Pol|Kendaraan|Mobil|Truck|Pol\.?)|Plat(?:\s*No)?|Truck\s*No\.?)\s*[:\-.]?\s*\b([A-Z]{1,2})[ \t\-]*(\d{1,4})[ \t\-]*([A-Z]{1,3})\b/i;
@@ -679,7 +765,7 @@ export function extractPlatTruk(text: string): string {
const prefix = labelMatch[1].toUpperCase();
const num = labelMatch[2];
const suffix = labelMatch[3].toUpperCase();
if (isValidPrefix(prefix)) {
if (isValidPrefix(prefix) && !isForbiddenSuffix(suffix)) {
return `${prefix} ${num} ${suffix}`;
}
}
@@ -692,7 +778,7 @@ export function extractPlatTruk(text: string): string {
const num = match[2];
const suffix = match[3].toUpperCase();
if (isValidPrefix(prefix)) {
if (isValidPrefix(prefix) && !isForbiddenSuffix(suffix)) {
return `${prefix} ${num} ${suffix}`;
}
}
@@ -771,9 +857,10 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
const day = parseInt(dm[1], 10);
const year = parseInt(dm[3], 10);
if (day >= 1 && day <= 31 && year >= 2010 && year <= currentFullYear + 1) {
// Valid — normalize capitalization
// Valid — normalize capitalization and pad day with leading zero
const month = dm[2].charAt(0).toUpperCase() + dm[2].slice(1).toLowerCase();
result.tanggal = `${dm[1]} ${month} ${dm[3]}`;
const paddedDay = dm[1].padStart(2, '0');
result.tanggal = `${paddedDay} ${month} ${dm[3]}`;
} else {
result.tanggal = "Not Found";
}
@@ -783,7 +870,10 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
// --- noPO ---
// Get year from parsed date (or default to current year if date not found)
const docYY = result.tanggal && result.tanggal !== "Not Found" ? getYearFromDate(result.tanggal) : currentYY;
let docYY = result.tanggal && result.tanggal !== "Not Found" ? getYearFromDate(result.tanggal) : currentYY;
if (docYY !== currentYY) {
docYY = currentYY;
}
// --- noPO ---
// Must match PO/YY/NNNN+ where YY is the document-specific year segment, NNNN = 4+ digits