feat: update backend OCR parser, web app, mobile app camera/preview UI, tests, and documentation with sample images

This commit is contained in:
Rafhan Mazaya Fathurrahman committed 2026-07-02 13:33:57 +07:00
1 parent aa3233e411
commit bdb3a49742
71 files changed
+3360 -764

No files matched your search

+42 -14
View File
@@ -15,37 +15,65 @@ export interface ActiveUploadLog {
frontend_response?: any;
}
// Store active log in global context to persist across Next.js hot-reloads
// Store active logs in global context as a Map keyed by filename
// This supports concurrent uploads without race conditions
const globalForActiveLog = global as unknown as {
activeLog: ActiveUploadLog | null;
activeLogs: Map<string, ActiveUploadLog>;
};
// Initialise the map once (survives Next.js hot-reloads on the same process)
if (!globalForActiveLog.activeLogs) {
globalForActiveLog.activeLogs = new Map();
}
export function startActiveLog(filename: string) {
globalForActiveLog.activeLog = {
globalForActiveLog.activeLogs.set(filename, {
filename,
vllm_calls: []
};
});
console.log(`[ActiveLog] Started tracking log for ${filename}`);
}
export function logVllmCall(request: any, response: any) {
if (globalForActiveLog.activeLog) {
globalForActiveLog.activeLog.vllm_calls.push({
export function logVllmCall(filename: string, request: any, response: any) {
const log = globalForActiveLog.activeLogs.get(filename);
if (log) {
log.vllm_calls.push({
request,
response,
timestamp: new Date().toISOString()
});
console.log(`[ActiveLog] Logged vLLM call for ${globalForActiveLog.activeLog.filename} (total calls: ${globalForActiveLog.activeLog.vllm_calls.length})`);
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
} else {
console.log("[ActiveLog] Warning: Attempted to log vLLM call but no active log session is running.");
console.log(`[ActiveLog] Warning: Attempted to log vLLM call for "${filename}" but no active log session is running.`);
}
}
export function getActiveLog(): ActiveUploadLog | null {
return globalForActiveLog.activeLog;
export function getActiveLog(filename: string): ActiveUploadLog | null {
return globalForActiveLog.activeLogs.get(filename) || null;
}
export function clearActiveLog() {
globalForActiveLog.activeLog = null;
console.log("[ActiveLog] Cleared active log tracking context");
export function clearActiveLog(filename: string) {
globalForActiveLog.activeLogs.delete(filename);
console.log(`[ActiveLog] Cleared active log tracking context for ${filename}`);
}
/**
* Log a vLLM call to ALL currently active upload sessions.
* Used by the vllm-proxy, which doesn't have per-upload filename context,
* since the pipeline-api processes exactly one upload at a time.
*/
export function logVllmCallToAll(request: any, response: any) {
const sessions = globalForActiveLog.activeLogs;
if (sessions.size === 0) {
console.log("[ActiveLog] Warning: Attempted to log vLLM call but no active log session is running.");
return;
}
for (const [filename, log] of sessions) {
log.vllm_calls.push({
request,
response,
timestamp: new Date().toISOString()
});
console.log(`[ActiveLog] Logged vLLM call for ${filename} (total calls: ${log.vllm_calls.length})`);
}
}
+343 -197
View File
@@ -18,31 +18,30 @@ function cleanFinalValue(val: string, preserveNewlines = false): string {
function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
if (!raw || raw === "Not Found") return "Not Found";
// Strip leading label noise like "No. PO : " before matching
// Strip leading label noise like "No. PO : " before matching, allowing common visual confusions
const stripped = raw
.replace(/^No\.?\s*PO\s*[:\-]?\s*/i, "")
.replace(/^No\.?\s*(?:PO|P0|07|70|F0|O0)\s*[:\-]?\s*/i, "")
.trim();
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn
// ALWAYS use currentYearLastTwo — never trust OCR year (can be corrupted)
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn, 07/26/nnn
// Structure: [PREFIX][optional_noise_digits][/][optional_year_segment][/]?[NUMBER]
// We find the LAST slash and take everything after it as the real number
const withSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)\d*[ \t]*[\/\-][ \t]*(?:\d{0,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
const withSlash = /^(?:[A-Z0-9]{1,4})[ \t]*[\/\-][ \t]*(?:\d{1,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
const m1 = stripped.match(withSlash);
if (m1) {
return `PO/${currentYearLastTwo}/${m1[1]}`;
}
// Pattern 2: No slashes — OCR fused: PO12070000190729 or F012070000170727
// Pattern 2: No slashes — OCR fused: PO12070000190729 or F012070000170727, 7012010000100029
// Structure: [PREFIX][digits_with_noise][real_number_starting_0000]
const noSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)(\d+)$/i;
const noSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0|07|70|11|17|76|0|7)(\d+)$/i;
const m2 = stripped.match(noSlash);
if (m2) {
const digits = m2[1];
// Real PO number starts with 0000 in observed patterns
const numberPart = digits.replace(/^\d{2,4}(0{4}\d+)$/, "$1");
if (numberPart && numberPart !== digits) {
return `PO/${currentYearLastTwo}/${numberPart}`;
// Real PO number starts with 0000 (or 000, 00)
const matchNum = digits.match(/(0{2,}\d+)$/);
if (matchNum) {
return `PO/${currentYearLastTwo}/${matchNum[1]}`;
}
// Fallback: strip up to 4 leading noise digits
const fallbackDigits = digits.replace(/^\d{2,4}/, "");
@@ -52,8 +51,12 @@ function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
return `PO/${currentYearLastTwo}/${digits}`;
}
// Pattern 3: Just a raw number (8+ digits) — not a valid PO format
// Pattern 3: Just a raw number (8+ digits) — not a valid PO format unless it has 0000
if (/^\d{8,}$/.test(stripped)) {
const matchNum = stripped.match(/(0{2,}\d+)$/);
if (matchNum) {
return `PO/${currentYearLastTwo}/${matchNum[1]}`;
}
return "Not Found";
}
@@ -70,26 +73,79 @@ function getYearFromDate(dateStr: string): string {
return new Date().getFullYear().toString().slice(-2);
}
const MONTHS_MAP: Record<string, string> = {
january: "January", januari: "January", janov: "January", jan: "January",
february: "February", februari: "February", feb: "February",
march: "March", maret: "March", mar: "March",
april: "April", apr: "April",
may: "May", mei: "May",
june: "June", juni: "June", jun: "June",
july: "July", juli: "July", jul: "July",
august: "August", agustus: "August", agt: "August", ags: "August", aug: "August",
september: "September", sept: "September", sep: "September",
oktober: "October", october: "October", okt: "October", oct: "October",
november: "November", nopember: "November", nov: "November",
desember: "December", december: "December", des: "December", dec: "December"
};
function cleanDateValue(raw: string): string {
if (!raw || raw === "Not Found") return "Not Found";
// Enforce dd Month yyyy pattern (digits, month letters, year digits)
// Permissive of various spacing/dashes/slashes
const pattern = /\b(\d{1,2})[ \t\-\/]*(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)([a-zA-Z]*)[ \t\-\/]*(\d{4})\b/i;
const match = raw.match(pattern);
if (match) {
const day = match[1];
const month = match[2] + match[3];
const year = match[4];
// Capitalize month first letter, keep rest lowercase (e.g. May, June)
const formattedMonth = month.charAt(0).toUpperCase() + month.slice(1).toLowerCase();
return `${day} ${formattedMonth} ${year}`;
if (!raw) return "Not Found";
const cleaned = raw.trim();
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
const today = new Date();
let day: number | null = null;
let monthStr: string | null = null;
let year: number | null = null;
// 1. Try to find 4-digit year (2010 to 2035)
const yearMatch = cleaned.match(/\b(20\d{2})\b/);
if (yearMatch) {
const parsedYear = parseInt(yearMatch[1], 10);
if (parsedYear >= 2010 && parsedYear <= 2035) {
year = parsedYear;
}
}
// Fallback: If no standard date pattern is found, return Not Found
return "Not Found";
// 2. Try to find month using keywords
const lowerRaw = cleaned.toLowerCase();
const monthsKeys = Object.keys(MONTHS_MAP);
monthsKeys.sort((a, b) => b.length - a.length);
for (const key of monthsKeys) {
if (lowerRaw.includes(key)) {
monthStr = MONTHS_MAP[key] || null;
break;
}
}
// 3. Try to find 1 or 2 digit day (not part of the year)
let textForDay = cleaned;
if (year) {
textForDay = textForDay.replace(year.toString(), "");
}
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
if (dayMatches) {
for (const matchStr of dayMatches) {
const parsedDay = parseInt(matchStr, 10);
if (parsedDay >= 1 && parsedDay <= 31) {
day = parsedDay;
break;
}
}
}
// 4. Fallback fill-in from current date
const currentYear = today.getFullYear();
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
const currentMonth = currentMonthNames[today.getMonth()];
const currentDay = today.getDate();
const finalDay = day !== null ? day : currentDay;
const finalMonth = monthStr !== null ? monthStr : currentMonth;
const finalYear = year !== null ? year : currentYear;
return `${finalDay} ${finalMonth} ${finalYear}`;
}
export function parseDOMetadata(markdown: string) {
@@ -363,185 +419,232 @@ export function parseDOMetadata(markdown: string) {
metadata.noSO = cleanFinalValue(metadata.noSO);
metadata.noDO = cleanFinalValue(metadata.noDO);
// Always use current year for fused (no-slash) PO patterns — OCR corrupts year digits
const currentYearLastTwo = new Date().getFullYear().toString().slice(-2);
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), currentYearLastTwo);
// Extract PO year dynamically from parsed document date (or default to current year if date not found)
const docYearLastTwo = getYearFromDate(metadata.tanggal);
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), docYearLastTwo);
// Parse HTML tables for items
const tableRegex = /<table[^>]*>([\s\S]*?)<\/table>/g;
let match;
while ((match = tableRegex.exec(markdown)) !== null) {
const tableHtml = match[1];
// Parse all raw rows and cells first to get td details, including rowspan and colspan
const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/g;
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
interface CellInfo {
text: string;
rowspan: number;
colspan: number;
}
const rawRows: CellInfo[][] = [];
let trMatch;
let rowIndex = 0;
while ((trMatch = trRegex.exec(tableHtml)) !== null) {
const rowHtml = trMatch[1];
const rowCells: CellInfo[] = [];
let tdMatch;
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
const tdHtml = tdMatch[0];
const cellContent = tdMatch[1];
const rsMatch = tdHtml.match(/rowspan=["']?(\d+)["']?/i);
const rowspan = rsMatch ? parseInt(rsMatch[1], 10) : 1;
const csMatch = tdHtml.match(/colspan=["']?(\d+)["']?/i);
const colspan = csMatch ? parseInt(csMatch[1], 10) : 1;
const cellText = cellContent.replace(/<[^>]*>/g, "").trim().replace(/\\n/g, "\n");
rowCells.push({
text: cellText,
rowspan,
colspan
});
}
if (rowCells.length > 0) {
rawRows.push(rowCells);
}
}
if (rawRows.length === 0) continue;
// Determine the maximum columns in the grid
let maxCols = 0;
for (const cell of rawRows[0]) {
maxCols += cell.colspan;
}
const numRows = rawRows.length;
const grid: string[][] = Array.from({ length: numRows }, () => new Array(maxCols).fill(""));
// Fill the grid, respecting rowspan and colspan
for (let r = 0; r < numRows; r++) {
const rowCells = rawRows[r];
let cellIndex = 0;
for (let c = 0; c < maxCols; c++) {
// Skip if already filled
if (grid[r][c] !== "") {
continue;
}
if (cellIndex >= rowCells.length) {
break;
}
const cell = rowCells[cellIndex++];
const lines = cell.text.split("\n").map(l => l.trim()).filter(Boolean);
for (let dr = 0; dr < cell.rowspan; dr++) {
if (r + dr >= numRows) break;
for (let dc = 0; dc < cell.colspan; dc++) {
if (c + dc >= maxCols) break;
let cellValue = cell.text;
if (cell.rowspan > 1 && lines.length > 0) {
cellValue = lines[dr] ?? lines[lines.length - 1] ?? "";
}
grid[r + dr][c + dc] = cellValue;
}
}
}
}
// Process the grid
let kIdx = 0;
let nIdx = 1;
let bIdx = 2;
let jIdx = 3;
let isItemsTable = false;
while ((trMatch = trRegex.exec(tableHtml)) !== null) {
const rowHtml = trMatch[1];
if (rowIndex === 0) {
// Parse header row
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
let tdMatch;
const headerCells: string[] = [];
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
headerCells.push(tdMatch[1].replace(/<[^>]*>/g, "").trim().toLowerCase());
// Check headers in grid[0]
const headerCells = grid[0].map(h => h.toLowerCase());
const foundKode = headerCells.findIndex(h => h.includes("kode") || h.includes("item code"));
const foundNama = headerCells.findIndex(h => h.includes("nama") || h.includes("item name") || h.includes("description"));
const foundBanyak = headerCells.findIndex(h => h.includes("banyak") || h.includes("qty") || h.includes("quantity"));
const foundJumlah = headerCells.findIndex(h => h.includes("jumlah") || h.includes("total"));
if (foundKode !== -1 || foundNama !== -1) {
isItemsTable = true;
kIdx = foundKode !== -1 ? foundKode : 0;
nIdx = foundNama !== -1 ? foundNama : 1;
bIdx = foundBanyak !== -1 ? foundBanyak : 2;
jIdx = foundJumlah !== -1 ? foundJumlah : 3;
}
if (isItemsTable) {
for (let r = 1; r < numRows; r++) {
const cells = grid[r];
const kodeCell = cells[kIdx] || "";
const namaCell = cells[nIdx] || "";
let banyakCell = "";
let jumlahCell = "";
// Check if there is an extra column before banyak that we should merge with banyak
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
const qtyCell = cells[bIdx - 1] || "";
const unitCell = cells[bIdx] || "";
banyakCell = `${qtyCell} ${unitCell}`.trim();
} else {
banyakCell = cells[bIdx] || "";
}
const foundKode = headerCells.findIndex(h => h.includes("kode") || h.includes("item code"));
const foundNama = headerCells.findIndex(h => h.includes("nama") || h.includes("item name") || h.includes("description"));
const foundBanyak = headerCells.findIndex(h => h.includes("banyak") || h.includes("qty") || h.includes("quantity"));
const foundJumlah = headerCells.findIndex(h => h.includes("jumlah") || h.includes("total"));
if (foundKode !== -1 || foundNama !== -1) {
isItemsTable = true;
kIdx = foundKode !== -1 ? foundKode : 0;
nIdx = foundNama !== -1 ? foundNama : 1;
bIdx = foundBanyak !== -1 ? foundBanyak : 2;
jIdx = foundJumlah !== -1 ? foundJumlah : 3;
}
} else {
if (isItemsTable) {
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
let tdMatch;
const cells: string[] = [];
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
// Normalize literal \n text if returned as literal string "\n"
const cellText = tdMatch[1].replace(/<[^>]*>/g, "").trim().replace(/\\n/g, "\n");
cells.push(cellText);
if (jIdx !== -1) {
jumlahCell = cells[jIdx] || "";
} else {
if (cells.length === 5 && bIdx === 3) {
jumlahCell = cells[4] || "";
} else {
jumlahCell = cells[3] || "";
}
if (cells.length >= 3) {
const kodeCell = cells[kIdx] || "";
const namaCell = cells[nIdx] || "";
let banyakCell = "";
let jumlahCell = "";
// Check if there is an extra column before banyak that we should merge with banyak
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
const qtyCell = cells[bIdx - 1] || "";
const unitCell = cells[bIdx] || "";
const qtyLines = qtyCell.split("\n").map(l => l.trim());
const unitLines = unitCell.split("\n").map(l => l.trim());
const combinedLines: string[] = [];
const maxQLen = Math.max(qtyLines.length, unitLines.length);
for (let idx = 0; idx < maxQLen; idx++) {
let q = qtyLines[idx] || "";
const u = unitLines[idx] || "";
// Default to "1" if quantity is missing for a valid item row
const numItems = kodeCell.split("\n").map(p => p.trim()).filter(Boolean).length;
if (!q && idx < numItems) {
q = "1";
}
combinedLines.push(`${q} ${u}`.trim());
}
banyakCell = combinedLines.join("\n");
} else {
banyakCell = cells[bIdx] || "";
}
if (jIdx !== -1) {
jumlahCell = cells[jIdx] || "";
} else {
if (cells.length === 5 && bIdx === 3) {
jumlahCell = cells[4] || "";
} else {
jumlahCell = cells[3] || "";
}
}
// Split cell contents by newlines to support combined rows
const kodeParts = kodeCell.split("\n").map(p => p.trim()).filter(Boolean);
const namaParts = namaCell.split("\n").map(p => p.trim()).filter(Boolean);
const banyakParts = banyakCell.split("\n").map(p => p.trim()).filter(Boolean);
const jumlahParts = jumlahCell.split("\n").map(p => p.trim()).filter(Boolean);
const isWatermark = (s: string) => {
const sl = s.toLowerCase();
return (
sl === "asli" ||
sl === "copy" ||
sl === "nama barang" ||
sl === "tanda tangan supir" ||
sl === "penerima barang" ||
sl === "barang dikirim dalam keadaan baik" ||
sl === "jumlah"
);
};
// Filter watermark keywords from each parts array
const cleanKodes = kodeParts.filter(p => !isWatermark(p));
const cleanNamas = namaParts.filter(p => !isWatermark(p));
let cleanBanyaks = banyakParts.filter(p => !isWatermark(p));
const cleanJumlahs = jumlahParts.filter(p => !isWatermark(p));
if (cleanBanyaks.length === 2 * cleanKodes.length) {
const halved: string[] = [];
const half = cleanKodes.length;
for (let i = 0; i < half; i++) {
const qty = cleanBanyaks[i] || "";
const unit = cleanBanyaks[i + half] || "";
halved.push(`${qty} ${unit}`.trim());
}
cleanBanyaks = halved;
}
const maxLen = Math.max(cleanKodes.length, cleanNamas.length, cleanBanyaks.length, cleanJumlahs.length);
for (let i = 0; i < maxLen; i++) {
const k = cleanKodes[i] || "";
const n = cleanNamas[i] || "";
let b = cleanBanyaks[i] || "";
const j = cleanJumlahs[i] || "";
if (isWatermark(k) || isWatermark(n)) {
continue;
}
// Clean checkmarks and extra spaces from banyak
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
// Fallback for Banyak if empty or purely alphabetical unit
if (!b) {
b = "1";
} else if (/^[a-zA-Z]+$/.test(b)) {
b = `1 ${b}`;
}
// Autocomplete packaging units if Banyak is purely numeric
if (b && /^\d+$/.test(b)) {
const code = k.trim();
const name = n.toLowerCase();
if (code === "11310024" || name.includes("griller")) {
b = `${b} KRG`;
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
b = `${b} BAG`;
}
}
// Validate kodeBarang: must not be blank and must match exactly 8 digits
const cleanKode = k.trim();
if (cleanKode === "" || !/^\d{8}$/.test(cleanKode)) {
continue;
}
metadata.items.push({
kodeBarang: k,
namaBarang: n,
banyak: b,
jumlah: j
});
}
// Split cell contents by newlines to support combined rows
const kodeParts = kodeCell.split("\n").map(p => p.trim()).filter(Boolean);
const namaParts = namaCell.split("\n").map(p => p.trim()).filter(Boolean);
const banyakParts = banyakCell.split("\n").map(p => p.trim()).filter(Boolean);
const jumlahParts = jumlahCell.split("\n").map(p => p.trim()).filter(Boolean);
const isWatermark = (s: string) => {
const sl = s.toLowerCase();
return (
sl === "asli" ||
sl === "copy" ||
sl === "nama barang" ||
sl === "tanda tangan supir" ||
sl === "penerima barang" ||
sl === "barang dikirim dalam keadaan baik" ||
sl === "jumlah"
);
};
// Filter watermark keywords from each parts array
const cleanKodes = kodeParts.filter(p => !isWatermark(p));
const cleanNamas = namaParts.filter(p => !isWatermark(p));
let cleanBanyaks = banyakParts.filter(p => !isWatermark(p));
const cleanJumlahs = jumlahParts.filter(p => !isWatermark(p));
if (cleanBanyaks.length === 2 * cleanKodes.length) {
const halved: string[] = [];
const half = cleanKodes.length;
for (let i = 0; i < half; i++) {
const qty = cleanBanyaks[i] || "";
const unit = cleanBanyaks[i + half] || "";
halved.push(`${qty} ${unit}`.trim());
}
cleanBanyaks = halved;
}
const maxLen = Math.max(cleanKodes.length, cleanNamas.length, cleanBanyaks.length, cleanJumlahs.length);
for (let i = 0; i < maxLen; i++) {
const k = cleanKodes[i] || "";
const n = cleanNamas[i] || "";
let b = cleanBanyaks[i] || "";
const j = cleanJumlahs[i] || "";
if (isWatermark(k) || isWatermark(n)) {
continue;
}
// Clean checkmarks and extra spaces from banyak
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
// Fallback for Banyak if empty or purely alphabetical unit
if (!b) {
b = "1";
} else if (/^[a-zA-Z]+$/.test(b)) {
b = `1 ${b}`;
}
// Autocomplete packaging units if Banyak is purely numeric
if (b && /^\d+$/.test(b)) {
const code = k.trim();
const name = n.toLowerCase();
if (code === "11310024" || name.includes("griller")) {
b = `${b} KRG`;
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
b = `${b} BAG`;
}
}
const cleanKode = k.trim();
const cleanNama = n.trim();
if (cleanKode === "" && cleanNama === "") {
continue;
}
metadata.items.push({
kodeBarang: k,
namaBarang: n,
banyak: b,
jumlah: j
});
}
}
rowIndex++;
}
}
@@ -615,6 +718,44 @@ function formatPlatNumber(raw: string): string {
const VALID_MONTHS = ["January","February","March","April","May","June","July","August","September","October","November","December"];
const MONTH_SHORT = ["Jan","Feb","Mar","Apr","May","Jun","Jul","Aug","Sep","Oct","Nov","Dec"];
export function correctVisualDigits(val: string): string {
if (!val || val === "Not Found") return "";
let cleaned = val.trim();
const parts = cleaned.split(/[^a-zA-Z0-9]+/);
let bestPart = parts[0] || "";
let maxDigitsCount = 0;
for (const part of parts) {
const digitsCount = (part.match(/[0-9]/g) || []).length;
if (digitsCount > maxDigitsCount) {
maxDigitsCount = digitsCount;
bestPart = part;
}
}
if (maxDigitsCount === 0) {
let maxLen = 0;
for (const part of parts) {
if (part.length > maxLen) {
maxLen = part.length;
bestPart = part;
}
}
}
bestPart = bestPart
.replace(/[Oo]/g, "0")
.replace(/[Ii|l]/g, "1")
.replace(/[Bb]/g, "8")
.replace(/[Ss]/g, "5")
.replace(/[Zz]/g, "2")
.replace(/[Gg]/g, "9")
.replace(/[^0-9]/g, "");
return bestPart;
}
export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata> & Record<string, any>): typeof meta {
const currentYY = new Date().getFullYear().toString().slice(-2);
const currentFullYear = new Date().getFullYear();
@@ -641,22 +782,26 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
}
// --- noPO ---
// Must match PO/YY/NNNN+ where YY = current year, NNNN = 4+ digits
// If year segment doesn't match current year, auto-correct it (parser already forces current year,
// but this is a safety net in case anything slipped through)
// Get year from parsed date (or default to current year if date not found)
const docYY = result.tanggal && result.tanggal !== "Not Found" ? getYearFromDate(result.tanggal) : currentYY;
// --- noPO ---
// Must match PO/YY/NNNN+ where YY is the document-specific year segment, NNNN = 4+ digits
const noPO = (result.noPO || "").trim();
const poPattern = /^PO\/(\d{2})\/(\d{4,})$/i;
const pm = noPO.match(poPattern);
if (pm) {
// Auto-correct year to current year regardless of what was parsed
result.noPO = `PO/${currentYY}/${pm[2]}`;
// Keep the parsed year segment if it matches docYY, or fall back to docYY
const yearSegment = pm[1] === docYY ? pm[1] : docYY;
result.noPO = `PO/${yearSegment}/${pm[2]}`;
} else {
result.noPO = "Not Found";
}
// --- noSO ---
// Must be numeric string, 7-12 digits
const noSO = (result.noSO || "").trim();
let noSO = (result.noSO || "").trim();
noSO = correctVisualDigits(noSO);
if (/^\d{7,12}$/.test(noSO)) {
result.noSO = noSO;
} else {
@@ -665,7 +810,8 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
// --- noDO ---
// Must be numeric string, 7-12 digits
const noDO = (result.noDO || "").trim();
let noDO = (result.noDO || "").trim();
noDO = correctVisualDigits(noDO);
if (/^\d{7,12}$/.test(noDO)) {
result.noDO = noDO;
} else {