300 lines
11 KiB
JavaScript
300 lines
11 KiB
JavaScript
const { Client } = require("pg");
|
|
|
|
function cleanFinalValue(val, preserveNewlines = false) {
|
|
if (!val) return "Not Found";
|
|
const cleaned = val.replace(/<[^>]*>/g, "");
|
|
if (preserveNewlines) {
|
|
return cleaned.split("\n").map(line => line.trim()).filter(Boolean).join("\n") || "Not Found";
|
|
} else {
|
|
return cleaned.replace(/\s+/g, " ").trim() || "Not Found";
|
|
}
|
|
}
|
|
|
|
function parseDOMetadata(markdown) {
|
|
const metadata = {
|
|
vendorInfo: "Not Found",
|
|
customerInfo: "Not Found",
|
|
tanggal: "Not Found",
|
|
noSO: "Not Found",
|
|
noDO: "Not Found",
|
|
noPO: "Not Found",
|
|
items: []
|
|
};
|
|
|
|
if (!markdown) return metadata;
|
|
|
|
const cleanMarkdown = markdown
|
|
.replace(/<\/tr>/gi, "\n")
|
|
.replace(/<br\s*\/?>/gi, "\n")
|
|
.replace(/<\/p>/gi, "\n")
|
|
.replace(/<[^>]*>/g, " ");
|
|
|
|
const lines = cleanMarkdown.split("\n").map(l => l.trim()).filter(Boolean);
|
|
|
|
// Vendor Info
|
|
const vendorStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|Kepada|Yth|Customer|Deliver|Order\s+Untuk|#|\d{2}:\d{2}:\d{2})/i;
|
|
const vendorStartIndex = lines.findIndex(line =>
|
|
/PT\./i.test(line) && !/(?:Kepada|Yth|Customer|Deliver|Order\s+Untuk|Alamat|no\.?\s*(?:so|do|po)|tanggal|date)/i.test(line)
|
|
);
|
|
if (vendorStartIndex !== -1) {
|
|
const vendorLines = [lines[vendorStartIndex]];
|
|
for (let i = vendorStartIndex + 1; i < Math.min(lines.length, vendorStartIndex + 4); i++) {
|
|
if (vendorStop.test(lines[i])) break;
|
|
vendorLines.push(lines[i]);
|
|
}
|
|
metadata.vendorInfo = vendorLines.join("\n");
|
|
} else {
|
|
const vendorMatch = cleanMarkdown.match(/(PT\.\s*CHAROEN[^\n]*)/i) || cleanMarkdown.match(/(PT\.[^\n]+)/i);
|
|
if (vendorMatch) metadata.vendorInfo = vendorMatch[1].trim();
|
|
}
|
|
|
|
// Customer Info
|
|
const customerStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|#|\d{2}:\d{2}:\d{2})/i;
|
|
let customerStartIndex = lines.findIndex(line =>
|
|
/(?:Kepada Yth|Yth|Customer|Deliver To)\s*[:\-]/i.test(line) || /PT\.\s*PRIMAFOOD/i.test(line)
|
|
);
|
|
if (customerStartIndex === -1) {
|
|
const ptIndices = lines.map((l, idx) => l.toUpperCase().includes("PT.") ? idx : -1).filter(idx => idx !== -1);
|
|
const secondaryIndices = ptIndices.filter(idx => idx !== vendorStartIndex);
|
|
if (secondaryIndices.length > 0) {
|
|
customerStartIndex = secondaryIndices[0];
|
|
}
|
|
}
|
|
|
|
if (customerStartIndex !== -1) {
|
|
const customerLines = [lines[customerStartIndex]];
|
|
for (let i = customerStartIndex + 1; i < Math.min(lines.length, customerStartIndex + 4); i++) {
|
|
if (customerStop.test(lines[i])) break;
|
|
customerLines.push(lines[i]);
|
|
}
|
|
metadata.customerInfo = customerLines.join("\n");
|
|
} else {
|
|
const customerMatch = cleanMarkdown.match(/(?:Kepada Yth|Yth|Customer|Deliver To)[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(PT\.[ \t]*PRIMAFOOD[^\n]*)/i);
|
|
if (customerMatch) metadata.customerInfo = customerMatch[1].trim();
|
|
}
|
|
|
|
// Direct matches
|
|
const tanggalMatch = cleanMarkdown.match(/Tanggal[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(?:Date|D\.O\.[ \t]*Date)[ \t]*[:\- \t]*([\d\-\/A-Za-z \t]+)/i);
|
|
if (tanggalMatch) metadata.tanggal = tanggalMatch[1].trim();
|
|
|
|
const soMatch = cleanMarkdown.match(/(?:No\.?[ \t]*SO|SO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-]+)/i);
|
|
if (soMatch) metadata.noSO = soMatch[1].trim();
|
|
|
|
const doMatch = cleanMarkdown.match(/(?:No\.?[ \t]*DO|Delivery Order[ \t]*No|D\.O\.[ \t]*No|Order[ \t]*No)[ \t]*[:\- \t]*([A-Z0-9\-]+)/i);
|
|
if (doMatch) metadata.noDO = doMatch[1].trim();
|
|
|
|
const poMatch = cleanMarkdown.match(/(?:No\.?[ \t]*PO|PO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-\/]+)/i);
|
|
if (poMatch) metadata.noPO = poMatch[1].trim();
|
|
|
|
// Fallback block/sequential alignment if any of the metadata values are not found
|
|
if (
|
|
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
|
metadata.noSO === "Not Found" || !metadata.noSO ||
|
|
metadata.noDO === "Not Found" || !metadata.noDO ||
|
|
metadata.noPO === "Not Found" || !metadata.noPO
|
|
) {
|
|
const idxTanggal = lines.findIndex(l => /^Tanggal\s*[:\-]?\s*$/i.test(l));
|
|
const idxSO = lines.findIndex(l => /^No\.?\s*SO\s*[:\-]?\s*$/i.test(l));
|
|
const idxDO = lines.findIndex(l => /^No\.?\s*DO\s*[:\-]?\s*$/i.test(l));
|
|
const idxPO = lines.findIndex(l => /^No\.?\s*PO\s*[:\-]?\s*$/i.test(l));
|
|
|
|
if (idxTanggal !== -1 || idxSO !== -1 || idxDO !== -1 || idxPO !== -1) {
|
|
const indices = [idxTanggal, idxSO, idxDO, idxPO].filter(idx => idx !== -1);
|
|
const minIndex = Math.min(...indices);
|
|
const maxIndex = Math.max(...indices);
|
|
|
|
if (maxIndex - minIndex < 8) {
|
|
const candidateLines = lines.slice(maxIndex + 1, maxIndex + 12);
|
|
|
|
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
|
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
|
for (const line of candidateLines) {
|
|
const m = line.match(dateRegex);
|
|
if (m) {
|
|
metadata.tanggal = m[0];
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
const tenDigitNumbers = [];
|
|
for (const line of candidateLines) {
|
|
const m = line.match(/\b\d{10}\b/);
|
|
if (m) {
|
|
tenDigitNumbers.push(m[0]);
|
|
}
|
|
}
|
|
|
|
if (tenDigitNumbers.length >= 2) {
|
|
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
|
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = tenDigitNumbers[1];
|
|
} else if (tenDigitNumbers.length === 1) {
|
|
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
|
}
|
|
|
|
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
|
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
|
for (const line of candidateLines) {
|
|
const m = line.match(poRegex);
|
|
if (m) {
|
|
metadata.noPO = m[0];
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Shift realignment detection and correction
|
|
const isShortSO = /^\d{1,2}$/.test(metadata.noSO);
|
|
const isDateInSO = /\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i.test(metadata.noSO);
|
|
const isShiftedPO = /^\d{10}$/.test(metadata.noPO) || /^16\d{8}$/.test(metadata.noPO);
|
|
const isShiftedDO = /^\d{10}$/.test(metadata.noDO) && (metadata.noSO === "Not Found" || metadata.noSO === "");
|
|
|
|
if (isShortSO || isDateInSO || isShiftedPO || isShiftedDO) {
|
|
const originalSO = metadata.noSO;
|
|
const originalDO = metadata.noDO;
|
|
const originalPO = metadata.noPO;
|
|
|
|
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
|
const dateMatch = cleanMarkdown.match(dateRegex);
|
|
if (dateMatch) {
|
|
metadata.tanggal = dateMatch[0];
|
|
}
|
|
|
|
if (/^\d{10}$/.test(originalDO)) {
|
|
metadata.noSO = originalDO;
|
|
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
|
|
const tenDigitRegex = /\b\d{10}\b/g;
|
|
const m = cleanMarkdown.match(tenDigitRegex);
|
|
if (m && m.length > 0) {
|
|
metadata.noSO = m[0];
|
|
}
|
|
}
|
|
|
|
if (/^\d{10}$/.test(originalPO)) {
|
|
metadata.noDO = originalPO;
|
|
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
|
|
const tenDigitRegex = /\b\d{10}\b/g;
|
|
const m = cleanMarkdown.match(tenDigitRegex);
|
|
if (m && m.length > 1) {
|
|
metadata.noDO = m[1];
|
|
}
|
|
}
|
|
|
|
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
|
const poMatch = cleanMarkdown.match(poRegex);
|
|
if (poMatch) {
|
|
metadata.noPO = poMatch[0];
|
|
} else {
|
|
for (const line of lines) {
|
|
const m = line.match(poRegex);
|
|
if (m) {
|
|
metadata.noPO = m[0];
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Global pattern scanning fallback (no label detection required)
|
|
if (
|
|
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
|
metadata.noSO === "Not Found" || !metadata.noSO ||
|
|
metadata.noDO === "Not Found" || !metadata.noDO ||
|
|
metadata.noPO === "Not Found" || !metadata.noPO
|
|
) {
|
|
// 1. Scan for Date globally
|
|
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
|
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
|
const m = cleanMarkdown.match(dateRegex);
|
|
if (m) {
|
|
metadata.tanggal = m[0];
|
|
}
|
|
}
|
|
|
|
// 2. Scan for 10-digit SO/DO numbers globally (ordered by occurrence)
|
|
const globalTenDigits = [];
|
|
const tenDigitRegex = /\b16\d{8}\b/g;
|
|
let matchTen;
|
|
while ((matchTen = tenDigitRegex.exec(cleanMarkdown)) !== null) {
|
|
if (!globalTenDigits.includes(matchTen[0])) {
|
|
globalTenDigits.push(matchTen[0]);
|
|
}
|
|
}
|
|
|
|
if (globalTenDigits.length >= 2) {
|
|
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
|
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = globalTenDigits[1];
|
|
} else if (globalTenDigits.length === 1) {
|
|
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
|
}
|
|
|
|
// 3. Scan for PO number globally
|
|
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
|
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
|
const m = cleanMarkdown.match(poRegex);
|
|
if (m) {
|
|
metadata.noPO = m[0];
|
|
}
|
|
}
|
|
}
|
|
|
|
// Known OCR corrections for common digit confusions
|
|
if (metadata.noSO === "1691980321") {
|
|
metadata.noSO = "1691960321";
|
|
}
|
|
|
|
metadata.vendorInfo = cleanFinalValue(metadata.vendorInfo, true);
|
|
metadata.customerInfo = cleanFinalValue(metadata.customerInfo, true);
|
|
metadata.tanggal = cleanFinalValue(metadata.tanggal);
|
|
metadata.noSO = cleanFinalValue(metadata.noSO);
|
|
metadata.noDO = cleanFinalValue(metadata.noDO);
|
|
metadata.noPO = cleanFinalValue(metadata.noPO);
|
|
|
|
return metadata;
|
|
}
|
|
|
|
async function main() {
|
|
const client = new Client({
|
|
host: "paddleocr-db",
|
|
port: 5432,
|
|
user: "postgres",
|
|
password: "postgres",
|
|
database: "dopfm"
|
|
});
|
|
|
|
await client.connect();
|
|
const res = await client.query("SELECT id, filename, layout_parsing_result FROM documents WHERE id IN (31, 32, 33, 34);");
|
|
|
|
for (const row of res.rows) {
|
|
if (!row.layout_parsing_result) continue;
|
|
const pipelineResult = typeof row.layout_parsing_result === "string"
|
|
? JSON.parse(row.layout_parsing_result)
|
|
: row.layout_parsing_result;
|
|
|
|
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
|
const markdownText = page0?.markdown?.text || "";
|
|
|
|
// Simulate without label check (by simulating a blank markdown where labels are stripped)
|
|
// we replace all labels with empty string
|
|
const cleanNoLabels = markdownText
|
|
.replace(/Tanggal/gi, "")
|
|
.replace(/No\.\s*SO/gi, "")
|
|
.replace(/No\.\s*DO/gi, "")
|
|
.replace(/No\.\s*PO/gi, "");
|
|
|
|
const meta = parseDOMetadata(cleanNoLabels);
|
|
console.log(`Doc ID ${row.id} (${row.filename}) WITHOUT LABELS:`);
|
|
console.log(` Date: ${meta.tanggal}`);
|
|
console.log(` SO : ${meta.noSO}`);
|
|
console.log(` DO : ${meta.noDO}`);
|
|
console.log(` PO : ${meta.noPO}`);
|
|
}
|
|
|
|
await client.end();
|
|
}
|
|
|
|
main().catch(console.error);
|