fix: resolve table column shift, normalize units and standardize dates

This commit is contained in:
Rafhan Mazaya Fathurrahman committed 2026-07-03 16:04:57 +07:00
1 parent 095d4b799a
commit 2900eb670b
47 files changed
+11881 -2724

No files matched your search

+1
View File
@@ -7,6 +7,7 @@ const nextConfig: NextConfig = {
"*.trycloudflare.com",
"*.ngrok.io",
"*.ngrok-free.app",
"*.ngrok-free.dev",
"*.ngrok.app",
"*.loca.lt",
"*.serveo.net",
Binary file not shown.
+15 -17
View File
@@ -7,35 +7,19 @@ const BASE_URL = 'http://localhost:3000/api/parse';
const testFiles = [
"1782870899198-sample_do.jpeg",
"1782872141829-rotated_1782872118246_rotated_1782872112296_rotated_1782872107513_CAP3481952560331460805.jpg",
"1782875271815-rotated_1782875258464_rotated_1782875254091_rotated_1782875250123_CAP1599226410575173446.jpg",
"1782884664859-rotated_1782884661516_rotated_1782884658097_rotated_1782884654697_CAP5956452738616729026.jpg",
"1782888127716-CAP6161747431193129837.jpg",
"1782888211457-rotated_1782888204591_rotated_1782888199962_rotated_1782888195402_CAP7202641176939787142.jpg",
"1782888609885-rotated_1782888593813_1000000454.jpg",
"1782890303600-1000000465.jpg",
"1782892728794-1000000466.jpg",
"1782909791970-sample_do.jpeg",
"IMG_20260630_145445.jpg",
"IMG_20260701_134446.jpg",
"IMG_20260701_134504.jpg",
"IMG_20260701_134533.jpg",
"IMG_20260701_134555.jpg",
"IMG_20260701_134555~2.jpg",
"IMG_20260701_134635.jpg",
"IMG_20260701_134636.jpg",
"IMG_20260701_134646.jpg",
"IMG_20260701_134646~2.jpg",
"IMG_20260701_134704.jpg",
"IMG_20260701_134704~2.jpg",
"IMG_20260701_134810.jpg",
"IMG_20260701_134817.jpg",
"IMG_20260701_135900.jpg",
"IMG_20260701_135910.jpg",
"IMG_20260701_135923.jpg",
"IMG_20260701_135938.jpg",
"IMG_20260701_135950.jpg",
"IMG_20260701_140005.jpg"
"IMG_20260701_134810.jpg"
];
function postJSON(url, body) {
@@ -203,6 +187,10 @@ async function main() {
status: 'Success',
tilt: tiltStr,
unwarped: unwarpedStr,
rawMarkdown: resData.postProcessingDetails?.rawMarkdown || "",
layer1RawRegex: resData.postProcessingDetails?.layer1RawRegex || {},
layer2Sanitized: resData.postProcessingDetails?.layer2Sanitized || {},
layer3Final: resData.postProcessingDetails?.layer3Final || {},
metadata: docMeta,
items: resData.items || []
}) + '\n');
@@ -240,6 +228,16 @@ async function main() {
console.error('Failed to combine test reports:', combineErr);
}
// Compile JSONL into the final JSON v2
try {
const lines = fs.readFileSync(jsonlFile, 'utf8').split('\n').filter(Boolean);
const results = lines.map(line => JSON.parse(line));
fs.writeFileSync('/uploads/ai_results_v2.json', JSON.stringify(results, null, 2));
console.log('Compiled results saved to /uploads/ai_results_v2.json');
} catch (compileErr) {
console.error('Failed to compile results into JSON v2:', compileErr);
}
console.log('Batch test completed. Report written to /uploads/test_images_report.md');
}
+403
View File
@@ -0,0 +1,403 @@
/**
* run_full_test.js
*
* Runs OCR parsing against ALL images in backend/sources/test-images/
* and captures every pipeline stage for analysis:
* - rawMarkdown : raw text from PaddleOCR layout parser
* - layer1RawRegex: output of parseDOMetadata (regex extraction)
* - layer2Sanitized: output of sanitizeParsedMetadata (format checks)
* - layer3Final : final metadata after SKU triple-check + store resolution
*
* Outputs:
* backend/sources/ai_results.json — machine-readable per-file results
* backend/sources/ai_results.md — human-readable stage-by-stage breakdown
*
* Usage (from host machine, Docker must be running):
* node run_full_test.js
*
* The script talks to the nginx gateway on port 8000.
* To override: set env var BASE_URL=http://localhost:3000/api/parse
*/
const fs = require('fs');
const path = require('path');
const http = require('http');
const https = require('https');
// ─── Config ──────────────────────────────────────────────────────────────────
const BASE_URL = process.env.BASE_URL || 'http://localhost:8000/api/parse';
const TEST_IMAGES_DIR = path.resolve(__dirname, '../sources/test-images');
const OUTPUT_JSON = path.resolve(__dirname, '../sources/ai_results.json');
const OUTPUT_MD = path.resolve(__dirname, '../sources/ai_results.md');
const REQUEST_TIMEOUT_MS = 20 * 60 * 1000; // 20 minutes per image
// ─── HTTP Helper ─────────────────────────────────────────────────────────────
function postJSON(url, body) {
return new Promise((resolve, reject) => {
const parsedUrl = new URL(url);
const bodyStr = JSON.stringify(body);
const lib = parsedUrl.protocol === 'https:' ? https : http;
const options = {
hostname: parsedUrl.hostname,
port: parsedUrl.port || (parsedUrl.protocol === 'https:' ? 443 : 80),
path: parsedUrl.pathname + parsedUrl.search,
method: 'POST',
headers: {
'Content-Type': 'application/json',
'Content-Length': Buffer.byteLength(bodyStr),
},
timeout: REQUEST_TIMEOUT_MS,
};
const req = lib.request(options, (res) => {
let data = '';
res.on('data', (chunk) => { data += chunk; });
res.on('end', () => {
resolve({
ok: res.statusCode >= 200 && res.statusCode < 300,
status: res.statusCode,
body: data,
});
});
});
req.on('timeout', () => {
req.destroy(new Error(`Request timed out after ${REQUEST_TIMEOUT_MS / 60000}m`));
});
req.on('error', reject);
req.write(bodyStr);
req.end();
});
}
// ─── Markdown Helpers ─────────────────────────────────────────────────────────
function mdSection(title, level = 2) {
return `${'#'.repeat(level)} ${title}\n\n`;
}
function mdCode(content, lang = '') {
if (content === null || content === undefined) return '*null*\n\n';
const str = typeof content === 'string' ? content : JSON.stringify(content, null, 2);
return `\`\`\`${lang}\n${str}\n\`\`\`\n\n`;
}
function mdField(label, value) {
const display = (value === null || value === undefined || value === '') ? '*empty*' : `\`${value}\``;
return `- **${label}**: ${display}\n`;
}
function mdTable(headers, rows) {
if (!rows || rows.length === 0) return '*No items.*\n\n';
const sep = headers.map(() => '---');
const lines = [
`| ${headers.join(' | ')} |`,
`| ${sep.join(' | ')} |`,
...rows.map(r => `| ${r.map(c => String(c ?? '').replace(/\|/g, '\\|')).join(' | ')} |`),
];
return lines.join('\n') + '\n\n';
}
// ─── Main ─────────────────────────────────────────────────────────────────────
async function main() {
// Discover all image files
let files;
try {
files = fs.readdirSync(TEST_IMAGES_DIR).filter(f =>
/\.(jpe?g|png|webp|bmp)$/i.test(f)
).sort();
} catch (e) {
console.error(`Cannot read test-images directory: ${TEST_IMAGES_DIR}`);
console.error(e.message);
process.exit(1);
}
if (files.length === 0) {
console.error('No image files found in', TEST_IMAGES_DIR);
process.exit(1);
}
console.log(`\n🚀 Starting batch test`);
console.log(` API endpoint : ${BASE_URL}`);
console.log(` Images found : ${files.length}`);
console.log(` Output JSON : ${OUTPUT_JSON}`);
console.log(` Output MD : ${OUTPUT_MD}`);
console.log('─'.repeat(60));
const jsonResults = [];
const mdParts = [];
const summaryRows = [];
// ── Markdown document header ──────────────────────────────────────────────
mdParts.push(
`# OCR Batch Test Report\n\n`,
`> Generated: ${new Date().toISOString()}\n`,
`> API: \`${BASE_URL}\`\n`,
`> Images: **${files.length}** files from \`backend/sources/test-images/\`\n\n`,
`---\n\n`,
`## Summary\n\n`,
'<!-- summary_table_placeholder -->\n\n',
`---\n\n`,
`## Stage-by-Stage Results\n\n`,
);
const summaryPlaceholderIndex = mdParts.indexOf('<!-- summary_table_placeholder -->\n\n');
// ── Process each file ────────────────────────────────────────────────────
for (let idx = 0; idx < files.length; idx++) {
const file = files[idx];
const num = `[${String(idx + 1).padStart(2, '0')}/${files.length}]`;
process.stdout.write(`${num} ${file} ... `);
const entry = {
index: idx + 1,
filename: file,
status: 'pending',
tilt: null,
unwarped: null,
// pipeline stages
rawMarkdown: null,
layer1RawRegex: null,
layer2Sanitized: null,
layer3Final: null,
items: [],
error: null,
};
try {
const t0 = Date.now();
const res = await postJSON(BASE_URL, { filename: file });
const elapsed = ((Date.now() - t0) / 1000).toFixed(1);
if (!res.ok) {
process.stdout.write(`❌ HTTP ${res.status} (${elapsed}s)\n`);
entry.status = 'http_error';
entry.error = `HTTP ${res.status}: ${res.body}`;
} else {
let data;
try {
data = JSON.parse(res.body);
} catch (_) {
entry.status = 'json_parse_error';
entry.error = 'Response is not valid JSON';
process.stdout.write(`❌ JSON parse error (${elapsed}s)\n`);
data = null;
}
if (data) {
if (data.error) {
process.stdout.write(`⚠️ API error: ${data.error} (${elapsed}s)\n`);
entry.status = 'api_error';
entry.error = data.error;
} else {
const pipelineInfo = (data.result || {}).pipeline_info || {};
entry.status = 'success';
entry.tilt = pipelineInfo.tilt !== undefined ? +parseFloat(pipelineInfo.tilt).toFixed(2) : null;
entry.unwarped = pipelineInfo.unwarped ?? null;
const ppd = data.postProcessingDetails || {};
entry.rawMarkdown = ppd.rawMarkdown ?? null;
entry.layer1RawRegex = ppd.layer1RawRegex ?? null;
entry.layer2Sanitized = ppd.layer2Sanitized ?? null;
entry.layer3Final = ppd.layer3Final ?? null;
entry.items = data.items ?? [];
const itemCount = entry.items.length;
process.stdout.write(`✅ ${itemCount} item(s), tilt=${entry.tilt ?? 'N/A'}° (${elapsed}s)\n`);
}
}
}
} catch (err) {
process.stdout.write(`💥 ${err.message}\n`);
entry.status = 'exception';
entry.error = err.message;
}
jsonResults.push(entry);
// ── Build per-file markdown section ─────────────────────────────────────
const statusEmoji = {
success: '✅',
http_error: '❌',
api_error: '⚠️',
json_parse_error: '❌',
exception: '💥',
}[entry.status] || '❓';
let fileMd = '';
fileMd += `### ${idx + 1}. \`${file}\`\n\n`;
fileMd += `**Status**: ${statusEmoji} \`${entry.status}\`\n\n`;
if (entry.status !== 'success') {
fileMd += `> **Error**: ${entry.error}\n\n`;
fileMd += `---\n\n`;
summaryRows.push([idx + 1, `\`${file}\``, `${statusEmoji} ${entry.status}`, 'N/A', 'N/A', 'N/A', 'N/A']);
mdParts.push(fileMd);
continue;
}
// ── Stage 0: Pipeline Info ────────────────────────────────────────────────
fileMd += `#### 📐 Stage 0 — Pipeline Info\n\n`;
fileMd += mdField('Tilt detected', entry.tilt !== null ? `${entry.tilt}°` : 'N/A');
fileMd += mdField('Auto-unwarped', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A');
fileMd += '\n';
// ── Stage 1: Raw Markdown from OCR ───────────────────────────────────────
fileMd += `#### 📄 Stage 1 — Raw OCR Markdown\n\n`;
fileMd += `*This is the raw text extracted by PaddleOCR layout parser before any post-processing.*\n\n`;
if (entry.rawMarkdown) {
fileMd += mdCode(entry.rawMarkdown, 'markdown');
} else {
fileMd += '*No raw markdown captured.*\n\n';
}
// ── Stage 2: Layer 1 — Regex Extraction ──────────────────────────────────
fileMd += `#### 🔍 Stage 2 — Layer 1: Regex Extraction (\`parseDOMetadata\`)\n\n`;
fileMd += `*Regex patterns are applied to raw markdown to extract header fields and item rows.*\n\n`;
if (entry.layer1RawRegex) {
const l1 = entry.layer1RawRegex;
fileMd += `**Header fields (raw regex output):**\n\n`;
fileMd += mdField('noDO', l1.noDO);
fileMd += mdField('noPO', l1.noPO);
fileMd += mdField('noSO', l1.noSO);
fileMd += mdField('tanggal', l1.tanggal);
fileMd += mdField('vendorInfo', l1.vendorInfo);
fileMd += mdField('customerInfo', l1.customerInfo);
fileMd += mdField('alamat', l1.alamat);
fileMd += mdField('orderUntuk', l1.orderUntuk);
fileMd += mdField('platTruk', l1.platTruk);
fileMd += '\n';
fileMd += `**Raw items (${(l1.items || []).length} row(s)):**\n\n`;
fileMd += mdTable(
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
(l1.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
);
} else {
fileMd += '*Layer 1 data not captured.*\n\n';
}
// ── Stage 3: Layer 2 — Sanitized ─────────────────────────────────────────
fileMd += `#### 🧹 Stage 3 — Layer 2: Sanitized (\`sanitizeParsedMetadata\`)\n\n`;
fileMd += `*Strict format enforcement: corrects date formats, trims whitespace, enforces field constraints.*\n\n`;
if (entry.layer2Sanitized) {
const l2 = entry.layer2Sanitized;
fileMd += `**Header fields (after sanitization):**\n\n`;
fileMd += mdField('noDO', l2.noDO);
fileMd += mdField('noPO', l2.noPO);
fileMd += mdField('noSO', l2.noSO);
fileMd += mdField('tanggal', l2.tanggal);
fileMd += mdField('vendorInfo', l2.vendorInfo);
fileMd += mdField('customerInfo', l2.customerInfo);
fileMd += mdField('alamat', l2.alamat);
fileMd += mdField('orderUntuk', l2.orderUntuk);
fileMd += mdField('platTruk', l2.platTruk);
fileMd += '\n';
fileMd += `**Sanitized items (${(l2.items || []).length} row(s)):**\n\n`;
fileMd += mdTable(
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
(l2.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
);
} else {
fileMd += '*Layer 2 data not captured.*\n\n';
}
// ── Stage 4: Layer 3 — Final (SKU triple-check + store resolution) ────────
fileMd += `#### ✅ Stage 4 — Layer 3: Final (\`SKU triple-check + store resolution\`)\n\n`;
fileMd += `*SKU validated against master list (score ≥ 0.6 threshold). Items with noise SKU codes are filtered out. Store resolved from DB.*\n\n`;
if (entry.layer3Final) {
const l3 = entry.layer3Final;
fileMd += `**Final metadata:**\n\n`;
fileMd += mdField('noDO', l3.noDO);
fileMd += mdField('noPO', l3.noPO);
fileMd += mdField('noSO', l3.noSO);
fileMd += mdField('tanggal', l3.tanggal);
fileMd += mdField('vendorInfo', l3.vendorInfo);
fileMd += mdField('customerInfo', l3.customerInfo);
fileMd += mdField('alamat', l3.alamat);
fileMd += mdField('orderUntuk', l3.orderUntuk);
fileMd += mdField('platTruk', l3.platTruk);
fileMd += '\n';
fileMd += `**Final items after SKU validation (${(l3.items || []).length} row(s)):**\n\n`;
fileMd += mdTable(
['kodeBarangOriginal', 'kodeBarang (corrected)', 'namaBarang', 'banyak', 'jumlah'],
(l3.items || []).map(it => [
it.kodeBarangOriginal ?? it.kodeBarang,
it.kodeBarang,
it.namaBarang,
it.banyak,
it.jumlah
])
);
} else {
fileMd += '*Layer 3 data not captured.*\n\n';
}
// ── Stage 5: Final submitted items (from root items[]) ───────────────────
fileMd += `#### 🗃️ Stage 5 — Submitted Items (ready-to-use JSON)\n\n`;
fileMd += `*These are the items actually returned to the caller and saved to the database.*\n\n`;
fileMd += mdTable(
['kodeBarangOriginal', 'kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
(entry.items || []).map(it => [
it.kodeBarangOriginal ?? it.kodeBarang,
it.kodeBarang,
it.namaBarang,
it.banyak,
it.jumlah
])
);
fileMd += `---\n\n`;
// ── Summary row ──────────────────────────────────────────────────────────
const l3meta = entry.layer3Final || {};
summaryRows.push([
idx + 1,
`\`${file}\``,
`${statusEmoji} success`,
entry.tilt !== null ? `${entry.tilt}°` : 'N/A',
entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A',
`\`${l3meta.noDO ?? 'N/A'}\``,
`\`${l3meta.noPO ?? 'N/A'}\``,
`${entry.items.length}`,
]);
mdParts.push(fileMd);
}
// ── Inject summary table ──────────────────────────────────────────────────
const summaryTable = mdTable(
['#', 'Filename', 'Status', 'Tilt', 'Unwarped', 'DO', 'PO', 'Items'],
summaryRows
);
mdParts[summaryPlaceholderIndex] = summaryTable;
// ── Write outputs ─────────────────────────────────────────────────────────
const jsonOut = JSON.stringify(jsonResults, null, 2);
fs.writeFileSync(OUTPUT_JSON, jsonOut, 'utf8');
console.log(`\n✅ JSON saved → ${OUTPUT_JSON}`);
const mdOut = mdParts.join('');
fs.writeFileSync(OUTPUT_MD, mdOut, 'utf8');
console.log(`✅ MD saved → ${OUTPUT_MD}`);
// ── Final stats ───────────────────────────────────────────────────────────
const succeeded = jsonResults.filter(r => r.status === 'success').length;
const failed = jsonResults.length - succeeded;
console.log('\n─'.repeat(60));
console.log(` Total : ${jsonResults.length}`);
console.log(` Success: ${succeeded}`);
console.log(` Failed : ${failed}`);
console.log('─'.repeat(60));
}
main().catch(err => {
console.error('Fatal error:', err);
process.exit(1);
});
+46 -14
View File
@@ -51,12 +51,15 @@ export async function POST(req: NextRequest) {
let isSample = true;
if (!fs.existsSync(filePath)) {
filePath = path.join("/uploads", safeFile);
filePath = path.join(process.cwd(), "public", "test-images", safeFile);
if (!fs.existsSync(filePath)) {
return NextResponse.json({ error: "File not found" }, { status: 404 });
filePath = path.join("/uploads", safeFile);
if (!fs.existsSync(filePath)) {
return NextResponse.json({ error: "File not found" }, { status: 404 });
}
isUpload = true;
isSample = false;
}
isUpload = true;
isSample = false;
}
// Read file and compute content hash
@@ -156,6 +159,7 @@ export async function POST(req: NextRequest) {
const rawMetadata = parseDOMetadata(markdownText);
// Second-layer sanity check: enforces strict field formats and auto-corrects anomalies
const docMetadata = sanitizeParsedMetadata(rawMetadata as any);
const docMetadataSanitized = JSON.parse(JSON.stringify(docMetadata));
// Load SKU Master list from DB (including new packaging details for triple check validation)
const skuDbRes = await query("SELECT no_sku, nama_item, standar_jumlah, jenis_outer FROM sku_master");
@@ -256,15 +260,25 @@ export async function POST(req: NextRequest) {
const numericQtyMatch = ocrQty.match(/[\d.,]+/);
if (numericQtyMatch) {
const numericQty = numericQtyMatch[0];
const qtyUnitMatch = ocrQty.match(/[a-zA-Z]+/);
const parsedQtyUnit = qtyUnitMatch ? qtyUnitMatch[0].toUpperCase() : "";
const isQtyUnitValid = ["KRG", "BOX", "BAG", "PAC", "PC", "KG", "PCS"].includes(parsedQtyUnit);
const standardOuter = bestMatch.jenis_outer || "";
if (standardOuter.toLowerCase() === "karung") {
item.banyak = `${numericQty} KRG`;
} else if (standardOuter.toLowerCase() === "box") {
item.banyak = `${numericQty} BOX`;
} else if (standardOuter.toLowerCase() === "bag") {
item.banyak = `${numericQty} BAG`;
if (isQtyUnitValid) {
item.banyak = `${numericQty} ${parsedQtyUnit}`;
} else if (standardOuter) {
if (standardOuter.toLowerCase() === "karung") {
item.banyak = `${numericQty} KRG`;
} else if (standardOuter.toLowerCase() === "box") {
item.banyak = `${numericQty} BOX`;
} else if (standardOuter.toLowerCase() === "bag") {
item.banyak = `${numericQty} BAG`;
} else {
item.banyak = `${numericQty} ${standardOuter.toUpperCase()}`;
}
} else {
item.banyak = `${numericQty} ${standardOuter.toUpperCase()}`;
item.banyak = ocrQty;
}
}
@@ -272,8 +286,20 @@ export async function POST(req: NextRequest) {
const numericPriceMatch = ocrPrice.match(/[\d.,]+/);
if (numericPriceMatch) {
const numericPrice = numericPriceMatch[0];
const standardInner = (bestMatch.standar_jumlah || "").toUpperCase();
item.jumlah = `${numericPrice} ${standardInner}`;
const priceUnitMatch = ocrPrice.match(/[a-zA-Z]+/);
const parsedPriceUnit = priceUnitMatch ? priceUnitMatch[0].toUpperCase() : "";
const isPriceUnitValid = ["KRG", "BOX", "BAG", "PAC", "PC", "KG", "PCS"].includes(parsedPriceUnit);
const standardInner = bestMatch.standar_jumlah || "";
if (standardInner.toLowerCase() === "pc" || standardInner.toLowerCase() === "pcs") {
item.jumlah = `${numericPrice} ${standardInner.toUpperCase()}`;
} else if (isPriceUnitValid) {
item.jumlah = `${numericPrice} ${parsedPriceUnit}`;
} else if (standardInner) {
item.jumlah = `${numericPrice} ${standardInner.toUpperCase()}`;
} else {
item.jumlah = ocrPrice;
}
}
checkedItems.push(item);
@@ -371,12 +397,18 @@ export async function POST(req: NextRequest) {
]);
}
// Return wrapped success response with items, flagged, remarks
// Return wrapped success response with items, flagged, remarks, and intermediate post-processing details
return NextResponse.json({
errorCode: 0,
errorMsg: "Success",
result: pipelineResult,
items: docMetadata.items,
postProcessingDetails: {
rawMarkdown: markdownText,
layer1RawRegex: rawMetadata,
layer2Sanitized: docMetadataSanitized,
layer3Final: docMetadata
},
flagged: {},
remarks: {}
});
+26 -1
View File
@@ -29,19 +29,43 @@ function runTests() {
{ name: "Date (Asli/Copy) prefix noise", markdown: "Tanggal: (Asli/Copy) 15 May 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "15 May 2026" } },
{ name: "Date junk suffix cut", markdown: "Tanggal: 25 May 2020 (Printed by system)", expected: { tanggal: "25 May 2020" } },
{ name: "Date bad OCR month Hv -> Not Found", markdown: "Tanggal:25 Hv 2024\nNo.SO : 1601001206", expected: { tanggal: "Not Found" } },
{ name: "Date single digit 7 May 2026 -> 07 May 2026", markdown: "Tanggal: 7 May 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "07 May 2026" } },
{ name: "Date single digit 4 Apr 2026 -> 04 April 2026", markdown: "Tanggal: 4 Apr 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "04 April 2026" } },
{ name: "00117709 before Tanggal must not pollute date", markdown: "00117709\nTanggal:25May2020\nNo.SO : 1091721200\nNo. DO : 1657943004\nNo. PO : F0/26/0000190929", expected: { tanggal: "25 May 2020" } },
{ name: "Plate B 9427 UXT", markdown: "Truck No. B 9427 UXT\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9427 UXT" } },
{ name: "Plate B-9999-XYZ dash", markdown: "No. Polisi: B-9999-XYZ\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9999 XYZ" } },
{ name: "Plate ignore PO/SO prefix", markdown: "Plate is PO 1234 SO but real truck is A 123 B\nNo. PO : PO/26/0000178435", expected: { platTruk: "A 123 B" } },
{ name: "Plate B9427UXT adjacent", markdown: "No Polisi B9427UXT\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9427 UXT" } },
{ name: "Plate real doc B 9723 CXS", markdown: "Truck No.\nB 9723 CXS\nWH 01 / 01\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9723 CXS" } },
{
name: "Table column shift alignment correction",
markdown: "No. PO : PO/26/0000178435\n" +
"<table>" +
"<tr><td>Kode Barang</td><td>Nama Barang</td><td>Banyak</td><td>Jumlah</td></tr>" +
"<tr><td></td><td>Item A</td><td>2 KRG</td><td>40 PC</td></tr>" +
"<tr><td>11310014</td><td>Item B</td><td>2 KRG</td><td>40 PC</td></tr>" +
"<tr><td>11310024</td><td>Item C</td><td>1 BOX</td><td>10 KG</td></tr>" +
"<tr><td>11720055</td><td></td><td></td><td></td></tr>" +
"</table>",
expected: {
items: [
{ kodeBarang: "11310014", namaBarang: "Item A", banyak: "2 KRG", jumlah: "40 PC" },
{ kodeBarang: "11310024", namaBarang: "Item B", banyak: "2 KRG", jumlah: "40 PC" },
{ kodeBarang: "11720055", namaBarang: "Item C", banyak: "1 BOX", jumlah: "10 KG" }
]
} as any
}
];
for (const t of parseTests) {
try {
const result = parseDOMetadata(t.markdown) as any;
for (const [key, val] of Object.entries(t.expected)) {
assert.strictEqual(result[key], val, `field [${key}] expected "${val}" got "${result[key]}"`);
if (key === "items") {
assert.deepStrictEqual(result.items, val);
} else {
assert.strictEqual(result[key], val, `field [${key}] expected "${val}" got "${result[key]}"`);
}
}
console.log(`[PASS] ${t.name}`);
} catch (err: any) {
@@ -56,6 +80,7 @@ function runTests() {
// tanggal valid
{ name: "sanitize: valid tanggal 30 June 2026 passes", input: { tanggal: "30 June 2026" }, expected: { tanggal: "30 June 2026" } },
{ name: "sanitize: valid tanggal 25 May 2020 passes", input: { tanggal: "25 May 2020" }, expected: { tanggal: "25 May 2020" } },
{ name: "sanitize: single digit tanggal 4 April 2026 -> 04 April 2026", input: { tanggal: "4 April 2026" }, expected: { tanggal: "04 April 2026" } },
// tanggal invalid
{ name: "sanitize: tanggal bad month Hv -> Not Found", input: { tanggal: "25 Hv 2024" }, expected: { tanggal: "Not Found" } },
{ name: "sanitize: tanggal as number 0011770 -> Not Found", input: { tanggal: "0011770" }, expected: { tanggal: "Not Found" } },
+130 -40
View File
@@ -26,7 +26,7 @@ function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn, 07/26/nnn
// Structure: [PREFIX][optional_noise_digits][/][optional_year_segment][/]?[NUMBER]
// We find the LAST slash and take everything after it as the real number
const withSlash = /^(?:[A-Z0-9]{1,4})[ \t]*[\/\-][ \t]*(?:\d{1,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
const withSlash = /^(?:[A-Z0-9]+)[ \t]*[\/\-][ \t]*(?:\d{1,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
const m1 = stripped.match(withSlash);
if (m1) {
return `PO/${currentYearLastTwo}/${m1[1]}`;
@@ -93,7 +93,20 @@ function cleanDateValue(raw: string): string {
const cleaned = raw.trim();
if (cleaned === "Not Found" || cleaned === "") return "Not Found";
const today = new Date();
// Try unified regex first
const unifiedPattern = /(\d{1,2})[ \t\-\/]*(jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|maret|apr(?:il)?|may|mei|jun(?:[ei])?|jul(?:[ii])?|aug(?:ustus)?|agt|ags|sep(?:tember)?|oct(?:ober)?|okt|nov(?:ember)?|nop(?:ember)?|dec(?:ember)?|des(?:ember)?)[ \t\-\/]*(20\d{2})/i;
const match = cleaned.match(unifiedPattern);
if (match) {
const day = parseInt(match[1], 10);
const monthKey = match[2].toLowerCase();
const year = parseInt(match[3], 10);
const month = MONTHS_MAP[monthKey] || monthKey;
if (day >= 1 && day <= 31 && year >= 2010 && year <= 2035) {
const paddedDay = day.toString().padStart(2, '0');
return `${paddedDay} ${month} ${year}`;
}
}
let day: number | null = null;
let monthStr: string | null = null;
let year: number | null = null;
@@ -119,12 +132,14 @@ function cleanDateValue(raw: string): string {
}
}
if (!monthStr) return "Not Found";
// 3. Try to find 1 or 2 digit day (not part of the year)
let textForDay = cleaned;
if (year) {
textForDay = textForDay.replace(year.toString(), "");
}
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g);
const dayMatches = textForDay.match(/\b(\d{1,2})\b/g) || textForDay.match(/(\d{1,2})/g);
if (dayMatches) {
for (const matchStr of dayMatches) {
const parsedDay = parseInt(matchStr, 10);
@@ -135,17 +150,12 @@ function cleanDateValue(raw: string): string {
}
}
// 4. Fallback fill-in from current date
const currentYear = today.getFullYear();
const currentMonthNames = ["January", "February", "March", "April", "May", "June", "July", "August", "September", "October", "November", "December"];
const currentMonth = currentMonthNames[today.getMonth()];
const currentDay = today.getDate();
if (day !== null && year !== null) {
const paddedDay = day.toString().padStart(2, '0');
return `${paddedDay} ${monthStr} ${year}`;
}
const finalDay = day !== null ? day : currentDay;
const finalMonth = monthStr !== null ? monthStr : currentMonth;
const finalYear = year !== null ? year : currentYear;
return `${finalDay} ${finalMonth} ${finalYear}`;
return "Not Found";
}
export function parseDOMetadata(markdown: string) {
@@ -317,25 +327,38 @@ export function parseDOMetadata(markdown: string) {
metadata.tanggal = dateMatch[0];
}
// 2. Real SO is the value that was matched under No. DO
if (/^\d{10}$/.test(originalDO)) {
metadata.noSO = originalDO;
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex);
if (m && m.length > 0) {
metadata.noSO = m[0];
}
}
// Only shift values if the PO field is shifted (i.e. PO field is a 10-digit number instead of PO format).
// If the PO field has a valid PO format (e.g. PO/26/...), PO and DO are in correct places and not shifted.
const isPoValid = /^PO\/[A-Z0-9\-\/]+\b/i.test(originalPO);
// 3. Real DO is the value that was matched under No. PO
if (/^\d{10}$/.test(originalPO)) {
metadata.noDO = originalPO;
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex);
if (m && m.length > 1) {
metadata.noDO = m[1];
if (!isPoValid) {
// 2. Real SO is the value that was matched under No. DO
if (/^\d{10}$/.test(originalDO)) {
metadata.noSO = originalDO;
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex);
const cleanMatches = m ? m.filter(val => !originalPO.includes(val)) : [];
if (cleanMatches.length > 0) {
metadata.noSO = cleanMatches[0];
}
}
// 3. Real DO is the value that was matched under No. PO
if (/^\d{10}$/.test(originalPO)) {
metadata.noDO = originalPO;
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
const tenDigitRegex = /\b\d{10}\b/g;
const m = cleanMarkdown.match(tenDigitRegex);
const cleanMatches = m ? m.filter(val => !originalPO.includes(val)) : [];
if (cleanMatches.length > 1) {
metadata.noDO = cleanMatches[1];
}
}
} else {
// PO is valid and NOT shifted. SO was just empty or noise.
if (isShortSO || isDateInSO) {
metadata.noSO = "Not Found";
}
}
@@ -420,7 +443,11 @@ export function parseDOMetadata(markdown: string) {
metadata.noDO = cleanFinalValue(metadata.noDO);
// Extract PO year dynamically from parsed document date (or default to current year if date not found)
const docYearLastTwo = getYearFromDate(metadata.tanggal);
const currentYY = new Date().getFullYear().toString().slice(-2);
let docYearLastTwo = getYearFromDate(metadata.tanggal);
if (docYearLastTwo !== currentYY) {
docYearLastTwo = currentYY;
}
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), docYearLastTwo);
// Parse HTML tables for items
@@ -536,6 +563,45 @@ export function parseDOMetadata(markdown: string) {
}
if (isItemsTable) {
// Check for SKU column downward shift of 1 row:
// - First data row (r=1) has empty SKU, but has description
// - A downstream row (shiftBoundaryRow) has SKU, but description, banyak, jumlah are empty
if (numRows > 2 && kIdx !== -1 && nIdx !== -1) {
const firstRowSku = (grid[1][kIdx] || "").trim();
const firstRowDesc = (grid[1][nIdx] || "").trim();
if (firstRowSku === "" && firstRowDesc !== "") {
let shiftBoundaryRow = -1;
for (let r = 2; r < numRows; r++) {
const skuVal = (grid[r][kIdx] || "").trim();
let othersEmpty = true;
for (let c = 0; c < maxCols; c++) {
if (c !== kIdx) {
const val = (grid[r][c] || "").trim();
if (val !== "") {
othersEmpty = false;
break;
}
}
}
if (skuVal !== "" && /^\d{8}$/.test(skuVal) && othersEmpty) {
shiftBoundaryRow = r;
break;
}
}
if (shiftBoundaryRow !== -1) {
console.log(`Detected shifted SKU column in table. Shift boundary row: ${shiftBoundaryRow}. Correcting alignment...`);
for (let r = 1; r < shiftBoundaryRow; r++) {
grid[r][kIdx] = grid[r + 1][kIdx];
}
grid[shiftBoundaryRow][kIdx] = "";
}
}
}
for (let r = 1; r < numRows; r++) {
const cells = grid[r];
const kodeCell = cells[kIdx] || "";
@@ -544,7 +610,7 @@ export function parseDOMetadata(markdown: string) {
let jumlahCell = "";
// Check if there is an extra column before banyak that we should merge with banyak
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx && headerCells[bIdx - 1] !== headerCells[nIdx]) {
const qtyCell = cells[bIdx - 1] || "";
const unitCell = cells[bIdx] || "";
banyakCell = `${qtyCell} ${unitCell}`.trim();
@@ -612,6 +678,11 @@ export function parseDOMetadata(markdown: string) {
// Clean checkmarks and extra spaces from banyak
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
b = b.replace(/\b80[xX]\b/g, "BOX")
.replace(/\bB0[xX]\b/g, "BOX")
.replace(/\b8[aA][gG]\b/g, "BAG")
.replace(/\b[pP][aA][iI]\b/g, "PAC")
.trim();
// Fallback for Banyak if empty or purely alphabetical unit
if (!b) {
@@ -626,11 +697,21 @@ export function parseDOMetadata(markdown: string) {
const name = n.toLowerCase();
if (code === "11310024" || name.includes("griller")) {
b = `${b} KRG`;
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
} else if (code === "11640053" || name.includes("bone in leg")) {
b = `${b} BAG`;
} else if (code.startsWith("21") || code.startsWith("12") || name.includes("nasi") || name.includes("rice") || name.includes("nugget") || name.includes("sosis") || name.includes("sausage") || name.includes("fiesta") || name.includes("champ")) {
b = `${b} BOX`;
}
}
// Clean checkmarks and extra spaces from jumlah
let cleanJ = j.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
cleanJ = cleanJ.replace(/\b80[xX]\b/g, "BOX")
.replace(/\bB0[xX]\b/g, "BOX")
.replace(/\b8[aA][gG]\b/g, "BAG")
.replace(/\b[pP][aA][iI]\b/g, "PAC")
.trim();
const cleanKode = k.trim();
const cleanNama = n.trim();
if (cleanKode === "" && cleanNama === "") {
@@ -641,7 +722,7 @@ export function parseDOMetadata(markdown: string) {
kodeBarang: k,
namaBarang: n,
banyak: b,
jumlah: j
jumlah: cleanJ
});
}
}
@@ -671,6 +752,11 @@ export function extractPlatTruk(text: string): string {
];
const isValidPrefix = (p: string) => arrayPlat.includes(p.toUpperCase());
const FORBIDDEN_SUFFIXES = [
"BOX", "KRG", "BAG", "PAC", "KG", "PC", "PCS", "LGT", "K", "EKR", "LTR", "BKS",
"IDN", "WH", "PM", "TTD", "CPI", "HO", "NO", "GR", "G"
];
const isForbiddenSuffix = (s: string) => FORBIDDEN_SUFFIXES.includes(s.toUpperCase());
// 1. Look for explicit labels: No. Polisi, No. Pol, No. Polisi:, No. Kendaraan, Plat No, Plat, Truck No., etc.
const labelRegex = /(?:No\.?\s*(?:Polisi|Pol|Kendaraan|Mobil|Truck|Pol\.?)|Plat(?:\s*No)?|Truck\s*No\.?)\s*[:\-.]?\s*\b([A-Z]{1,2})[ \t\-]*(\d{1,4})[ \t\-]*([A-Z]{1,3})\b/i;
@@ -679,7 +765,7 @@ export function extractPlatTruk(text: string): string {
const prefix = labelMatch[1].toUpperCase();
const num = labelMatch[2];
const suffix = labelMatch[3].toUpperCase();
if (isValidPrefix(prefix)) {
if (isValidPrefix(prefix) && !isForbiddenSuffix(suffix)) {
return `${prefix} ${num} ${suffix}`;
}
}
@@ -692,7 +778,7 @@ export function extractPlatTruk(text: string): string {
const num = match[2];
const suffix = match[3].toUpperCase();
if (isValidPrefix(prefix)) {
if (isValidPrefix(prefix) && !isForbiddenSuffix(suffix)) {
return `${prefix} ${num} ${suffix}`;
}
}
@@ -771,9 +857,10 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
const day = parseInt(dm[1], 10);
const year = parseInt(dm[3], 10);
if (day >= 1 && day <= 31 && year >= 2010 && year <= currentFullYear + 1) {
// Valid — normalize capitalization
// Valid — normalize capitalization and pad day with leading zero
const month = dm[2].charAt(0).toUpperCase() + dm[2].slice(1).toLowerCase();
result.tanggal = `${dm[1]} ${month} ${dm[3]}`;
const paddedDay = dm[1].padStart(2, '0');
result.tanggal = `${paddedDay} ${month} ${dm[3]}`;
} else {
result.tanggal = "Not Found";
}
@@ -783,7 +870,10 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
// --- noPO ---
// Get year from parsed date (or default to current year if date not found)
const docYY = result.tanggal && result.tanggal !== "Not Found" ? getYearFromDate(result.tanggal) : currentYY;
let docYY = result.tanggal && result.tanggal !== "Not Found" ? getYearFromDate(result.tanggal) : currentYY;
if (docYY !== currentYY) {
docYY = currentYY;
}
// --- noPO ---
// Must match PO/YY/NNNN+ where YY is the document-specific year segment, NNNN = 4+ digits