404 lines
16 KiB
JavaScript
404 lines
16 KiB
JavaScript
/**
|
|
* run_full_test.js
|
|
*
|
|
* Runs OCR parsing against ALL images in backend/sources/test-images/
|
|
* and captures every pipeline stage for analysis:
|
|
* - rawMarkdown : raw text from PaddleOCR layout parser
|
|
* - layer1RawRegex: output of parseDOMetadata (regex extraction)
|
|
* - layer2Sanitized: output of sanitizeParsedMetadata (format checks)
|
|
* - layer3Final : final metadata after SKU triple-check + store resolution
|
|
*
|
|
* Outputs:
|
|
* backend/sources/ai_results.json — machine-readable per-file results
|
|
* backend/sources/ai_results.md — human-readable stage-by-stage breakdown
|
|
*
|
|
* Usage (from host machine, Docker must be running):
|
|
* node run_full_test.js
|
|
*
|
|
* The script talks to the nginx gateway on port 8000.
|
|
* To override: set env var BASE_URL=http://localhost:3000/api/parse
|
|
*/
|
|
|
|
const fs = require('fs');
|
|
const path = require('path');
|
|
const http = require('http');
|
|
const https = require('https');
|
|
|
|
// ─── Config ──────────────────────────────────────────────────────────────────
|
|
|
|
const BASE_URL = process.env.BASE_URL || 'http://localhost:8000/api/parse';
|
|
const TEST_IMAGES_DIR = path.resolve(__dirname, '../sources/test-images');
|
|
const OUTPUT_JSON = path.resolve(__dirname, '../sources/ai_results.json');
|
|
const OUTPUT_MD = path.resolve(__dirname, '../sources/ai_results.md');
|
|
const REQUEST_TIMEOUT_MS = 20 * 60 * 1000; // 20 minutes per image
|
|
|
|
// ─── HTTP Helper ─────────────────────────────────────────────────────────────
|
|
|
|
function postJSON(url, body) {
|
|
return new Promise((resolve, reject) => {
|
|
const parsedUrl = new URL(url);
|
|
const bodyStr = JSON.stringify(body);
|
|
const lib = parsedUrl.protocol === 'https:' ? https : http;
|
|
|
|
const options = {
|
|
hostname: parsedUrl.hostname,
|
|
port: parsedUrl.port || (parsedUrl.protocol === 'https:' ? 443 : 80),
|
|
path: parsedUrl.pathname + parsedUrl.search,
|
|
method: 'POST',
|
|
headers: {
|
|
'Content-Type': 'application/json',
|
|
'Content-Length': Buffer.byteLength(bodyStr),
|
|
},
|
|
timeout: REQUEST_TIMEOUT_MS,
|
|
};
|
|
|
|
const req = lib.request(options, (res) => {
|
|
let data = '';
|
|
res.on('data', (chunk) => { data += chunk; });
|
|
res.on('end', () => {
|
|
resolve({
|
|
ok: res.statusCode >= 200 && res.statusCode < 300,
|
|
status: res.statusCode,
|
|
body: data,
|
|
});
|
|
});
|
|
});
|
|
|
|
req.on('timeout', () => {
|
|
req.destroy(new Error(`Request timed out after ${REQUEST_TIMEOUT_MS / 60000}m`));
|
|
});
|
|
req.on('error', reject);
|
|
|
|
req.write(bodyStr);
|
|
req.end();
|
|
});
|
|
}
|
|
|
|
// ─── Markdown Helpers ─────────────────────────────────────────────────────────
|
|
|
|
function mdSection(title, level = 2) {
|
|
return `${'#'.repeat(level)} ${title}\n\n`;
|
|
}
|
|
|
|
function mdCode(content, lang = '') {
|
|
if (content === null || content === undefined) return '*null*\n\n';
|
|
const str = typeof content === 'string' ? content : JSON.stringify(content, null, 2);
|
|
return `\`\`\`${lang}\n${str}\n\`\`\`\n\n`;
|
|
}
|
|
|
|
function mdField(label, value) {
|
|
const display = (value === null || value === undefined || value === '') ? '*empty*' : `\`${value}\``;
|
|
return `- **${label}**: ${display}\n`;
|
|
}
|
|
|
|
function mdTable(headers, rows) {
|
|
if (!rows || rows.length === 0) return '*No items.*\n\n';
|
|
const sep = headers.map(() => '---');
|
|
const lines = [
|
|
`| ${headers.join(' | ')} |`,
|
|
`| ${sep.join(' | ')} |`,
|
|
...rows.map(r => `| ${r.map(c => String(c ?? '').replace(/\|/g, '\\|')).join(' | ')} |`),
|
|
];
|
|
return lines.join('\n') + '\n\n';
|
|
}
|
|
|
|
// ─── Main ─────────────────────────────────────────────────────────────────────
|
|
|
|
async function main() {
|
|
// Discover all image files
|
|
let files;
|
|
try {
|
|
files = fs.readdirSync(TEST_IMAGES_DIR).filter(f =>
|
|
/\.(jpe?g|png|webp|bmp)$/i.test(f)
|
|
).sort();
|
|
} catch (e) {
|
|
console.error(`Cannot read test-images directory: ${TEST_IMAGES_DIR}`);
|
|
console.error(e.message);
|
|
process.exit(1);
|
|
}
|
|
|
|
if (files.length === 0) {
|
|
console.error('No image files found in', TEST_IMAGES_DIR);
|
|
process.exit(1);
|
|
}
|
|
|
|
console.log(`\n🚀 Starting batch test`);
|
|
console.log(` API endpoint : ${BASE_URL}`);
|
|
console.log(` Images found : ${files.length}`);
|
|
console.log(` Output JSON : ${OUTPUT_JSON}`);
|
|
console.log(` Output MD : ${OUTPUT_MD}`);
|
|
console.log('─'.repeat(60));
|
|
|
|
const jsonResults = [];
|
|
const mdParts = [];
|
|
const summaryRows = [];
|
|
|
|
// ── Markdown document header ──────────────────────────────────────────────
|
|
mdParts.push(
|
|
`# OCR Batch Test Report\n\n`,
|
|
`> Generated: ${new Date().toISOString()}\n`,
|
|
`> API: \`${BASE_URL}\`\n`,
|
|
`> Images: **${files.length}** files from \`backend/sources/test-images/\`\n\n`,
|
|
`---\n\n`,
|
|
`## Summary\n\n`,
|
|
'<!-- summary_table_placeholder -->\n\n',
|
|
`---\n\n`,
|
|
`## Stage-by-Stage Results\n\n`,
|
|
);
|
|
const summaryPlaceholderIndex = mdParts.indexOf('<!-- summary_table_placeholder -->\n\n');
|
|
|
|
// ── Process each file ────────────────────────────────────────────────────
|
|
for (let idx = 0; idx < files.length; idx++) {
|
|
const file = files[idx];
|
|
const num = `[${String(idx + 1).padStart(2, '0')}/${files.length}]`;
|
|
process.stdout.write(`${num} ${file} ... `);
|
|
|
|
const entry = {
|
|
index: idx + 1,
|
|
filename: file,
|
|
status: 'pending',
|
|
tilt: null,
|
|
unwarped: null,
|
|
// pipeline stages
|
|
rawMarkdown: null,
|
|
layer1RawRegex: null,
|
|
layer2Sanitized: null,
|
|
layer3Final: null,
|
|
items: [],
|
|
error: null,
|
|
};
|
|
|
|
try {
|
|
const t0 = Date.now();
|
|
const res = await postJSON(BASE_URL, { filename: file });
|
|
const elapsed = ((Date.now() - t0) / 1000).toFixed(1);
|
|
|
|
if (!res.ok) {
|
|
process.stdout.write(`❌ HTTP ${res.status} (${elapsed}s)\n`);
|
|
entry.status = 'http_error';
|
|
entry.error = `HTTP ${res.status}: ${res.body}`;
|
|
} else {
|
|
let data;
|
|
try {
|
|
data = JSON.parse(res.body);
|
|
} catch (_) {
|
|
entry.status = 'json_parse_error';
|
|
entry.error = 'Response is not valid JSON';
|
|
process.stdout.write(`❌ JSON parse error (${elapsed}s)\n`);
|
|
data = null;
|
|
}
|
|
|
|
if (data) {
|
|
if (data.error) {
|
|
process.stdout.write(`⚠️ API error: ${data.error} (${elapsed}s)\n`);
|
|
entry.status = 'api_error';
|
|
entry.error = data.error;
|
|
} else {
|
|
const pipelineInfo = (data.result || {}).pipeline_info || {};
|
|
entry.status = 'success';
|
|
entry.tilt = pipelineInfo.tilt !== undefined ? +parseFloat(pipelineInfo.tilt).toFixed(2) : null;
|
|
entry.unwarped = pipelineInfo.unwarped ?? null;
|
|
|
|
const ppd = data.postProcessingDetails || {};
|
|
entry.rawMarkdown = ppd.rawMarkdown ?? null;
|
|
entry.layer1RawRegex = ppd.layer1RawRegex ?? null;
|
|
entry.layer2Sanitized = ppd.layer2Sanitized ?? null;
|
|
entry.layer3Final = ppd.layer3Final ?? null;
|
|
entry.items = data.items ?? [];
|
|
|
|
const itemCount = entry.items.length;
|
|
process.stdout.write(`✅ ${itemCount} item(s), tilt=${entry.tilt ?? 'N/A'}° (${elapsed}s)\n`);
|
|
}
|
|
}
|
|
}
|
|
} catch (err) {
|
|
process.stdout.write(`💥 ${err.message}\n`);
|
|
entry.status = 'exception';
|
|
entry.error = err.message;
|
|
}
|
|
|
|
jsonResults.push(entry);
|
|
|
|
// ── Build per-file markdown section ─────────────────────────────────────
|
|
const statusEmoji = {
|
|
success: '✅',
|
|
http_error: '❌',
|
|
api_error: '⚠️',
|
|
json_parse_error: '❌',
|
|
exception: '💥',
|
|
}[entry.status] || '❓';
|
|
|
|
let fileMd = '';
|
|
fileMd += `### ${idx + 1}. \`${file}\`\n\n`;
|
|
fileMd += `**Status**: ${statusEmoji} \`${entry.status}\`\n\n`;
|
|
|
|
if (entry.status !== 'success') {
|
|
fileMd += `> **Error**: ${entry.error}\n\n`;
|
|
fileMd += `---\n\n`;
|
|
summaryRows.push([idx + 1, `\`${file}\``, `${statusEmoji} ${entry.status}`, 'N/A', 'N/A', 'N/A', 'N/A']);
|
|
mdParts.push(fileMd);
|
|
continue;
|
|
}
|
|
|
|
// ── Stage 0: Pipeline Info ────────────────────────────────────────────────
|
|
fileMd += `#### 📐 Stage 0 — Pipeline Info\n\n`;
|
|
fileMd += mdField('Tilt detected', entry.tilt !== null ? `${entry.tilt}°` : 'N/A');
|
|
fileMd += mdField('Auto-unwarped', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A');
|
|
fileMd += '\n';
|
|
|
|
// ── Stage 1: Raw Markdown from OCR ───────────────────────────────────────
|
|
fileMd += `#### 📄 Stage 1 — Raw OCR Markdown\n\n`;
|
|
fileMd += `*This is the raw text extracted by PaddleOCR layout parser before any post-processing.*\n\n`;
|
|
if (entry.rawMarkdown) {
|
|
fileMd += mdCode(entry.rawMarkdown, 'markdown');
|
|
} else {
|
|
fileMd += '*No raw markdown captured.*\n\n';
|
|
}
|
|
|
|
// ── Stage 2: Layer 1 — Regex Extraction ──────────────────────────────────
|
|
fileMd += `#### 🔍 Stage 2 — Layer 1: Regex Extraction (\`parseDOMetadata\`)\n\n`;
|
|
fileMd += `*Regex patterns are applied to raw markdown to extract header fields and item rows.*\n\n`;
|
|
if (entry.layer1RawRegex) {
|
|
const l1 = entry.layer1RawRegex;
|
|
fileMd += `**Header fields (raw regex output):**\n\n`;
|
|
fileMd += mdField('noDO', l1.noDO);
|
|
fileMd += mdField('noPO', l1.noPO);
|
|
fileMd += mdField('noSO', l1.noSO);
|
|
fileMd += mdField('tanggal', l1.tanggal);
|
|
fileMd += mdField('vendorInfo', l1.vendorInfo);
|
|
fileMd += mdField('customerInfo', l1.customerInfo);
|
|
fileMd += mdField('alamat', l1.alamat);
|
|
fileMd += mdField('orderUntuk', l1.orderUntuk);
|
|
fileMd += mdField('platTruk', l1.platTruk);
|
|
fileMd += '\n';
|
|
|
|
fileMd += `**Raw items (${(l1.items || []).length} row(s)):**\n\n`;
|
|
fileMd += mdTable(
|
|
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
|
(l1.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
|
|
);
|
|
} else {
|
|
fileMd += '*Layer 1 data not captured.*\n\n';
|
|
}
|
|
|
|
// ── Stage 3: Layer 2 — Sanitized ─────────────────────────────────────────
|
|
fileMd += `#### 🧹 Stage 3 — Layer 2: Sanitized (\`sanitizeParsedMetadata\`)\n\n`;
|
|
fileMd += `*Strict format enforcement: corrects date formats, trims whitespace, enforces field constraints.*\n\n`;
|
|
if (entry.layer2Sanitized) {
|
|
const l2 = entry.layer2Sanitized;
|
|
fileMd += `**Header fields (after sanitization):**\n\n`;
|
|
fileMd += mdField('noDO', l2.noDO);
|
|
fileMd += mdField('noPO', l2.noPO);
|
|
fileMd += mdField('noSO', l2.noSO);
|
|
fileMd += mdField('tanggal', l2.tanggal);
|
|
fileMd += mdField('vendorInfo', l2.vendorInfo);
|
|
fileMd += mdField('customerInfo', l2.customerInfo);
|
|
fileMd += mdField('alamat', l2.alamat);
|
|
fileMd += mdField('orderUntuk', l2.orderUntuk);
|
|
fileMd += mdField('platTruk', l2.platTruk);
|
|
fileMd += '\n';
|
|
|
|
fileMd += `**Sanitized items (${(l2.items || []).length} row(s)):**\n\n`;
|
|
fileMd += mdTable(
|
|
['kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
|
(l2.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah])
|
|
);
|
|
} else {
|
|
fileMd += '*Layer 2 data not captured.*\n\n';
|
|
}
|
|
|
|
// ── Stage 4: Layer 3 — Final (SKU triple-check + store resolution) ────────
|
|
fileMd += `#### ✅ Stage 4 — Layer 3: Final (\`SKU triple-check + store resolution\`)\n\n`;
|
|
fileMd += `*SKU validated against master list (score ≥ 0.6 threshold). Items with noise SKU codes are filtered out. Store resolved from DB.*\n\n`;
|
|
if (entry.layer3Final) {
|
|
const l3 = entry.layer3Final;
|
|
fileMd += `**Final metadata:**\n\n`;
|
|
fileMd += mdField('noDO', l3.noDO);
|
|
fileMd += mdField('noPO', l3.noPO);
|
|
fileMd += mdField('noSO', l3.noSO);
|
|
fileMd += mdField('tanggal', l3.tanggal);
|
|
fileMd += mdField('vendorInfo', l3.vendorInfo);
|
|
fileMd += mdField('customerInfo', l3.customerInfo);
|
|
fileMd += mdField('alamat', l3.alamat);
|
|
fileMd += mdField('orderUntuk', l3.orderUntuk);
|
|
fileMd += mdField('platTruk', l3.platTruk);
|
|
fileMd += '\n';
|
|
|
|
fileMd += `**Final items after SKU validation (${(l3.items || []).length} row(s)):**\n\n`;
|
|
fileMd += mdTable(
|
|
['kodeBarangOriginal', 'kodeBarang (corrected)', 'namaBarang', 'banyak', 'jumlah'],
|
|
(l3.items || []).map(it => [
|
|
it.kodeBarangOriginal ?? it.kodeBarang,
|
|
it.kodeBarang,
|
|
it.namaBarang,
|
|
it.banyak,
|
|
it.jumlah
|
|
])
|
|
);
|
|
} else {
|
|
fileMd += '*Layer 3 data not captured.*\n\n';
|
|
}
|
|
|
|
// ── Stage 5: Final submitted items (from root items[]) ───────────────────
|
|
fileMd += `#### 🗃️ Stage 5 — Submitted Items (ready-to-use JSON)\n\n`;
|
|
fileMd += `*These are the items actually returned to the caller and saved to the database.*\n\n`;
|
|
fileMd += mdTable(
|
|
['kodeBarangOriginal', 'kodeBarang', 'namaBarang', 'banyak', 'jumlah'],
|
|
(entry.items || []).map(it => [
|
|
it.kodeBarangOriginal ?? it.kodeBarang,
|
|
it.kodeBarang,
|
|
it.namaBarang,
|
|
it.banyak,
|
|
it.jumlah
|
|
])
|
|
);
|
|
|
|
fileMd += `---\n\n`;
|
|
|
|
// ── Summary row ──────────────────────────────────────────────────────────
|
|
const l3meta = entry.layer3Final || {};
|
|
summaryRows.push([
|
|
idx + 1,
|
|
`\`${file}\``,
|
|
`${statusEmoji} success`,
|
|
entry.tilt !== null ? `${entry.tilt}°` : 'N/A',
|
|
entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A',
|
|
`\`${l3meta.noDO ?? 'N/A'}\``,
|
|
`\`${l3meta.noPO ?? 'N/A'}\``,
|
|
`${entry.items.length}`,
|
|
]);
|
|
|
|
mdParts.push(fileMd);
|
|
}
|
|
|
|
// ── Inject summary table ──────────────────────────────────────────────────
|
|
const summaryTable = mdTable(
|
|
['#', 'Filename', 'Status', 'Tilt', 'Unwarped', 'DO', 'PO', 'Items'],
|
|
summaryRows
|
|
);
|
|
mdParts[summaryPlaceholderIndex] = summaryTable;
|
|
|
|
// ── Write outputs ─────────────────────────────────────────────────────────
|
|
const jsonOut = JSON.stringify(jsonResults, null, 2);
|
|
fs.writeFileSync(OUTPUT_JSON, jsonOut, 'utf8');
|
|
console.log(`\n✅ JSON saved → ${OUTPUT_JSON}`);
|
|
|
|
const mdOut = mdParts.join('');
|
|
fs.writeFileSync(OUTPUT_MD, mdOut, 'utf8');
|
|
console.log(`✅ MD saved → ${OUTPUT_MD}`);
|
|
|
|
// ── Final stats ───────────────────────────────────────────────────────────
|
|
const succeeded = jsonResults.filter(r => r.status === 'success').length;
|
|
const failed = jsonResults.length - succeeded;
|
|
console.log('\n─'.repeat(60));
|
|
console.log(` Total : ${jsonResults.length}`);
|
|
console.log(` Success: ${succeeded}`);
|
|
console.log(` Failed : ${failed}`);
|
|
console.log('─'.repeat(60));
|
|
}
|
|
|
|
main().catch(err => {
|
|
console.error('Fatal error:', err);
|
|
process.exit(1);
|
|
});
|