/** * run_full_test.js * * Runs OCR parsing against ALL images in backend/sources/test-images/ * and captures every pipeline stage for analysis: * - rawMarkdown : raw text from PaddleOCR layout parser * - layer1RawRegex: output of parseDOMetadata (regex extraction) * - layer2Sanitized: output of sanitizeParsedMetadata (format checks) * - layer3Final : final metadata after SKU triple-check + store resolution * * Outputs: * backend/sources/ai_results.json — machine-readable per-file results * backend/sources/ai_results.md — human-readable stage-by-stage breakdown * * Usage (from host machine, Docker must be running): * node run_full_test.js * * The script talks to the nginx gateway on port 8000. * To override: set env var BASE_URL=http://localhost:3000/api/parse */ const fs = require('fs'); const path = require('path'); const http = require('http'); const https = require('https'); // ─── Config ────────────────────────────────────────────────────────────────── const BASE_URL = process.env.BASE_URL || 'http://localhost:8000/api/parse'; const TEST_IMAGES_DIR = path.resolve(__dirname, '../sources/test-images'); const OUTPUT_JSON = path.resolve(__dirname, '../sources/ai_results.json'); const OUTPUT_MD = path.resolve(__dirname, '../sources/ai_results.md'); const REQUEST_TIMEOUT_MS = 20 * 60 * 1000; // 20 minutes per image // ─── HTTP Helper ───────────────────────────────────────────────────────────── function postJSON(url, body) { return new Promise((resolve, reject) => { const parsedUrl = new URL(url); const bodyStr = JSON.stringify(body); const lib = parsedUrl.protocol === 'https:' ? https : http; const options = { hostname: parsedUrl.hostname, port: parsedUrl.port || (parsedUrl.protocol === 'https:' ? 443 : 80), path: parsedUrl.pathname + parsedUrl.search, method: 'POST', headers: { 'Content-Type': 'application/json', 'Content-Length': Buffer.byteLength(bodyStr), }, timeout: REQUEST_TIMEOUT_MS, }; const req = lib.request(options, (res) => { let data = ''; res.on('data', (chunk) => { data += chunk; }); res.on('end', () => { resolve({ ok: res.statusCode >= 200 && res.statusCode < 300, status: res.statusCode, body: data, }); }); }); req.on('timeout', () => { req.destroy(new Error(`Request timed out after ${REQUEST_TIMEOUT_MS / 60000}m`)); }); req.on('error', reject); req.write(bodyStr); req.end(); }); } // ─── Markdown Helpers ───────────────────────────────────────────────────────── function mdSection(title, level = 2) { return `${'#'.repeat(level)} ${title}\n\n`; } function mdCode(content, lang = '') { if (content === null || content === undefined) return '*null*\n\n'; const str = typeof content === 'string' ? content : JSON.stringify(content, null, 2); return `\`\`\`${lang}\n${str}\n\`\`\`\n\n`; } function mdField(label, value) { const display = (value === null || value === undefined || value === '') ? '*empty*' : `\`${value}\``; return `- **${label}**: ${display}\n`; } function mdTable(headers, rows) { if (!rows || rows.length === 0) return '*No items.*\n\n'; const sep = headers.map(() => '---'); const lines = [ `| ${headers.join(' | ')} |`, `| ${sep.join(' | ')} |`, ...rows.map(r => `| ${r.map(c => String(c ?? '').replace(/\|/g, '\\|')).join(' | ')} |`), ]; return lines.join('\n') + '\n\n'; } // ─── Main ───────────────────────────────────────────────────────────────────── async function main() { // Discover all image files let files; try { files = fs.readdirSync(TEST_IMAGES_DIR).filter(f => /\.(jpe?g|png|webp|bmp)$/i.test(f) ).sort(); } catch (e) { console.error(`Cannot read test-images directory: ${TEST_IMAGES_DIR}`); console.error(e.message); process.exit(1); } if (files.length === 0) { console.error('No image files found in', TEST_IMAGES_DIR); process.exit(1); } console.log(`\n🚀 Starting batch test`); console.log(` API endpoint : ${BASE_URL}`); console.log(` Images found : ${files.length}`); console.log(` Output JSON : ${OUTPUT_JSON}`); console.log(` Output MD : ${OUTPUT_MD}`); console.log('─'.repeat(60)); const jsonResults = []; const mdParts = []; const summaryRows = []; // ── Markdown document header ────────────────────────────────────────────── mdParts.push( `# OCR Batch Test Report\n\n`, `> Generated: ${new Date().toISOString()}\n`, `> API: \`${BASE_URL}\`\n`, `> Images: **${files.length}** files from \`backend/sources/test-images/\`\n\n`, `---\n\n`, `## Summary\n\n`, '\n\n', `---\n\n`, `## Stage-by-Stage Results\n\n`, ); const summaryPlaceholderIndex = mdParts.indexOf('\n\n'); // ── Process each file ──────────────────────────────────────────────────── for (let idx = 0; idx < files.length; idx++) { const file = files[idx]; const num = `[${String(idx + 1).padStart(2, '0')}/${files.length}]`; process.stdout.write(`${num} ${file} ... `); const entry = { index: idx + 1, filename: file, status: 'pending', tilt: null, unwarped: null, // pipeline stages rawMarkdown: null, layer1RawRegex: null, layer2Sanitized: null, layer3Final: null, items: [], error: null, }; try { const t0 = Date.now(); const res = await postJSON(BASE_URL, { filename: file }); const elapsed = ((Date.now() - t0) / 1000).toFixed(1); if (!res.ok) { process.stdout.write(`❌ HTTP ${res.status} (${elapsed}s)\n`); entry.status = 'http_error'; entry.error = `HTTP ${res.status}: ${res.body}`; } else { let data; try { data = JSON.parse(res.body); } catch (_) { entry.status = 'json_parse_error'; entry.error = 'Response is not valid JSON'; process.stdout.write(`❌ JSON parse error (${elapsed}s)\n`); data = null; } if (data) { if (data.error) { process.stdout.write(`⚠️ API error: ${data.error} (${elapsed}s)\n`); entry.status = 'api_error'; entry.error = data.error; } else { const pipelineInfo = (data.result || {}).pipeline_info || {}; entry.status = 'success'; entry.tilt = pipelineInfo.tilt !== undefined ? +parseFloat(pipelineInfo.tilt).toFixed(2) : null; entry.unwarped = pipelineInfo.unwarped ?? null; const ppd = data.postProcessingDetails || {}; entry.rawMarkdown = ppd.rawMarkdown ?? null; entry.layer1RawRegex = ppd.layer1RawRegex ?? null; entry.layer2Sanitized = ppd.layer2Sanitized ?? null; entry.layer3Final = ppd.layer3Final ?? null; entry.items = data.items ?? []; const itemCount = entry.items.length; process.stdout.write(`✅ ${itemCount} item(s), tilt=${entry.tilt ?? 'N/A'}° (${elapsed}s)\n`); } } } } catch (err) { process.stdout.write(`💥 ${err.message}\n`); entry.status = 'exception'; entry.error = err.message; } jsonResults.push(entry); // ── Build per-file markdown section ───────────────────────────────────── const statusEmoji = { success: '✅', http_error: '❌', api_error: '⚠️', json_parse_error: '❌', exception: '💥', }[entry.status] || '❓'; let fileMd = ''; fileMd += `### ${idx + 1}. \`${file}\`\n\n`; fileMd += `**Status**: ${statusEmoji} \`${entry.status}\`\n\n`; if (entry.status !== 'success') { fileMd += `> **Error**: ${entry.error}\n\n`; fileMd += `---\n\n`; summaryRows.push([idx + 1, `\`${file}\``, `${statusEmoji} ${entry.status}`, 'N/A', 'N/A', 'N/A', 'N/A']); mdParts.push(fileMd); continue; } // ── Stage 0: Pipeline Info ──────────────────────────────────────────────── fileMd += `#### 📐 Stage 0 — Pipeline Info\n\n`; fileMd += mdField('Tilt detected', entry.tilt !== null ? `${entry.tilt}°` : 'N/A'); fileMd += mdField('Auto-unwarped', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A'); fileMd += '\n'; // ── Stage 1: Raw Markdown from OCR ─────────────────────────────────────── fileMd += `#### 📄 Stage 1 — Raw OCR Markdown\n\n`; fileMd += `*This is the raw text extracted by PaddleOCR layout parser before any post-processing.*\n\n`; if (entry.rawMarkdown) { fileMd += mdCode(entry.rawMarkdown, 'markdown'); } else { fileMd += '*No raw markdown captured.*\n\n'; } // ── Stage 2: Layer 1 — Regex Extraction ────────────────────────────────── fileMd += `#### 🔍 Stage 2 — Layer 1: Regex Extraction (\`parseDOMetadata\`)\n\n`; fileMd += `*Regex patterns are applied to raw markdown to extract header fields and item rows.*\n\n`; if (entry.layer1RawRegex) { const l1 = entry.layer1RawRegex; fileMd += `**Header fields (raw regex output):**\n\n`; fileMd += mdField('noDO', l1.noDO); fileMd += mdField('noPO', l1.noPO); fileMd += mdField('noSO', l1.noSO); fileMd += mdField('tanggal', l1.tanggal); fileMd += mdField('vendorInfo', l1.vendorInfo); fileMd += mdField('customerInfo', l1.customerInfo); fileMd += mdField('alamat', l1.alamat); fileMd += mdField('orderUntuk', l1.orderUntuk); fileMd += mdField('platTruk', l1.platTruk); fileMd += '\n'; fileMd += `**Raw items (${(l1.items || []).length} row(s)):**\n\n`; fileMd += mdTable( ['kodeBarang', 'namaBarang', 'banyak', 'jumlah'], (l1.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah]) ); } else { fileMd += '*Layer 1 data not captured.*\n\n'; } // ── Stage 3: Layer 2 — Sanitized ───────────────────────────────────────── fileMd += `#### 🧹 Stage 3 — Layer 2: Sanitized (\`sanitizeParsedMetadata\`)\n\n`; fileMd += `*Strict format enforcement: corrects date formats, trims whitespace, enforces field constraints.*\n\n`; if (entry.layer2Sanitized) { const l2 = entry.layer2Sanitized; fileMd += `**Header fields (after sanitization):**\n\n`; fileMd += mdField('noDO', l2.noDO); fileMd += mdField('noPO', l2.noPO); fileMd += mdField('noSO', l2.noSO); fileMd += mdField('tanggal', l2.tanggal); fileMd += mdField('vendorInfo', l2.vendorInfo); fileMd += mdField('customerInfo', l2.customerInfo); fileMd += mdField('alamat', l2.alamat); fileMd += mdField('orderUntuk', l2.orderUntuk); fileMd += mdField('platTruk', l2.platTruk); fileMd += '\n'; fileMd += `**Sanitized items (${(l2.items || []).length} row(s)):**\n\n`; fileMd += mdTable( ['kodeBarang', 'namaBarang', 'banyak', 'jumlah'], (l2.items || []).map(it => [it.kodeBarang, it.namaBarang, it.banyak, it.jumlah]) ); } else { fileMd += '*Layer 2 data not captured.*\n\n'; } // ── Stage 4: Layer 3 — Final (SKU triple-check + store resolution) ──────── fileMd += `#### ✅ Stage 4 — Layer 3: Final (\`SKU triple-check + store resolution\`)\n\n`; fileMd += `*SKU validated against master list (score ≥ 0.6 threshold). Items with noise SKU codes are filtered out. Store resolved from DB.*\n\n`; if (entry.layer3Final) { const l3 = entry.layer3Final; fileMd += `**Final metadata:**\n\n`; fileMd += mdField('noDO', l3.noDO); fileMd += mdField('noPO', l3.noPO); fileMd += mdField('noSO', l3.noSO); fileMd += mdField('tanggal', l3.tanggal); fileMd += mdField('vendorInfo', l3.vendorInfo); fileMd += mdField('customerInfo', l3.customerInfo); fileMd += mdField('alamat', l3.alamat); fileMd += mdField('orderUntuk', l3.orderUntuk); fileMd += mdField('platTruk', l3.platTruk); fileMd += '\n'; fileMd += `**Final items after SKU validation (${(l3.items || []).length} row(s)):**\n\n`; fileMd += mdTable( ['kodeBarangOriginal', 'kodeBarang (corrected)', 'namaBarang', 'banyak', 'jumlah'], (l3.items || []).map(it => [ it.kodeBarangOriginal ?? it.kodeBarang, it.kodeBarang, it.namaBarang, it.banyak, it.jumlah ]) ); } else { fileMd += '*Layer 3 data not captured.*\n\n'; } // ── Stage 5: Final submitted items (from root items[]) ─────────────────── fileMd += `#### 🗃️ Stage 5 — Submitted Items (ready-to-use JSON)\n\n`; fileMd += `*These are the items actually returned to the caller and saved to the database.*\n\n`; fileMd += mdTable( ['kodeBarangOriginal', 'kodeBarang', 'namaBarang', 'banyak', 'jumlah'], (entry.items || []).map(it => [ it.kodeBarangOriginal ?? it.kodeBarang, it.kodeBarang, it.namaBarang, it.banyak, it.jumlah ]) ); fileMd += `---\n\n`; // ── Summary row ────────────────────────────────────────────────────────── const l3meta = entry.layer3Final || {}; summaryRows.push([ idx + 1, `\`${file}\``, `${statusEmoji} success`, entry.tilt !== null ? `${entry.tilt}°` : 'N/A', entry.unwarped !== null ? (entry.unwarped ? 'Yes' : 'No') : 'N/A', `\`${l3meta.noDO ?? 'N/A'}\``, `\`${l3meta.noPO ?? 'N/A'}\``, `${entry.items.length}`, ]); mdParts.push(fileMd); } // ── Inject summary table ────────────────────────────────────────────────── const summaryTable = mdTable( ['#', 'Filename', 'Status', 'Tilt', 'Unwarped', 'DO', 'PO', 'Items'], summaryRows ); mdParts[summaryPlaceholderIndex] = summaryTable; // ── Write outputs ───────────────────────────────────────────────────────── const jsonOut = JSON.stringify(jsonResults, null, 2); fs.writeFileSync(OUTPUT_JSON, jsonOut, 'utf8'); console.log(`\n✅ JSON saved → ${OUTPUT_JSON}`); const mdOut = mdParts.join(''); fs.writeFileSync(OUTPUT_MD, mdOut, 'utf8'); console.log(`✅ MD saved → ${OUTPUT_MD}`); // ── Final stats ─────────────────────────────────────────────────────────── const succeeded = jsonResults.filter(r => r.status === 'success').length; const failed = jsonResults.length - succeeded; console.log('\n─'.repeat(60)); console.log(` Total : ${jsonResults.length}`); console.log(` Success: ${succeeded}`); console.log(` Failed : ${failed}`); console.log('─'.repeat(60)); } main().catch(err => { console.error('Fatal error:', err); process.exit(1); });