Files
pfm-ocr/backend/pfm-web-app/run_batch_test.js
T
Rafhan Mazaya FathurrahmanandClaude Sonnet 5 4808a798fb Add mobile reliability fixes, Bahasa Indonesia UI, and continue OCR accuracy tuning
Reliability/PoC hardening: dedupe uploads by file_hash, surface editor sync
failures instead of a false success SnackBar with a retry-without-re-OCR path,
bound the OCR pipeline fetches with timeouts, share a single ApiClient/Dio
instance app-wide, tune capture JPEG quality, and add an opt-in
docker-compose.demo.yml for a production-mode run ahead of client demos.

Translate all Flutter-side user-facing text (screens, validators, SnackBars,
the printed delivery receipt, and shared API error messages) to Bahasa
Indonesia.

Also includes in-progress OCR parser/accuracy-tuning work from the same
session: table column/unit normalization fixes, store/customer master data,
accuracy history log, and test-image renaming/cleanup.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eRAsLqN9Sg1b9YPz22Lxz
2026-07-04 19:41:24 +07:00

245 lines
8.7 KiB
JavaScript

const fs = require('fs');
const path = require('path');
const http = require('http');
const { Client } = require('pg');
const BASE_URL = 'http://localhost:3000/api/parse';
const testFiles = [
"do-001.jpg",
"do-002.jpg",
"do-003.jpg",
"do-004.jpg",
"do-005.jpg",
"do-006.jpg",
"do-007.jpg",
"do-008.jpg",
"do-009.jpg",
"do-010.jpg",
"do-011.jpg",
"do-012.jpg",
"do-013.jpg",
"do-014.jpg"
];
function postJSON(url, body) {
return new Promise((resolve, reject) => {
const parsedUrl = new URL(url);
const bodyStr = JSON.stringify(body);
const options = {
hostname: parsedUrl.hostname,
port: parsedUrl.port,
path: parsedUrl.pathname + parsedUrl.search,
method: 'POST',
headers: {
'Content-Type': 'application/json',
'Content-Length': Buffer.byteLength(bodyStr)
},
timeout: 1200000 // 20 minutes
};
const req = http.request(options, (res) => {
let data = '';
res.on('data', (chunk) => { data += chunk; });
res.on('end', () => {
resolve({
ok: res.statusCode >= 200 && res.statusCode < 300,
status: res.statusCode,
json: async () => JSON.parse(data),
text: async () => data
});
});
});
req.on('timeout', () => {
req.destroy(new Error('Request Timeout (20m)'));
});
req.on('error', (err) => { reject(err); });
req.write(bodyStr);
req.end();
});
}
async function getDocumentMetadataFromDb(filename) {
const client = new Client({
host: 'paddleocr-db',
port: 5432,
user: 'postgres',
password: 'postgres',
database: 'dopfm'
});
try {
await client.connect();
const res = await client.query('SELECT metadata FROM documents WHERE filename = $1', [filename]);
return res.rows[0]?.metadata || {};
} catch (err) {
console.error('Database query failed:', err.message);
return {};
} finally {
await client.end();
}
}
async function main() {
console.log(`Starting single image test for ${testFiles.length} file...`);
const summaryTmpFile = '/uploads/test_images_report_summary.tmp';
const detailsTmpFile = '/uploads/test_images_report_details.tmp';
const jsonlFile = '/uploads/test_images_results.jsonl';
const finalReportFile = '/uploads/test_images_report.md';
// Initialize summary header
let summaryHeader = `# Batch OCR Parsing Test Report\n\n`;
summaryHeader += `Processed **${testFiles.length}** file from \`backend/sources/test-images\`.\n\n`;
summaryHeader += `## Summary Table\n\n`;
summaryHeader += `| No | Filename | Status | Tilt | Auto-Rotated | PO | SO | DO | Date | Store Match | Items Count |\n`;
summaryHeader += `|---|---|---|---|---|---|---|---|---|---|---|\n`;
fs.writeFileSync(summaryTmpFile, summaryHeader);
// Initialize details header
let detailsHeader = `\n\n## Detailed Results per Image\n\n`;
fs.writeFileSync(detailsTmpFile, detailsHeader);
// Clean jsonl
fs.writeFileSync(jsonlFile, '');
for (let idx = 0; idx < testFiles.length; idx++) {
const file = testFiles[idx];
console.log(`[${idx + 1}/${testFiles.length}] Processing file: ${file}`);
try {
const response = await postJSON(BASE_URL, { filename: file });
if (!response.ok) {
const errorText = await response.text();
console.error(`Error parsing file ${file}: ${errorText}`);
// Write fail state incrementally
const tableLine = `| ${idx + 1} | \`${file}\` | **Failed** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
fs.appendFileSync(summaryTmpFile, tableLine);
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
detailedText += `- **Status**: Failed\n`;
detailedText += `- **Error Detail**: \`${errorText || 'Unknown error'}\`\n`;
detailedText += `\n---\n\n`;
fs.appendFileSync(detailsTmpFile, detailedText);
fs.appendFileSync(jsonlFile, JSON.stringify({
filename: file,
status: 'Failed',
error: errorText || 'Unknown error'
}) + '\n');
continue;
}
const resData = await response.json();
const pipelineRes = resData.result || {};
const page0 = pipelineRes.layoutParsingResults?.[0] || {};
const rawMarkdown = page0.markdown?.text || "N/A";
const info = pipelineRes.pipeline_info || {};
// Direct DB query for accurate metadata (bypassing Auth)
const docMeta = await getDocumentMetadataFromDb(file);
const tiltStr = info.tilt !== undefined ? parseFloat(info.tilt).toFixed(2) : 'N/A';
const unwarpedStr = info.unwarped ? 'Yes' : 'No';
const itemsCount = (resData.items || []).length;
// Write success state incrementally
const tableLine = `| ${idx + 1} | \`${file}\` | **Success** | ${tiltStr}° | ${unwarpedStr} | \`${docMeta.noPO || 'N/A'}\` | \`${docMeta.noSO || 'N/A'}\` | \`${docMeta.noDO || 'N/A'}\` | \`${docMeta.tanggal || 'N/A'}\` | ${docMeta.orderUntuk || 'N/A'} | ${itemsCount} |\n`;
fs.appendFileSync(summaryTmpFile, tableLine);
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
detailedText += `- **Status**: Success\n`;
detailedText += `- **Tilt Detected**: ${tiltStr}°\n`;
detailedText += `- **Auto-Rotated/Unwarped**: ${unwarpedStr}\n`;
detailedText += `- **Extracted Metadata**:\n`;
detailedText += ` * **PO**: \`${docMeta.noPO || 'N/A'}\`\n`;
detailedText += ` * **SO**: \`${docMeta.noSO || 'N/A'}\`\n`;
detailedText += ` * **DO**: \`${docMeta.noDO || 'N/A'}\`\n`;
detailedText += ` * **Tanggal**: \`${docMeta.tanggal || 'N/A'}\`\n`;
detailedText += ` * **Customer**: \`${docMeta.customerInfo || 'N/A'}\`\n`;
detailedText += ` * **Store**: \`${docMeta.orderUntuk || 'N/A'}\`\n`;
detailedText += ` * **Alamat**: \`${docMeta.alamat || 'N/A'}\`\n`;
detailedText += ` * **Plat Nomor**: \`${docMeta.platTruk || 'N/A'}\`\n`;
detailedText += `- **Raw Layout Markdown**:\n`;
detailedText += `\`\`\`markdown\n${rawMarkdown}\n\`\`\`\n`;
detailedText += `- **Parsed Items (${itemsCount})**:\n`;
if (itemsCount > 0) {
detailedText += ` | Code (SKU) | Name | Qty | Price |\n`;
detailedText += ` |---|---|---|---|\n`;
(resData.items || []).forEach(item => {
detailedText += ` | \`${item.kodeBarang}\` | ${item.namaBarang} | \`${item.banyak}\` | \`${item.jumlah}\` |\n`;
});
} else {
detailedText += ` *No valid SKU items parsed.*\n`;
}
detailedText += `\n---\n\n`;
fs.appendFileSync(detailsTmpFile, detailedText);
fs.appendFileSync(jsonlFile, JSON.stringify({
filename: file,
status: 'Success',
tilt: tiltStr,
unwarped: unwarpedStr,
rawMarkdown: resData.postProcessingDetails?.rawMarkdown || "",
layer1RawRegex: resData.postProcessingDetails?.layer1RawRegex || {},
layer2Sanitized: resData.postProcessingDetails?.layer2Sanitized || {},
layer3Final: resData.postProcessingDetails?.layer3Final || {},
metadata: docMeta,
items: resData.items || []
}) + '\n');
} catch (err) {
console.error(`Exception during file ${file}:`, err);
const tableLine = `| ${idx + 1} | \`${file}\` | **Error** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
fs.appendFileSync(summaryTmpFile, tableLine);
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
detailedText += `- **Status**: Error\n`;
detailedText += `- **Error Detail**: \`${err.message}\`\n`;
detailedText += `\n---\n\n`;
fs.appendFileSync(detailsTmpFile, detailedText);
fs.appendFileSync(jsonlFile, JSON.stringify({
filename: file,
status: 'Error',
error: err.message
}) + '\n');
}
}
// Combine temporary files into the final report
try {
const summaryContent = fs.readFileSync(summaryTmpFile, 'utf8');
const detailsContent = fs.readFileSync(detailsTmpFile, 'utf8');
fs.writeFileSync(finalReportFile, summaryContent + '\n' + detailsContent);
// Clean up temporary files
fs.unlinkSync(summaryTmpFile);
fs.unlinkSync(detailsTmpFile);
} catch (combineErr) {
console.error('Failed to combine test reports:', combineErr);
}
// Compile JSONL into the final JSON v2
try {
const lines = fs.readFileSync(jsonlFile, 'utf8').split('\n').filter(Boolean);
const results = lines.map(line => JSON.parse(line));
fs.writeFileSync('/uploads/ai_results_v2.json', JSON.stringify(results, null, 2));
console.log('Compiled results saved to /uploads/ai_results_v2.json');
} catch (compileErr) {
console.error('Failed to compile results into JSON v2:', compileErr);
}
console.log('Batch test completed. Report written to /uploads/test_images_report.md');
}
main();