245 lines
9.1 KiB
JavaScript
245 lines
9.1 KiB
JavaScript
const fs = require('fs');
|
|
const path = require('path');
|
|
const http = require('http');
|
|
const { Client } = require('pg');
|
|
|
|
const BASE_URL = 'http://localhost:3000/api/parse';
|
|
|
|
const testFiles = [
|
|
"1782870899198-sample_do.jpeg",
|
|
"1782884664859-rotated_1782884661516_rotated_1782884658097_rotated_1782884654697_CAP5956452738616729026.jpg",
|
|
"1782888127716-CAP6161747431193129837.jpg",
|
|
"1782888211457-rotated_1782888204591_rotated_1782888199962_rotated_1782888195402_CAP7202641176939787142.jpg",
|
|
"1782888609885-rotated_1782888593813_1000000454.jpg",
|
|
"1782890303600-1000000465.jpg",
|
|
"1782892728794-1000000466.jpg",
|
|
"IMG_20260630_145445.jpg",
|
|
"IMG_20260701_134446.jpg",
|
|
"IMG_20260701_134504.jpg",
|
|
"IMG_20260701_134555.jpg",
|
|
"IMG_20260701_134555~2.jpg",
|
|
"IMG_20260701_134646~2.jpg",
|
|
"IMG_20260701_134810.jpg"
|
|
];
|
|
|
|
function postJSON(url, body) {
|
|
return new Promise((resolve, reject) => {
|
|
const parsedUrl = new URL(url);
|
|
const bodyStr = JSON.stringify(body);
|
|
|
|
const options = {
|
|
hostname: parsedUrl.hostname,
|
|
port: parsedUrl.port,
|
|
path: parsedUrl.pathname + parsedUrl.search,
|
|
method: 'POST',
|
|
headers: {
|
|
'Content-Type': 'application/json',
|
|
'Content-Length': Buffer.byteLength(bodyStr)
|
|
},
|
|
timeout: 1200000 // 20 minutes
|
|
};
|
|
|
|
const req = http.request(options, (res) => {
|
|
let data = '';
|
|
res.on('data', (chunk) => { data += chunk; });
|
|
res.on('end', () => {
|
|
resolve({
|
|
ok: res.statusCode >= 200 && res.statusCode < 300,
|
|
status: res.statusCode,
|
|
json: async () => JSON.parse(data),
|
|
text: async () => data
|
|
});
|
|
});
|
|
});
|
|
|
|
req.on('timeout', () => {
|
|
req.destroy(new Error('Request Timeout (20m)'));
|
|
});
|
|
|
|
req.on('error', (err) => { reject(err); });
|
|
req.write(bodyStr);
|
|
req.end();
|
|
});
|
|
}
|
|
|
|
async function getDocumentMetadataFromDb(filename) {
|
|
const client = new Client({
|
|
host: 'paddleocr-db',
|
|
port: 5432,
|
|
user: 'postgres',
|
|
password: 'postgres',
|
|
database: 'dopfm'
|
|
});
|
|
|
|
try {
|
|
await client.connect();
|
|
const res = await client.query('SELECT metadata FROM documents WHERE filename = $1', [filename]);
|
|
return res.rows[0]?.metadata || {};
|
|
} catch (err) {
|
|
console.error('Database query failed:', err.message);
|
|
return {};
|
|
} finally {
|
|
await client.end();
|
|
}
|
|
}
|
|
|
|
async function main() {
|
|
console.log(`Starting single image test for ${testFiles.length} file...`);
|
|
|
|
const summaryTmpFile = '/uploads/test_images_report_summary.tmp';
|
|
const detailsTmpFile = '/uploads/test_images_report_details.tmp';
|
|
const jsonlFile = '/uploads/test_images_results.jsonl';
|
|
const finalReportFile = '/uploads/test_images_report.md';
|
|
|
|
// Initialize summary header
|
|
let summaryHeader = `# Batch OCR Parsing Test Report\n\n`;
|
|
summaryHeader += `Processed **${testFiles.length}** file from \`backend/sources/test-images\`.\n\n`;
|
|
summaryHeader += `## Summary Table\n\n`;
|
|
summaryHeader += `| No | Filename | Status | Tilt | Auto-Rotated | PO | SO | DO | Date | Store Match | Items Count |\n`;
|
|
summaryHeader += `|---|---|---|---|---|---|---|---|---|---|---|\n`;
|
|
fs.writeFileSync(summaryTmpFile, summaryHeader);
|
|
|
|
// Initialize details header
|
|
let detailsHeader = `\n\n## Detailed Results per Image\n\n`;
|
|
fs.writeFileSync(detailsTmpFile, detailsHeader);
|
|
|
|
// Clean jsonl
|
|
fs.writeFileSync(jsonlFile, '');
|
|
|
|
for (let idx = 0; idx < testFiles.length; idx++) {
|
|
const file = testFiles[idx];
|
|
console.log(`[${idx + 1}/${testFiles.length}] Processing file: ${file}`);
|
|
|
|
try {
|
|
const response = await postJSON(BASE_URL, { filename: file });
|
|
|
|
if (!response.ok) {
|
|
const errorText = await response.text();
|
|
console.error(`Error parsing file ${file}: ${errorText}`);
|
|
|
|
// Write fail state incrementally
|
|
const tableLine = `| ${idx + 1} | \`${file}\` | **Failed** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
|
|
fs.appendFileSync(summaryTmpFile, tableLine);
|
|
|
|
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
|
detailedText += `- **Status**: Failed\n`;
|
|
detailedText += `- **Error Detail**: \`${errorText || 'Unknown error'}\`\n`;
|
|
detailedText += `\n---\n\n`;
|
|
fs.appendFileSync(detailsTmpFile, detailedText);
|
|
|
|
fs.appendFileSync(jsonlFile, JSON.stringify({
|
|
filename: file,
|
|
status: 'Failed',
|
|
error: errorText || 'Unknown error'
|
|
}) + '\n');
|
|
|
|
continue;
|
|
}
|
|
|
|
const resData = await response.json();
|
|
const pipelineRes = resData.result || {};
|
|
const page0 = pipelineRes.layoutParsingResults?.[0] || {};
|
|
const rawMarkdown = page0.markdown?.text || "N/A";
|
|
const info = pipelineRes.pipeline_info || {};
|
|
|
|
// Direct DB query for accurate metadata (bypassing Auth)
|
|
const docMeta = await getDocumentMetadataFromDb(file);
|
|
|
|
const tiltStr = info.tilt !== undefined ? parseFloat(info.tilt).toFixed(2) : 'N/A';
|
|
const unwarpedStr = info.unwarped ? 'Yes' : 'No';
|
|
const itemsCount = (resData.items || []).length;
|
|
|
|
// Write success state incrementally
|
|
const tableLine = `| ${idx + 1} | \`${file}\` | **Success** | ${tiltStr}° | ${unwarpedStr} | \`${docMeta.noPO || 'N/A'}\` | \`${docMeta.noSO || 'N/A'}\` | \`${docMeta.noDO || 'N/A'}\` | \`${docMeta.tanggal || 'N/A'}\` | ${docMeta.orderUntuk || 'N/A'} | ${itemsCount} |\n`;
|
|
fs.appendFileSync(summaryTmpFile, tableLine);
|
|
|
|
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
|
detailedText += `- **Status**: Success\n`;
|
|
detailedText += `- **Tilt Detected**: ${tiltStr}°\n`;
|
|
detailedText += `- **Auto-Rotated/Unwarped**: ${unwarpedStr}\n`;
|
|
detailedText += `- **Extracted Metadata**:\n`;
|
|
detailedText += ` * **PO**: \`${docMeta.noPO || 'N/A'}\`\n`;
|
|
detailedText += ` * **SO**: \`${docMeta.noSO || 'N/A'}\`\n`;
|
|
detailedText += ` * **DO**: \`${docMeta.noDO || 'N/A'}\`\n`;
|
|
detailedText += ` * **Tanggal**: \`${docMeta.tanggal || 'N/A'}\`\n`;
|
|
detailedText += ` * **Customer**: \`${docMeta.customerInfo || 'N/A'}\`\n`;
|
|
detailedText += ` * **Store**: \`${docMeta.orderUntuk || 'N/A'}\`\n`;
|
|
detailedText += ` * **Alamat**: \`${docMeta.alamat || 'N/A'}\`\n`;
|
|
detailedText += ` * **Plat Nomor**: \`${docMeta.platTruk || 'N/A'}\`\n`;
|
|
detailedText += `- **Raw Layout Markdown**:\n`;
|
|
detailedText += `\`\`\`markdown\n${rawMarkdown}\n\`\`\`\n`;
|
|
detailedText += `- **Parsed Items (${itemsCount})**:\n`;
|
|
|
|
if (itemsCount > 0) {
|
|
detailedText += ` | Code (SKU) | Name | Qty | Price |\n`;
|
|
detailedText += ` |---|---|---|---|\n`;
|
|
(resData.items || []).forEach(item => {
|
|
detailedText += ` | \`${item.kodeBarang}\` | ${item.namaBarang} | \`${item.banyak}\` | \`${item.jumlah}\` |\n`;
|
|
});
|
|
} else {
|
|
detailedText += ` *No valid SKU items parsed.*\n`;
|
|
}
|
|
detailedText += `\n---\n\n`;
|
|
fs.appendFileSync(detailsTmpFile, detailedText);
|
|
|
|
fs.appendFileSync(jsonlFile, JSON.stringify({
|
|
filename: file,
|
|
status: 'Success',
|
|
tilt: tiltStr,
|
|
unwarped: unwarpedStr,
|
|
rawMarkdown: resData.postProcessingDetails?.rawMarkdown || "",
|
|
layer1RawRegex: resData.postProcessingDetails?.layer1RawRegex || {},
|
|
layer2Sanitized: resData.postProcessingDetails?.layer2Sanitized || {},
|
|
layer3Final: resData.postProcessingDetails?.layer3Final || {},
|
|
metadata: docMeta,
|
|
items: resData.items || []
|
|
}) + '\n');
|
|
|
|
} catch (err) {
|
|
console.error(`Exception during file ${file}:`, err);
|
|
|
|
const tableLine = `| ${idx + 1} | \`${file}\` | **Error** | N/A | N/A | N/A | N/A | N/A | N/A | N/A | N/A |\n`;
|
|
fs.appendFileSync(summaryTmpFile, tableLine);
|
|
|
|
let detailedText = `### ${idx + 1}. \`${file}\`\n`;
|
|
detailedText += `- **Status**: Error\n`;
|
|
detailedText += `- **Error Detail**: \`${err.message}\`\n`;
|
|
detailedText += `\n---\n\n`;
|
|
fs.appendFileSync(detailsTmpFile, detailedText);
|
|
|
|
fs.appendFileSync(jsonlFile, JSON.stringify({
|
|
filename: file,
|
|
status: 'Error',
|
|
error: err.message
|
|
}) + '\n');
|
|
}
|
|
}
|
|
|
|
// Combine temporary files into the final report
|
|
try {
|
|
const summaryContent = fs.readFileSync(summaryTmpFile, 'utf8');
|
|
const detailsContent = fs.readFileSync(detailsTmpFile, 'utf8');
|
|
fs.writeFileSync(finalReportFile, summaryContent + '\n' + detailsContent);
|
|
|
|
// Clean up temporary files
|
|
fs.unlinkSync(summaryTmpFile);
|
|
fs.unlinkSync(detailsTmpFile);
|
|
} catch (combineErr) {
|
|
console.error('Failed to combine test reports:', combineErr);
|
|
}
|
|
|
|
// Compile JSONL into the final JSON v2
|
|
try {
|
|
const lines = fs.readFileSync(jsonlFile, 'utf8').split('\n').filter(Boolean);
|
|
const results = lines.map(line => JSON.parse(line));
|
|
fs.writeFileSync('/uploads/ai_results_v2.json', JSON.stringify(results, null, 2));
|
|
console.log('Compiled results saved to /uploads/ai_results_v2.json');
|
|
} catch (compileErr) {
|
|
console.error('Failed to compile results into JSON v2:', compileErr);
|
|
}
|
|
|
|
console.log('Batch test completed. Report written to /uploads/test_images_report.md');
|
|
}
|
|
|
|
main();
|