chore: reduce layout threshold to 0.2, tune vLLM memory, and add test scripts and reports

This commit is contained in:
Rafhan Mazaya Fathurrahman committed 2026-07-02 15:30:19 +07:00
1 parent bdb3a49742
commit 5c7c64e2f4
6 files changed
+2090 -9

No files matched your search

+1 -1
View File
@@ -25,7 +25,7 @@ SubModules:
model_name: PP-DocLayoutV3
model_dir: null
batch_size: 8
threshold: 0.3
threshold: 0.2
layout_nms: True
layout_unclip_ratio: [1.0, 1.0]
layout_merge_bboxes_mode:
+1 -1
View File
@@ -1,6 +1,6 @@
# vLLM backend tuning for paddleocr genai_server
# Docs: https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment
gpu-memory-utilization: 0.6
gpu-memory-utilization: 0.35
max-num-seqs: 4
enforce-eager: true
max-model-len: 2048
+44
View File
@@ -0,0 +1,44 @@
const { Client } = require('pg');
async function main() {
const client = new Client({
host: process.env.PGHOST || "paddleocr-db",
port: parseInt(process.env.PGPORT || "5432"),
user: process.env.PGUSER || "postgres",
password: process.env.PGPASSWORD || "postgres",
database: process.env.PGDATABASE || "dopfm",
});
await client.connect();
console.log('Connected to PG database.');
const res = await client.query("SELECT id, filename FROM documents WHERE parsed = false;");
console.log(`Found ${res.rows.length} documents to parse.`);
for (let i = 0; i < res.rows.length; i++) {
const row = res.rows[i];
console.log(`[${i+1}/${res.rows.length}] Reparsing ${row.filename} (ID: ${row.id})...`);
try {
const response = await fetch('http://localhost:3000/api/parse', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ filename: row.filename })
});
if (response.ok) {
console.log(`Successfully triggered parse for ${row.filename}. Status: ${response.status}`);
} else {
console.error(`Failed to parse ${row.filename}. Status: ${response.status}, Error: ${await response.text()}`);
}
} catch (err) {
console.error(`Fetch error for ${row.filename}:`, err.message);
}
}
await client.end();
console.log('Done reparsing.');
}
main().catch(err => {
console.error('Fatal error:', err);
process.exit(1);
});
+113
View File
@@ -0,0 +1,113 @@
const fs = require('fs');
const path = require('path');
const PROXY_URL = 'http://localhost:8000/api/vllm-proxy/v1/chat/completions';
const IMAGE_PATH = path.join(__dirname, '..', 'sources', 'test-images', '1782870899198-sample_do.jpeg');
async function testGuidedDecoding() {
console.log('Reading test image from:', IMAGE_PATH);
if (!fs.existsSync(IMAGE_PATH)) {
console.error('Test image not found!');
process.exit(1);
}
const imageBuffer = fs.readFileSync(IMAGE_PATH);
const base64Image = imageBuffer.toString('base64');
const imageUrl = `data:image/jpeg;base64,${base64Image}`;
// JSON Schema for DO metadata
const jsonSchema = {
type: "object",
properties: {
tanggal: { type: "string" },
noPo: { type: "string" },
noSo: { type: "string" },
noDo: { type: "string" },
kepadaYth: { type: "string" },
orderUntuk: { type: "string" },
alamat: { type: "string" },
platTruk: { type: "string" },
namaDriver: { type: "string" },
namaPenerima: { type: "string" },
items: {
type: "array",
items: {
type: "object",
properties: {
nomor_sku: { type: "string" },
nama_barang: { type: "string" },
banyak: { type: "string" },
jumlah: { type: "string" }
},
required: ["nomor_sku", "nama_barang", "banyak", "jumlah"]
}
}
},
required: ["tanggal", "noPo", "noSo", "noDo", "kepadaYth", "items"]
};
const payload = {
model: "PaddleOCR-VL-1.6-0.9B",
messages: [
{
role: "user",
content: [
{
type: "image_url",
image_url: {
url: imageUrl
}
},
{
type: "text",
text: "Extract all structural details from this Delivery Order. Match the requested JSON Schema exactly."
}
]
}
],
temperature: 0.1,
max_tokens: 1024,
guided_json: JSON.stringify(jsonSchema) // standard vLLM guided JSON schema format
};
console.log('Sending request to vLLM proxy with JSON Schema...');
try {
const startTime = Date.now();
const response = await fetch(PROXY_URL, {
method: 'POST',
headers: {
'Content-Type': 'application/json'
},
body: JSON.stringify(payload)
});
console.log(`Response status: ${response.status} (${response.statusText})`);
const duration = ((Date.now() - startTime) / 1000).toFixed(2);
console.log(`Request completed in ${duration}s`);
const result = await response.json();
if (response.ok) {
console.log('=== SUCCESS RESPONSE ===');
console.log(JSON.stringify(result, null, 2));
const content = result.choices?.[0]?.message?.content;
console.log('\n=== EXTRACTED CONTENT ===');
console.log(content);
try {
const parsed = JSON.parse(content);
console.log('\nValid JSON parsed successfully! ✅');
console.log(parsed);
} catch (err) {
console.error('\nFailed to parse content as JSON! ❌', err.message);
}
} else {
console.error('=== ERROR RESPONSE ===');
console.error(result);
}
} catch (error) {
console.error('Request failed:', error);
}
}
testGuidedDecoding();
+109
View File
@@ -0,0 +1,109 @@
const fs = require('fs');
const path = require('path');
const PIPELINE_URL = 'http://localhost:8000/layout-parsing';
const PROXY_URL = 'http://localhost:8000/api/vllm-proxy/v1/chat/completions';
const UPLOADS_DIR = path.join(__dirname, '..', 'uploads');
const notFoundImages = [
'1782888211457-rotated_1782888204591_rotated_1782888199962_rotated_1782888195402_CAP7202641176939787142.jpg',
'1782875064223-CAP2007290974474760139.jpg',
'1782884586047-CAP9169719214882332189.jpg',
'1782893409879-CAP4330863738757813156.jpg'
];
async function testImage(filename) {
console.log(`\n========================================`);
console.log(`TESTING FILE: ${filename}`);
console.log(`========================================`);
const filePath = path.join(UPLOADS_DIR, filename);
if (!fs.existsSync(filePath)) {
console.error(`File does not exist on disk: ${filePath}`);
return;
}
const fileBuffer = fs.readFileSync(filePath);
const base64Image = fileBuffer.toString('base64');
// Test 1: Hit the Layout Parsing Pipeline API
console.log('\n--- Test 1: Layout Parsing Pipeline (Standard) ---');
try {
const res = await fetch(PIPELINE_URL, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
file: base64Image,
matchHistoryJob: false,
useLayoutDetection: true,
fileType: 1,
useDocUnwarping: false,
useDocOrientationClassify: true
})
});
if (res.ok) {
const data = await res.json();
const markdown = data.result?.layoutParsingResults?.[0]?.markdown?.text || data.layoutParsingResults?.[0]?.markdown?.text || '';
console.log('Resulting Markdown snippet (first 300 chars):');
console.log(markdown.substring(0, 300));
console.log(`\nDoes it contain PO, SO, DO or Tanggal?`);
console.log(`- "Tanggal": ${/Tanggal/i.test(markdown)}`);
console.log(`- "SO": ${/SO/i.test(markdown)}`);
console.log(`- "DO": ${/DO/i.test(markdown)}`);
console.log(`- "PO": ${/PO/i.test(markdown)}`);
} else {
console.error(`Pipeline returned status ${res.status}: ${await res.text()}`);
}
} catch (err) {
console.error('Pipeline test failed:', err.message);
}
// Test 2: Direct vLLM completions with simple extraction prompt
console.log('\n--- Test 2: Direct vLLM Simple Extraction ---');
try {
const res = await fetch(PROXY_URL, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model: "PaddleOCR-VL-1.6-0.9B",
messages: [
{
role: "user",
content: [
{
type: "image_url",
image_url: { url: `data:image/jpeg;base64,${base64Image}` }
},
{
type: "text",
text: "Read the top right section of the document. Extract Tanggal, No. SO, No. DO, and No. PO."
}
]
}
],
temperature: 0.1,
max_tokens: 300
})
});
if (res.ok) {
const data = await res.json();
const content = data.choices?.[0]?.message?.content;
console.log('vLLM Response:');
console.log(content);
} else {
console.error(`vLLM proxy returned status ${res.status}: ${await res.text()}`);
}
} catch (err) {
console.error('vLLM test failed:', err.message);
}
}
async function runAll() {
for (const filename of notFoundImages) {
await testImage(filename);
}
}
runAll();
File diff suppressed because it is too large. Load diff