chore: reduce layout threshold to 0.2, tune vLLM memory, and add test scripts and reports
This commit is contained in:
1 parent
bdb3a49742
commit
5c7c64e2f4
6 files changed
+2090
-9
No files matched your search
@@ -25,7 +25,7 @@ SubModules:
|
||||
model_name: PP-DocLayoutV3
|
||||
model_dir: null
|
||||
batch_size: 8
|
||||
threshold: 0.3
|
||||
threshold: 0.2
|
||||
layout_nms: True
|
||||
layout_unclip_ratio: [1.0, 1.0]
|
||||
layout_merge_bboxes_mode:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# vLLM backend tuning for paddleocr genai_server
|
||||
# Docs: https://www.paddleocr.ai/latest/en/version3.x/pipeline_usage/PaddleOCR-VL.html#331-server-side-parameter-adjustment
|
||||
gpu-memory-utilization: 0.6
|
||||
gpu-memory-utilization: 0.35
|
||||
max-num-seqs: 4
|
||||
enforce-eager: true
|
||||
max-model-len: 2048
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
const { Client } = require('pg');
|
||||
|
||||
async function main() {
|
||||
const client = new Client({
|
||||
host: process.env.PGHOST || "paddleocr-db",
|
||||
port: parseInt(process.env.PGPORT || "5432"),
|
||||
user: process.env.PGUSER || "postgres",
|
||||
password: process.env.PGPASSWORD || "postgres",
|
||||
database: process.env.PGDATABASE || "dopfm",
|
||||
});
|
||||
|
||||
await client.connect();
|
||||
console.log('Connected to PG database.');
|
||||
|
||||
const res = await client.query("SELECT id, filename FROM documents WHERE parsed = false;");
|
||||
console.log(`Found ${res.rows.length} documents to parse.`);
|
||||
|
||||
for (let i = 0; i < res.rows.length; i++) {
|
||||
const row = res.rows[i];
|
||||
console.log(`[${i+1}/${res.rows.length}] Reparsing ${row.filename} (ID: ${row.id})...`);
|
||||
try {
|
||||
const response = await fetch('http://localhost:3000/api/parse', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ filename: row.filename })
|
||||
});
|
||||
if (response.ok) {
|
||||
console.log(`Successfully triggered parse for ${row.filename}. Status: ${response.status}`);
|
||||
} else {
|
||||
console.error(`Failed to parse ${row.filename}. Status: ${response.status}, Error: ${await response.text()}`);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(`Fetch error for ${row.filename}:`, err.message);
|
||||
}
|
||||
}
|
||||
|
||||
await client.end();
|
||||
console.log('Done reparsing.');
|
||||
}
|
||||
|
||||
main().catch(err => {
|
||||
console.error('Fatal error:', err);
|
||||
process.exit(1);
|
||||
});
|
||||
@@ -0,0 +1,113 @@
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const PROXY_URL = 'http://localhost:8000/api/vllm-proxy/v1/chat/completions';
|
||||
const IMAGE_PATH = path.join(__dirname, '..', 'sources', 'test-images', '1782870899198-sample_do.jpeg');
|
||||
|
||||
async function testGuidedDecoding() {
|
||||
console.log('Reading test image from:', IMAGE_PATH);
|
||||
if (!fs.existsSync(IMAGE_PATH)) {
|
||||
console.error('Test image not found!');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const imageBuffer = fs.readFileSync(IMAGE_PATH);
|
||||
const base64Image = imageBuffer.toString('base64');
|
||||
const imageUrl = `data:image/jpeg;base64,${base64Image}`;
|
||||
|
||||
// JSON Schema for DO metadata
|
||||
const jsonSchema = {
|
||||
type: "object",
|
||||
properties: {
|
||||
tanggal: { type: "string" },
|
||||
noPo: { type: "string" },
|
||||
noSo: { type: "string" },
|
||||
noDo: { type: "string" },
|
||||
kepadaYth: { type: "string" },
|
||||
orderUntuk: { type: "string" },
|
||||
alamat: { type: "string" },
|
||||
platTruk: { type: "string" },
|
||||
namaDriver: { type: "string" },
|
||||
namaPenerima: { type: "string" },
|
||||
items: {
|
||||
type: "array",
|
||||
items: {
|
||||
type: "object",
|
||||
properties: {
|
||||
nomor_sku: { type: "string" },
|
||||
nama_barang: { type: "string" },
|
||||
banyak: { type: "string" },
|
||||
jumlah: { type: "string" }
|
||||
},
|
||||
required: ["nomor_sku", "nama_barang", "banyak", "jumlah"]
|
||||
}
|
||||
}
|
||||
},
|
||||
required: ["tanggal", "noPo", "noSo", "noDo", "kepadaYth", "items"]
|
||||
};
|
||||
|
||||
const payload = {
|
||||
model: "PaddleOCR-VL-1.6-0.9B",
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: {
|
||||
url: imageUrl
|
||||
}
|
||||
},
|
||||
{
|
||||
type: "text",
|
||||
text: "Extract all structural details from this Delivery Order. Match the requested JSON Schema exactly."
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
temperature: 0.1,
|
||||
max_tokens: 1024,
|
||||
guided_json: JSON.stringify(jsonSchema) // standard vLLM guided JSON schema format
|
||||
};
|
||||
|
||||
console.log('Sending request to vLLM proxy with JSON Schema...');
|
||||
try {
|
||||
const startTime = Date.now();
|
||||
const response = await fetch(PROXY_URL, {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json'
|
||||
},
|
||||
body: JSON.stringify(payload)
|
||||
});
|
||||
|
||||
console.log(`Response status: ${response.status} (${response.statusText})`);
|
||||
const duration = ((Date.now() - startTime) / 1000).toFixed(2);
|
||||
console.log(`Request completed in ${duration}s`);
|
||||
|
||||
const result = await response.json();
|
||||
if (response.ok) {
|
||||
console.log('=== SUCCESS RESPONSE ===');
|
||||
console.log(JSON.stringify(result, null, 2));
|
||||
|
||||
const content = result.choices?.[0]?.message?.content;
|
||||
console.log('\n=== EXTRACTED CONTENT ===');
|
||||
console.log(content);
|
||||
|
||||
try {
|
||||
const parsed = JSON.parse(content);
|
||||
console.log('\nValid JSON parsed successfully! ✅');
|
||||
console.log(parsed);
|
||||
} catch (err) {
|
||||
console.error('\nFailed to parse content as JSON! ❌', err.message);
|
||||
}
|
||||
} else {
|
||||
console.error('=== ERROR RESPONSE ===');
|
||||
console.error(result);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Request failed:', error);
|
||||
}
|
||||
}
|
||||
|
||||
testGuidedDecoding();
|
||||
@@ -0,0 +1,109 @@
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const PIPELINE_URL = 'http://localhost:8000/layout-parsing';
|
||||
const PROXY_URL = 'http://localhost:8000/api/vllm-proxy/v1/chat/completions';
|
||||
const UPLOADS_DIR = path.join(__dirname, '..', 'uploads');
|
||||
|
||||
const notFoundImages = [
|
||||
'1782888211457-rotated_1782888204591_rotated_1782888199962_rotated_1782888195402_CAP7202641176939787142.jpg',
|
||||
'1782875064223-CAP2007290974474760139.jpg',
|
||||
'1782884586047-CAP9169719214882332189.jpg',
|
||||
'1782893409879-CAP4330863738757813156.jpg'
|
||||
];
|
||||
|
||||
async function testImage(filename) {
|
||||
console.log(`\n========================================`);
|
||||
console.log(`TESTING FILE: ${filename}`);
|
||||
console.log(`========================================`);
|
||||
|
||||
const filePath = path.join(UPLOADS_DIR, filename);
|
||||
if (!fs.existsSync(filePath)) {
|
||||
console.error(`File does not exist on disk: ${filePath}`);
|
||||
return;
|
||||
}
|
||||
|
||||
const fileBuffer = fs.readFileSync(filePath);
|
||||
const base64Image = fileBuffer.toString('base64');
|
||||
|
||||
// Test 1: Hit the Layout Parsing Pipeline API
|
||||
console.log('\n--- Test 1: Layout Parsing Pipeline (Standard) ---');
|
||||
try {
|
||||
const res = await fetch(PIPELINE_URL, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
file: base64Image,
|
||||
matchHistoryJob: false,
|
||||
useLayoutDetection: true,
|
||||
fileType: 1,
|
||||
useDocUnwarping: false,
|
||||
useDocOrientationClassify: true
|
||||
})
|
||||
});
|
||||
|
||||
if (res.ok) {
|
||||
const data = await res.json();
|
||||
const markdown = data.result?.layoutParsingResults?.[0]?.markdown?.text || data.layoutParsingResults?.[0]?.markdown?.text || '';
|
||||
console.log('Resulting Markdown snippet (first 300 chars):');
|
||||
console.log(markdown.substring(0, 300));
|
||||
console.log(`\nDoes it contain PO, SO, DO or Tanggal?`);
|
||||
console.log(`- "Tanggal": ${/Tanggal/i.test(markdown)}`);
|
||||
console.log(`- "SO": ${/SO/i.test(markdown)}`);
|
||||
console.log(`- "DO": ${/DO/i.test(markdown)}`);
|
||||
console.log(`- "PO": ${/PO/i.test(markdown)}`);
|
||||
} else {
|
||||
console.error(`Pipeline returned status ${res.status}: ${await res.text()}`);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Pipeline test failed:', err.message);
|
||||
}
|
||||
|
||||
// Test 2: Direct vLLM completions with simple extraction prompt
|
||||
console.log('\n--- Test 2: Direct vLLM Simple Extraction ---');
|
||||
try {
|
||||
const res = await fetch(PROXY_URL, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model: "PaddleOCR-VL-1.6-0.9B",
|
||||
messages: [
|
||||
{
|
||||
role: "user",
|
||||
content: [
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: `data:image/jpeg;base64,${base64Image}` }
|
||||
},
|
||||
{
|
||||
type: "text",
|
||||
text: "Read the top right section of the document. Extract Tanggal, No. SO, No. DO, and No. PO."
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
temperature: 0.1,
|
||||
max_tokens: 300
|
||||
})
|
||||
});
|
||||
|
||||
if (res.ok) {
|
||||
const data = await res.json();
|
||||
const content = data.choices?.[0]?.message?.content;
|
||||
console.log('vLLM Response:');
|
||||
console.log(content);
|
||||
} else {
|
||||
console.error(`vLLM proxy returned status ${res.status}: ${await res.text()}`);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('vLLM test failed:', err.message);
|
||||
}
|
||||
}
|
||||
|
||||
async function runAll() {
|
||||
for (const filename of notFoundImages) {
|
||||
await testImage(filename);
|
||||
}
|
||||
}
|
||||
|
||||
runAll();
|
||||
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user