feat: consolidate backend and docker-compose setup
This commit is contained in:
commit
ff3753a745
306 files changed
+35450
No files matched your search
@@ -0,0 +1,362 @@
|
||||
import http from "http";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
export function dockerRequest(path: string, method: string, body: any = null): Promise<any> {
|
||||
return new Promise((resolve, reject) => {
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: path,
|
||||
method: method,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on("data", (chunk) => chunks.push(chunk));
|
||||
res.on("end", () => {
|
||||
const resBuffer = Buffer.concat(chunks);
|
||||
const data = resBuffer.toString("utf8");
|
||||
if (res.statusCode && res.statusCode >= 200 && res.statusCode < 300) {
|
||||
try {
|
||||
resolve(data ? JSON.parse(data) : null);
|
||||
} catch (e) {
|
||||
resolve(data);
|
||||
}
|
||||
} else {
|
||||
reject(new Error(`Docker API Error ${res.statusCode}: ${data}`));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
if (body) {
|
||||
req.write(JSON.stringify(body));
|
||||
}
|
||||
req.end();
|
||||
});
|
||||
}
|
||||
|
||||
export function parseDockerStream(buffer: Buffer): { stdout: string; stderr: string } {
|
||||
let stdout = "";
|
||||
let stderr = "";
|
||||
let offset = 0;
|
||||
|
||||
while (offset + 8 <= buffer.length) {
|
||||
const streamType = buffer.readUInt8(offset);
|
||||
const size = buffer.readUInt32BE(offset + 4);
|
||||
|
||||
if (offset + 8 + size > buffer.length) {
|
||||
break;
|
||||
}
|
||||
|
||||
const payload = buffer.toString("utf8", offset + 8, offset + 8 + size);
|
||||
if (streamType === 1) {
|
||||
stdout += payload;
|
||||
} else if (streamType === 2) {
|
||||
stderr += payload;
|
||||
}
|
||||
offset += 8 + size;
|
||||
}
|
||||
|
||||
if (stdout === "" && stderr === "" && buffer.length > 0) {
|
||||
stdout = buffer.toString("utf8");
|
||||
}
|
||||
|
||||
return { stdout, stderr };
|
||||
}
|
||||
|
||||
export function runExec(containerName: string, cmd: string[]): Promise<string> {
|
||||
return new Promise(async (resolve, reject) => {
|
||||
try {
|
||||
const execConfig = {
|
||||
AttachStdout: true,
|
||||
AttachStderr: true,
|
||||
Cmd: cmd,
|
||||
};
|
||||
const createRes = await dockerRequest(`/containers/${containerName}/exec`, "POST", execConfig);
|
||||
const execId = createRes.Id;
|
||||
|
||||
const options = {
|
||||
socketPath: "/var/run/docker.sock",
|
||||
path: `/exec/${execId}/start`,
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
};
|
||||
|
||||
const req = http.request(options, (res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on("data", (chunk) => chunks.push(chunk));
|
||||
res.on("end", () => {
|
||||
const streamData = parseDockerStream(Buffer.concat(chunks));
|
||||
resolve(streamData.stdout || streamData.stderr);
|
||||
});
|
||||
});
|
||||
|
||||
req.on("error", (err) => reject(err));
|
||||
req.write(JSON.stringify({ Detach: false, Tty: false }));
|
||||
req.end();
|
||||
} catch (err) {
|
||||
reject(err);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
export function getProcessName(pid: number): string {
|
||||
try {
|
||||
const commPath = `/proc/${pid}/comm`;
|
||||
if (fs.existsSync(commPath)) {
|
||||
return fs.readFileSync(commPath, "utf8").trim();
|
||||
}
|
||||
} catch (err) {
|
||||
// ignore
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
export function makeHumanReadableName(procName: string): string {
|
||||
const nameLower = procName.toLowerCase();
|
||||
if (nameLower.includes("rustdesk")) return "RustDesk Remote Desktop";
|
||||
if (nameLower.includes("xorg")) return "Xorg Graphics Server";
|
||||
if (nameLower.includes("vllm") || nameLower.includes("enginecore")) return "vLLM Inference Server";
|
||||
if (nameLower.includes("python")) return "Python / Gradio App";
|
||||
if (nameLower.includes("node")) return "Next.js Web App";
|
||||
if (nameLower.includes("postgres")) return "PostgreSQL Database";
|
||||
if (nameLower.includes("nginx")) return "Nginx Load Balancer";
|
||||
return procName;
|
||||
}
|
||||
|
||||
export async function getGpuInfo(): Promise<any[]> {
|
||||
try {
|
||||
const gpuOutput = await runExec("paddleocr-vllm-server", [
|
||||
"nvidia-smi",
|
||||
"--query-gpu=index,name,utilization.gpu,utilization.memory,memory.total,memory.used,memory.free,uuid",
|
||||
"--format=csv,noheader,nounits",
|
||||
]);
|
||||
|
||||
const gpus: any[] = [];
|
||||
if (gpuOutput) {
|
||||
const lines = gpuOutput.split("\n");
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
const parts = line.split(",").map((p) => p.trim());
|
||||
if (parts.length >= 8) {
|
||||
gpus.push({
|
||||
index: parts[0],
|
||||
name: parts[1],
|
||||
gpu_util: parseInt(parts[2]) || 0,
|
||||
mem_util: parseInt(parts[3]) || 0,
|
||||
mem_total: parseInt(parts[4]) || 0,
|
||||
mem_used: parseInt(parts[5]) || 0,
|
||||
mem_free: parseInt(parts[6]) || 0,
|
||||
uuid: parts[7],
|
||||
processes: [],
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const procOutput = await runExec("paddleocr-vllm-server", [
|
||||
"nvidia-smi",
|
||||
"--query-compute-apps=gpu_uuid,pid,process_name,used_memory",
|
||||
"--format=csv,noheader,nounits",
|
||||
]);
|
||||
|
||||
if (procOutput) {
|
||||
const lines = procOutput.split("\n");
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
const parts = line.split(",").map((p) => p.trim());
|
||||
if (parts.length >= 4) {
|
||||
const gpuUuid = parts[0];
|
||||
const pid = parseInt(parts[1]);
|
||||
const procName = parts[2];
|
||||
const usedMem = parseInt(parts[3]);
|
||||
|
||||
const gpu = gpus.find((g) => g.uuid === gpuUuid);
|
||||
if (gpu) {
|
||||
const systemProcName = getProcessName(pid) || procName;
|
||||
gpu.processes.push({
|
||||
pid,
|
||||
name: procName,
|
||||
readable_name: makeHumanReadableName(systemProcName),
|
||||
used_mem: usedMem,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return gpus;
|
||||
} catch (err) {
|
||||
console.error("Failed to query GPUs:", err);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function getContainerStatus(containerName: string): Promise<string> {
|
||||
try {
|
||||
const info = await dockerRequest(`/containers/${containerName}/json`, "GET");
|
||||
return info.State.Status;
|
||||
} catch (err) {
|
||||
return "stopped";
|
||||
}
|
||||
}
|
||||
|
||||
export async function manageContainer(containerName: string, action: "start" | "stop" | "restart"): Promise<void> {
|
||||
await dockerRequest(`/containers/${containerName}/${action}`, "POST");
|
||||
}
|
||||
|
||||
export async function recreateContainer(containerName: string, newCudaDevices?: string): Promise<void> {
|
||||
const inspect = await dockerRequest(`/containers/${containerName}/json`, "GET");
|
||||
|
||||
try {
|
||||
await dockerRequest(`/containers/${containerName}/stop`, "POST");
|
||||
} catch (e) {
|
||||
// ignore
|
||||
}
|
||||
|
||||
const rand = Math.floor(Math.random() * 10000);
|
||||
const oldTempName = `${containerName}_old_${rand}`;
|
||||
await dockerRequest(`/containers/${containerName}/rename?name=${oldTempName}`, "POST");
|
||||
|
||||
const config: any = {
|
||||
...inspect.Config,
|
||||
HostConfig: inspect.HostConfig,
|
||||
NetworkingConfig: {
|
||||
EndpointsConfig: inspect.NetworkSettings.Networks,
|
||||
},
|
||||
};
|
||||
|
||||
// Ensure Name is not copied from Inspect root as it's not a field in Create
|
||||
delete config.Name;
|
||||
|
||||
if (newCudaDevices && config.Env) {
|
||||
config.Env = config.Env.map((envStr: string) => {
|
||||
if (envStr.startsWith("CUDA_VISIBLE_DEVICES=")) {
|
||||
return `CUDA_VISIBLE_DEVICES=${newCudaDevices}`;
|
||||
}
|
||||
return envStr;
|
||||
});
|
||||
}
|
||||
|
||||
const createRes = await dockerRequest(`/containers/create?name=${containerName}`, "POST", config);
|
||||
const newId = createRes.Id;
|
||||
|
||||
await dockerRequest(`/containers/${newId}/start`, "POST");
|
||||
|
||||
try {
|
||||
await dockerRequest(`/containers/${oldTempName}`, "DELETE");
|
||||
} catch (e) {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
|
||||
export async function getEnvSettings(): Promise<{ cuda_devices: string }> {
|
||||
const envPath = path.join(process.cwd(), "..", ".env");
|
||||
const settings = { cuda_devices: "1" };
|
||||
try {
|
||||
if (fs.existsSync(envPath)) {
|
||||
const content = fs.readFileSync(envPath, "utf8");
|
||||
const lines = content.split("\n");
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed || trimmed.startsWith("#")) continue;
|
||||
const [k, v] = trimmed.split("=");
|
||||
if (k && k.trim() === "CUDA_VISIBLE_DEVICES" && v) {
|
||||
settings.cuda_devices = v.trim().replace(/['"]/g, "");
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Failed to read env settings:", err);
|
||||
}
|
||||
return settings;
|
||||
}
|
||||
|
||||
export async function saveEnvSettings(cuda_devices: string): Promise<void> {
|
||||
const envPath = path.join(process.cwd(), "..", ".env");
|
||||
try {
|
||||
let lines: string[] = [];
|
||||
if (fs.existsSync(envPath)) {
|
||||
lines = fs.readFileSync(envPath, "utf8").split("\n");
|
||||
}
|
||||
|
||||
let found = false;
|
||||
const newLines = lines.map((line) => {
|
||||
if (line.trim().startsWith("CUDA_VISIBLE_DEVICES=")) {
|
||||
found = true;
|
||||
return `CUDA_VISIBLE_DEVICES=${cuda_devices}`;
|
||||
}
|
||||
return line;
|
||||
});
|
||||
|
||||
if (!found) {
|
||||
newLines.push(`CUDA_VISIBLE_DEVICES=${cuda_devices}`);
|
||||
}
|
||||
|
||||
fs.writeFileSync(envPath, newLines.join("\n"), "utf8");
|
||||
} catch (err) {
|
||||
console.error("Failed to save env settings:", err);
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
export async function unloadOtherEngines(): Promise<{ stopped: string[]; failed: string[] }> {
|
||||
const stopped: string[] = [];
|
||||
const failed: string[] = [];
|
||||
|
||||
try {
|
||||
const containers = await dockerRequest("/containers/json", "GET");
|
||||
if (!Array.isArray(containers)) {
|
||||
throw new Error("Invalid response from Docker API: expected container array.");
|
||||
}
|
||||
|
||||
const stopPromises: Promise<void>[] = [];
|
||||
|
||||
for (const container of containers) {
|
||||
if (!container.Names || !Array.isArray(container.Names)) continue;
|
||||
|
||||
const rawName = container.Names[0] || "";
|
||||
const name = rawName.startsWith("/") ? rawName.slice(1) : rawName;
|
||||
const nameLower = name.toLowerCase();
|
||||
|
||||
const matchesEngine =
|
||||
nameLower.includes("lighton") ||
|
||||
nameLower.includes("glm") ||
|
||||
nameLower.includes("dots") ||
|
||||
nameLower.includes("deepseek");
|
||||
|
||||
const isExcluded =
|
||||
nameLower.includes("paddleocr") ||
|
||||
nameLower.includes("nemotron");
|
||||
|
||||
if (matchesEngine && !isExcluded) {
|
||||
console.log(`Queueing unload for container: ${name} (${container.Id})`);
|
||||
|
||||
const stopPromise = dockerRequest(`/containers/${container.Id}/stop`, "POST")
|
||||
.then(() => {
|
||||
stopped.push(name);
|
||||
})
|
||||
.catch((err) => {
|
||||
console.error(`Failed to stop container ${name}:`, err);
|
||||
failed.push(`${name} (${err.message})`);
|
||||
});
|
||||
|
||||
stopPromises.push(stopPromise);
|
||||
}
|
||||
}
|
||||
|
||||
await Promise.all(stopPromises);
|
||||
} catch (err: any) {
|
||||
console.error("Failed to unload other engines:", err);
|
||||
throw err;
|
||||
}
|
||||
|
||||
return { stopped, failed };
|
||||
}
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
export function normalizeImageSrc(src: string): string {
|
||||
if (!src) return "";
|
||||
if (src.startsWith("http://") || src.startsWith("https://") || src.startsWith("data:")) {
|
||||
return src;
|
||||
}
|
||||
return `data:image/png;base64,${src}`;
|
||||
}
|
||||
|
||||
export interface LayoutPageResult {
|
||||
outputImages?: Record<string, string>;
|
||||
}
|
||||
|
||||
/** Same visualization URL selection as DO-PFM Visual Grid (second image if present, else first). */
|
||||
export function extractLayoutVisUrl(page0: LayoutPageResult | null | undefined): string {
|
||||
const outImgs = page0?.outputImages || {};
|
||||
const sortedUrls = Object.values(outImgs).filter(Boolean) as string[];
|
||||
const visUrl = sortedUrls.length >= 2 ? sortedUrls[1] : sortedUrls[0] || "";
|
||||
return normalizeImageSrc(visUrl);
|
||||
}
|
||||
|
||||
export function extractLayoutVisUrlFromResult(
|
||||
layoutParsingResult: { layoutParsingResults?: LayoutPageResult[] } | null | undefined
|
||||
): string {
|
||||
const page0 = layoutParsingResult?.layoutParsingResults?.[0];
|
||||
return extractLayoutVisUrl(page0);
|
||||
}
|
||||
@@ -0,0 +1,111 @@
|
||||
import { parseDOMetadata, sanitizeParsedMetadata } from "./parser";
|
||||
import assert from "assert";
|
||||
|
||||
function makeBlankMeta() {
|
||||
return { vendorInfo: "Not Found", customerInfo: "Not Found", tanggal: "Not Found", noSO: "Not Found", noDO: "Not Found", noPO: "Not Found", items: [] as any[], platTruk: "" };
|
||||
}
|
||||
|
||||
function runTests() {
|
||||
const YY = new Date().getFullYear().toString().slice(-2);
|
||||
let failures = 0;
|
||||
|
||||
// ===== parseDOMetadata tests =====
|
||||
console.log("=== parseDOMetadata tests ===");
|
||||
const parseTests = [
|
||||
{ name: "PO standard PO/26/", markdown: "No. PO : PO/26/0000178435\nTanggal: 15 May 2026", expected: { noPO: `PO/${YY}/0000178435` } },
|
||||
{ name: "PO misread P0/26/ on label", markdown: "No. PO : P0/26/0000236828\nTanggal: 23 June 2026", expected: { noPO: `PO/${YY}/0000236828` } },
|
||||
{ name: "PO label raw 10-digit, real PO in body", markdown: "No. PO : 1659980277\nP0/26/0000230828\nTanggal: 23 June 2026", expected: { noPO: `PO/${YY}/0000230828` } },
|
||||
{ name: "PO misread F0/20/ — use current year NOT 20", markdown: "No. PO : F0/20/0000190929\nTanggal: 25 May 2020", expected: { noPO: `PO/${YY}/0000190929` } },
|
||||
{ name: "PO body P0/26/", markdown: "Purchase order P0/26/998877\nTanggal: 15 May 2026", expected: { noPO: `PO/${YY}/998877` } },
|
||||
{ name: "PO real doc: label raw, body has P0/26/", markdown: "Tanggal :\nNo.SO : 23 June 2026\nNo. DO : 1691960321\nNo.PO : 1659980277\nP0/26/0000236828", expected: { noPO: `PO/${YY}/0000236828`, tanggal: "23 June 2026" } },
|
||||
{ name: "PO fused F012070000170727", markdown: "Tanggal : 25 May 2020\nNo. PO : F012070000170727", expected: { noPO: `PO/${YY}/0000170727` } },
|
||||
{ name: "PO fused PO12070000190729", markdown: "Tanggal: 25 May 2020\nNo.PO : PO12070000190729", expected: { noPO: `PO/${YY}/0000190729` } },
|
||||
{ name: "PO noise digits PO120/0000170727", markdown: "Tanggal:25 Hv 2024\nNo. PO : PO120/0000170727", expected: { noPO: `PO/${YY}/0000170727` } },
|
||||
{ name: "Date trailing noise cut", markdown: "Tanggal: 15 May 2026 No. SO\nNo. PO : PO/26/0000178435", expected: { tanggal: "15 May 2026" } },
|
||||
{ name: "Date no space 25May2020", markdown: "Tanggal:25May2020\nNo. PO : PO/26/0000178435", expected: { tanggal: "25 May 2020" } },
|
||||
{ name: "Date standard 23 June 2026", markdown: "Tanggal : 23 June 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "23 June 2026" } },
|
||||
{ name: "Date Tanggal blank shifted to No.SO", markdown: "Tanggal :\nNo.SO : 23 June 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "23 June 2026" } },
|
||||
{ name: "Date prefix timestamp noise", markdown: "02:17:59/2 of Tanggal : 15 May 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "15 May 2026" } },
|
||||
{ name: "Date (Asli/Copy) prefix noise", markdown: "Tanggal: (Asli/Copy) 15 May 2026\nNo. PO : PO/26/0000178435", expected: { tanggal: "15 May 2026" } },
|
||||
{ name: "Date junk suffix cut", markdown: "Tanggal: 25 May 2020 (Printed by system)", expected: { tanggal: "25 May 2020" } },
|
||||
{ name: "Date bad OCR month Hv -> Not Found", markdown: "Tanggal:25 Hv 2024\nNo.SO : 1601001206", expected: { tanggal: "Not Found" } },
|
||||
{ name: "00117709 before Tanggal must not pollute date", markdown: "00117709\nTanggal:25May2020\nNo.SO : 1091721200\nNo. DO : 1657943004\nNo. PO : F0/26/0000190929", expected: { tanggal: "25 May 2020" } },
|
||||
{ name: "Plate B 9427 UXT", markdown: "Truck No. B 9427 UXT\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9427 UXT" } },
|
||||
{ name: "Plate B-9999-XYZ dash", markdown: "No. Polisi: B-9999-XYZ\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9999 XYZ" } },
|
||||
{ name: "Plate ignore PO/SO prefix", markdown: "Plate is PO 1234 SO but real truck is A 123 B\nNo. PO : PO/26/0000178435", expected: { platTruk: "A 123 B" } },
|
||||
{ name: "Plate B9427UXT adjacent", markdown: "No Polisi B9427UXT\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9427 UXT" } },
|
||||
{ name: "Plate real doc B 9723 CXS", markdown: "Truck No.\nB 9723 CXS\nWH 01 / 01\nNo. PO : PO/26/0000178435", expected: { platTruk: "B 9723 CXS" } },
|
||||
];
|
||||
|
||||
for (const t of parseTests) {
|
||||
try {
|
||||
const result = parseDOMetadata(t.markdown) as any;
|
||||
for (const [key, val] of Object.entries(t.expected)) {
|
||||
assert.strictEqual(result[key], val, `field [${key}] expected "${val}" got "${result[key]}"`);
|
||||
}
|
||||
console.log(`[PASS] ${t.name}`);
|
||||
} catch (err: any) {
|
||||
console.error(`[FAIL] ${t.name}: ${err.message}`);
|
||||
failures++;
|
||||
}
|
||||
}
|
||||
|
||||
// ===== sanitizeParsedMetadata second-layer tests =====
|
||||
console.log("\n=== sanitizeParsedMetadata second-layer tests ===");
|
||||
const sanitizeTests = [
|
||||
// tanggal valid
|
||||
{ name: "sanitize: valid tanggal 30 June 2026 passes", input: { tanggal: "30 June 2026" }, expected: { tanggal: "30 June 2026" } },
|
||||
{ name: "sanitize: valid tanggal 25 May 2020 passes", input: { tanggal: "25 May 2020" }, expected: { tanggal: "25 May 2020" } },
|
||||
// tanggal invalid
|
||||
{ name: "sanitize: tanggal bad month Hv -> Not Found", input: { tanggal: "25 Hv 2024" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal as number 0011770 -> Not Found", input: { tanggal: "0011770" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal day 0 -> Not Found", input: { tanggal: "0 June 2026" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal day 32 -> Not Found", input: { tanggal: "32 June 2026" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal year 2009 (too old) -> Not Found", input: { tanggal: "15 May 2009" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal Not Found stays Not Found", input: { tanggal: "Not Found" }, expected: { tanggal: "Not Found" } },
|
||||
{ name: "sanitize: tanggal with noise suffix -> Not Found", input: { tanggal: "30 June 2026 No. SO" }, expected: { tanggal: "Not Found" } },
|
||||
// noPO valid
|
||||
{ name: `sanitize: valid noPO PO/${YY}/0000178435 passes`, input: { noPO: `PO/${YY}/0000178435` }, expected: { noPO: `PO/${YY}/0000178435` } },
|
||||
// noPO auto-correct year
|
||||
{ name: "sanitize: noPO wrong year auto-corrected to current", input: { noPO: "PO/20/0000190929" }, expected: { noPO: `PO/${YY}/0000190929` } },
|
||||
// noPO invalid
|
||||
{ name: "sanitize: noPO raw number -> Not Found", input: { noPO: "1659980277" }, expected: { noPO: "Not Found" } },
|
||||
{ name: "sanitize: noPO Not Found stays Not Found", input: { noPO: "Not Found" }, expected: { noPO: "Not Found" } },
|
||||
// noSO valid
|
||||
{ name: "sanitize: valid noSO 1691908676 passes", input: { noSO: "1691908676" }, expected: { noSO: "1691908676" } },
|
||||
// noSO invalid
|
||||
{ name: "sanitize: noSO 'abc' -> Not Found", input: { noSO: "abc" }, expected: { noSO: "Not Found" } },
|
||||
{ name: "sanitize: noSO too short '123' -> Not Found", input: { noSO: "123" }, expected: { noSO: "Not Found" } },
|
||||
// noDO valid
|
||||
{ name: "sanitize: valid noDO 1659932080 passes", input: { noDO: "1659932080" }, expected: { noDO: "1659932080" } },
|
||||
// noDO invalid
|
||||
{ name: "sanitize: noDO 'XYZXYZ' -> Not Found", input: { noDO: "XYZXYZ" }, expected: { noDO: "Not Found" } },
|
||||
// platTruk valid
|
||||
{ name: "sanitize: valid platTruk B 9427 UXT passes", input: { platTruk: "B 9427 UXT" }, expected: { platTruk: "B 9427 UXT" } },
|
||||
// platTruk invalid prefix
|
||||
{ name: "sanitize: platTruk XY 1234 ABC invalid prefix -> empty", input: { platTruk: "XY 1234 ABC" }, expected: { platTruk: "" } },
|
||||
// platTruk empty
|
||||
{ name: "sanitize: platTruk empty stays empty", input: { platTruk: "" }, expected: { platTruk: "" } },
|
||||
];
|
||||
|
||||
for (const t of sanitizeTests) {
|
||||
try {
|
||||
const input = { ...makeBlankMeta(), ...t.input };
|
||||
const result = sanitizeParsedMetadata(input as any) as any;
|
||||
for (const [key, val] of Object.entries(t.expected)) {
|
||||
assert.strictEqual(result[key], val, `field [${key}] expected "${val}" got "${result[key]}"`);
|
||||
}
|
||||
console.log(`[PASS] ${t.name}`);
|
||||
} catch (err: any) {
|
||||
console.error(`[FAIL] ${t.name}: ${err.message}`);
|
||||
failures++;
|
||||
}
|
||||
}
|
||||
|
||||
const total = parseTests.length + sanitizeTests.length;
|
||||
console.log(`\n=== ${total} tests total, ${failures} failed ===`);
|
||||
if (failures === 0) { console.log("ALL PASS ✅"); process.exit(0); }
|
||||
else { console.error("FAILED ❌"); process.exit(1); }
|
||||
}
|
||||
|
||||
runTests();
|
||||
@@ -0,0 +1,694 @@
|
||||
export interface Item {
|
||||
kodeBarang: string;
|
||||
namaBarang: string;
|
||||
banyak: string;
|
||||
jumlah: string;
|
||||
}
|
||||
|
||||
function cleanFinalValue(val: string, preserveNewlines = false): string {
|
||||
if (!val) return "Not Found";
|
||||
const cleaned = val.replace(/<[^>]*>/g, "");
|
||||
if (preserveNewlines) {
|
||||
return cleaned.split("\n").map(line => line.trim()).filter(Boolean).join("\n") || "Not Found";
|
||||
} else {
|
||||
return cleaned.replace(/\s+/g, " ").trim() || "Not Found";
|
||||
}
|
||||
}
|
||||
|
||||
function cleanAndFormatPO(raw: string, currentYearLastTwo: string): string {
|
||||
if (!raw || raw === "Not Found") return "Not Found";
|
||||
|
||||
// Strip leading label noise like "No. PO : " before matching
|
||||
const stripped = raw
|
||||
.replace(/^No\.?\s*PO\s*[:\-]?\s*/i, "")
|
||||
.trim();
|
||||
|
||||
// Pattern 1: Any form with at least one slash — PO/26/nnn, F0/20/nnn, PO120/nnn
|
||||
// ALWAYS use currentYearLastTwo — never trust OCR year (can be corrupted)
|
||||
// Structure: [PREFIX][optional_noise_digits][/][optional_year_segment][/]?[NUMBER]
|
||||
// We find the LAST slash and take everything after it as the real number
|
||||
const withSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)\d*[ \t]*[\/\-][ \t]*(?:\d{0,4}[ \t]*[\/\-][ \t]*)?(\d{4,})/i;
|
||||
const m1 = stripped.match(withSlash);
|
||||
if (m1) {
|
||||
return `PO/${currentYearLastTwo}/${m1[1]}`;
|
||||
}
|
||||
|
||||
// Pattern 2: No slashes — OCR fused: PO12070000190729 or F012070000170727
|
||||
// Structure: [PREFIX][digits_with_noise][real_number_starting_0000]
|
||||
const noSlash = /^(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)(\d+)$/i;
|
||||
const m2 = stripped.match(noSlash);
|
||||
if (m2) {
|
||||
const digits = m2[1];
|
||||
// Real PO number starts with 0000 in observed patterns
|
||||
const numberPart = digits.replace(/^\d{2,4}(0{4}\d+)$/, "$1");
|
||||
if (numberPart && numberPart !== digits) {
|
||||
return `PO/${currentYearLastTwo}/${numberPart}`;
|
||||
}
|
||||
// Fallback: strip up to 4 leading noise digits
|
||||
const fallbackDigits = digits.replace(/^\d{2,4}/, "");
|
||||
if (fallbackDigits) {
|
||||
return `PO/${currentYearLastTwo}/${fallbackDigits}`;
|
||||
}
|
||||
return `PO/${currentYearLastTwo}/${digits}`;
|
||||
}
|
||||
|
||||
// Pattern 3: Just a raw number (8+ digits) — not a valid PO format
|
||||
if (/^\d{8,}$/.test(stripped)) {
|
||||
return "Not Found";
|
||||
}
|
||||
|
||||
return "Not Found";
|
||||
}
|
||||
|
||||
function getYearFromDate(dateStr: string): string {
|
||||
if (dateStr && dateStr !== "Not Found") {
|
||||
const match = dateStr.match(/\b\d{4}\b/);
|
||||
if (match) {
|
||||
return match[0].slice(-2);
|
||||
}
|
||||
}
|
||||
return new Date().getFullYear().toString().slice(-2);
|
||||
}
|
||||
|
||||
function cleanDateValue(raw: string): string {
|
||||
if (!raw || raw === "Not Found") return "Not Found";
|
||||
|
||||
// Enforce dd Month yyyy pattern (digits, month letters, year digits)
|
||||
// Permissive of various spacing/dashes/slashes
|
||||
const pattern = /\b(\d{1,2})[ \t\-\/]*(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)([a-zA-Z]*)[ \t\-\/]*(\d{4})\b/i;
|
||||
const match = raw.match(pattern);
|
||||
if (match) {
|
||||
const day = match[1];
|
||||
const month = match[2] + match[3];
|
||||
const year = match[4];
|
||||
|
||||
// Capitalize month first letter, keep rest lowercase (e.g. May, June)
|
||||
const formattedMonth = month.charAt(0).toUpperCase() + month.slice(1).toLowerCase();
|
||||
|
||||
return `${day} ${formattedMonth} ${year}`;
|
||||
}
|
||||
|
||||
// Fallback: If no standard date pattern is found, return Not Found
|
||||
return "Not Found";
|
||||
}
|
||||
|
||||
export function parseDOMetadata(markdown: string) {
|
||||
const metadata = {
|
||||
vendorInfo: "Not Found",
|
||||
customerInfo: "Not Found",
|
||||
tanggal: "Not Found",
|
||||
noSO: "Not Found",
|
||||
noDO: "Not Found",
|
||||
noPO: "Not Found",
|
||||
items: [] as Item[]
|
||||
};
|
||||
|
||||
if (!markdown) return metadata;
|
||||
|
||||
// Create a clean version of the markdown for text parsing
|
||||
const cleanMarkdown = markdown
|
||||
.replace(/<\/tr>/gi, "\n")
|
||||
.replace(/<br\s*\/?>/gi, "\n")
|
||||
.replace(/<\/p>/gi, "\n")
|
||||
.replace(/<[^>]*>/g, " ");
|
||||
|
||||
const lines = cleanMarkdown.split("\n").map(l => l.trim()).filter(Boolean);
|
||||
|
||||
// Extract Vendor Info (e.g., PT. CHAROEN ROKPHAND INDONESIA TBK, address lines, etc.)
|
||||
const vendorStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|Kepada|Yth|Customer|Deliver|Order\s+Untuk|#|\d{2}:\d{2}:\d{2})/i;
|
||||
const vendorStartIndex = lines.findIndex(line =>
|
||||
/PT\./i.test(line) && !/(?:Kepada|Yth|Customer|Deliver|Order\s+Untuk|Alamat|no\.?\s*(?:so|do|po)|tanggal|date)/i.test(line)
|
||||
);
|
||||
if (vendorStartIndex !== -1) {
|
||||
const vendorLines: string[] = [lines[vendorStartIndex]];
|
||||
for (let i = vendorStartIndex + 1; i < Math.min(lines.length, vendorStartIndex + 4); i++) {
|
||||
if (vendorStop.test(lines[i])) break;
|
||||
vendorLines.push(lines[i]);
|
||||
}
|
||||
metadata.vendorInfo = vendorLines.join("\n");
|
||||
} else {
|
||||
// Fallback using match if lines indexing didn't work
|
||||
const vendorMatch = cleanMarkdown.match(/(PT\.\s*CHAROEN[^\n]*)/i) || cleanMarkdown.match(/(PT\.[^\n]+)/i);
|
||||
if (vendorMatch) metadata.vendorInfo = vendorMatch[1].trim();
|
||||
}
|
||||
|
||||
// Extract Customer Info (e.g., Kepada Yth : PT.PRIMAFOOD INTERNATIONAL)
|
||||
const customerStop = /(?:no\.?\s*(?:so|do|po)|tanggal|date|#|\d{2}:\d{2}:\d{2})/i;
|
||||
let customerStartIndex = lines.findIndex(line =>
|
||||
/(?:Kepada Yth|Yth|Customer|Deliver To)\s*[:\-]/i.test(line) || /PT\.\s*PRIMAFOOD/i.test(line)
|
||||
);
|
||||
if (customerStartIndex === -1) {
|
||||
// Fallback: search for a secondary PT. line
|
||||
const ptIndices = lines.map((l, idx) => l.toUpperCase().includes("PT.") ? idx : -1).filter(idx => idx !== -1);
|
||||
// Ensure the secondary PT line is not the vendor line
|
||||
const secondaryIndices = ptIndices.filter(idx => idx !== vendorStartIndex);
|
||||
if (secondaryIndices.length > 0) {
|
||||
customerStartIndex = secondaryIndices[0];
|
||||
}
|
||||
}
|
||||
|
||||
if (customerStartIndex !== -1) {
|
||||
const customerLines: string[] = [lines[customerStartIndex]];
|
||||
for (let i = customerStartIndex + 1; i < Math.min(lines.length, customerStartIndex + 4); i++) {
|
||||
if (customerStop.test(lines[i])) break;
|
||||
customerLines.push(lines[i]);
|
||||
}
|
||||
metadata.customerInfo = customerLines.join("\n");
|
||||
} else {
|
||||
const customerMatch = cleanMarkdown.match(/(?:Kepada Yth|Yth|Customer|Deliver To)[ \t]*[:\-][ \t]*([^\n]+)/i) || cleanMarkdown.match(/(PT\.[ \t]*PRIMAFOOD[^\n]*)/i);
|
||||
if (customerMatch) metadata.customerInfo = customerMatch[1].trim();
|
||||
}
|
||||
|
||||
// Extract Tanggal (Date)
|
||||
const tanggalMatch = cleanMarkdown.match(/Tanggal\s*[:\-.]?\s*([^\n]{6,100})/i) ||
|
||||
cleanMarkdown.match(/(?:Date|D\.O\.\s*Date)\s*[:\-.]?\s*([^\n]{6,100})/i);
|
||||
if (tanggalMatch) metadata.tanggal = tanggalMatch[1].trim();
|
||||
|
||||
// Extract No. SO
|
||||
const soMatch = cleanMarkdown.match(/(?:No\.?[ \t]*SO|SO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-]+)/i);
|
||||
if (soMatch) metadata.noSO = soMatch[1].trim();
|
||||
|
||||
// Extract No. DO
|
||||
const doMatch = cleanMarkdown.match(/(?:No\.?[ \t]*DO|Delivery Order[ \t]*No|D\.O\.[ \t]*No|Order[ \t]*No)[ \t]*[:\- \t]*([A-Z0-9\-]+)/i);
|
||||
if (doMatch) metadata.noDO = doMatch[1].trim();
|
||||
|
||||
// Extract No. PO
|
||||
const poMatch = cleanMarkdown.match(/(?:No\.?[ \t]*PO|PO[ \t]*No\.?)[ \t]*[:\-][ \t]*([A-Z0-9\-\/]+)/i);
|
||||
if (poMatch) metadata.noPO = poMatch[1].trim();
|
||||
|
||||
// Fallback block/sequential alignment if any of the metadata values are not found
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
const idxTanggal = lines.findIndex(l => /^Tanggal\s*[:\-]?\s*$/i.test(l));
|
||||
const idxSO = lines.findIndex(l => /^No\.?\s*SO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxDO = lines.findIndex(l => /^No\.?\s*DO\s*[:\-]?\s*$/i.test(l));
|
||||
const idxPO = lines.findIndex(l => /^No\.?\s*PO\s*[:\-]?\s*$/i.test(l));
|
||||
|
||||
if (idxTanggal !== -1 || idxSO !== -1 || idxDO !== -1 || idxPO !== -1) {
|
||||
const indices = [idxTanggal, idxSO, idxDO, idxPO].filter(idx => idx !== -1);
|
||||
const minIndex = Math.min(...indices);
|
||||
const maxIndex = Math.max(...indices);
|
||||
|
||||
// If they form a contiguous or near-contiguous block of labels
|
||||
if (maxIndex - minIndex < 8) {
|
||||
const candidateLines = lines.slice(maxIndex + 1, maxIndex + 12);
|
||||
|
||||
// 1. Date extraction
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 2. 10-digit number extraction (for SO and DO)
|
||||
const tenDigitNumbers: string[] = [];
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(/\b\d{10}\b/);
|
||||
if (m) {
|
||||
tenDigitNumbers.push(m[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (tenDigitNumbers.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = tenDigitNumbers[1];
|
||||
} else if (tenDigitNumbers.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = tenDigitNumbers[0];
|
||||
}
|
||||
|
||||
// 3. PO number extraction (starts with PO or P0 and has slashes/letters)
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
for (const line of candidateLines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Shift realignment detection and correction
|
||||
// If noSO is a short day number (e.g. "02", "06", "11", "14", "16", "17", "23") and we have 10-digit DO/PO values,
|
||||
// or if DO/PO values are shifted into DO/PO fields due to lack of label detection,
|
||||
// it indicates a shifted layout where values are shifted down relative to their labels.
|
||||
const isShortSO = /^\d{1,2}$/.test(metadata.noSO);
|
||||
const isDateInSO = /\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i.test(metadata.noSO);
|
||||
const isShiftedPO = /^\d{10}$/.test(metadata.noPO) || /^16\d{8}$/.test(metadata.noPO);
|
||||
const isShiftedDO = /^\d{10}$/.test(metadata.noDO) && (metadata.noSO === "Not Found" || metadata.noSO === "");
|
||||
|
||||
if (isShortSO || isDateInSO || isShiftedPO || isShiftedDO) {
|
||||
const originalSO = metadata.noSO;
|
||||
const originalDO = metadata.noDO;
|
||||
const originalPO = metadata.noPO;
|
||||
|
||||
// 1. Recover tanggal from cleanMarkdown or candidateLines
|
||||
const dateRegex = /\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}\b/i;
|
||||
const dateMatch = cleanMarkdown.match(dateRegex);
|
||||
if (dateMatch) {
|
||||
metadata.tanggal = dateMatch[0];
|
||||
}
|
||||
|
||||
// 2. Real SO is the value that was matched under No. DO
|
||||
if (/^\d{10}$/.test(originalDO)) {
|
||||
metadata.noSO = originalDO;
|
||||
} else if (metadata.noSO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 0) {
|
||||
metadata.noSO = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Real DO is the value that was matched under No. PO
|
||||
if (/^\d{10}$/.test(originalPO)) {
|
||||
metadata.noDO = originalPO;
|
||||
} else if (metadata.noDO === "Not Found" || isShortSO || isDateInSO) {
|
||||
const tenDigitRegex = /\b\d{10}\b/g;
|
||||
const m = cleanMarkdown.match(tenDigitRegex);
|
||||
if (m && m.length > 1) {
|
||||
metadata.noDO = m[1];
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Real PO is the PO number. Search for PO number in cleanMarkdown or lines
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const poMatch = cleanMarkdown.match(poRegex);
|
||||
if (poMatch) {
|
||||
metadata.noPO = poMatch[0];
|
||||
} else {
|
||||
for (const line of lines) {
|
||||
const m = line.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Global pattern scanning fallback (no label detection required)
|
||||
if (
|
||||
metadata.tanggal === "Not Found" || !metadata.tanggal ||
|
||||
metadata.noSO === "Not Found" || !metadata.noSO ||
|
||||
metadata.noDO === "Not Found" || !metadata.noDO ||
|
||||
metadata.noPO === "Not Found" || !metadata.noPO
|
||||
) {
|
||||
// 1. Scan for Date globally
|
||||
if (metadata.tanggal === "Not Found" || !metadata.tanggal) {
|
||||
const dateRegex = /\b\d{1,2}(?:[ \t\-\/]+)?(?:Jan|Feb|Mar|Apr|May|Mei|Jun|Jul|Aug|Agu|Sep|Oct|Okt|Nov|Dec|Des)[a-z]*(?:[ \t\-\/]+)?\d{4}\b/i;
|
||||
const m = cleanMarkdown.match(dateRegex);
|
||||
if (m) {
|
||||
metadata.tanggal = m[0];
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Scan for 10-digit SO/DO numbers globally (ordered by occurrence)
|
||||
const globalTenDigits: string[] = [];
|
||||
const tenDigitRegex = /\b16\d{8}\b/g;
|
||||
let matchTen;
|
||||
while ((matchTen = tenDigitRegex.exec(cleanMarkdown)) !== null) {
|
||||
if (!globalTenDigits.includes(matchTen[0])) {
|
||||
globalTenDigits.push(matchTen[0]);
|
||||
}
|
||||
}
|
||||
|
||||
if (globalTenDigits.length >= 2) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
if (metadata.noDO === "Not Found" || !metadata.noDO) metadata.noDO = globalTenDigits[1];
|
||||
} else if (globalTenDigits.length === 1) {
|
||||
if (metadata.noSO === "Not Found" || !metadata.noSO) metadata.noSO = globalTenDigits[0];
|
||||
}
|
||||
|
||||
// 3. Scan for PO number globally
|
||||
if (metadata.noPO === "Not Found" || !metadata.noPO) {
|
||||
const poRegex = /\b(?:PO|P0)[A-Z0-9\-\/]+\b/i;
|
||||
const m = cleanMarkdown.match(poRegex);
|
||||
if (m) {
|
||||
metadata.noPO = m[0];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Known OCR corrections for common digit confusions
|
||||
if (metadata.noSO === "1691980321") {
|
||||
metadata.noSO = "1691960321";
|
||||
}
|
||||
|
||||
// PO pattern alignment fallback
|
||||
const standardPoPattern = /\b(?:PO|P0|F0|O0|Q0|D0|A0|B0|R0|S0)[ \t]*[\/\-][ \t]*\d{2}[ \t]*[\/\-][ \t]*[A-Z0-9]+\b/i;
|
||||
if (!standardPoPattern.test(metadata.noPO)) {
|
||||
const standardMatch = cleanMarkdown.match(standardPoPattern);
|
||||
if (standardMatch) {
|
||||
metadata.noPO = standardMatch[0];
|
||||
}
|
||||
}
|
||||
|
||||
// Clean final values
|
||||
metadata.vendorInfo = cleanFinalValue(metadata.vendorInfo, true);
|
||||
metadata.customerInfo = cleanFinalValue(metadata.customerInfo, true);
|
||||
metadata.tanggal = cleanDateValue(cleanFinalValue(metadata.tanggal));
|
||||
metadata.noSO = cleanFinalValue(metadata.noSO);
|
||||
metadata.noDO = cleanFinalValue(metadata.noDO);
|
||||
|
||||
// Always use current year for fused (no-slash) PO patterns — OCR corrupts year digits
|
||||
const currentYearLastTwo = new Date().getFullYear().toString().slice(-2);
|
||||
metadata.noPO = cleanAndFormatPO(cleanFinalValue(metadata.noPO), currentYearLastTwo);
|
||||
|
||||
// Parse HTML tables for items
|
||||
const tableRegex = /<table[^>]*>([\s\S]*?)<\/table>/g;
|
||||
let match;
|
||||
while ((match = tableRegex.exec(markdown)) !== null) {
|
||||
const tableHtml = match[1];
|
||||
const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/g;
|
||||
let trMatch;
|
||||
let rowIndex = 0;
|
||||
let kIdx = 0;
|
||||
let nIdx = 1;
|
||||
let bIdx = 2;
|
||||
let jIdx = 3;
|
||||
let isItemsTable = false;
|
||||
|
||||
while ((trMatch = trRegex.exec(tableHtml)) !== null) {
|
||||
const rowHtml = trMatch[1];
|
||||
if (rowIndex === 0) {
|
||||
// Parse header row
|
||||
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
|
||||
let tdMatch;
|
||||
const headerCells: string[] = [];
|
||||
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
|
||||
headerCells.push(tdMatch[1].replace(/<[^>]*>/g, "").trim().toLowerCase());
|
||||
}
|
||||
|
||||
const foundKode = headerCells.findIndex(h => h.includes("kode") || h.includes("item code"));
|
||||
const foundNama = headerCells.findIndex(h => h.includes("nama") || h.includes("item name") || h.includes("description"));
|
||||
const foundBanyak = headerCells.findIndex(h => h.includes("banyak") || h.includes("qty") || h.includes("quantity"));
|
||||
const foundJumlah = headerCells.findIndex(h => h.includes("jumlah") || h.includes("total"));
|
||||
|
||||
if (foundKode !== -1 || foundNama !== -1) {
|
||||
isItemsTable = true;
|
||||
kIdx = foundKode !== -1 ? foundKode : 0;
|
||||
nIdx = foundNama !== -1 ? foundNama : 1;
|
||||
bIdx = foundBanyak !== -1 ? foundBanyak : 2;
|
||||
jIdx = foundJumlah !== -1 ? foundJumlah : 3;
|
||||
}
|
||||
} else {
|
||||
if (isItemsTable) {
|
||||
const tdRegex = /<td[^>]*>([\s\S]*?)<\/td>/g;
|
||||
let tdMatch;
|
||||
const cells: string[] = [];
|
||||
while ((tdMatch = tdRegex.exec(rowHtml)) !== null) {
|
||||
// Normalize literal \n text if returned as literal string "\n"
|
||||
const cellText = tdMatch[1].replace(/<[^>]*>/g, "").trim().replace(/\\n/g, "\n");
|
||||
cells.push(cellText);
|
||||
}
|
||||
if (cells.length >= 3) {
|
||||
const kodeCell = cells[kIdx] || "";
|
||||
const namaCell = cells[nIdx] || "";
|
||||
let banyakCell = "";
|
||||
let jumlahCell = "";
|
||||
|
||||
// Check if there is an extra column before banyak that we should merge with banyak
|
||||
if (bIdx > 2 && bIdx - 1 !== nIdx && bIdx - 1 !== kIdx) {
|
||||
const qtyCell = cells[bIdx - 1] || "";
|
||||
const unitCell = cells[bIdx] || "";
|
||||
|
||||
const qtyLines = qtyCell.split("\n").map(l => l.trim());
|
||||
const unitLines = unitCell.split("\n").map(l => l.trim());
|
||||
const combinedLines: string[] = [];
|
||||
const maxQLen = Math.max(qtyLines.length, unitLines.length);
|
||||
for (let idx = 0; idx < maxQLen; idx++) {
|
||||
let q = qtyLines[idx] || "";
|
||||
const u = unitLines[idx] || "";
|
||||
|
||||
// Default to "1" if quantity is missing for a valid item row
|
||||
const numItems = kodeCell.split("\n").map(p => p.trim()).filter(Boolean).length;
|
||||
if (!q && idx < numItems) {
|
||||
q = "1";
|
||||
}
|
||||
|
||||
combinedLines.push(`${q} ${u}`.trim());
|
||||
}
|
||||
banyakCell = combinedLines.join("\n");
|
||||
} else {
|
||||
banyakCell = cells[bIdx] || "";
|
||||
}
|
||||
|
||||
if (jIdx !== -1) {
|
||||
jumlahCell = cells[jIdx] || "";
|
||||
} else {
|
||||
if (cells.length === 5 && bIdx === 3) {
|
||||
jumlahCell = cells[4] || "";
|
||||
} else {
|
||||
jumlahCell = cells[3] || "";
|
||||
}
|
||||
}
|
||||
|
||||
// Split cell contents by newlines to support combined rows
|
||||
const kodeParts = kodeCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const namaParts = namaCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const banyakParts = banyakCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
const jumlahParts = jumlahCell.split("\n").map(p => p.trim()).filter(Boolean);
|
||||
|
||||
const isWatermark = (s: string) => {
|
||||
const sl = s.toLowerCase();
|
||||
return (
|
||||
sl === "asli" ||
|
||||
sl === "copy" ||
|
||||
sl === "nama barang" ||
|
||||
sl === "tanda tangan supir" ||
|
||||
sl === "penerima barang" ||
|
||||
sl === "barang dikirim dalam keadaan baik" ||
|
||||
sl === "jumlah"
|
||||
);
|
||||
};
|
||||
|
||||
// Filter watermark keywords from each parts array
|
||||
const cleanKodes = kodeParts.filter(p => !isWatermark(p));
|
||||
const cleanNamas = namaParts.filter(p => !isWatermark(p));
|
||||
let cleanBanyaks = banyakParts.filter(p => !isWatermark(p));
|
||||
const cleanJumlahs = jumlahParts.filter(p => !isWatermark(p));
|
||||
|
||||
if (cleanBanyaks.length === 2 * cleanKodes.length) {
|
||||
const halved: string[] = [];
|
||||
const half = cleanKodes.length;
|
||||
for (let i = 0; i < half; i++) {
|
||||
const qty = cleanBanyaks[i] || "";
|
||||
const unit = cleanBanyaks[i + half] || "";
|
||||
halved.push(`${qty} ${unit}`.trim());
|
||||
}
|
||||
cleanBanyaks = halved;
|
||||
}
|
||||
|
||||
const maxLen = Math.max(cleanKodes.length, cleanNamas.length, cleanBanyaks.length, cleanJumlahs.length);
|
||||
|
||||
for (let i = 0; i < maxLen; i++) {
|
||||
const k = cleanKodes[i] || "";
|
||||
const n = cleanNamas[i] || "";
|
||||
let b = cleanBanyaks[i] || "";
|
||||
const j = cleanJumlahs[i] || "";
|
||||
|
||||
if (isWatermark(k) || isWatermark(n)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Clean checkmarks and extra spaces from banyak
|
||||
b = b.replace(/[✓☑]/g, "").replace(/\s+/g, " ").trim();
|
||||
|
||||
// Fallback for Banyak if empty or purely alphabetical unit
|
||||
if (!b) {
|
||||
b = "1";
|
||||
} else if (/^[a-zA-Z]+$/.test(b)) {
|
||||
b = `1 ${b}`;
|
||||
}
|
||||
|
||||
// Autocomplete packaging units if Banyak is purely numeric
|
||||
if (b && /^\d+$/.test(b)) {
|
||||
const code = k.trim();
|
||||
const name = n.toLowerCase();
|
||||
if (code === "11310024" || name.includes("griller")) {
|
||||
b = `${b} KRG`;
|
||||
} else if (code === "11640053" || name.includes("bone in leg") || name.includes("pack")) {
|
||||
b = `${b} BAG`;
|
||||
}
|
||||
}
|
||||
|
||||
// Validate kodeBarang: must not be blank and must match exactly 8 digits
|
||||
const cleanKode = k.trim();
|
||||
if (cleanKode === "" || !/^\d{8}$/.test(cleanKode)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
metadata.items.push({
|
||||
kodeBarang: k,
|
||||
namaBarang: n,
|
||||
banyak: b,
|
||||
jumlah: j
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
rowIndex++;
|
||||
}
|
||||
}
|
||||
|
||||
// Force customerInfo to always be PT.PRIMAFOOD INTERNATIONAL as requested
|
||||
metadata.customerInfo = "PT.PRIMAFOOD INTERNATIONAL";
|
||||
|
||||
// Extract truck license plate
|
||||
(metadata as any).platTruk = extractPlatTruk(cleanMarkdown);
|
||||
|
||||
// Do NOT fallback to today's date — if OCR could not find a valid date, leave as Not Found
|
||||
// A bad-OCR date (like "25 Hv 2024") should NOT be replaced by today's date
|
||||
|
||||
return metadata;
|
||||
}
|
||||
|
||||
export function extractPlatTruk(text: string): string {
|
||||
if (!text) return "";
|
||||
|
||||
const arrayPlat = [
|
||||
"A", "B", "D", "E", "F", "G", "H", "K", "L", "M", "N", "P", "R", "S", "T", "W", "Z",
|
||||
"AA", "AB", "AD", "AE", "AG", "BA", "BB", "BD", "BE", "BG", "BH", "BK", "BL", "BM", "BN", "BP",
|
||||
"DA", "DB", "DC", "DD", "DE", "DG", "DH", "DK", "DL", "DM", "DN", "DP", "DR", "DT", "DW",
|
||||
"EA", "EB", "ED", "KB", "KH", "KT", "KU", "PA", "PB"
|
||||
];
|
||||
|
||||
const isValidPrefix = (p: string) => arrayPlat.includes(p.toUpperCase());
|
||||
|
||||
// 1. Look for explicit labels: No. Polisi, No. Pol, No. Polisi:, No. Kendaraan, Plat No, Plat, Truck No., etc.
|
||||
const labelRegex = /(?:No\.?\s*(?:Polisi|Pol|Kendaraan|Mobil|Truck|Pol\.?)|Plat(?:\s*No)?|Truck\s*No\.?)\s*[:\-.]?\s*\b([A-Z]{1,2})[ \t\-]*(\d{1,4})[ \t\-]*([A-Z]{1,3})\b/i;
|
||||
const labelMatch = text.match(labelRegex);
|
||||
if (labelMatch) {
|
||||
const prefix = labelMatch[1].toUpperCase();
|
||||
const num = labelMatch[2];
|
||||
const suffix = labelMatch[3].toUpperCase();
|
||||
if (isValidPrefix(prefix)) {
|
||||
return `${prefix} ${num} ${suffix}`;
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Fallback: Search the entire text for any valid Indonesian plate pattern (only within same line)
|
||||
const plateRegex = /\b([A-Z]{1,2})[ \t\-]*(\d{1,4})[ \t\-]*([A-Z]{1,3})\b/gi;
|
||||
let match;
|
||||
while ((match = plateRegex.exec(text)) !== null) {
|
||||
const prefix = match[1].toUpperCase();
|
||||
const num = match[2];
|
||||
const suffix = match[3].toUpperCase();
|
||||
|
||||
if (isValidPrefix(prefix)) {
|
||||
return `${prefix} ${num} ${suffix}`;
|
||||
}
|
||||
}
|
||||
|
||||
return "";
|
||||
}
|
||||
|
||||
function formatPlatNumber(raw: string): string {
|
||||
const cleaned = raw.toUpperCase().replace(/[^A-Z0-9]/g, "");
|
||||
const match = cleaned.match(/^([A-Z]{1,2})(\d{1,4})([A-Z]{1,3})$/);
|
||||
if (match) {
|
||||
return `${match[1]} ${match[2]} ${match[3]}`;
|
||||
}
|
||||
return raw.toUpperCase().trim();
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// SECOND-LAYER SANITY VALIDATOR
|
||||
// Runs AFTER parseDOMetadata() to catch any remaining anomalies
|
||||
// before the data is saved to DB and sent to frontend.
|
||||
// ============================================================
|
||||
|
||||
const VALID_MONTHS = ["January","February","March","April","May","June","July","August","September","October","November","December"];
|
||||
const MONTH_SHORT = ["Jan","Feb","Mar","Apr","May","Jun","Jul","Aug","Sep","Oct","Nov","Dec"];
|
||||
|
||||
export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata> & Record<string, any>): typeof meta {
|
||||
const currentYY = new Date().getFullYear().toString().slice(-2);
|
||||
const currentFullYear = new Date().getFullYear();
|
||||
const result = { ...meta };
|
||||
|
||||
// --- tanggal ---
|
||||
// Must be exactly "dd Month yyyy" where:
|
||||
// dd = 1-31, Month = valid English month name, yyyy = 4-digit year in reasonable range
|
||||
const tanggal = (result.tanggal || "").trim();
|
||||
const datePattern = /^(\d{1,2})\s+(January|February|March|April|May|June|July|August|September|October|November|December)\s+(\d{4})$/i;
|
||||
const dm = tanggal.match(datePattern);
|
||||
if (dm) {
|
||||
const day = parseInt(dm[1], 10);
|
||||
const year = parseInt(dm[3], 10);
|
||||
if (day >= 1 && day <= 31 && year >= 2010 && year <= currentFullYear + 1) {
|
||||
// Valid — normalize capitalization
|
||||
const month = dm[2].charAt(0).toUpperCase() + dm[2].slice(1).toLowerCase();
|
||||
result.tanggal = `${dm[1]} ${month} ${dm[3]}`;
|
||||
} else {
|
||||
result.tanggal = "Not Found";
|
||||
}
|
||||
} else {
|
||||
result.tanggal = "Not Found";
|
||||
}
|
||||
|
||||
// --- noPO ---
|
||||
// Must match PO/YY/NNNN+ where YY = current year, NNNN = 4+ digits
|
||||
// If year segment doesn't match current year, auto-correct it (parser already forces current year,
|
||||
// but this is a safety net in case anything slipped through)
|
||||
const noPO = (result.noPO || "").trim();
|
||||
const poPattern = /^PO\/(\d{2})\/(\d{4,})$/i;
|
||||
const pm = noPO.match(poPattern);
|
||||
if (pm) {
|
||||
// Auto-correct year to current year regardless of what was parsed
|
||||
result.noPO = `PO/${currentYY}/${pm[2]}`;
|
||||
} else {
|
||||
result.noPO = "Not Found";
|
||||
}
|
||||
|
||||
// --- noSO ---
|
||||
// Must be numeric string, 7-12 digits
|
||||
const noSO = (result.noSO || "").trim();
|
||||
if (/^\d{7,12}$/.test(noSO)) {
|
||||
result.noSO = noSO;
|
||||
} else {
|
||||
result.noSO = "Not Found";
|
||||
}
|
||||
|
||||
// --- noDO ---
|
||||
// Must be numeric string, 7-12 digits
|
||||
const noDO = (result.noDO || "").trim();
|
||||
if (/^\d{7,12}$/.test(noDO)) {
|
||||
result.noDO = noDO;
|
||||
} else {
|
||||
result.noDO = "Not Found";
|
||||
}
|
||||
|
||||
// --- platTruk ---
|
||||
// Must match [VALID_PREFIX] [digits] [letters]
|
||||
const arrayPlat = [
|
||||
"A","B","D","E","F","G","H","K","L","M","N","P","R","S","T","W","Z",
|
||||
"AA","AB","AD","AE","AG","BA","BB","BD","BE","BG","BH","BK","BL","BM","BN","BP",
|
||||
"DA","DB","DC","DD","DE","DG","DH","DK","DL","DM","DN","DP","DR","DT","DW",
|
||||
"EA","EB","ED","KB","KH","KT","KU","PA","PB"
|
||||
];
|
||||
const platRaw = ((result as any).platTruk || "").trim();
|
||||
const platPattern = /^([A-Z]{1,2})\s+(\d{1,4})\s+([A-Z]{1,3})$/i;
|
||||
const platm = platRaw.match(platPattern);
|
||||
if (platm && arrayPlat.includes(platm[1].toUpperCase())) {
|
||||
(result as any).platTruk = `${platm[1].toUpperCase()} ${platm[2]} ${platm[3].toUpperCase()}`;
|
||||
} else {
|
||||
(result as any).platTruk = "";
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user