Files
pfm-ocr/backend/pfm-web-app/src/db/index.ts
T
Rafhan Mazaya FathurrahmanandClaude Sonnet 5 e60ab63154 Adopt agents-settings kit, ship Product/SKU scan models, harden auth, verify OCR accuracy
Backend (app-pfm-ocr-v2/backend):
- Product/SKU scan feature complete: trained DINOv2 index (118 reference
  photos, 16 SKU classes) and YOLO classifier (83.3% top-1 val accuracy),
  fixed scripts/install-pipeline.sh (was missing ultralytics/torch), fully
  browser-verified end-to-end on /scan-pfm. Mobile m-scan-pfm page cancelled
  (Flutter app handles mobile; web UI is desktop-only for pipeline testing).
- Fixed a real data-loss bug: Save Ground Truth (scan-pfm and the DO-flow's
  manual-label) was silently writing into the pfm-web-app container's
  ephemeral filesystem instead of the host, because /sources wasn't
  bind-mounted in docker-compose.yml. Added the mount, recovered an
  orphaned entry.
- accounts.password is now bcrypt-hashed (bcryptjs, idempotent migration
  in db/init.ts) instead of plaintext; login route compares hashes.
- /api/v1/documents/* (list, PUT, upload) now enforces real 401 auth,
  matching what the Flutter client already sends. The "classic" routes
  deliberately stay open — they're dev-only web UI with no login flow and
  won't exist in production.
- OCR accuracy investigated end-to-end: real baseline is 95.10% overall
  (target met; accuracy_report.md was stale at 75.04%, now flagged). Fixed
  one genuine parser.ts bug (SO/DO field duplication in the global fallback
  regex); remaining gaps are OCR/layout-model limitations, not parser bugs.
- Adopted a standalone copy of the fhanyuh/agents-settings e/n workflow
  scoped to backend/ (AGENTS.md Part A/B split, SKILLS.md, plans/, docs/),
  independent of the root copy which now covers Flutter only.
- next-implementation.md deleted; content folded into
  backend/plans/next-enhancements.md for traceability.

Root:
- Adopted fhanyuh/agents-settings kit (AGENTS.md, SKILLS.md, plans/,
  docs/feature-list.md), scoped to the Flutter app only.
- Pending documents queue now persists to Hive (lib/core/storage) instead
  of memory-only, surviving an app kill mid-upload.

Removed backend_backup/ (stale Express/Prisma prototype, superseded by
pfm-web-app) and the completed plans/next-enhancement-plan.md checklist.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-07-08 11:56:32 +07:00

256 lines
9.3 KiB
TypeScript

import { Pool, PoolClient } from "pg";
import { initDb } from "./init";
const pool = new Pool({
host: process.env.PGHOST || "localhost",
port: parseInt(process.env.PGPORT || "5432"),
user: process.env.PGUSER || "postgres",
password: process.env.PGPASSWORD || "postgres",
database: process.env.PGDATABASE || "dopfm",
});
let initialized = false;
let initPromise: Promise<Pool> | null = null;
export async function getPool(): Promise<Pool> {
if (initialized) {
return pool;
}
if (!initPromise) {
initPromise = (async () => {
try {
await initDb(pool);
initialized = true;
} catch (err) {
console.error("Failed to initialize database:", err);
}
return pool;
})();
}
return initPromise;
}
export async function query(text: string, params?: unknown[]) {
const p = await getPool();
return p.query(text, params);
}
/** Runs `fn` inside a BEGIN/COMMIT transaction on a single held connection, rolling back and rethrowing on any failure. */
export async function withTransaction<T>(
fn: (client: PoolClient) => Promise<T>
): Promise<T> {
const p = await getPool();
const client = await p.connect();
try {
await client.query("BEGIN");
const result = await fn(client);
await client.query("COMMIT");
return result;
} catch (err) {
await client.query("ROLLBACK");
throw err;
} finally {
client.release();
}
}
export async function cleanupAndReindexItems(docId: number) {
// 1. Delete rows where kode_barang is blank/null or doesn't match an 8-digit number
await query(
`DELETE FROM ocr_items
WHERE document_id = $1
AND (kode_barang IS NULL OR TRIM(kode_barang) = '' OR NOT (kode_barang ~ '^[0-9]{8}$'))`,
[docId]
);
// 2. Fetch remaining rows ordered by row_index
const res = await query(
`SELECT id, row_index
FROM ocr_items
WHERE document_id = $1
ORDER BY row_index`,
[docId]
);
// 3. Update row_index to be sequential
for (let i = 0; i < res.rows.length; i++) {
const row = res.rows[i];
if (row.row_index !== i) {
await query(
`UPDATE ocr_items
SET row_index = $1
WHERE id = $2`,
[i, row.id]
);
}
}
}
const STORE_STOPWORDS = new Set([
"dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw",
"jalan", "raya", "blok", "nomor", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"
]);
function tokenize(text: string): string[] {
return text.toLowerCase()
.replace(/[^a-z0-9\s]/g, " ")
.split(/\s+/)
.filter(w => w.length > 2 && !STORE_STOPWORDS.has(w));
}
// customers.name is stored as "CUSTOMER NAME, JL. street address..." - split on the first
// street-address marker to get just the canonical address portion.
function splitCustomerAddress(name: string): string {
const m = name.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
return (m ? m[0] : name).replace(/\s+/g, " ").trim();
}
// A noisy OCR'd address (varying per document due to misread letters) that recognizably
// belongs to a known customer should be reported as that customer's clean canonical address,
// rather than whatever garbled text this particular scan happened to produce.
async function canonicalizeCustomerAddress(extracted: string): Promise<string> {
if (!extracted) return extracted;
const extractedTokens = new Set(tokenize(extracted));
if (extractedTokens.size === 0) return extracted;
const customersRes = await query("SELECT name FROM customers");
let bestAddress: string | null = null;
let bestMatchCount = 0;
let bestScore = 0;
for (const row of customersRes.rows) {
const canonicalAddress = splitCustomerAddress(row.name);
const addressTokens = tokenize(canonicalAddress);
if (addressTokens.length === 0) continue;
const uniqueAddressTokens = new Set(addressTokens);
let matchCount = 0;
for (const token of uniqueAddressTokens) {
if (extractedTokens.has(token)) matchCount++;
}
const score = matchCount / uniqueAddressTokens.size;
if (matchCount >= 3 && score >= 0.45 && (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore))) {
bestMatchCount = matchCount;
bestScore = score;
bestAddress = canonicalAddress;
}
}
return bestAddress ?? extracted;
}
// The delivery truck/signature line near the bottom of the table ("Truck No. B 9427 UXT
// PX HEAD OFFICE ANCOL : JL. ANCOL BARAT VIII...") names the actual destination store, when
// present. Scoping the match to just this line (and just nama_toko, not nama_toko+alamat)
// avoids the customer's own fixed head-office address elsewhere in the document being
// mistaken for the destination - that address is present on every document regardless of
// which store it's actually going to, so matching against it produces confident false
// positives for documents that don't specify a destination store name at all.
function extractTruckLineSnippet(fullText: string): string {
const m = fullText.match(/Truck\s*No\.?[\s\S]{0,180}/i);
return m ? m[0] : "";
}
// True when the printed "Order Untuk" text is actually the customer's company name - a common
// OCR layout jumble where the "Kepada Yth" and "Order Untuk" fields merge, meaning the real
// destination value was lost and the truck line is the better signal.
async function looksLikeCustomerName(text: string): Promise<boolean> {
if (!text) return false;
const textTokens = new Set(tokenize(text));
if (textTokens.size === 0) return false;
const customersRes = await query("SELECT name FROM customers");
for (const row of customersRes.rows) {
const companyName = String(row.name).split(/\bJL\.?\b|\bJALAN\b/i)[0];
const nameTokens = tokenize(companyName);
if (nameTokens.length === 0) continue;
let matchCount = 0;
for (const token of new Set(nameTokens)) {
if (textTokens.has(token)) matchCount++;
}
if (matchCount >= 1 && matchCount / new Set(nameTokens).size >= 0.5) return true;
}
return false;
}
export async function resolveStoreFromText(fullMarkdown: string): Promise<{ orderUntuk: string; alamat: string }> {
if (!fullMarkdown || fullMarkdown === "Not Found") {
return { orderUntuk: "", alamat: "" };
}
// The printed "Alamat" field is the customer's own (fixed) address, not the destination
// store's registered address - it stays the same across documents regardless of which
// store the truck line names. So alamat always comes from the literal printed text; only
// the store name itself benefits from being resolved to its canonical store_master form.
// The line right after "Alamat:" sometimes holds a region code ("DKI AREA") rather than the
// street address, with the real address following on the next line(s) - capture the whole
// block up to the item table and prefer the "JL./JALAN ..." street-address line within it.
const alamatBlockMatch = fullMarkdown.match(/Alamat\s*[:\-]?\s*([\s\S]+?)(?=<table|$)/i);
let literalAlamat = "";
if (alamatBlockMatch) {
const block = alamatBlockMatch[1];
const streetMatch = block.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
literalAlamat = (streetMatch ? streetMatch[0] : block).replace(/\s+/g, " ").trim();
}
literalAlamat = await canonicalizeCustomerAddress(literalAlamat);
// The printed "Order Untuk" value is the primary source for the store field: it usually
// holds a region designator ("DKI AREA", "PFM-KU") or a store name, and that's what the
// document actually says. Only when OCR jumbled it with the customer's company name (or
// lost it entirely) do we fall back to matching the truck/signature line against
// store_master to recover the destination store.
const orderMatch = fullMarkdown.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
const literalOrder = orderMatch ? orderMatch[1].trim() : "";
const orderIsUsable = literalOrder !== "" && !(await looksLikeCustomerName(literalOrder));
if (orderIsUsable) {
return { orderUntuk: literalOrder, alamat: literalAlamat };
}
const storeRes = await query("SELECT nama_toko, kode_toko, alamat FROM store_master");
const stores = storeRes.rows;
const truckSnippet = extractTruckLineSnippet(fullMarkdown);
const snippetTokens = new Set(tokenize(truckSnippet));
let bestStore: any = null;
let bestScore = 0;
let bestMatchCount = 0;
if (snippetTokens.size > 0) {
for (const store of stores) {
const storeTokens = tokenize(store.nama_toko);
if (storeTokens.length === 0) continue;
const uniqueStoreTokens = new Set(storeTokens);
let matchCount = 0;
for (const token of uniqueStoreTokens) {
if (snippetTokens.has(token)) matchCount++;
}
const score = matchCount / uniqueStoreTokens.size;
// Two distinct matching tokens minimum: single-token overlaps (e.g. a store whose only
// distinctive token is a common street/area word appearing in the snippet's address
// text) produce far too many confident false positives.
if (matchCount >= 2) {
if (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore)) {
bestMatchCount = matchCount;
bestScore = score;
bestStore = store;
}
}
}
}
if (bestStore) {
return { orderUntuk: bestStore.nama_toko, alamat: literalAlamat };
}
return { orderUntuk: literalOrder, alamat: literalAlamat };
}
export { pool };