Add mobile reliability fixes, Bahasa Indonesia UI, and continue OCR accuracy tuning

Reliability/PoC hardening: dedupe uploads by file_hash, surface editor sync
failures instead of a false success SnackBar with a retry-without-re-OCR path,
bound the OCR pipeline fetches with timeouts, share a single ApiClient/Dio
instance app-wide, tune capture JPEG quality, and add an opt-in
docker-compose.demo.yml for a production-mode run ahead of client demos.

Translate all Flutter-side user-facing text (screens, validators, SnackBars,
the printed delivery receipt, and shared API error messages) to Bahasa
Indonesia.

Also includes in-progress OCR parser/accuracy-tuning work from the same
session: table column/unit normalization fixes, store/customer master data,
accuracy history log, and test-image renaming/cleanup.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eRAsLqN9Sg1b9YPz22Lxz
This commit is contained in:
Rafhan Mazaya FathurrahmanandClaude Sonnet 5 committed 2026-07-04 19:41:24 +07:00
1 parent 095dd4cb8b
commit 4808a798fb
78 files changed
+8069 -584

No files matched your search

+132 -29
View File
@@ -67,45 +67,155 @@ export async function cleanupAndReindexItems(docId: number) {
}
}
export async function resolveStoreFromText(custInfo: string): Promise<{ orderUntuk: string; alamat: string }> {
if (!custInfo || custInfo === "Not Found") {
const STORE_STOPWORDS = new Set([
"dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw",
"jalan", "raya", "blok", "nomor", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"
]);
function tokenize(text: string): string[] {
return text.toLowerCase()
.replace(/[^a-z0-9\s]/g, " ")
.split(/\s+/)
.filter(w => w.length > 2 && !STORE_STOPWORDS.has(w));
}
// customers.name is stored as "CUSTOMER NAME, JL. street address..." - split on the first
// street-address marker to get just the canonical address portion.
function splitCustomerAddress(name: string): string {
const m = name.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
return (m ? m[0] : name).replace(/\s+/g, " ").trim();
}
// A noisy OCR'd address (varying per document due to misread letters) that recognizably
// belongs to a known customer should be reported as that customer's clean canonical address,
// rather than whatever garbled text this particular scan happened to produce.
async function canonicalizeCustomerAddress(extracted: string): Promise<string> {
if (!extracted) return extracted;
const extractedTokens = new Set(tokenize(extracted));
if (extractedTokens.size === 0) return extracted;
const customersRes = await query("SELECT name FROM customers");
let bestAddress: string | null = null;
let bestMatchCount = 0;
let bestScore = 0;
for (const row of customersRes.rows) {
const canonicalAddress = splitCustomerAddress(row.name);
const addressTokens = tokenize(canonicalAddress);
if (addressTokens.length === 0) continue;
const uniqueAddressTokens = new Set(addressTokens);
let matchCount = 0;
for (const token of uniqueAddressTokens) {
if (extractedTokens.has(token)) matchCount++;
}
const score = matchCount / uniqueAddressTokens.size;
if (matchCount >= 3 && score >= 0.45 && (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore))) {
bestMatchCount = matchCount;
bestScore = score;
bestAddress = canonicalAddress;
}
}
return bestAddress ?? extracted;
}
// The delivery truck/signature line near the bottom of the table ("Truck No. B 9427 UXT
// PX HEAD OFFICE ANCOL : JL. ANCOL BARAT VIII...") names the actual destination store, when
// present. Scoping the match to just this line (and just nama_toko, not nama_toko+alamat)
// avoids the customer's own fixed head-office address elsewhere in the document being
// mistaken for the destination - that address is present on every document regardless of
// which store it's actually going to, so matching against it produces confident false
// positives for documents that don't specify a destination store name at all.
function extractTruckLineSnippet(fullText: string): string {
const m = fullText.match(/Truck\s*No\.?[\s\S]{0,180}/i);
return m ? m[0] : "";
}
// True when the printed "Order Untuk" text is actually the customer's company name - a common
// OCR layout jumble where the "Kepada Yth" and "Order Untuk" fields merge, meaning the real
// destination value was lost and the truck line is the better signal.
async function looksLikeCustomerName(text: string): Promise<boolean> {
if (!text) return false;
const textTokens = new Set(tokenize(text));
if (textTokens.size === 0) return false;
const customersRes = await query("SELECT name FROM customers");
for (const row of customersRes.rows) {
const companyName = String(row.name).split(/\bJL\.?\b|\bJALAN\b/i)[0];
const nameTokens = tokenize(companyName);
if (nameTokens.length === 0) continue;
let matchCount = 0;
for (const token of new Set(nameTokens)) {
if (textTokens.has(token)) matchCount++;
}
if (matchCount >= 1 && matchCount / new Set(nameTokens).size >= 0.5) return true;
}
return false;
}
export async function resolveStoreFromText(fullMarkdown: string): Promise<{ orderUntuk: string; alamat: string }> {
if (!fullMarkdown || fullMarkdown === "Not Found") {
return { orderUntuk: "", alamat: "" };
}
// Load all stores
// The printed "Alamat" field is the customer's own (fixed) address, not the destination
// store's registered address - it stays the same across documents regardless of which
// store the truck line names. So alamat always comes from the literal printed text; only
// the store name itself benefits from being resolved to its canonical store_master form.
// The line right after "Alamat:" sometimes holds a region code ("DKI AREA") rather than the
// street address, with the real address following on the next line(s) - capture the whole
// block up to the item table and prefer the "JL./JALAN ..." street-address line within it.
const alamatBlockMatch = fullMarkdown.match(/Alamat\s*[:\-]?\s*([\s\S]+?)(?=<table|$)/i);
let literalAlamat = "";
if (alamatBlockMatch) {
const block = alamatBlockMatch[1];
const streetMatch = block.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
literalAlamat = (streetMatch ? streetMatch[0] : block).replace(/\s+/g, " ").trim();
}
literalAlamat = await canonicalizeCustomerAddress(literalAlamat);
// The printed "Order Untuk" value is the primary source for the store field: it usually
// holds a region designator ("DKI AREA", "PFM-KU") or a store name, and that's what the
// document actually says. Only when OCR jumbled it with the customer's company name (or
// lost it entirely) do we fall back to matching the truck/signature line against
// store_master to recover the destination store.
const orderMatch = fullMarkdown.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
const literalOrder = orderMatch ? orderMatch[1].trim() : "";
const orderIsUsable = literalOrder !== "" && !(await looksLikeCustomerName(literalOrder));
if (orderIsUsable) {
return { orderUntuk: literalOrder, alamat: literalAlamat };
}
const storeRes = await query("SELECT nama_toko, kode_toko, alamat FROM store_master");
const stores = storeRes.rows;
const ocrTokens = new Set(
custInfo.toLowerCase()
.replace(/[^a-z0-9\s]/g, " ")
.split(/\s+/)
.filter(w => w.length > 2 && !["dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw"].includes(w))
);
const truckSnippet = extractTruckLineSnippet(fullMarkdown);
const snippetTokens = new Set(tokenize(truckSnippet));
let bestStore: any = null;
let bestScore = 0;
let bestMatchCount = 0;
if (ocrTokens.size > 0) {
if (snippetTokens.size > 0) {
for (const store of stores) {
const searchStr = `${store.nama_toko} ${store.alamat}`.toLowerCase();
const storeTokens = searchStr
.replace(/[^a-z0-9\s]/g, " ")
.split(/\s+/)
.filter(w => w.length > 2 && !["dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw", "jalan", "raya", "blok", "nomor", "rt", "rw", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"].includes(w));
const storeTokens = tokenize(store.nama_toko);
if (storeTokens.length === 0) continue;
let matchCount = 0;
const uniqueStoreTokens = new Set(storeTokens);
let matchCount = 0;
for (const token of uniqueStoreTokens) {
if (ocrTokens.has(token)) {
matchCount++;
}
if (snippetTokens.has(token)) matchCount++;
}
const score = matchCount / uniqueStoreTokens.size;
// Two distinct matching tokens minimum: single-token overlaps (e.g. a store whose only
// distinctive token is a common street/area word appearing in the snippet's address
// text) produce far too many confident false positives.
if (matchCount >= 2) {
if (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore)) {
bestMatchCount = matchCount;
@@ -117,17 +227,10 @@ export async function resolveStoreFromText(custInfo: string): Promise<{ orderUnt
}
if (bestStore) {
return { orderUntuk: bestStore.nama_toko, alamat: bestStore.alamat };
return { orderUntuk: bestStore.nama_toko, alamat: literalAlamat };
}
// Fallback pattern matching
const orderMatch = custInfo.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
const alamatMatch = custInfo.match(/Alamat\s*[:\-]\s*([^\n]+)/i);
return {
orderUntuk: orderMatch ? orderMatch[1].trim() : "",
alamat: alamatMatch ? alamatMatch[1].trim() : ""
};
return { orderUntuk: literalOrder, alamat: literalAlamat };
}
export { pool };