Add mobile reliability fixes, Bahasa Indonesia UI, and continue OCR accuracy tuning
Reliability/PoC hardening: dedupe uploads by file_hash, surface editor sync failures instead of a false success SnackBar with a retry-without-re-OCR path, bound the OCR pipeline fetches with timeouts, share a single ApiClient/Dio instance app-wide, tune capture JPEG quality, and add an opt-in docker-compose.demo.yml for a production-mode run ahead of client demos. Translate all Flutter-side user-facing text (screens, validators, SnackBars, the printed delivery receipt, and shared API error messages) to Bahasa Indonesia. Also includes in-progress OCR parser/accuracy-tuning work from the same session: table column/unit normalization fixes, store/customer master data, accuracy history log, and test-image renaming/cleanup. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eRAsLqN9Sg1b9YPz22Lxz
This commit is contained in:
1 parent
095dd4cb8b
commit
4808a798fb
78 files changed
+8069
-584
No files matched your search
@@ -67,45 +67,155 @@ export async function cleanupAndReindexItems(docId: number) {
|
||||
}
|
||||
}
|
||||
|
||||
export async function resolveStoreFromText(custInfo: string): Promise<{ orderUntuk: string; alamat: string }> {
|
||||
if (!custInfo || custInfo === "Not Found") {
|
||||
const STORE_STOPWORDS = new Set([
|
||||
"dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw",
|
||||
"jalan", "raya", "blok", "nomor", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"
|
||||
]);
|
||||
|
||||
function tokenize(text: string): string[] {
|
||||
return text.toLowerCase()
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !STORE_STOPWORDS.has(w));
|
||||
}
|
||||
|
||||
// customers.name is stored as "CUSTOMER NAME, JL. street address..." - split on the first
|
||||
// street-address marker to get just the canonical address portion.
|
||||
function splitCustomerAddress(name: string): string {
|
||||
const m = name.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
|
||||
return (m ? m[0] : name).replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
// A noisy OCR'd address (varying per document due to misread letters) that recognizably
|
||||
// belongs to a known customer should be reported as that customer's clean canonical address,
|
||||
// rather than whatever garbled text this particular scan happened to produce.
|
||||
async function canonicalizeCustomerAddress(extracted: string): Promise<string> {
|
||||
if (!extracted) return extracted;
|
||||
|
||||
const extractedTokens = new Set(tokenize(extracted));
|
||||
if (extractedTokens.size === 0) return extracted;
|
||||
|
||||
const customersRes = await query("SELECT name FROM customers");
|
||||
|
||||
let bestAddress: string | null = null;
|
||||
let bestMatchCount = 0;
|
||||
let bestScore = 0;
|
||||
|
||||
for (const row of customersRes.rows) {
|
||||
const canonicalAddress = splitCustomerAddress(row.name);
|
||||
const addressTokens = tokenize(canonicalAddress);
|
||||
if (addressTokens.length === 0) continue;
|
||||
|
||||
const uniqueAddressTokens = new Set(addressTokens);
|
||||
let matchCount = 0;
|
||||
for (const token of uniqueAddressTokens) {
|
||||
if (extractedTokens.has(token)) matchCount++;
|
||||
}
|
||||
const score = matchCount / uniqueAddressTokens.size;
|
||||
|
||||
if (matchCount >= 3 && score >= 0.45 && (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore))) {
|
||||
bestMatchCount = matchCount;
|
||||
bestScore = score;
|
||||
bestAddress = canonicalAddress;
|
||||
}
|
||||
}
|
||||
|
||||
return bestAddress ?? extracted;
|
||||
}
|
||||
|
||||
// The delivery truck/signature line near the bottom of the table ("Truck No. B 9427 UXT
|
||||
// PX HEAD OFFICE ANCOL : JL. ANCOL BARAT VIII...") names the actual destination store, when
|
||||
// present. Scoping the match to just this line (and just nama_toko, not nama_toko+alamat)
|
||||
// avoids the customer's own fixed head-office address elsewhere in the document being
|
||||
// mistaken for the destination - that address is present on every document regardless of
|
||||
// which store it's actually going to, so matching against it produces confident false
|
||||
// positives for documents that don't specify a destination store name at all.
|
||||
function extractTruckLineSnippet(fullText: string): string {
|
||||
const m = fullText.match(/Truck\s*No\.?[\s\S]{0,180}/i);
|
||||
return m ? m[0] : "";
|
||||
}
|
||||
|
||||
// True when the printed "Order Untuk" text is actually the customer's company name - a common
|
||||
// OCR layout jumble where the "Kepada Yth" and "Order Untuk" fields merge, meaning the real
|
||||
// destination value was lost and the truck line is the better signal.
|
||||
async function looksLikeCustomerName(text: string): Promise<boolean> {
|
||||
if (!text) return false;
|
||||
const textTokens = new Set(tokenize(text));
|
||||
if (textTokens.size === 0) return false;
|
||||
|
||||
const customersRes = await query("SELECT name FROM customers");
|
||||
for (const row of customersRes.rows) {
|
||||
const companyName = String(row.name).split(/\bJL\.?\b|\bJALAN\b/i)[0];
|
||||
const nameTokens = tokenize(companyName);
|
||||
if (nameTokens.length === 0) continue;
|
||||
let matchCount = 0;
|
||||
for (const token of new Set(nameTokens)) {
|
||||
if (textTokens.has(token)) matchCount++;
|
||||
}
|
||||
if (matchCount >= 1 && matchCount / new Set(nameTokens).size >= 0.5) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
export async function resolveStoreFromText(fullMarkdown: string): Promise<{ orderUntuk: string; alamat: string }> {
|
||||
if (!fullMarkdown || fullMarkdown === "Not Found") {
|
||||
return { orderUntuk: "", alamat: "" };
|
||||
}
|
||||
|
||||
// Load all stores
|
||||
// The printed "Alamat" field is the customer's own (fixed) address, not the destination
|
||||
// store's registered address - it stays the same across documents regardless of which
|
||||
// store the truck line names. So alamat always comes from the literal printed text; only
|
||||
// the store name itself benefits from being resolved to its canonical store_master form.
|
||||
// The line right after "Alamat:" sometimes holds a region code ("DKI AREA") rather than the
|
||||
// street address, with the real address following on the next line(s) - capture the whole
|
||||
// block up to the item table and prefer the "JL./JALAN ..." street-address line within it.
|
||||
const alamatBlockMatch = fullMarkdown.match(/Alamat\s*[:\-]?\s*([\s\S]+?)(?=<table|$)/i);
|
||||
let literalAlamat = "";
|
||||
if (alamatBlockMatch) {
|
||||
const block = alamatBlockMatch[1];
|
||||
const streetMatch = block.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
|
||||
literalAlamat = (streetMatch ? streetMatch[0] : block).replace(/\s+/g, " ").trim();
|
||||
}
|
||||
literalAlamat = await canonicalizeCustomerAddress(literalAlamat);
|
||||
|
||||
// The printed "Order Untuk" value is the primary source for the store field: it usually
|
||||
// holds a region designator ("DKI AREA", "PFM-KU") or a store name, and that's what the
|
||||
// document actually says. Only when OCR jumbled it with the customer's company name (or
|
||||
// lost it entirely) do we fall back to matching the truck/signature line against
|
||||
// store_master to recover the destination store.
|
||||
const orderMatch = fullMarkdown.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
|
||||
const literalOrder = orderMatch ? orderMatch[1].trim() : "";
|
||||
const orderIsUsable = literalOrder !== "" && !(await looksLikeCustomerName(literalOrder));
|
||||
|
||||
if (orderIsUsable) {
|
||||
return { orderUntuk: literalOrder, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
const storeRes = await query("SELECT nama_toko, kode_toko, alamat FROM store_master");
|
||||
const stores = storeRes.rows;
|
||||
|
||||
const ocrTokens = new Set(
|
||||
custInfo.toLowerCase()
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !["dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw"].includes(w))
|
||||
);
|
||||
const truckSnippet = extractTruckLineSnippet(fullMarkdown);
|
||||
const snippetTokens = new Set(tokenize(truckSnippet));
|
||||
|
||||
let bestStore: any = null;
|
||||
let bestScore = 0;
|
||||
let bestMatchCount = 0;
|
||||
|
||||
if (ocrTokens.size > 0) {
|
||||
if (snippetTokens.size > 0) {
|
||||
for (const store of stores) {
|
||||
const searchStr = `${store.nama_toko} ${store.alamat}`.toLowerCase();
|
||||
const storeTokens = searchStr
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !["dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw", "jalan", "raya", "blok", "nomor", "rt", "rw", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"].includes(w));
|
||||
|
||||
const storeTokens = tokenize(store.nama_toko);
|
||||
if (storeTokens.length === 0) continue;
|
||||
|
||||
let matchCount = 0;
|
||||
const uniqueStoreTokens = new Set(storeTokens);
|
||||
let matchCount = 0;
|
||||
for (const token of uniqueStoreTokens) {
|
||||
if (ocrTokens.has(token)) {
|
||||
matchCount++;
|
||||
}
|
||||
if (snippetTokens.has(token)) matchCount++;
|
||||
}
|
||||
|
||||
const score = matchCount / uniqueStoreTokens.size;
|
||||
// Two distinct matching tokens minimum: single-token overlaps (e.g. a store whose only
|
||||
// distinctive token is a common street/area word appearing in the snippet's address
|
||||
// text) produce far too many confident false positives.
|
||||
if (matchCount >= 2) {
|
||||
if (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore)) {
|
||||
bestMatchCount = matchCount;
|
||||
@@ -117,17 +227,10 @@ export async function resolveStoreFromText(custInfo: string): Promise<{ orderUnt
|
||||
}
|
||||
|
||||
if (bestStore) {
|
||||
return { orderUntuk: bestStore.nama_toko, alamat: bestStore.alamat };
|
||||
return { orderUntuk: bestStore.nama_toko, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
// Fallback pattern matching
|
||||
const orderMatch = custInfo.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
|
||||
const alamatMatch = custInfo.match(/Alamat\s*[:\-]\s*([^\n]+)/i);
|
||||
|
||||
return {
|
||||
orderUntuk: orderMatch ? orderMatch[1].trim() : "",
|
||||
alamat: alamatMatch ? alamatMatch[1].trim() : ""
|
||||
};
|
||||
return { orderUntuk: literalOrder, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
export { pool };
|
||||
Reference in new issue
Block a user