Add mobile reliability fixes, Bahasa Indonesia UI, and continue OCR accuracy tuning
Reliability/PoC hardening: dedupe uploads by file_hash, surface editor sync failures instead of a false success SnackBar with a retry-without-re-OCR path, bound the OCR pipeline fetches with timeouts, share a single ApiClient/Dio instance app-wide, tune capture JPEG quality, and add an opt-in docker-compose.demo.yml for a production-mode run ahead of client demos. Translate all Flutter-side user-facing text (screens, validators, SnackBars, the printed delivery receipt, and shared API error messages) to Bahasa Indonesia. Also includes in-progress OCR parser/accuracy-tuning work from the same session: table column/unit normalization fixes, store/customer master data, accuracy history log, and test-image renaming/cleanup. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eRAsLqN9Sg1b9YPz22Lxz
This commit is contained in:
1 parent
095dd4cb8b
commit
4808a798fb
78 files changed
+8069
-584
No files matched your search
@@ -1,86 +0,0 @@
|
||||
import os
|
||||
import json
|
||||
import subprocess
|
||||
|
||||
def main():
|
||||
to_delete = [
|
||||
"1782872141829-rotated_1782872118246_rotated_1782872112296_rotated_1782872107513_CAP3481952560331460805.jpg",
|
||||
"1782909791970-sample_do.jpeg",
|
||||
"IMG_20260701_134646.jpg",
|
||||
"IMG_20260701_135910.jpg",
|
||||
"IMG_20260701_134704.jpg",
|
||||
"IMG_20260701_134704~2.jpg",
|
||||
"IMG_20260701_135950.jpg",
|
||||
"IMG_20260701_134533.jpg",
|
||||
"IMG_20260701_140005.jpg",
|
||||
"IMG_20260701_134817.jpg",
|
||||
"IMG_20260701_135923.jpg",
|
||||
"IMG_20260701_135900.jpg",
|
||||
"IMG_20260701_135938.jpg"
|
||||
]
|
||||
|
||||
base_dir = "backend"
|
||||
if not os.path.exists(base_dir):
|
||||
base_dir = "."
|
||||
|
||||
uploads_dir = os.path.join(base_dir, "uploads")
|
||||
sources_dir = os.path.join(base_dir, "sources")
|
||||
public_dir = os.path.join(base_dir, "pfm-web-app/public")
|
||||
|
||||
deleted_counts = {
|
||||
"manual_labels_json": 0,
|
||||
"source_images": 0,
|
||||
"public_images": 0
|
||||
}
|
||||
|
||||
# 1. Delete physical files
|
||||
for filename in to_delete:
|
||||
# manual label json
|
||||
ml_path = os.path.join(uploads_dir, f"manual_label_{filename}.json")
|
||||
if os.path.exists(ml_path):
|
||||
os.remove(ml_path)
|
||||
deleted_counts["manual_labels_json"] += 1
|
||||
print(f"Removed manual label: {ml_path}")
|
||||
|
||||
# source image
|
||||
img_source_path = os.path.join(sources_dir, "test-images", filename)
|
||||
if os.path.exists(img_source_path):
|
||||
os.remove(img_source_path)
|
||||
deleted_counts["source_images"] += 1
|
||||
print(f"Removed source image: {img_source_path}")
|
||||
|
||||
# public image
|
||||
img_public_path = os.path.join(public_dir, "test-images", filename)
|
||||
if os.path.exists(img_public_path):
|
||||
os.remove(img_public_path)
|
||||
deleted_counts["public_images"] += 1
|
||||
print(f"Removed public image: {img_public_path}")
|
||||
|
||||
# 2. Filter test_images_results.jsonl
|
||||
jsonl_path = os.path.join(uploads_dir, "test_images_results.jsonl")
|
||||
if os.path.exists(jsonl_path):
|
||||
remaining_lines = []
|
||||
with open(jsonl_path, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
data = json.loads(line)
|
||||
fn = data.get("filename")
|
||||
if fn not in to_delete:
|
||||
remaining_lines.append(line)
|
||||
except Exception as e:
|
||||
print(f"Error parsing line: {e}")
|
||||
remaining_lines.append(line)
|
||||
|
||||
with open(jsonl_path, "w", encoding="utf-8") as f:
|
||||
f.writelines(remaining_lines)
|
||||
print(f"Filtered results JSONL. Remaining entries: {len(remaining_lines)}")
|
||||
|
||||
print(f"\nCleanup results:")
|
||||
print(f"- Manual label JSON files deleted: {deleted_counts['manual_labels_json']}")
|
||||
print(f"- Source images deleted: {deleted_counts['source_images']}")
|
||||
print(f"- Public served images deleted: {deleted_counts['public_images']}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,52 +0,0 @@
|
||||
import os
|
||||
import json
|
||||
|
||||
def main():
|
||||
to_delete = [
|
||||
"IMG_20260701_134635.jpg",
|
||||
"IMG_20260701_134636.jpg"
|
||||
]
|
||||
|
||||
base_dir = "backend"
|
||||
if not os.path.exists(base_dir):
|
||||
base_dir = "."
|
||||
|
||||
uploads_dir = os.path.join(base_dir, "uploads")
|
||||
public_dir = os.path.join(base_dir, "pfm-web-app/public")
|
||||
|
||||
for filename in to_delete:
|
||||
# Delete manual label json
|
||||
ml_path = os.path.join(uploads_dir, f"manual_label_{filename}.json")
|
||||
if os.path.exists(ml_path):
|
||||
os.remove(ml_path)
|
||||
print(f"Deleted: {ml_path}")
|
||||
|
||||
# Delete public image
|
||||
img_public_path = os.path.join(public_dir, "test-images", filename)
|
||||
if os.path.exists(img_public_path):
|
||||
os.remove(img_public_path)
|
||||
print(f"Deleted: {img_public_path}")
|
||||
|
||||
# Filter test_images_results.jsonl
|
||||
jsonl_path = os.path.join(uploads_dir, "test_images_results.jsonl")
|
||||
if os.path.exists(jsonl_path):
|
||||
remaining_lines = []
|
||||
with open(jsonl_path, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
data = json.loads(line)
|
||||
fn = data.get("filename")
|
||||
if fn not in to_delete:
|
||||
remaining_lines.append(line)
|
||||
except Exception as e:
|
||||
print(f"Error parsing line: {e}")
|
||||
remaining_lines.append(line)
|
||||
|
||||
with open(jsonl_path, "w", encoding="utf-8") as f:
|
||||
f.writelines(remaining_lines)
|
||||
print(f"Filtered results JSONL. Remaining entries: {len(remaining_lines)}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,56 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import glob
|
||||
|
||||
def main():
|
||||
uploads_dir = "backend/uploads"
|
||||
sources_dir = "backend/sources"
|
||||
|
||||
if not os.path.exists(uploads_dir):
|
||||
# Fallback to local paths if run from backend folder
|
||||
uploads_dir = "uploads"
|
||||
sources_dir = "sources"
|
||||
|
||||
manual_pattern = os.path.join(uploads_dir, "manual_label_*.json")
|
||||
results_jsonl = os.path.join(uploads_dir, "test_images_results.jsonl")
|
||||
|
||||
output_manual = os.path.join(sources_dir, "manual_labels.json")
|
||||
output_ai = os.path.join(sources_dir, "ai_results.json")
|
||||
|
||||
# 1. Compile Manual Labels
|
||||
manual_files = glob.glob(manual_pattern)
|
||||
manual_list = []
|
||||
|
||||
for mf in sorted(manual_files):
|
||||
try:
|
||||
with open(mf, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
manual_list.append(data)
|
||||
except Exception as e:
|
||||
print(f"Error reading manual label {mf}: {e}")
|
||||
|
||||
with open(output_manual, "w", encoding="utf-8") as f:
|
||||
json.dump(manual_list, f, indent=2, ensure_ascii=False)
|
||||
print(f"Successfully compiled {len(manual_list)} manual labels into: {output_manual}")
|
||||
|
||||
# 2. Compile AI Results
|
||||
ai_list = []
|
||||
if os.path.exists(results_jsonl):
|
||||
try:
|
||||
with open(results_jsonl, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
if not line.strip():
|
||||
continue
|
||||
data = json.loads(line)
|
||||
ai_list.append(data)
|
||||
except Exception as e:
|
||||
print(f"Error reading JSONL results: {e}")
|
||||
|
||||
with open(output_ai, "w", encoding="utf-8") as f:
|
||||
json.dump(ai_list, f, indent=2, ensure_ascii=False)
|
||||
print(f"Successfully compiled {len(ai_list)} AI results into: {output_ai}")
|
||||
else:
|
||||
print(f"Warning: JSONL results file not found at {results_jsonl}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,9 @@
|
||||
services:
|
||||
db:
|
||||
ports:
|
||||
- "5432:5432"
|
||||
pipeline-api:
|
||||
ports:
|
||||
- "8090:8090"
|
||||
environment:
|
||||
- VLLM_SERVER_URL=http://paddleocr-vllm-server:8118/v1
|
||||
@@ -1,64 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import glob
|
||||
import csv
|
||||
|
||||
def main():
|
||||
uploads_dir = "backend/uploads"
|
||||
if not os.path.exists(uploads_dir):
|
||||
uploads_dir = "uploads"
|
||||
|
||||
csv_path = os.path.join(uploads_dir, "sku_master.csv")
|
||||
if not os.path.exists(csv_path):
|
||||
print(f"Error: sku_master.csv not found at {csv_path}")
|
||||
return
|
||||
|
||||
# Load SKU map
|
||||
sku_map = {}
|
||||
with open(csv_path, "r", encoding="utf-8") as f:
|
||||
reader = csv.reader(f)
|
||||
header = next(reader, None) # skip header
|
||||
for row in reader:
|
||||
if len(row) >= 2:
|
||||
sku_map[row[0].strip()] = row[1].strip()
|
||||
|
||||
print(f"Loaded {len(sku_map)} SKU mappings from master list.")
|
||||
|
||||
manual_pattern = os.path.join(uploads_dir, "manual_label_*.json")
|
||||
manual_files = glob.glob(manual_pattern)
|
||||
modified_files = 0
|
||||
filled_items = 0
|
||||
|
||||
for mf in manual_files:
|
||||
try:
|
||||
with open(mf, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
is_modified = False
|
||||
items = data.get("items", [])
|
||||
for item in items:
|
||||
sku = item.get("kodeBarang", "").strip()
|
||||
name = item.get("namaBarang", "").strip()
|
||||
|
||||
# If name is empty, try to fill it
|
||||
if not name and sku:
|
||||
master_name = sku_map.get(sku)
|
||||
if master_name:
|
||||
item["namaBarang"] = master_name
|
||||
is_modified = True
|
||||
filled_items += 1
|
||||
|
||||
if is_modified:
|
||||
with open(mf, "w", encoding="utf-8") as f:
|
||||
json.dump(data, f, indent=2, ensure_ascii=False)
|
||||
modified_files += 1
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error processing manual label {mf}: {e}")
|
||||
|
||||
print(f"Finished item name fill process.")
|
||||
print(f"Files updated: {modified_files}")
|
||||
print(f"Item names filled: {filled_items}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,33 +0,0 @@
|
||||
import os
|
||||
import glob
|
||||
|
||||
def main():
|
||||
uploads_dir = "backend/uploads"
|
||||
if not os.path.exists(uploads_dir):
|
||||
uploads_dir = "uploads"
|
||||
|
||||
manual_pattern = os.path.join(uploads_dir, "manual_label_*.json")
|
||||
manual_files = glob.glob(manual_pattern)
|
||||
fixed_count = 0
|
||||
|
||||
for mf in manual_files:
|
||||
try:
|
||||
with open(mf, "r", encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
|
||||
if "✓" in content:
|
||||
# Replace with the correct checkmark character
|
||||
content = content.replace("✓", "✓")
|
||||
|
||||
with open(mf, "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
|
||||
print(f"Fixed encoding corruption in: {mf}")
|
||||
fixed_count += 1
|
||||
except Exception as e:
|
||||
print(f"Error fixing {mf}: {e}")
|
||||
|
||||
print(f"Encoding fix completed. Files fixed: {fixed_count}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,50 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import glob
|
||||
|
||||
def main():
|
||||
uploads_dir = "backend/uploads"
|
||||
if not os.path.exists(uploads_dir):
|
||||
uploads_dir = "uploads"
|
||||
|
||||
manual_pattern = os.path.join(uploads_dir, "manual_label_*.json")
|
||||
manual_files = glob.glob(manual_pattern)
|
||||
|
||||
target_store = "PX HEAD OFFICE ANCOL"
|
||||
target_alamat = "JL. ANCOL BARAT VIII/1 KEL. ANCOL, KEC. PADEMANGAN JAKARTA UTARA, DKI JAKARTA"
|
||||
target_customer = "PT. PRIMAFOOD INTERNATIONAL"
|
||||
fixed_count = 0
|
||||
|
||||
for mf in manual_files:
|
||||
try:
|
||||
with open(mf, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
is_modified = False
|
||||
|
||||
# Fix customer
|
||||
current_customer = data.get("customer", "")
|
||||
if current_customer != target_customer:
|
||||
data["customer"] = target_customer
|
||||
is_modified = True
|
||||
|
||||
# Fix store
|
||||
store = data.get("store", "")
|
||||
if store == "DKI AREA":
|
||||
data["store"] = target_store
|
||||
data["alamat"] = target_alamat
|
||||
is_modified = True
|
||||
|
||||
if is_modified:
|
||||
with open(mf, "w", encoding="utf-8") as f:
|
||||
json.dump(data, f, indent=2, ensure_ascii=False)
|
||||
|
||||
print(f"Updated manual label: {mf}")
|
||||
fixed_count += 1
|
||||
except Exception as e:
|
||||
print(f"Error updating {mf}: {e}")
|
||||
|
||||
print(f"Finished updating manual label files. Total files updated: {fixed_count}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Binary file not shown.
@@ -6,20 +6,20 @@ const { Client } = require('pg');
|
||||
const BASE_URL = 'http://localhost:3000/api/parse';
|
||||
|
||||
const testFiles = [
|
||||
"1782870899198-sample_do.jpeg",
|
||||
"1782884664859-rotated_1782884661516_rotated_1782884658097_rotated_1782884654697_CAP5956452738616729026.jpg",
|
||||
"1782888127716-CAP6161747431193129837.jpg",
|
||||
"1782888211457-rotated_1782888204591_rotated_1782888199962_rotated_1782888195402_CAP7202641176939787142.jpg",
|
||||
"1782888609885-rotated_1782888593813_1000000454.jpg",
|
||||
"1782890303600-1000000465.jpg",
|
||||
"1782892728794-1000000466.jpg",
|
||||
"IMG_20260630_145445.jpg",
|
||||
"IMG_20260701_134446.jpg",
|
||||
"IMG_20260701_134504.jpg",
|
||||
"IMG_20260701_134555.jpg",
|
||||
"IMG_20260701_134555~2.jpg",
|
||||
"IMG_20260701_134646~2.jpg",
|
||||
"IMG_20260701_134810.jpg"
|
||||
"do-001.jpg",
|
||||
"do-002.jpg",
|
||||
"do-003.jpg",
|
||||
"do-004.jpg",
|
||||
"do-005.jpg",
|
||||
"do-006.jpg",
|
||||
"do-007.jpg",
|
||||
"do-008.jpg",
|
||||
"do-009.jpg",
|
||||
"do-010.jpg",
|
||||
"do-011.jpg",
|
||||
"do-012.jpg",
|
||||
"do-013.jpg",
|
||||
"do-014.jpg"
|
||||
];
|
||||
|
||||
function postJSON(url, body) {
|
||||
|
||||
@@ -0,0 +1,642 @@
|
||||
// OCR/parsing accuracy regression tool.
|
||||
//
|
||||
// Hits the live /api/parse endpoint for every image in backend/sources/test-images,
|
||||
// diffs the result against backend/sources/manual_labels.json at all three
|
||||
// post-processing stages (layer1RawRegex -> layer2Sanitized -> layer3Final) so a
|
||||
// mismatch can be attributed to the stage that introduced it, and appends a summary
|
||||
// to backend/sources/accuracy_history.jsonl for run-over-run regression tracking.
|
||||
//
|
||||
// Usage:
|
||||
// node scripts/accuracy-check.mts # reuse cached OCR (fast)
|
||||
// node scripts/accuracy-check.mts --refresh-ocr # force a fresh pipeline run for every image
|
||||
// node scripts/accuracy-check.mts --detail <filename> # print full per-stage breakdown for one image
|
||||
// node scripts/accuracy-check.mts --base-url http://localhost:3000
|
||||
|
||||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { execSync } from "node:child_process";
|
||||
|
||||
const __filename = fileURLToPath(import.meta.url);
|
||||
const __dirname = path.dirname(__filename);
|
||||
|
||||
const APP_ROOT = path.join(__dirname, "..");
|
||||
const SOURCES_DIR = path.join(APP_ROOT, "..", "sources");
|
||||
const TEST_IMAGES_DIR = path.join(SOURCES_DIR, "test-images");
|
||||
const LABELS_PATH = path.join(SOURCES_DIR, "manual_labels.json");
|
||||
const HISTORY_PATH = path.join(SOURCES_DIR, "accuracy_history.jsonl");
|
||||
|
||||
// A fresh (non-cached) pipeline run does the whole layout+recognition pass twice for any
|
||||
// image with >1deg tilt (see route.ts), and dense tables can push a single pass past 120s
|
||||
// on modest hardware (one observed case took ~194s) - keep this comfortably above that.
|
||||
const FETCH_TIMEOUT_MS = 300_000;
|
||||
|
||||
interface Item {
|
||||
kodeBarang: string;
|
||||
namaBarang: string;
|
||||
banyak: string;
|
||||
jumlah: string;
|
||||
}
|
||||
|
||||
interface GroundTruth {
|
||||
filename: string;
|
||||
noPO: string;
|
||||
noSO: string;
|
||||
noDO: string;
|
||||
tanggal: string;
|
||||
customer: string;
|
||||
store: string;
|
||||
alamat: string;
|
||||
plat: string;
|
||||
items: Item[];
|
||||
}
|
||||
|
||||
type StageName = "layer1RawRegex" | "layer2Sanitized" | "layer3Final";
|
||||
const STAGES: StageName[] = ["layer1RawRegex", "layer2Sanitized", "layer3Final"];
|
||||
const STAGE_LABEL: Record<StageName, string> = {
|
||||
layer1RawRegex: "L1raw",
|
||||
layer2Sanitized: "L2san",
|
||||
layer3Final: "L3final",
|
||||
};
|
||||
|
||||
// Header fields present at every stage (untouched or normalized by parseDOMetadata/sanitizeParsedMetadata).
|
||||
const HEADER_FIELD_MAP: [gtKey: keyof GroundTruth, stageKey: string][] = [
|
||||
["noPO", "noPO"],
|
||||
["noSO", "noSO"],
|
||||
["noDO", "noDO"],
|
||||
["tanggal", "tanggal"],
|
||||
["customer", "customerInfo"],
|
||||
["plat", "platTruk"],
|
||||
];
|
||||
|
||||
// Only resolved on layer3Final (store/address lookup against the store master DB happens after sanitize).
|
||||
const STORE_FIELD_MAP: [gtKey: keyof GroundTruth, stageKey: string][] = [
|
||||
["store", "orderUntuk"],
|
||||
["alamat", "alamat"],
|
||||
];
|
||||
|
||||
const ITEM_FIELDS: (keyof Item)[] = ["kodeBarang", "namaBarang", "banyak", "jumlah"];
|
||||
|
||||
const FIELD_ORDER = [
|
||||
"noPO", "noSO", "noDO", "tanggal", "customer", "plat", "store", "alamat",
|
||||
"itemCount", "kodeBarang", "namaBarang", "banyak", "jumlah",
|
||||
];
|
||||
|
||||
interface Check {
|
||||
field: string;
|
||||
stage: StageName;
|
||||
match: boolean;
|
||||
}
|
||||
|
||||
interface DetailFieldRow {
|
||||
field: string;
|
||||
gt: string;
|
||||
values: Record<StageName, string | null>;
|
||||
matches: Record<StageName, boolean | null>;
|
||||
}
|
||||
|
||||
interface DetailItemRow {
|
||||
rowIndex: number;
|
||||
field: string;
|
||||
gt: string;
|
||||
values: Record<StageName, string | null>;
|
||||
matches: Record<StageName, boolean>;
|
||||
}
|
||||
|
||||
interface ImageDetail {
|
||||
headerRows: DetailFieldRow[];
|
||||
itemRows: DetailItemRow[];
|
||||
}
|
||||
|
||||
interface HistoryEntry {
|
||||
timestamp: string;
|
||||
commit: string;
|
||||
refreshOcr: boolean;
|
||||
imageCount: number;
|
||||
failedImages: string[];
|
||||
overall: Record<StageName, number>;
|
||||
fields: Record<string, Record<StageName, number>>;
|
||||
perImage: Record<string, number>;
|
||||
}
|
||||
|
||||
interface CliArgs {
|
||||
refreshOcr: boolean;
|
||||
detail: string | null;
|
||||
baseUrl: string;
|
||||
dumpJson: string | null;
|
||||
}
|
||||
|
||||
function parseArgs(argv: string[]): CliArgs {
|
||||
const args: CliArgs = {
|
||||
refreshOcr: false,
|
||||
detail: null,
|
||||
baseUrl: process.env.ACCURACY_BASE_URL || "http://localhost:3000",
|
||||
dumpJson: null,
|
||||
};
|
||||
for (let i = 0; i < argv.length; i++) {
|
||||
const a = argv[i];
|
||||
if (a === "--refresh-ocr") {
|
||||
args.refreshOcr = true;
|
||||
} else if (a === "--detail") {
|
||||
args.detail = argv[++i] ?? null;
|
||||
} else if (a === "--base-url") {
|
||||
args.baseUrl = argv[++i] ?? args.baseUrl;
|
||||
} else if (a === "--dump-json") {
|
||||
args.dumpJson = argv[++i] ?? null;
|
||||
} else if (a === "--help" || a === "-h") {
|
||||
printHelp();
|
||||
process.exit(0);
|
||||
}
|
||||
}
|
||||
return args;
|
||||
}
|
||||
|
||||
function printHelp() {
|
||||
console.log(`OCR accuracy regression tool
|
||||
|
||||
Usage:
|
||||
node scripts/accuracy-check.mts [options]
|
||||
|
||||
Options:
|
||||
--refresh-ocr Force a fresh pipeline run for every test image (requires the
|
||||
pipeline-api service to be up), instead of reusing cached OCR.
|
||||
--detail <filename> Print the full per-stage breakdown for one image.
|
||||
--base-url <url> Base URL of the running Next dev server (default http://localhost:3000).
|
||||
--dump-json <path> Write full per-image ground-truth-vs-AI detail (all stages) as JSON.
|
||||
--help Show this message.
|
||||
`);
|
||||
}
|
||||
|
||||
function norm(v: unknown): string {
|
||||
if (v === null || v === undefined) return "";
|
||||
// Spacing around punctuation ("PT. PRIMAFOOD" vs "PT.PRIMAFOOD") is a formatting
|
||||
// difference, not an extraction error - the ground-truth labels themselves are
|
||||
// inconsistent about it, so neutralize it before comparing.
|
||||
return String(v)
|
||||
.replace(/\s*([.,:;\/])\s*/g, "$1")
|
||||
.replace(/\s+/g, " ")
|
||||
.trim()
|
||||
.toUpperCase();
|
||||
}
|
||||
|
||||
function isMatch(a: unknown, b: unknown): boolean {
|
||||
return norm(a) === norm(b);
|
||||
}
|
||||
|
||||
function loadGroundTruth(): Map<string, GroundTruth> {
|
||||
const raw = fs.existsSync(LABELS_PATH) ? fs.readFileSync(LABELS_PATH, "utf8") : "";
|
||||
const arr: GroundTruth[] = raw.trim() ? JSON.parse(raw) : [];
|
||||
const map = new Map<string, GroundTruth>();
|
||||
for (const entry of arr) map.set(entry.filename, entry);
|
||||
return map;
|
||||
}
|
||||
|
||||
function listTestImages(): string[] {
|
||||
if (!fs.existsSync(TEST_IMAGES_DIR)) return [];
|
||||
return fs
|
||||
.readdirSync(TEST_IMAGES_DIR)
|
||||
.filter(f => [".jpg", ".jpeg", ".png"].includes(path.extname(f).toLowerCase()))
|
||||
.sort();
|
||||
}
|
||||
|
||||
async function checkServerReachable(baseUrl: string): Promise<void> {
|
||||
try {
|
||||
const res = await fetchWithTimeout(`${baseUrl}/api/manual-images`, {}, 10_000);
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
} catch (err) {
|
||||
throw new Error(
|
||||
`Next dev server not reachable at ${baseUrl} (${(err as Error).message}). ` +
|
||||
`Start it with "npm run dev" in backend/pfm-web-app.`
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchWithTimeout(url: string, init: RequestInit, timeoutMs: number): Promise<Response> {
|
||||
const controller = new AbortController();
|
||||
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
||||
try {
|
||||
return await fetch(url, { ...init, signal: controller.signal });
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchParse(baseUrl: string, filename: string): Promise<any> {
|
||||
const res = await fetchWithTimeout(
|
||||
`${baseUrl}/api/parse`,
|
||||
{
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ filename }),
|
||||
},
|
||||
FETCH_TIMEOUT_MS
|
||||
);
|
||||
const json = await res.json();
|
||||
if (!res.ok) {
|
||||
throw new Error(json?.error || `HTTP ${res.status}`);
|
||||
}
|
||||
if (!json.postProcessingDetails) {
|
||||
throw new Error("Response missing postProcessingDetails (parse route may have hit its DB-save fallback)");
|
||||
}
|
||||
return json;
|
||||
}
|
||||
|
||||
async function refreshOcrCache(filenames: string[]): Promise<void> {
|
||||
const { Pool } = await import("pg");
|
||||
const pool = new Pool({
|
||||
host: process.env.PGHOST || "localhost",
|
||||
port: parseInt(process.env.PGPORT || "5432", 10),
|
||||
user: process.env.PGUSER || "postgres",
|
||||
password: process.env.PGPASSWORD || "postgres",
|
||||
database: process.env.PGDATABASE || "dopfm",
|
||||
});
|
||||
try {
|
||||
for (const filename of filenames) {
|
||||
await pool.query("DELETE FROM documents WHERE filename = $1", [filename]);
|
||||
}
|
||||
} finally {
|
||||
await pool.end();
|
||||
}
|
||||
}
|
||||
|
||||
function getStageObj(parsed: any, stage: StageName): any {
|
||||
return parsed?.postProcessingDetails?.[stage] || {};
|
||||
}
|
||||
|
||||
function compareImage(gt: GroundTruth, parsed: any): { checks: Check[]; detail: ImageDetail } {
|
||||
const checks: Check[] = [];
|
||||
const headerRows: DetailFieldRow[] = [];
|
||||
|
||||
for (const [gtKey, stageKey] of HEADER_FIELD_MAP) {
|
||||
const gtVal = String(gt[gtKey] ?? "");
|
||||
const row: DetailFieldRow = { field: gtKey, gt: gtVal, values: {} as any, matches: {} as any };
|
||||
for (const stage of STAGES) {
|
||||
const obj = getStageObj(parsed, stage);
|
||||
const val = obj?.[stageKey] ?? null;
|
||||
const match = isMatch(gtVal, val);
|
||||
checks.push({ field: gtKey, stage, match });
|
||||
row.values[stage] = val;
|
||||
row.matches[stage] = match;
|
||||
}
|
||||
headerRows.push(row);
|
||||
}
|
||||
|
||||
for (const [gtKey, stageKey] of STORE_FIELD_MAP) {
|
||||
const gtVal = String(gt[gtKey] ?? "");
|
||||
const row: DetailFieldRow = { field: gtKey, gt: gtVal, values: {} as any, matches: {} as any };
|
||||
for (const stage of STAGES) {
|
||||
const obj = getStageObj(parsed, stage);
|
||||
if (!(stageKey in obj)) {
|
||||
// Not resolved yet at this stage (store/address lookup only runs once, for layer3Final).
|
||||
row.values[stage] = null;
|
||||
row.matches[stage] = null;
|
||||
continue;
|
||||
}
|
||||
const val = obj[stageKey];
|
||||
const match = isMatch(gtVal, val);
|
||||
checks.push({ field: gtKey, stage, match });
|
||||
row.values[stage] = val;
|
||||
row.matches[stage] = match;
|
||||
}
|
||||
headerRows.push(row);
|
||||
}
|
||||
|
||||
const gtCount = gt.items.length;
|
||||
const countRow: DetailFieldRow = { field: "itemCount", gt: String(gtCount), values: {} as any, matches: {} as any };
|
||||
for (const stage of STAGES) {
|
||||
const obj = getStageObj(parsed, stage);
|
||||
const items: Item[] = obj?.items || [];
|
||||
const match = items.length === gtCount;
|
||||
checks.push({ field: "itemCount", stage, match });
|
||||
countRow.values[stage] = String(items.length);
|
||||
countRow.matches[stage] = match;
|
||||
}
|
||||
headerRows.push(countRow);
|
||||
|
||||
const itemRows: DetailItemRow[] = [];
|
||||
for (let i = 0; i < gtCount; i++) {
|
||||
const gtItem = gt.items[i];
|
||||
for (const field of ITEM_FIELDS) {
|
||||
const rowDetail: DetailItemRow = {
|
||||
rowIndex: i,
|
||||
field,
|
||||
gt: String(gtItem[field] ?? ""),
|
||||
values: {} as any,
|
||||
matches: {} as any,
|
||||
};
|
||||
for (const stage of STAGES) {
|
||||
const obj = getStageObj(parsed, stage);
|
||||
const items: Item[] = obj?.items || [];
|
||||
const stageItem = items[i];
|
||||
const val = stageItem ? stageItem[field] : null;
|
||||
const match = !!stageItem && isMatch(gtItem[field], val);
|
||||
checks.push({ field, stage, match });
|
||||
rowDetail.values[stage] = val ?? null;
|
||||
rowDetail.matches[stage] = match;
|
||||
}
|
||||
itemRows.push(rowDetail);
|
||||
}
|
||||
}
|
||||
|
||||
return { checks, detail: { headerRows, itemRows } };
|
||||
}
|
||||
|
||||
function pct(correct: number, total: number): number {
|
||||
return total === 0 ? 0 : (correct / total) * 100;
|
||||
}
|
||||
|
||||
function aggregateByField(allChecks: Check[]): Record<string, Record<StageName, { correct: number; total: number }>> {
|
||||
const agg: Record<string, Record<StageName, { correct: number; total: number }>> = {};
|
||||
for (const c of allChecks) {
|
||||
if (!agg[c.field]) {
|
||||
agg[c.field] = {
|
||||
layer1RawRegex: { correct: 0, total: 0 },
|
||||
layer2Sanitized: { correct: 0, total: 0 },
|
||||
layer3Final: { correct: 0, total: 0 },
|
||||
};
|
||||
}
|
||||
agg[c.field][c.stage].total++;
|
||||
if (c.match) agg[c.field][c.stage].correct++;
|
||||
}
|
||||
return agg;
|
||||
}
|
||||
|
||||
function aggregateOverall(allChecks: Check[]): Record<StageName, { correct: number; total: number }> {
|
||||
const overall: Record<StageName, { correct: number; total: number }> = {
|
||||
layer1RawRegex: { correct: 0, total: 0 },
|
||||
layer2Sanitized: { correct: 0, total: 0 },
|
||||
layer3Final: { correct: 0, total: 0 },
|
||||
};
|
||||
for (const c of allChecks) {
|
||||
overall[c.stage].total++;
|
||||
if (c.match) overall[c.stage].correct++;
|
||||
}
|
||||
return overall;
|
||||
}
|
||||
|
||||
function getGitCommit(): string {
|
||||
try {
|
||||
return execSync("git rev-parse --short HEAD", { cwd: APP_ROOT }).toString().trim();
|
||||
} catch {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
function loadLastHistoryEntry(): HistoryEntry | null {
|
||||
if (!fs.existsSync(HISTORY_PATH)) return null;
|
||||
const lines = fs.readFileSync(HISTORY_PATH, "utf8").trim().split("\n").filter(Boolean);
|
||||
if (lines.length === 0) return null;
|
||||
try {
|
||||
return JSON.parse(lines[lines.length - 1]);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function appendHistory(entry: HistoryEntry) {
|
||||
fs.appendFileSync(HISTORY_PATH, JSON.stringify(entry) + "\n", "utf8");
|
||||
}
|
||||
|
||||
function fmtPct(n: number): string {
|
||||
return `${n.toFixed(1)}%`;
|
||||
}
|
||||
|
||||
function fmtDelta(curr: number, prev: number | undefined): string {
|
||||
if (prev === undefined) return "";
|
||||
const d = curr - prev;
|
||||
if (Math.abs(d) < 0.05) return " ±0.0";
|
||||
const sign = d > 0 ? "+" : "";
|
||||
return `${sign}${d.toFixed(1)}`;
|
||||
}
|
||||
|
||||
function printSummary(
|
||||
fieldAgg: Record<string, Record<StageName, { correct: number; total: number }>>,
|
||||
overallAgg: Record<StageName, { correct: number; total: number }>,
|
||||
perImagePct: Record<string, number>,
|
||||
prev: HistoryEntry | null,
|
||||
imageCount: number,
|
||||
failedImages: string[]
|
||||
) {
|
||||
console.log("");
|
||||
console.log(`OCR accuracy summary (${imageCount} images${failedImages.length ? `, ${failedImages.length} failed` : ""})`);
|
||||
console.log("");
|
||||
|
||||
const fieldCol = 12;
|
||||
const numCol = 9;
|
||||
const header =
|
||||
"Field".padEnd(fieldCol) +
|
||||
STAGES.map(s => STAGE_LABEL[s].padStart(numCol)).join("") +
|
||||
" Δ(L3 vs prev)".padStart(16);
|
||||
console.log(header);
|
||||
console.log("-".repeat(header.length));
|
||||
|
||||
for (const field of FIELD_ORDER) {
|
||||
const stat = fieldAgg[field];
|
||||
if (!stat) continue;
|
||||
const cells = STAGES.map(s => {
|
||||
const st = stat[s];
|
||||
return st.total === 0 ? "n/a".padStart(numCol) : fmtPct(pct(st.correct, st.total)).padStart(numCol);
|
||||
}).join("");
|
||||
const currL3 = pct(stat.layer3Final.correct, stat.layer3Final.total);
|
||||
const prevL3 = prev?.fields?.[field]?.layer3Final;
|
||||
const delta = fmtDelta(currL3, prevL3);
|
||||
console.log(field.padEnd(fieldCol) + cells + delta.padStart(16));
|
||||
}
|
||||
|
||||
console.log("-".repeat(header.length));
|
||||
const overallCells = STAGES.map(s => fmtPct(pct(overallAgg[s].correct, overallAgg[s].total)).padStart(numCol)).join("");
|
||||
const overallL3 = pct(overallAgg.layer3Final.correct, overallAgg.layer3Final.total);
|
||||
const overallDelta = fmtDelta(overallL3, prev?.overall?.layer3Final);
|
||||
console.log("OVERALL".padEnd(fieldCol) + overallCells + overallDelta.padStart(16));
|
||||
|
||||
if (failedImages.length) {
|
||||
console.log("");
|
||||
console.log(`Failed to parse: ${failedImages.join(", ")}`);
|
||||
}
|
||||
|
||||
if (prev) {
|
||||
const fieldRegressions: string[] = [];
|
||||
const fieldImprovements: string[] = [];
|
||||
for (const field of FIELD_ORDER) {
|
||||
const stat = fieldAgg[field];
|
||||
if (!stat) continue;
|
||||
const currL3 = pct(stat.layer3Final.correct, stat.layer3Final.total);
|
||||
const prevL3 = prev.fields?.[field]?.layer3Final;
|
||||
if (prevL3 === undefined) continue;
|
||||
const d = currL3 - prevL3;
|
||||
if (d <= -0.05) fieldRegressions.push(`${field} ${fmtDelta(currL3, prevL3)}`);
|
||||
else if (d >= 0.05) fieldImprovements.push(`${field} ${fmtDelta(currL3, prevL3)}`);
|
||||
}
|
||||
if (fieldRegressions.length) console.log(`\nField regressions (L3): ${fieldRegressions.join(", ")}`);
|
||||
if (fieldImprovements.length) console.log(`Field improvements (L3): ${fieldImprovements.join(", ")}`);
|
||||
|
||||
const imageRegressions: string[] = [];
|
||||
const imageImprovements: string[] = [];
|
||||
for (const [filename, currPct] of Object.entries(perImagePct)) {
|
||||
const prevPct = prev.perImage?.[filename];
|
||||
if (prevPct === undefined) continue;
|
||||
const d = currPct - prevPct;
|
||||
if (d <= -0.5) imageRegressions.push(`${filename} ${fmtDelta(currPct, prevPct)}`);
|
||||
else if (d >= 0.5) imageImprovements.push(`${filename} ${fmtDelta(currPct, prevPct)}`);
|
||||
}
|
||||
if (imageRegressions.length) console.log(`\nImage regressions: ${imageRegressions.join(", ")}`);
|
||||
if (imageImprovements.length) console.log(`Image improvements: ${imageImprovements.join(", ")}`);
|
||||
|
||||
console.log(`\n(vs run at ${prev.timestamp}${prev.commit !== "unknown" ? `, commit ${prev.commit}` : ""})`);
|
||||
} else {
|
||||
console.log("\n(no previous run in accuracy_history.jsonl — this is the baseline)");
|
||||
}
|
||||
console.log("");
|
||||
}
|
||||
|
||||
function checkMark(v: boolean | null): string {
|
||||
if (v === null) return "-";
|
||||
return v ? "✓" : "✗";
|
||||
}
|
||||
|
||||
function printDetail(filename: string, detail: ImageDetail | undefined) {
|
||||
console.log(`\n=== Detail: ${filename} ===\n`);
|
||||
if (!detail) {
|
||||
console.log("No data for this file (it may not exist in test-images/ or manual_labels.json, or parsing failed this run).\n");
|
||||
return;
|
||||
}
|
||||
|
||||
const fieldCol = 12;
|
||||
const valCol = 34;
|
||||
console.log("Field".padEnd(fieldCol) + STAGES.map(s => STAGE_LABEL[s].padEnd(valCol)).join(""));
|
||||
console.log(`(ground truth shown per row)`);
|
||||
for (const row of detail.headerRows) {
|
||||
console.log(`- ${row.field} = "${row.gt}"`);
|
||||
for (const stage of STAGES) {
|
||||
const val = row.values[stage];
|
||||
const mark = checkMark(row.matches[stage]);
|
||||
const label = STAGE_LABEL[stage].padEnd(8);
|
||||
console.log(` ${mark} ${label} ${val === null ? "(n/a)" : `"${val}"`}`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log("\nItems:");
|
||||
let currentRow = -1;
|
||||
for (const row of detail.itemRows) {
|
||||
if (row.rowIndex !== currentRow) {
|
||||
currentRow = row.rowIndex;
|
||||
console.log(` Row ${currentRow + 1}:`);
|
||||
}
|
||||
console.log(` - ${row.field} = "${row.gt}"`);
|
||||
for (const stage of STAGES) {
|
||||
const val = row.values[stage];
|
||||
const mark = checkMark(row.matches[stage]);
|
||||
const label = STAGE_LABEL[stage].padEnd(8);
|
||||
console.log(` ${mark} ${label} ${val === null ? "(missing row)" : `"${val}"`}`);
|
||||
}
|
||||
}
|
||||
console.log("");
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const args = parseArgs(process.argv.slice(2));
|
||||
|
||||
await checkServerReachable(args.baseUrl);
|
||||
|
||||
const gtMap = loadGroundTruth();
|
||||
const testImages = listTestImages();
|
||||
|
||||
if (testImages.length === 0) {
|
||||
console.error(`No test images found in ${TEST_IMAGES_DIR}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const missingGt = testImages.filter(f => !gtMap.has(f));
|
||||
if (missingGt.length) {
|
||||
console.warn(`Warning: ${missingGt.length} test image(s) have no ground truth entry and will be skipped: ${missingGt.join(", ")}`);
|
||||
}
|
||||
|
||||
if (args.refreshOcr) {
|
||||
console.log(`--refresh-ocr: clearing cached OCR for ${testImages.length} images (requires pipeline-api to be reachable)...`);
|
||||
await refreshOcrCache(testImages);
|
||||
}
|
||||
|
||||
const allChecks: Check[] = [];
|
||||
const perImagePct: Record<string, number> = {};
|
||||
const detailsByFile: Record<string, ImageDetail> = {};
|
||||
const failedImages: string[] = [];
|
||||
|
||||
for (const filename of testImages) {
|
||||
const gt = gtMap.get(filename);
|
||||
if (!gt) continue;
|
||||
process.stdout.write(`Parsing ${filename}... `);
|
||||
try {
|
||||
const parsed = await fetchParse(args.baseUrl, filename);
|
||||
const { checks, detail } = compareImage(gt, parsed);
|
||||
allChecks.push(...checks);
|
||||
detailsByFile[filename] = detail;
|
||||
const l3Checks = checks.filter(c => c.stage === "layer3Final");
|
||||
const imgPct = pct(l3Checks.filter(c => c.match).length, l3Checks.length);
|
||||
perImagePct[filename] = imgPct;
|
||||
console.log(`done (${imgPct.toFixed(0)}%)`);
|
||||
} catch (err) {
|
||||
console.log(`FAILED (${(err as Error).message})`);
|
||||
failedImages.push(filename);
|
||||
}
|
||||
}
|
||||
|
||||
const fieldAgg = aggregateByField(allChecks);
|
||||
const overallAgg = aggregateOverall(allChecks);
|
||||
const prev = loadLastHistoryEntry();
|
||||
|
||||
printSummary(fieldAgg, overallAgg, perImagePct, prev, testImages.length - missingGt.length, failedImages);
|
||||
|
||||
if (args.detail) {
|
||||
printDetail(args.detail, detailsByFile[args.detail]);
|
||||
}
|
||||
|
||||
if (args.dumpJson) {
|
||||
fs.writeFileSync(
|
||||
args.dumpJson,
|
||||
JSON.stringify(
|
||||
{
|
||||
timestamp: new Date().toISOString(),
|
||||
commit: getGitCommit(),
|
||||
imageCount: testImages.length - missingGt.length,
|
||||
overallL3: pct(overallAgg.layer3Final.correct, overallAgg.layer3Final.total),
|
||||
perImagePct,
|
||||
details: detailsByFile,
|
||||
},
|
||||
null,
|
||||
2
|
||||
),
|
||||
"utf8"
|
||||
);
|
||||
console.log(`\nWrote full per-image detail to ${args.dumpJson}`);
|
||||
}
|
||||
|
||||
const historyEntry: HistoryEntry = {
|
||||
timestamp: new Date().toISOString(),
|
||||
commit: getGitCommit(),
|
||||
refreshOcr: args.refreshOcr,
|
||||
imageCount: testImages.length - missingGt.length,
|
||||
failedImages,
|
||||
overall: {
|
||||
layer1RawRegex: pct(overallAgg.layer1RawRegex.correct, overallAgg.layer1RawRegex.total),
|
||||
layer2Sanitized: pct(overallAgg.layer2Sanitized.correct, overallAgg.layer2Sanitized.total),
|
||||
layer3Final: pct(overallAgg.layer3Final.correct, overallAgg.layer3Final.total),
|
||||
},
|
||||
fields: Object.fromEntries(
|
||||
Object.entries(fieldAgg).map(([field, stat]) => [
|
||||
field,
|
||||
{
|
||||
layer1RawRegex: pct(stat.layer1RawRegex.correct, stat.layer1RawRegex.total),
|
||||
layer2Sanitized: pct(stat.layer2Sanitized.correct, stat.layer2Sanitized.total),
|
||||
layer3Final: pct(stat.layer3Final.correct, stat.layer3Final.total),
|
||||
},
|
||||
])
|
||||
),
|
||||
perImage: perImagePct,
|
||||
};
|
||||
appendHistory(historyEntry);
|
||||
}
|
||||
|
||||
main().catch(err => {
|
||||
console.error("Fatal error:", err);
|
||||
process.exit(1);
|
||||
});
|
||||
@@ -4,7 +4,7 @@ import path from "path";
|
||||
|
||||
export async function GET() {
|
||||
try {
|
||||
const dirPath = path.join(process.cwd(), "public/test-images");
|
||||
const dirPath = path.join(process.cwd(), "..", "sources", "test-images");
|
||||
if (!fs.existsSync(dirPath)) {
|
||||
return NextResponse.json({ files: [] });
|
||||
}
|
||||
|
||||
@@ -3,33 +3,39 @@ import { query } from "../../../db";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
const UPLOADS_DIR = "/uploads";
|
||||
const LABELS_PATH = path.join(process.cwd(), "..", "sources", "manual_labels.json");
|
||||
|
||||
function getFilePath(filename: string) {
|
||||
// Sanitize filename to prevent directory traversal
|
||||
const safeFilename = path.basename(filename);
|
||||
return path.join(UPLOADS_DIR, `manual_label_${safeFilename}.json`);
|
||||
function readLabels(): any[] {
|
||||
if (!fs.existsSync(LABELS_PATH)) {
|
||||
return [];
|
||||
}
|
||||
const raw = fs.readFileSync(LABELS_PATH, "utf8");
|
||||
return raw.trim() ? JSON.parse(raw) : [];
|
||||
}
|
||||
|
||||
function writeLabels(labels: any[]) {
|
||||
fs.writeFileSync(LABELS_PATH, JSON.stringify(labels, null, 2), "utf8");
|
||||
}
|
||||
|
||||
export async function GET(req: NextRequest) {
|
||||
try {
|
||||
const { searchParams } = new URL(req.url);
|
||||
const filename = searchParams.get("filename");
|
||||
|
||||
|
||||
if (!filename) {
|
||||
return NextResponse.json({ error: "Filename parameter is required" }, { status: 400 });
|
||||
}
|
||||
|
||||
const filePath = getFilePath(filename);
|
||||
|
||||
if (fs.existsSync(filePath)) {
|
||||
const fileData = fs.readFileSync(filePath, "utf8");
|
||||
return NextResponse.json(JSON.parse(fileData));
|
||||
|
||||
const safeFilename = path.basename(filename);
|
||||
const labels = readLabels();
|
||||
const existing = labels.find(l => l.filename === safeFilename);
|
||||
|
||||
if (existing) {
|
||||
return NextResponse.json(existing);
|
||||
}
|
||||
|
||||
// Try fallback to automated parser results in database
|
||||
try {
|
||||
const safeFilename = path.basename(filename);
|
||||
const docRes = await query(
|
||||
"SELECT id, metadata FROM documents WHERE filename = $1",
|
||||
[safeFilename]
|
||||
@@ -93,20 +99,25 @@ export async function POST(req: NextRequest) {
|
||||
try {
|
||||
const body = await req.json();
|
||||
const { filename } = body;
|
||||
|
||||
|
||||
if (!filename) {
|
||||
return NextResponse.json({ error: "Filename is required in request body" }, { status: 400 });
|
||||
}
|
||||
|
||||
// Ensure uploads directory exists (just in case)
|
||||
if (!fs.existsSync(UPLOADS_DIR)) {
|
||||
fs.mkdirSync(UPLOADS_DIR, { recursive: true });
|
||||
|
||||
const safeFilename = path.basename(filename);
|
||||
const labels = readLabels();
|
||||
const index = labels.findIndex(l => l.filename === safeFilename);
|
||||
const entry = { ...body, filename: safeFilename };
|
||||
|
||||
if (index >= 0) {
|
||||
labels[index] = entry;
|
||||
} else {
|
||||
labels.push(entry);
|
||||
}
|
||||
|
||||
const filePath = getFilePath(filename);
|
||||
fs.writeFileSync(filePath, JSON.stringify(body, null, 2), "utf8");
|
||||
|
||||
return NextResponse.json({ success: true, filePath });
|
||||
|
||||
writeLabels(labels);
|
||||
|
||||
return NextResponse.json({ success: true, filePath: LABELS_PATH });
|
||||
} catch (err: any) {
|
||||
console.error("Error in POST manual-label:", err);
|
||||
return NextResponse.json({ error: err.message || "Failed to save manual label" }, { status: 500 });
|
||||
|
||||
@@ -5,6 +5,10 @@ import crypto from "crypto";
|
||||
import { query, cleanupAndReindexItems, resolveStoreFromText } from "../../../db";
|
||||
import { parseDOMetadata, sanitizeParsedMetadata } from "../../../utils/parser";
|
||||
|
||||
// Bounds each pipeline call so a wedged GPU container fails fast into the existing
|
||||
// graceful fallback path instead of hanging the request indefinitely.
|
||||
const PIPELINE_TIMEOUT_MS = 90_000;
|
||||
|
||||
function levenshteinDistance(s1: string, s2: string): number {
|
||||
const len1 = s1.length;
|
||||
const len2 = s2.length;
|
||||
@@ -51,7 +55,7 @@ export async function POST(req: NextRequest) {
|
||||
let isSample = true;
|
||||
|
||||
if (!fs.existsSync(filePath)) {
|
||||
filePath = path.join(process.cwd(), "public", "test-images", safeFile);
|
||||
filePath = path.join(process.cwd(), "..", "sources", "test-images", safeFile);
|
||||
if (!fs.existsSync(filePath)) {
|
||||
filePath = path.join("/uploads", safeFile);
|
||||
if (!fs.existsSync(filePath)) {
|
||||
@@ -110,7 +114,8 @@ export async function POST(req: NextRequest) {
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(payload)
|
||||
body: JSON.stringify(payload),
|
||||
signal: AbortSignal.timeout(PIPELINE_TIMEOUT_MS)
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
@@ -137,28 +142,63 @@ export async function POST(req: NextRequest) {
|
||||
|
||||
data = await response.json();
|
||||
|
||||
// Check if the image is not straight (tilt > 1.0 degree)
|
||||
// Retry with document unwarping when the first pass looks deficient - either the page
|
||||
// is visibly tilted, or key header labels are missing from the recognized text (photos
|
||||
// with perspective warp can lose entire regions in the first pass while still measuring
|
||||
// as "straight" because too few blocks survive for the tilt average to be meaningful).
|
||||
tilt = calculateAverageTilt(data);
|
||||
if (tilt > 1.0) {
|
||||
console.log(`Parsed document ${safeFile} is not straight (average tilt: ${tilt.toFixed(2)} deg). Re-running with unwarping and orientation classification enabled...`);
|
||||
const firstScore = scoreKeyContent(data);
|
||||
if (tilt > 1.0 || firstScore < KEY_CONTENT_PATTERNS.length - 1) {
|
||||
console.log(`First pass for ${safeFile} looks deficient (tilt: ${tilt.toFixed(2)} deg, key content: ${firstScore}/${KEY_CONTENT_PATTERNS.length}). Re-running with unwarping and orientation classification enabled...`);
|
||||
const unwarpPayload = {
|
||||
...payload,
|
||||
useDocUnwarping: true,
|
||||
useDocOrientationClassify: true
|
||||
};
|
||||
const unwarpResponse = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(unwarpPayload)
|
||||
});
|
||||
if (unwarpResponse.ok) {
|
||||
data = await unwarpResponse.json();
|
||||
console.log(`Document unwarped successfully.`);
|
||||
unwarped = true;
|
||||
} else {
|
||||
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
|
||||
// A timeout/network error on this retry pass must not discard an already-good
|
||||
// first-pass result - fall back to keeping `data` as-is, same as the "not ok"
|
||||
// branch below, instead of letting the error bubble up to the outer catch
|
||||
// (which would overwrite the whole document with hard "Not Found" defaults).
|
||||
try {
|
||||
const unwarpResponse = await fetch(pipelineUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json"
|
||||
},
|
||||
body: JSON.stringify(unwarpPayload),
|
||||
signal: AbortSignal.timeout(PIPELINE_TIMEOUT_MS)
|
||||
});
|
||||
if (unwarpResponse.ok) {
|
||||
// Keep whichever pass recognized more of the document. Unwarping usually recovers
|
||||
// lost regions on warped photos, but on some documents it degrades an already-good
|
||||
// first pass (observed: a doc losing its Tanggal label after unwarping) - so this
|
||||
// must be a comparison, not an unconditional replacement. On a tie, prefer the
|
||||
// unwarped pass: markdown length is not a reliable proxy for correctness (a longer
|
||||
// first pass sometimes just means more duplicated/garbled boilerplate text, which
|
||||
// was observed shifting the PO/SO/DO numbers on a genuinely warped photo).
|
||||
const unwarpData = await unwarpResponse.json();
|
||||
const secondScore = scoreKeyContent(unwarpData);
|
||||
const firstLen = extractMarkdownText(data).length;
|
||||
const secondLen = extractMarkdownText(unwarpData).length;
|
||||
// Header presence alone can't tell a well-formed item table from a garbled one (observed:
|
||||
// a first pass with all 5 header labels but a table row that swallowed a "Total Qty" line
|
||||
// into the SKU/name/quantity columns). When one pass's table is clearly more intact, that
|
||||
// takes priority even if the other pass narrowly wins on header count.
|
||||
const firstItemQuality = scoreItemQuality(data);
|
||||
const secondItemQuality = scoreItemQuality(unwarpData);
|
||||
const itemQualityGap = secondItemQuality - firstItemQuality;
|
||||
if (secondScore >= firstScore || itemQualityGap >= 0.4) {
|
||||
data = unwarpData;
|
||||
unwarped = true;
|
||||
console.log(`Unwarped result kept (key content ${secondScore} vs ${firstScore}, item quality ${secondItemQuality.toFixed(2)} vs ${firstItemQuality.toFixed(2)}, length ${secondLen} vs ${firstLen}).`);
|
||||
} else {
|
||||
console.log(`First-pass result kept (key content ${firstScore} vs ${secondScore}, item quality ${firstItemQuality.toFixed(2)} vs ${secondItemQuality.toFixed(2)}, length ${firstLen} vs ${secondLen}).`);
|
||||
}
|
||||
} else {
|
||||
console.error(`Unwarping failed with status ${unwarpResponse.status}`);
|
||||
}
|
||||
} catch (unwarpErr) {
|
||||
console.error(`Unwarping pass timed out or failed for ${safeFile}, keeping first-pass result:`, unwarpErr);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -180,6 +220,11 @@ export async function POST(req: NextRequest) {
|
||||
const page0 = pipelineResult?.layoutParsingResults?.[0] || {};
|
||||
const markdownText = page0?.markdown?.text || "";
|
||||
const rawMetadata = parseDOMetadata(markdownText);
|
||||
// Snapshot before sanitize/triple-check: sanitizeParsedMetadata shallow-copies its input,
|
||||
// so rawMetadata.items shares object references with docMetadata.items, and the triple-check
|
||||
// below mutates those items in place. Without this clone, rawMetadata would silently pick up
|
||||
// triple-check corrections and no longer reflect the true raw-regex output.
|
||||
const rawMetadataSnapshot = JSON.parse(JSON.stringify(rawMetadata));
|
||||
// Second-layer sanity check: enforces strict field formats and auto-corrects anomalies
|
||||
const docMetadata = sanitizeParsedMetadata(rawMetadata as any);
|
||||
const docMetadataSanitized = JSON.parse(JSON.stringify(docMetadata));
|
||||
@@ -287,10 +332,12 @@ export async function POST(req: NextRequest) {
|
||||
const parsedQtyUnit = qtyUnitMatch ? qtyUnitMatch[0].toUpperCase() : "";
|
||||
const isQtyUnitValid = ["KRG", "BOX", "BAG", "PAC", "PC", "KG", "PCS"].includes(parsedQtyUnit);
|
||||
|
||||
// Master data (jenis_outer) is the canonical packaging unit for a matched SKU, so it
|
||||
// takes priority over the OCR-parsed unit even when that unit happens to also be one
|
||||
// of the generically "valid" units (e.g. OCR reading "KRG" for a SKU whose master
|
||||
// says "Box" should still be corrected to "Box", not trusted just because KRG is valid).
|
||||
const standardOuter = bestMatch.jenis_outer || "";
|
||||
if (isQtyUnitValid) {
|
||||
item.banyak = `${numericQty} ${parsedQtyUnit}`;
|
||||
} else if (standardOuter) {
|
||||
if (standardOuter) {
|
||||
if (standardOuter.toLowerCase() === "karung") {
|
||||
item.banyak = `${numericQty} KRG`;
|
||||
} else if (standardOuter.toLowerCase() === "box") {
|
||||
@@ -300,6 +347,8 @@ export async function POST(req: NextRequest) {
|
||||
} else {
|
||||
item.banyak = `${numericQty} ${standardOuter.toUpperCase()}`;
|
||||
}
|
||||
} else if (isQtyUnitValid) {
|
||||
item.banyak = `${numericQty} ${parsedQtyUnit}`;
|
||||
} else {
|
||||
item.banyak = ocrQty;
|
||||
}
|
||||
@@ -313,13 +362,13 @@ export async function POST(req: NextRequest) {
|
||||
const parsedPriceUnit = priceUnitMatch ? priceUnitMatch[0].toUpperCase() : "";
|
||||
const isPriceUnitValid = ["KRG", "BOX", "BAG", "PAC", "PC", "KG", "PCS"].includes(parsedPriceUnit);
|
||||
|
||||
// Same priority fix as banyak/jenis_outer above: standar_jumlah is the canonical unit
|
||||
// for a matched SKU and must win over a merely-"valid" OCR-parsed unit.
|
||||
const standardInner = bestMatch.standar_jumlah || "";
|
||||
if (standardInner.toLowerCase() === "pc" || standardInner.toLowerCase() === "pcs") {
|
||||
if (standardInner) {
|
||||
item.jumlah = `${numericPrice} ${standardInner.toUpperCase()}`;
|
||||
} else if (isPriceUnitValid) {
|
||||
item.jumlah = `${numericPrice} ${parsedPriceUnit}`;
|
||||
} else if (standardInner) {
|
||||
item.jumlah = `${numericPrice} ${standardInner.toUpperCase()}`;
|
||||
} else {
|
||||
item.jumlah = ocrPrice;
|
||||
}
|
||||
@@ -428,7 +477,7 @@ export async function POST(req: NextRequest) {
|
||||
items: docMetadata.items,
|
||||
postProcessingDetails: {
|
||||
rawMarkdown: markdownText,
|
||||
layer1RawRegex: rawMetadata,
|
||||
layer1RawRegex: rawMetadataSnapshot,
|
||||
layer2Sanitized: docMetadataSanitized,
|
||||
layer3Final: docMetadata
|
||||
},
|
||||
@@ -463,6 +512,54 @@ export async function POST(req: NextRequest) {
|
||||
}
|
||||
}
|
||||
|
||||
function extractMarkdownText(data: any): string {
|
||||
const results = data?.result?.layoutParsingResults || data?.layoutParsingResults || [];
|
||||
return results[0]?.markdown?.text || "";
|
||||
}
|
||||
|
||||
// Labels that appear on every delivery order; how many are recognized is a cheap proxy for
|
||||
// whether the OCR pass captured the whole page or lost regions to perspective warp.
|
||||
const KEY_CONTENT_PATTERNS = [
|
||||
/Tanggal/i,
|
||||
/No\.?\s*SO/i,
|
||||
/No\.?\s*DO/i,
|
||||
/No\.?\s*PO/i,
|
||||
/Truck\s*No/i,
|
||||
];
|
||||
|
||||
function scoreKeyContent(data: any): number {
|
||||
const text = extractMarkdownText(data);
|
||||
return KEY_CONTENT_PATTERNS.reduce((n, re) => n + (re.test(text) ? 1 : 0), 0);
|
||||
}
|
||||
|
||||
// How early the first genuinely intact item row appears (1.0 = first row, lower = buried under
|
||||
// noise). A pass can recognize the same correct item as another pass yet still score worse
|
||||
// downstream, because a duplicate/spurious table earlier in the document (e.g. a garbled summary
|
||||
// box with a "Total Qty" row) gets extracted as extra leading item rows and the real item ends up
|
||||
// competing with that noise for the "first item" slot ground truth is compared against. Presence
|
||||
// alone (as a fraction) doesn't catch this - two passes can have the same fraction of intact rows
|
||||
// while one buries the real item under 3 leading noise rows and the other doesn't bury it at all.
|
||||
// "Intact" requires a short SKU-like code (not a long phrase like a truck/signature line), a unit
|
||||
// on the quantity, and a non-empty name.
|
||||
function scoreItemQuality(data: any): number {
|
||||
const text = extractMarkdownText(data);
|
||||
if (!text) return 0;
|
||||
let items: { kodeBarang: string; banyak: string; namaBarang: string }[] = [];
|
||||
try {
|
||||
items = parseDOMetadata(text)?.items || [];
|
||||
} catch {
|
||||
return 0;
|
||||
}
|
||||
if (items.length === 0) return 0;
|
||||
const isIntact = (it: { kodeBarang: string; banyak: string; namaBarang: string }) =>
|
||||
/^[A-Za-z0-9]{4,12}$/.test((it.kodeBarang || "").trim()) &&
|
||||
/[a-zA-Z]/.test(it.banyak || "") &&
|
||||
(it.namaBarang || "").trim().length > 0;
|
||||
const firstIntactIndex = items.findIndex(isIntact);
|
||||
if (firstIntactIndex === -1) return 0;
|
||||
return 1 / (1 + firstIntactIndex);
|
||||
}
|
||||
|
||||
function getBlockAngle(points: number[][]) {
|
||||
if (!points || points.length < 2) return 0;
|
||||
const p0 = points[0];
|
||||
|
||||
@@ -35,12 +35,11 @@ export async function POST(req: NextRequest) {
|
||||
const filename = `${Date.now()}-${safeName}`;
|
||||
const filePath = path.join(UPLOADS_DIR, filename);
|
||||
|
||||
// Save file
|
||||
// Compute hash before writing/inserting anything, so we can detect a duplicate
|
||||
// upload (e.g. the client retrying after a perceived timeout on a slow OCR pass)
|
||||
// without creating a second document row or re-running the pipeline on it.
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const buffer = Buffer.from(arrayBuffer);
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
// Compute hash
|
||||
const fileHash = crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
|
||||
// Geolocation tags
|
||||
@@ -49,6 +48,42 @@ export async function POST(req: NextRequest) {
|
||||
const latitude = latVal ? parseFloat(latVal.toString()) : null;
|
||||
const longitude = lngVal ? parseFloat(lngVal.toString()) : null;
|
||||
|
||||
const existing = await query(
|
||||
"SELECT id, latitude, longitude, upload_time FROM documents WHERE file_hash = $1 ORDER BY upload_time ASC LIMIT 1",
|
||||
[fileHash]
|
||||
);
|
||||
|
||||
if (existing.rows.length > 0) {
|
||||
const existingDoc = existing.rows[0];
|
||||
console.log(`[Dedup] Identical content already uploaded as document ${existingDoc.id}. Skipping duplicate insert and re-parse.`);
|
||||
|
||||
const mappedData = {
|
||||
id: existingDoc.id.toString(),
|
||||
header: { tanggal: "", no_po: "", no_so: "", no_do: "" },
|
||||
shipment: {
|
||||
kepada_yth: "PT.PRIMAFOOD INTERNATIONAL",
|
||||
order_untuk: "",
|
||||
alamat: "",
|
||||
plat_truk: "",
|
||||
nama_driver: "",
|
||||
nama_penerima: ""
|
||||
},
|
||||
items: [] as any[],
|
||||
latitude: existingDoc.latitude ? parseFloat(existingDoc.latitude.toString()) : latitude,
|
||||
longitude: existingDoc.longitude ? parseFloat(existingDoc.longitude.toString()) : longitude,
|
||||
createdAt: new Date(existingDoc.upload_time || Date.now()).toISOString()
|
||||
};
|
||||
|
||||
return NextResponse.json({
|
||||
status: "success",
|
||||
message: "Document already uploaded",
|
||||
data: mappedData
|
||||
}, { status: 201, headers: corsHeaders });
|
||||
}
|
||||
|
||||
// Save file
|
||||
fs.writeFileSync(filePath, buffer);
|
||||
|
||||
let docId: number;
|
||||
let finalFilename = filename;
|
||||
|
||||
@@ -68,12 +103,17 @@ export async function POST(req: NextRequest) {
|
||||
]);
|
||||
docId = insertRes.rows[0].id;
|
||||
|
||||
// Trigger parsing synchronously to ensure it is processed immediately on receiving the image
|
||||
// Trigger parsing synchronously to ensure it is processed immediately on receiving the image.
|
||||
// Bounded well above /api/parse's own per-pass pipeline timeout (2 passes worst case) so a
|
||||
// wedged GPU container doesn't hang this request forever - it still won't fit under the
|
||||
// mobile client's 2-minute receive timeout in the worst case, but bounds the hang to a fixed,
|
||||
// known ceiling instead of an indefinite one.
|
||||
try {
|
||||
await fetch("http://127.0.0.1:3000/api/parse", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ filename: finalFilename })
|
||||
body: JSON.stringify({ filename: finalFilename }),
|
||||
signal: AbortSignal.timeout(210_000)
|
||||
});
|
||||
} catch (err) {
|
||||
console.error("Error triggering parse synchronously:", err);
|
||||
|
||||
@@ -67,45 +67,155 @@ export async function cleanupAndReindexItems(docId: number) {
|
||||
}
|
||||
}
|
||||
|
||||
export async function resolveStoreFromText(custInfo: string): Promise<{ orderUntuk: string; alamat: string }> {
|
||||
if (!custInfo || custInfo === "Not Found") {
|
||||
const STORE_STOPWORDS = new Set([
|
||||
"dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw",
|
||||
"jalan", "raya", "blok", "nomor", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"
|
||||
]);
|
||||
|
||||
function tokenize(text: string): string[] {
|
||||
return text.toLowerCase()
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !STORE_STOPWORDS.has(w));
|
||||
}
|
||||
|
||||
// customers.name is stored as "CUSTOMER NAME, JL. street address..." - split on the first
|
||||
// street-address marker to get just the canonical address portion.
|
||||
function splitCustomerAddress(name: string): string {
|
||||
const m = name.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
|
||||
return (m ? m[0] : name).replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
// A noisy OCR'd address (varying per document due to misread letters) that recognizably
|
||||
// belongs to a known customer should be reported as that customer's clean canonical address,
|
||||
// rather than whatever garbled text this particular scan happened to produce.
|
||||
async function canonicalizeCustomerAddress(extracted: string): Promise<string> {
|
||||
if (!extracted) return extracted;
|
||||
|
||||
const extractedTokens = new Set(tokenize(extracted));
|
||||
if (extractedTokens.size === 0) return extracted;
|
||||
|
||||
const customersRes = await query("SELECT name FROM customers");
|
||||
|
||||
let bestAddress: string | null = null;
|
||||
let bestMatchCount = 0;
|
||||
let bestScore = 0;
|
||||
|
||||
for (const row of customersRes.rows) {
|
||||
const canonicalAddress = splitCustomerAddress(row.name);
|
||||
const addressTokens = tokenize(canonicalAddress);
|
||||
if (addressTokens.length === 0) continue;
|
||||
|
||||
const uniqueAddressTokens = new Set(addressTokens);
|
||||
let matchCount = 0;
|
||||
for (const token of uniqueAddressTokens) {
|
||||
if (extractedTokens.has(token)) matchCount++;
|
||||
}
|
||||
const score = matchCount / uniqueAddressTokens.size;
|
||||
|
||||
if (matchCount >= 3 && score >= 0.45 && (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore))) {
|
||||
bestMatchCount = matchCount;
|
||||
bestScore = score;
|
||||
bestAddress = canonicalAddress;
|
||||
}
|
||||
}
|
||||
|
||||
return bestAddress ?? extracted;
|
||||
}
|
||||
|
||||
// The delivery truck/signature line near the bottom of the table ("Truck No. B 9427 UXT
|
||||
// PX HEAD OFFICE ANCOL : JL. ANCOL BARAT VIII...") names the actual destination store, when
|
||||
// present. Scoping the match to just this line (and just nama_toko, not nama_toko+alamat)
|
||||
// avoids the customer's own fixed head-office address elsewhere in the document being
|
||||
// mistaken for the destination - that address is present on every document regardless of
|
||||
// which store it's actually going to, so matching against it produces confident false
|
||||
// positives for documents that don't specify a destination store name at all.
|
||||
function extractTruckLineSnippet(fullText: string): string {
|
||||
const m = fullText.match(/Truck\s*No\.?[\s\S]{0,180}/i);
|
||||
return m ? m[0] : "";
|
||||
}
|
||||
|
||||
// True when the printed "Order Untuk" text is actually the customer's company name - a common
|
||||
// OCR layout jumble where the "Kepada Yth" and "Order Untuk" fields merge, meaning the real
|
||||
// destination value was lost and the truck line is the better signal.
|
||||
async function looksLikeCustomerName(text: string): Promise<boolean> {
|
||||
if (!text) return false;
|
||||
const textTokens = new Set(tokenize(text));
|
||||
if (textTokens.size === 0) return false;
|
||||
|
||||
const customersRes = await query("SELECT name FROM customers");
|
||||
for (const row of customersRes.rows) {
|
||||
const companyName = String(row.name).split(/\bJL\.?\b|\bJALAN\b/i)[0];
|
||||
const nameTokens = tokenize(companyName);
|
||||
if (nameTokens.length === 0) continue;
|
||||
let matchCount = 0;
|
||||
for (const token of new Set(nameTokens)) {
|
||||
if (textTokens.has(token)) matchCount++;
|
||||
}
|
||||
if (matchCount >= 1 && matchCount / new Set(nameTokens).size >= 0.5) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
export async function resolveStoreFromText(fullMarkdown: string): Promise<{ orderUntuk: string; alamat: string }> {
|
||||
if (!fullMarkdown || fullMarkdown === "Not Found") {
|
||||
return { orderUntuk: "", alamat: "" };
|
||||
}
|
||||
|
||||
// Load all stores
|
||||
// The printed "Alamat" field is the customer's own (fixed) address, not the destination
|
||||
// store's registered address - it stays the same across documents regardless of which
|
||||
// store the truck line names. So alamat always comes from the literal printed text; only
|
||||
// the store name itself benefits from being resolved to its canonical store_master form.
|
||||
// The line right after "Alamat:" sometimes holds a region code ("DKI AREA") rather than the
|
||||
// street address, with the real address following on the next line(s) - capture the whole
|
||||
// block up to the item table and prefer the "JL./JALAN ..." street-address line within it.
|
||||
const alamatBlockMatch = fullMarkdown.match(/Alamat\s*[:\-]?\s*([\s\S]+?)(?=<table|$)/i);
|
||||
let literalAlamat = "";
|
||||
if (alamatBlockMatch) {
|
||||
const block = alamatBlockMatch[1];
|
||||
const streetMatch = block.match(/\b(?:JL\.?|JALAN)\b[\s\S]*/i);
|
||||
literalAlamat = (streetMatch ? streetMatch[0] : block).replace(/\s+/g, " ").trim();
|
||||
}
|
||||
literalAlamat = await canonicalizeCustomerAddress(literalAlamat);
|
||||
|
||||
// The printed "Order Untuk" value is the primary source for the store field: it usually
|
||||
// holds a region designator ("DKI AREA", "PFM-KU") or a store name, and that's what the
|
||||
// document actually says. Only when OCR jumbled it with the customer's company name (or
|
||||
// lost it entirely) do we fall back to matching the truck/signature line against
|
||||
// store_master to recover the destination store.
|
||||
const orderMatch = fullMarkdown.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
|
||||
const literalOrder = orderMatch ? orderMatch[1].trim() : "";
|
||||
const orderIsUsable = literalOrder !== "" && !(await looksLikeCustomerName(literalOrder));
|
||||
|
||||
if (orderIsUsable) {
|
||||
return { orderUntuk: literalOrder, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
const storeRes = await query("SELECT nama_toko, kode_toko, alamat FROM store_master");
|
||||
const stores = storeRes.rows;
|
||||
|
||||
const ocrTokens = new Set(
|
||||
custInfo.toLowerCase()
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !["dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw"].includes(w))
|
||||
);
|
||||
const truckSnippet = extractTruckLineSnippet(fullMarkdown);
|
||||
const snippetTokens = new Set(tokenize(truckSnippet));
|
||||
|
||||
let bestStore: any = null;
|
||||
let bestScore = 0;
|
||||
let bestMatchCount = 0;
|
||||
|
||||
if (ocrTokens.size > 0) {
|
||||
if (snippetTokens.size > 0) {
|
||||
for (const store of stores) {
|
||||
const searchStr = `${store.nama_toko} ${store.alamat}`.toLowerCase();
|
||||
const storeTokens = searchStr
|
||||
.replace(/[^a-z0-9\s]/g, " ")
|
||||
.split(/\s+/)
|
||||
.filter(w => w.length > 2 && !["dan", "dki", "area", "yth", "kepada", "order", "untuk", "alamat", "kel", "kec", "rt", "rw", "jalan", "raya", "blok", "nomor", "rt", "rw", "kelurahan", "kecamatan", "kota", "kabupaten", "provinsi"].includes(w));
|
||||
|
||||
const storeTokens = tokenize(store.nama_toko);
|
||||
if (storeTokens.length === 0) continue;
|
||||
|
||||
let matchCount = 0;
|
||||
const uniqueStoreTokens = new Set(storeTokens);
|
||||
let matchCount = 0;
|
||||
for (const token of uniqueStoreTokens) {
|
||||
if (ocrTokens.has(token)) {
|
||||
matchCount++;
|
||||
}
|
||||
if (snippetTokens.has(token)) matchCount++;
|
||||
}
|
||||
|
||||
const score = matchCount / uniqueStoreTokens.size;
|
||||
// Two distinct matching tokens minimum: single-token overlaps (e.g. a store whose only
|
||||
// distinctive token is a common street/area word appearing in the snippet's address
|
||||
// text) produce far too many confident false positives.
|
||||
if (matchCount >= 2) {
|
||||
if (matchCount > bestMatchCount || (matchCount === bestMatchCount && score > bestScore)) {
|
||||
bestMatchCount = matchCount;
|
||||
@@ -117,17 +227,10 @@ export async function resolveStoreFromText(custInfo: string): Promise<{ orderUnt
|
||||
}
|
||||
|
||||
if (bestStore) {
|
||||
return { orderUntuk: bestStore.nama_toko, alamat: bestStore.alamat };
|
||||
return { orderUntuk: bestStore.nama_toko, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
// Fallback pattern matching
|
||||
const orderMatch = custInfo.match(/Order\s+Untuk\s*[:\-]\s*([^\n]+)/i);
|
||||
const alamatMatch = custInfo.match(/Alamat\s*[:\-]\s*([^\n]+)/i);
|
||||
|
||||
return {
|
||||
orderUntuk: orderMatch ? orderMatch[1].trim() : "",
|
||||
alamat: alamatMatch ? alamatMatch[1].trim() : ""
|
||||
};
|
||||
return { orderUntuk: literalOrder, alamat: literalAlamat };
|
||||
}
|
||||
|
||||
export { pool };
|
||||
@@ -78,7 +78,7 @@ export async function initDb(pool: Pool) {
|
||||
// Seed customer
|
||||
await pool.query(`
|
||||
INSERT INTO customers (name)
|
||||
VALUES ('PT.PRIMAFOOD INTERNATIONAL, JL. ANCOL BARAT VIII/1, ANCOL, PADEMANGAN, JAKARTA UTARA, 14430')
|
||||
VALUES ('PT.PRIMAFOOD INTERNATIONAL, JL. ANCOL BARAT VIII/1 KEL. ANCOL, KEC. PADEMANGAN JAKARTA UTARA, DKI JAKARTA')
|
||||
ON CONFLICT (name) DO NOTHING;
|
||||
`);
|
||||
|
||||
|
||||
@@ -847,6 +847,14 @@ export function sanitizeParsedMetadata(meta: ReturnType<typeof parseDOMetadata>
|
||||
const currentFullYear = new Date().getFullYear();
|
||||
const result = { ...meta };
|
||||
|
||||
// --- customerInfo / vendorInfo ---
|
||||
// OCR typically drops the space after Indonesian business-entity prefixes ("PT.PRIMAFOOD"
|
||||
// instead of "PT. PRIMAFOOD"). Normalize the standard prefixes to always have one space.
|
||||
const normalizeEntityPrefix = (v: string) =>
|
||||
v && v !== "Not Found" ? v.replace(/\b(PT|CV|UD|PD|TB)\.(?=\S)/gi, (_, p) => `${p.toUpperCase()}. `) : v;
|
||||
result.customerInfo = normalizeEntityPrefix(result.customerInfo);
|
||||
result.vendorInfo = normalizeEntityPrefix(result.vendorInfo);
|
||||
|
||||
// --- tanggal ---
|
||||
// Must be exactly "dd Month yyyy" where:
|
||||
// dd = 1-31, Month = valid English month name, yyyy = 4-digit year in reasonable range
|
||||
|
||||
@@ -1 +1 @@
|
||||
{"filename": "Sample DO PFM-page-00001.jpg"}
|
||||
{"filename": "do-015.jpg"}
|
||||
@@ -2,7 +2,7 @@ const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const PROXY_URL = 'http://localhost:8000/api/vllm-proxy/v1/chat/completions';
|
||||
const IMAGE_PATH = path.join(__dirname, '..', 'sources', 'test-images', '1782870899198-sample_do.jpeg');
|
||||
const IMAGE_PATH = path.join(__dirname, '..', 'sources', 'test-images', 'do-001.jpg');
|
||||
|
||||
async function testGuidedDecoding() {
|
||||
console.log('Reading test image from:', IMAGE_PATH);
|
||||
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,7 @@
|
||||
{"timestamp":"2026-07-04T05:57:04.582Z","commit":"095dd4c","refreshOcr":false,"imageCount":37,"failedImages":[],"overall":{"layer1RawRegex":66.79841897233202,"layer2Sanitized":66.79841897233202,"layer3Final":87.87515006002401},"fields":{"noPO":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"noSO":{"layer1RawRegex":83.78378378378379,"layer2Sanitized":83.78378378378379,"layer3Final":83.78378378378379},"noDO":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"tanggal":{"layer1RawRegex":83.78378378378379,"layer2Sanitized":83.78378378378379,"layer3Final":83.78378378378379},"customer":{"layer1RawRegex":100,"layer2Sanitized":100,"layer3Final":100},"plat":{"layer1RawRegex":59.45945945945946,"layer2Sanitized":59.45945945945946,"layer3Final":59.45945945945946},"store":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":64.86486486486487},"alamat":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":32.432432432432435},"itemCount":{"layer1RawRegex":8.108108108108109,"layer2Sanitized":8.108108108108109,"layer3Final":97.2972972972973},"kodeBarang":{"layer1RawRegex":94.39999999999999,"layer2Sanitized":94.39999999999999,"layer3Final":98.4},"namaBarang":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":94.39999999999999},"banyak":{"layer1RawRegex":80.80000000000001,"layer2Sanitized":80.80000000000001,"layer3Final":94.39999999999999},"jumlah":{"layer1RawRegex":77.60000000000001,"layer2Sanitized":77.60000000000001,"layer3Final":90.4}},"perImage":{"do-001.jpg":100,"do-002.jpg":92.3076923076923,"do-003.jpg":92.3076923076923,"do-004.jpg":84.61538461538461,"do-005.jpg":87.87878787878788,"do-006.jpg":95.1219512195122,"do-007.jpg":92.3076923076923,"do-008.jpg":61.53846153846154,"do-009.jpg":53.84615384615385,"do-010.jpg":84.61538461538461,"do-011.jpg":76.92307692307693,"do-012.jpg":76.92307692307693,"do-013.jpg":100,"do-014.jpg":78.78787878787878,"do-015.jpg":92.3076923076923,"do-016.jpg":94.11764705882352,"do-017.jpg":88.23529411764706,"do-018.jpg":95.23809523809523,"do-019.jpg":95.1219512195122,"do-020.jpg":96,"do-021.jpg":95.1219512195122,"do-022.jpg":95.23809523809523,"do-023.jpg":96.55172413793103,"do-024.jpg":96,"do-025.jpg":86.48648648648648,"do-026.jpg":80,"do-027.jpg":85.36585365853658,"do-028.jpg":92,"do-029.jpg":90.47619047619048,"do-030.jpg":76.47058823529412,"do-031.jpg":88.23529411764706,"do-032.jpg":65.51724137931035,"do-033.jpg":71.42857142857143,"do-034.jpg":69.23076923076923,"do-035.jpg":90.47619047619048,"do-036.jpg":90.47619047619048,"do-037.jpg":92.3076923076923}}
|
||||
{"timestamp":"2026-07-04T06:47:47.559Z","commit":"095dd4c","refreshOcr":true,"imageCount":37,"failedImages":[],"overall":{"layer1RawRegex":66.93017127799736,"layer2Sanitized":66.93017127799736,"layer3Final":88.5954381752701},"fields":{"noPO":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"noSO":{"layer1RawRegex":83.78378378378379,"layer2Sanitized":83.78378378378379,"layer3Final":83.78378378378379},"noDO":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"tanggal":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"customer":{"layer1RawRegex":100,"layer2Sanitized":100,"layer3Final":100},"plat":{"layer1RawRegex":62.16216216216216,"layer2Sanitized":62.16216216216216,"layer3Final":62.16216216216216},"store":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":64.86486486486487},"alamat":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":37.83783783783784},"itemCount":{"layer1RawRegex":10.81081081081081,"layer2Sanitized":10.81081081081081,"layer3Final":94.5945945945946},"kodeBarang":{"layer1RawRegex":97.6,"layer2Sanitized":97.6,"layer3Final":97.6},"namaBarang":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":93.60000000000001},"banyak":{"layer1RawRegex":76.8,"layer2Sanitized":76.8,"layer3Final":94.39999999999999},"jumlah":{"layer1RawRegex":76,"layer2Sanitized":76,"layer3Final":93.60000000000001}},"perImage":{"do-001.jpg":100,"do-002.jpg":92.3076923076923,"do-003.jpg":69.23076923076923,"do-004.jpg":84.61538461538461,"do-005.jpg":87.87878787878788,"do-006.jpg":95.1219512195122,"do-007.jpg":92.3076923076923,"do-008.jpg":61.53846153846154,"do-009.jpg":69.23076923076923,"do-010.jpg":46.15384615384615,"do-011.jpg":76.92307692307693,"do-012.jpg":76.92307692307693,"do-013.jpg":100,"do-014.jpg":93.93939393939394,"do-015.jpg":92.3076923076923,"do-016.jpg":94.11764705882352,"do-017.jpg":88.23529411764706,"do-018.jpg":95.23809523809523,"do-019.jpg":95.1219512195122,"do-020.jpg":96,"do-021.jpg":95.1219512195122,"do-022.jpg":95.23809523809523,"do-023.jpg":96.55172413793103,"do-024.jpg":96,"do-025.jpg":89.1891891891892,"do-026.jpg":80,"do-027.jpg":85.36585365853658,"do-028.jpg":92,"do-029.jpg":90.47619047619048,"do-030.jpg":76.47058823529412,"do-031.jpg":88.23529411764706,"do-032.jpg":86.20689655172413,"do-033.jpg":71.42857142857143,"do-034.jpg":69.23076923076923,"do-035.jpg":90.47619047619048,"do-036.jpg":90.47619047619048,"do-037.jpg":92.3076923076923}}
|
||||
{"timestamp":"2026-07-04T06:48:56.651Z","commit":"095dd4c","refreshOcr":false,"imageCount":37,"failedImages":[],"overall":{"layer1RawRegex":66.93017127799736,"layer2Sanitized":66.93017127799736,"layer3Final":88.5954381752701},"fields":{"noPO":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"noSO":{"layer1RawRegex":83.78378378378379,"layer2Sanitized":83.78378378378379,"layer3Final":83.78378378378379},"noDO":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"tanggal":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"customer":{"layer1RawRegex":100,"layer2Sanitized":100,"layer3Final":100},"plat":{"layer1RawRegex":62.16216216216216,"layer2Sanitized":62.16216216216216,"layer3Final":62.16216216216216},"store":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":64.86486486486487},"alamat":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":37.83783783783784},"itemCount":{"layer1RawRegex":10.81081081081081,"layer2Sanitized":10.81081081081081,"layer3Final":94.5945945945946},"kodeBarang":{"layer1RawRegex":97.6,"layer2Sanitized":97.6,"layer3Final":97.6},"namaBarang":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":93.60000000000001},"banyak":{"layer1RawRegex":76.8,"layer2Sanitized":76.8,"layer3Final":94.39999999999999},"jumlah":{"layer1RawRegex":76,"layer2Sanitized":76,"layer3Final":93.60000000000001}},"perImage":{"do-001.jpg":100,"do-002.jpg":92.3076923076923,"do-003.jpg":69.23076923076923,"do-004.jpg":84.61538461538461,"do-005.jpg":87.87878787878788,"do-006.jpg":95.1219512195122,"do-007.jpg":92.3076923076923,"do-008.jpg":61.53846153846154,"do-009.jpg":69.23076923076923,"do-010.jpg":46.15384615384615,"do-011.jpg":76.92307692307693,"do-012.jpg":76.92307692307693,"do-013.jpg":100,"do-014.jpg":93.93939393939394,"do-015.jpg":92.3076923076923,"do-016.jpg":94.11764705882352,"do-017.jpg":88.23529411764706,"do-018.jpg":95.23809523809523,"do-019.jpg":95.1219512195122,"do-020.jpg":96,"do-021.jpg":95.1219512195122,"do-022.jpg":95.23809523809523,"do-023.jpg":96.55172413793103,"do-024.jpg":96,"do-025.jpg":89.1891891891892,"do-026.jpg":80,"do-027.jpg":85.36585365853658,"do-028.jpg":92,"do-029.jpg":90.47619047619048,"do-030.jpg":76.47058823529412,"do-031.jpg":88.23529411764706,"do-032.jpg":86.20689655172413,"do-033.jpg":71.42857142857143,"do-034.jpg":69.23076923076923,"do-035.jpg":90.47619047619048,"do-036.jpg":90.47619047619048,"do-037.jpg":92.3076923076923}}
|
||||
{"timestamp":"2026-07-04T06:49:36.163Z","commit":"095dd4c","refreshOcr":false,"imageCount":37,"failedImages":[],"overall":{"layer1RawRegex":66.93017127799736,"layer2Sanitized":66.93017127799736,"layer3Final":88.5954381752701},"fields":{"noPO":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"noSO":{"layer1RawRegex":83.78378378378379,"layer2Sanitized":83.78378378378379,"layer3Final":83.78378378378379},"noDO":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"tanggal":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"customer":{"layer1RawRegex":100,"layer2Sanitized":100,"layer3Final":100},"plat":{"layer1RawRegex":62.16216216216216,"layer2Sanitized":62.16216216216216,"layer3Final":62.16216216216216},"store":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":64.86486486486487},"alamat":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":37.83783783783784},"itemCount":{"layer1RawRegex":10.81081081081081,"layer2Sanitized":10.81081081081081,"layer3Final":94.5945945945946},"kodeBarang":{"layer1RawRegex":97.6,"layer2Sanitized":97.6,"layer3Final":97.6},"namaBarang":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":93.60000000000001},"banyak":{"layer1RawRegex":76.8,"layer2Sanitized":76.8,"layer3Final":94.39999999999999},"jumlah":{"layer1RawRegex":76,"layer2Sanitized":76,"layer3Final":93.60000000000001}},"perImage":{"do-001.jpg":100,"do-002.jpg":92.3076923076923,"do-003.jpg":69.23076923076923,"do-004.jpg":84.61538461538461,"do-005.jpg":87.87878787878788,"do-006.jpg":95.1219512195122,"do-007.jpg":92.3076923076923,"do-008.jpg":61.53846153846154,"do-009.jpg":69.23076923076923,"do-010.jpg":46.15384615384615,"do-011.jpg":76.92307692307693,"do-012.jpg":76.92307692307693,"do-013.jpg":100,"do-014.jpg":93.93939393939394,"do-015.jpg":92.3076923076923,"do-016.jpg":94.11764705882352,"do-017.jpg":88.23529411764706,"do-018.jpg":95.23809523809523,"do-019.jpg":95.1219512195122,"do-020.jpg":96,"do-021.jpg":95.1219512195122,"do-022.jpg":95.23809523809523,"do-023.jpg":96.55172413793103,"do-024.jpg":96,"do-025.jpg":89.1891891891892,"do-026.jpg":80,"do-027.jpg":85.36585365853658,"do-028.jpg":92,"do-029.jpg":90.47619047619048,"do-030.jpg":76.47058823529412,"do-031.jpg":88.23529411764706,"do-032.jpg":86.20689655172413,"do-033.jpg":71.42857142857143,"do-034.jpg":69.23076923076923,"do-035.jpg":90.47619047619048,"do-036.jpg":90.47619047619048,"do-037.jpg":92.3076923076923}}
|
||||
{"timestamp":"2026-07-04T07:29:39.005Z","commit":"095dd4c","refreshOcr":true,"imageCount":37,"failedImages":[],"overall":{"layer1RawRegex":67.19367588932806,"layer2Sanitized":67.19367588932806,"layer3Final":88.83553421368548},"fields":{"noPO":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"noSO":{"layer1RawRegex":86.48648648648648,"layer2Sanitized":86.48648648648648,"layer3Final":86.48648648648648},"noDO":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"tanggal":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"customer":{"layer1RawRegex":100,"layer2Sanitized":100,"layer3Final":100},"plat":{"layer1RawRegex":62.16216216216216,"layer2Sanitized":62.16216216216216,"layer3Final":62.16216216216216},"store":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":64.86486486486487},"alamat":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":37.83783783783784},"itemCount":{"layer1RawRegex":10.81081081081081,"layer2Sanitized":10.81081081081081,"layer3Final":94.5945945945946},"kodeBarang":{"layer1RawRegex":97.6,"layer2Sanitized":97.6,"layer3Final":97.6},"namaBarang":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":93.60000000000001},"banyak":{"layer1RawRegex":76.8,"layer2Sanitized":76.8,"layer3Final":94.39999999999999},"jumlah":{"layer1RawRegex":76,"layer2Sanitized":76,"layer3Final":93.60000000000001}},"perImage":{"do-001.jpg":100,"do-002.jpg":92.3076923076923,"do-003.jpg":92.3076923076923,"do-004.jpg":84.61538461538461,"do-005.jpg":87.87878787878788,"do-006.jpg":95.1219512195122,"do-007.jpg":92.3076923076923,"do-008.jpg":61.53846153846154,"do-009.jpg":69.23076923076923,"do-010.jpg":46.15384615384615,"do-011.jpg":76.92307692307693,"do-012.jpg":76.92307692307693,"do-013.jpg":100,"do-014.jpg":93.93939393939394,"do-015.jpg":92.3076923076923,"do-016.jpg":94.11764705882352,"do-017.jpg":88.23529411764706,"do-018.jpg":95.23809523809523,"do-019.jpg":95.1219512195122,"do-020.jpg":96,"do-021.jpg":95.1219512195122,"do-022.jpg":95.23809523809523,"do-023.jpg":96.55172413793103,"do-024.jpg":96,"do-025.jpg":86.48648648648648,"do-026.jpg":80,"do-027.jpg":85.36585365853658,"do-028.jpg":92,"do-029.jpg":90.47619047619048,"do-030.jpg":76.47058823529412,"do-031.jpg":88.23529411764706,"do-032.jpg":86.20689655172413,"do-033.jpg":71.42857142857143,"do-034.jpg":69.23076923076923,"do-035.jpg":90.47619047619048,"do-036.jpg":90.47619047619048,"do-037.jpg":92.3076923076923}}
|
||||
{"timestamp":"2026-07-04T08:29:06.624Z","commit":"095dd4c","refreshOcr":true,"imageCount":37,"failedImages":[],"overall":{"layer1RawRegex":67.72068511198947,"layer2Sanitized":67.72068511198947,"layer3Final":89.4357743097239},"fields":{"noPO":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"noSO":{"layer1RawRegex":86.48648648648648,"layer2Sanitized":86.48648648648648,"layer3Final":86.48648648648648},"noDO":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"tanggal":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"customer":{"layer1RawRegex":100,"layer2Sanitized":100,"layer3Final":100},"plat":{"layer1RawRegex":62.16216216216216,"layer2Sanitized":62.16216216216216,"layer3Final":62.16216216216216},"store":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":64.86486486486487},"alamat":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":37.83783783783784},"itemCount":{"layer1RawRegex":13.513513513513514,"layer2Sanitized":13.513513513513514,"layer3Final":97.2972972972973},"kodeBarang":{"layer1RawRegex":98.4,"layer2Sanitized":98.4,"layer3Final":98.4},"namaBarang":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":94.39999999999999},"banyak":{"layer1RawRegex":77.60000000000001,"layer2Sanitized":77.60000000000001,"layer3Final":95.19999999999999},"jumlah":{"layer1RawRegex":76.8,"layer2Sanitized":76.8,"layer3Final":94.39999999999999}},"perImage":{"do-001.jpg":100,"do-002.jpg":92.3076923076923,"do-003.jpg":92.3076923076923,"do-004.jpg":84.61538461538461,"do-005.jpg":87.87878787878788,"do-006.jpg":95.1219512195122,"do-007.jpg":92.3076923076923,"do-008.jpg":61.53846153846154,"do-009.jpg":69.23076923076923,"do-010.jpg":84.61538461538461,"do-011.jpg":76.92307692307693,"do-012.jpg":76.92307692307693,"do-013.jpg":100,"do-014.jpg":93.93939393939394,"do-015.jpg":92.3076923076923,"do-016.jpg":94.11764705882352,"do-017.jpg":88.23529411764706,"do-018.jpg":95.23809523809523,"do-019.jpg":95.1219512195122,"do-020.jpg":96,"do-021.jpg":95.1219512195122,"do-022.jpg":95.23809523809523,"do-023.jpg":96.55172413793103,"do-024.jpg":96,"do-025.jpg":86.48648648648648,"do-026.jpg":80,"do-027.jpg":85.36585365853658,"do-028.jpg":92,"do-029.jpg":90.47619047619048,"do-030.jpg":76.47058823529412,"do-031.jpg":88.23529411764706,"do-032.jpg":86.20689655172413,"do-033.jpg":71.42857142857143,"do-034.jpg":69.23076923076923,"do-035.jpg":90.47619047619048,"do-036.jpg":90.47619047619048,"do-037.jpg":92.3076923076923}}
|
||||
{"timestamp":"2026-07-04T09:58:40.843Z","commit":"095dd4c","refreshOcr":false,"imageCount":37,"failedImages":[],"overall":{"layer1RawRegex":67.72068511198947,"layer2Sanitized":67.72068511198947,"layer3Final":89.4357743097239},"fields":{"noPO":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"noSO":{"layer1RawRegex":86.48648648648648,"layer2Sanitized":86.48648648648648,"layer3Final":86.48648648648648},"noDO":{"layer1RawRegex":91.8918918918919,"layer2Sanitized":91.8918918918919,"layer3Final":91.8918918918919},"tanggal":{"layer1RawRegex":89.1891891891892,"layer2Sanitized":89.1891891891892,"layer3Final":89.1891891891892},"customer":{"layer1RawRegex":100,"layer2Sanitized":100,"layer3Final":100},"plat":{"layer1RawRegex":62.16216216216216,"layer2Sanitized":62.16216216216216,"layer3Final":62.16216216216216},"store":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":64.86486486486487},"alamat":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":37.83783783783784},"itemCount":{"layer1RawRegex":13.513513513513514,"layer2Sanitized":13.513513513513514,"layer3Final":97.2972972972973},"kodeBarang":{"layer1RawRegex":98.4,"layer2Sanitized":98.4,"layer3Final":98.4},"namaBarang":{"layer1RawRegex":0,"layer2Sanitized":0,"layer3Final":94.39999999999999},"banyak":{"layer1RawRegex":77.60000000000001,"layer2Sanitized":77.60000000000001,"layer3Final":95.19999999999999},"jumlah":{"layer1RawRegex":76.8,"layer2Sanitized":76.8,"layer3Final":94.39999999999999}},"perImage":{"do-001.jpg":100,"do-002.jpg":92.3076923076923,"do-003.jpg":92.3076923076923,"do-004.jpg":84.61538461538461,"do-005.jpg":87.87878787878788,"do-006.jpg":95.1219512195122,"do-007.jpg":92.3076923076923,"do-008.jpg":61.53846153846154,"do-009.jpg":69.23076923076923,"do-010.jpg":84.61538461538461,"do-011.jpg":76.92307692307693,"do-012.jpg":76.92307692307693,"do-013.jpg":100,"do-014.jpg":93.93939393939394,"do-015.jpg":92.3076923076923,"do-016.jpg":94.11764705882352,"do-017.jpg":88.23529411764706,"do-018.jpg":95.23809523809523,"do-019.jpg":95.1219512195122,"do-020.jpg":96,"do-021.jpg":95.1219512195122,"do-022.jpg":95.23809523809523,"do-023.jpg":96.55172413793103,"do-024.jpg":96,"do-025.jpg":86.48648648648648,"do-026.jpg":80,"do-027.jpg":85.36585365853658,"do-028.jpg":92,"do-029.jpg":90.47619047619048,"do-030.jpg":76.47058823529412,"do-031.jpg":88.23529411764706,"do-032.jpg":86.20689655172413,"do-033.jpg":71.42857142857143,"do-034.jpg":69.23076923076923,"do-035.jpg":90.47619047619048,"do-036.jpg":90.47619047619048,"do-037.jpg":92.3076923076923}}
|
||||
Binary file not shown.
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user