Adopt agents-settings kit, ship Product/SKU scan models, harden auth, verify OCR accuracy
Backend (app-pfm-ocr-v2/backend): - Product/SKU scan feature complete: trained DINOv2 index (118 reference photos, 16 SKU classes) and YOLO classifier (83.3% top-1 val accuracy), fixed scripts/install-pipeline.sh (was missing ultralytics/torch), fully browser-verified end-to-end on /scan-pfm. Mobile m-scan-pfm page cancelled (Flutter app handles mobile; web UI is desktop-only for pipeline testing). - Fixed a real data-loss bug: Save Ground Truth (scan-pfm and the DO-flow's manual-label) was silently writing into the pfm-web-app container's ephemeral filesystem instead of the host, because /sources wasn't bind-mounted in docker-compose.yml. Added the mount, recovered an orphaned entry. - accounts.password is now bcrypt-hashed (bcryptjs, idempotent migration in db/init.ts) instead of plaintext; login route compares hashes. - /api/v1/documents/* (list, PUT, upload) now enforces real 401 auth, matching what the Flutter client already sends. The "classic" routes deliberately stay open — they're dev-only web UI with no login flow and won't exist in production. - OCR accuracy investigated end-to-end: real baseline is 95.10% overall (target met; accuracy_report.md was stale at 75.04%, now flagged). Fixed one genuine parser.ts bug (SO/DO field duplication in the global fallback regex); remaining gaps are OCR/layout-model limitations, not parser bugs. - Adopted a standalone copy of the fhanyuh/agents-settings e/n workflow scoped to backend/ (AGENTS.md Part A/B split, SKILLS.md, plans/, docs/), independent of the root copy which now covers Flutter only. - next-implementation.md deleted; content folded into backend/plans/next-enhancements.md for traceability. Root: - Adopted fhanyuh/agents-settings kit (AGENTS.md, SKILLS.md, plans/, docs/feature-list.md), scoped to the Flutter app only. - Pending documents queue now persists to Hive (lib/core/storage) instead of memory-only, surviving an app kill mid-upload. Removed backend_backup/ (stale Express/Prisma prototype, superseded by pfm-web-app) and the completed plans/next-enhancement-plan.md checklist. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
3df9f6ec5d
commit
e60ab63154
129 files changed
+8520
-6684
No files matched your search
@@ -0,0 +1,132 @@
|
||||
#!/usr/bin/env python3
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
import pickle
|
||||
import numpy as np
|
||||
import torch
|
||||
from PIL import Image
|
||||
from torchvision import transforms
|
||||
from pathlib import Path
|
||||
|
||||
# Setup directories
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
|
||||
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
|
||||
DEFAULT_INDEX_PATH = DEFAULT_MODELS_DIR / "dinov2_index.pkl"
|
||||
|
||||
# Allowed image extensions
|
||||
IMAGE_EXTS = (".jpg", ".jpeg", ".png", ".webp", ".bmp")
|
||||
|
||||
# DINOv2 Image preprocessing
|
||||
DINOV2_TRANSFORMS = transforms.Compose([
|
||||
transforms.Resize((224, 224)),
|
||||
transforms.ToTensor(),
|
||||
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
|
||||
])
|
||||
|
||||
|
||||
def get_embedding(dinov2_model, image: Image.Image, device):
|
||||
if image.mode != "RGB":
|
||||
image = image.convert("RGB")
|
||||
|
||||
tensor = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
|
||||
|
||||
with torch.no_grad():
|
||||
embedding = dinov2_model(tensor)
|
||||
# L2 normalization for dot product similarity
|
||||
embedding = embedding / embedding.norm(dim=-1, keepdim=True)
|
||||
|
||||
return embedding.squeeze(0).cpu().numpy()
|
||||
|
||||
|
||||
def run_indexing(src_dir=DEFAULT_DATASET_DIR, out_path=DEFAULT_INDEX_PATH):
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
src_path = Path(src_dir).resolve()
|
||||
out_file_path = Path(out_path).resolve()
|
||||
|
||||
if not src_path.is_dir():
|
||||
print(f"Error: Source dataset directory not found: {src_path}")
|
||||
return False
|
||||
|
||||
out_file_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Load DINOv2 Model from Torch Hub
|
||||
print("Loading DINOv2 model (dinov2_vits14)...")
|
||||
t0 = time.perf_counter()
|
||||
dinov2_model = torch.hub.load("facebookresearch/dinov2", "dinov2_vits14").to(device)
|
||||
dinov2_model.eval()
|
||||
print(f"DINOv2 loaded in {time.perf_counter() - t0:.2f}s")
|
||||
|
||||
# Scan dataset directory
|
||||
class_dirs = [d for d in src_path.iterdir() if d.is_dir()]
|
||||
class_dirs.sort()
|
||||
|
||||
embeddings_list = []
|
||||
metadata_list = []
|
||||
|
||||
total_images = 0
|
||||
indexed_images = 0
|
||||
|
||||
for c_dir in class_dirs:
|
||||
class_name = c_dir.name
|
||||
|
||||
images = sorted(
|
||||
[f for f in c_dir.iterdir() if f.suffix.lower() in IMAGE_EXTS],
|
||||
key=lambda p: p.name
|
||||
)
|
||||
|
||||
if not images:
|
||||
continue
|
||||
|
||||
print(f"Processing class: {class_name} ({len(images)} images)")
|
||||
total_images += len(images)
|
||||
|
||||
for img_file in images:
|
||||
try:
|
||||
# Load image
|
||||
image = Image.open(img_file).convert("RGB")
|
||||
|
||||
# Extract DINOv2 embedding (using whole image as reference photo)
|
||||
embedding = get_embedding(dinov2_model, image, device)
|
||||
|
||||
embeddings_list.append(embedding)
|
||||
metadata_list.append({
|
||||
"class_name": class_name,
|
||||
"image_path": str(img_file.relative_to(src_path.parent)),
|
||||
"file_name": img_file.name
|
||||
})
|
||||
indexed_images += 1
|
||||
|
||||
except Exception as e:
|
||||
print(f" [Error] Failed to process {img_file.name}: {e}")
|
||||
|
||||
# Save the index
|
||||
if embeddings_list:
|
||||
embeddings_arr = np.vstack(embeddings_list)
|
||||
index_data = {
|
||||
"embeddings": embeddings_arr,
|
||||
"metadata": metadata_list
|
||||
}
|
||||
|
||||
with open(out_file_path, "wb") as f:
|
||||
pickle.dump(index_data, f)
|
||||
|
||||
print(f"\nSuccess! Indexed {indexed_images}/{total_images} images.")
|
||||
print(f"DINOv2 Vector Index saved to: {out_file_path}")
|
||||
return True
|
||||
else:
|
||||
print("\n[Warning] No images were successfully indexed.")
|
||||
return False
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
parser = argparse.ArgumentParser(description="Build DINOv2 image vector index for PFM products")
|
||||
parser.add_argument("--src-dir", default=str(DEFAULT_DATASET_DIR), help="Source directory of classes")
|
||||
parser.add_argument("--output", default=str(DEFAULT_INDEX_PATH), help="Output pickle index path")
|
||||
args = parser.parse_args()
|
||||
|
||||
run_indexing(src_dir=args.src_dir, out_path=args.output)
|
||||
Reference in new issue
Block a user