Adopt agents-settings kit, ship Product/SKU scan models, harden auth, verify OCR accuracy

Backend (app-pfm-ocr-v2/backend):
- Product/SKU scan feature complete: trained DINOv2 index (118 reference
  photos, 16 SKU classes) and YOLO classifier (83.3% top-1 val accuracy),
  fixed scripts/install-pipeline.sh (was missing ultralytics/torch), fully
  browser-verified end-to-end on /scan-pfm. Mobile m-scan-pfm page cancelled
  (Flutter app handles mobile; web UI is desktop-only for pipeline testing).
- Fixed a real data-loss bug: Save Ground Truth (scan-pfm and the DO-flow's
  manual-label) was silently writing into the pfm-web-app container's
  ephemeral filesystem instead of the host, because /sources wasn't
  bind-mounted in docker-compose.yml. Added the mount, recovered an
  orphaned entry.
- accounts.password is now bcrypt-hashed (bcryptjs, idempotent migration
  in db/init.ts) instead of plaintext; login route compares hashes.
- /api/v1/documents/* (list, PUT, upload) now enforces real 401 auth,
  matching what the Flutter client already sends. The "classic" routes
  deliberately stay open — they're dev-only web UI with no login flow and
  won't exist in production.
- OCR accuracy investigated end-to-end: real baseline is 95.10% overall
  (target met; accuracy_report.md was stale at 75.04%, now flagged). Fixed
  one genuine parser.ts bug (SO/DO field duplication in the global fallback
  regex); remaining gaps are OCR/layout-model limitations, not parser bugs.
- Adopted a standalone copy of the fhanyuh/agents-settings e/n workflow
  scoped to backend/ (AGENTS.md Part A/B split, SKILLS.md, plans/, docs/),
  independent of the root copy which now covers Flutter only.
- next-implementation.md deleted; content folded into
  backend/plans/next-enhancements.md for traceability.

Root:
- Adopted fhanyuh/agents-settings kit (AGENTS.md, SKILLS.md, plans/,
  docs/feature-list.md), scoped to the Flutter app only.
- Pending documents queue now persists to Hive (lib/core/storage) instead
  of memory-only, surviving an app kill mid-upload.

Removed backend_backup/ (stale Express/Prisma prototype, superseded by
pfm-web-app) and the completed plans/next-enhancement-plan.md checklist.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Rafhan Mazaya FathurrahmanandClaude Sonnet 5 committed 2026-07-08 11:56:32 +07:00
1 parent 3df9f6ec5d
commit e60ab63154
129 files changed
+8520 -6684

No files matched your search

@@ -0,0 +1,132 @@
#!/usr/bin/env python3
import os
import re
import time
import pickle
import numpy as np
import torch
from PIL import Image
from torchvision import transforms
from pathlib import Path
# Setup directories
SCRIPT_DIR = Path(__file__).resolve().parent
DEFAULT_DATASET_DIR = SCRIPT_DIR / "foto-kemasan-v2"
DEFAULT_MODELS_DIR = SCRIPT_DIR / "models"
DEFAULT_INDEX_PATH = DEFAULT_MODELS_DIR / "dinov2_index.pkl"
# Allowed image extensions
IMAGE_EXTS = (".jpg", ".jpeg", ".png", ".webp", ".bmp")
# DINOv2 Image preprocessing
DINOV2_TRANSFORMS = transforms.Compose([
transforms.Resize((224, 224)),
transforms.ToTensor(),
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
])
def get_embedding(dinov2_model, image: Image.Image, device):
if image.mode != "RGB":
image = image.convert("RGB")
tensor = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
with torch.no_grad():
embedding = dinov2_model(tensor)
# L2 normalization for dot product similarity
embedding = embedding / embedding.norm(dim=-1, keepdim=True)
return embedding.squeeze(0).cpu().numpy()
def run_indexing(src_dir=DEFAULT_DATASET_DIR, out_path=DEFAULT_INDEX_PATH):
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"Using device: {device}")
src_path = Path(src_dir).resolve()
out_file_path = Path(out_path).resolve()
if not src_path.is_dir():
print(f"Error: Source dataset directory not found: {src_path}")
return False
out_file_path.parent.mkdir(parents=True, exist_ok=True)
# Load DINOv2 Model from Torch Hub
print("Loading DINOv2 model (dinov2_vits14)...")
t0 = time.perf_counter()
dinov2_model = torch.hub.load("facebookresearch/dinov2", "dinov2_vits14").to(device)
dinov2_model.eval()
print(f"DINOv2 loaded in {time.perf_counter() - t0:.2f}s")
# Scan dataset directory
class_dirs = [d for d in src_path.iterdir() if d.is_dir()]
class_dirs.sort()
embeddings_list = []
metadata_list = []
total_images = 0
indexed_images = 0
for c_dir in class_dirs:
class_name = c_dir.name
images = sorted(
[f for f in c_dir.iterdir() if f.suffix.lower() in IMAGE_EXTS],
key=lambda p: p.name
)
if not images:
continue
print(f"Processing class: {class_name} ({len(images)} images)")
total_images += len(images)
for img_file in images:
try:
# Load image
image = Image.open(img_file).convert("RGB")
# Extract DINOv2 embedding (using whole image as reference photo)
embedding = get_embedding(dinov2_model, image, device)
embeddings_list.append(embedding)
metadata_list.append({
"class_name": class_name,
"image_path": str(img_file.relative_to(src_path.parent)),
"file_name": img_file.name
})
indexed_images += 1
except Exception as e:
print(f" [Error] Failed to process {img_file.name}: {e}")
# Save the index
if embeddings_list:
embeddings_arr = np.vstack(embeddings_list)
index_data = {
"embeddings": embeddings_arr,
"metadata": metadata_list
}
with open(out_file_path, "wb") as f:
pickle.dump(index_data, f)
print(f"\nSuccess! Indexed {indexed_images}/{total_images} images.")
print(f"DINOv2 Vector Index saved to: {out_file_path}")
return True
else:
print("\n[Warning] No images were successfully indexed.")
return False
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser(description="Build DINOv2 image vector index for PFM products")
parser.add_argument("--src-dir", default=str(DEFAULT_DATASET_DIR), help="Source directory of classes")
parser.add_argument("--output", default=str(DEFAULT_INDEX_PATH), help="Output pickle index path")
args = parser.parse_args()
run_indexing(src_dir=args.src_dir, out_path=args.output)