Adopt agents-settings kit, ship Product/SKU scan models, harden auth, verify OCR accuracy

Backend (app-pfm-ocr-v2/backend):
- Product/SKU scan feature complete: trained DINOv2 index (118 reference
  photos, 16 SKU classes) and YOLO classifier (83.3% top-1 val accuracy),
  fixed scripts/install-pipeline.sh (was missing ultralytics/torch), fully
  browser-verified end-to-end on /scan-pfm. Mobile m-scan-pfm page cancelled
  (Flutter app handles mobile; web UI is desktop-only for pipeline testing).
- Fixed a real data-loss bug: Save Ground Truth (scan-pfm and the DO-flow's
  manual-label) was silently writing into the pfm-web-app container's
  ephemeral filesystem instead of the host, because /sources wasn't
  bind-mounted in docker-compose.yml. Added the mount, recovered an
  orphaned entry.
- accounts.password is now bcrypt-hashed (bcryptjs, idempotent migration
  in db/init.ts) instead of plaintext; login route compares hashes.
- /api/v1/documents/* (list, PUT, upload) now enforces real 401 auth,
  matching what the Flutter client already sends. The "classic" routes
  deliberately stay open — they're dev-only web UI with no login flow and
  won't exist in production.
- OCR accuracy investigated end-to-end: real baseline is 95.10% overall
  (target met; accuracy_report.md was stale at 75.04%, now flagged). Fixed
  one genuine parser.ts bug (SO/DO field duplication in the global fallback
  regex); remaining gaps are OCR/layout-model limitations, not parser bugs.
- Adopted a standalone copy of the fhanyuh/agents-settings e/n workflow
  scoped to backend/ (AGENTS.md Part A/B split, SKILLS.md, plans/, docs/),
  independent of the root copy which now covers Flutter only.
- next-implementation.md deleted; content folded into
  backend/plans/next-enhancements.md for traceability.

Root:
- Adopted fhanyuh/agents-settings kit (AGENTS.md, SKILLS.md, plans/,
  docs/feature-list.md), scoped to the Flutter app only.
- Pending documents queue now persists to Hive (lib/core/storage) instead
  of memory-only, surviving an app kill mid-upload.

Removed backend_backup/ (stale Express/Prisma prototype, superseded by
pfm-web-app) and the completed plans/next-enhancement-plan.md checklist.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Rafhan Mazaya FathurrahmanandClaude Sonnet 5 committed 2026-07-08 11:56:32 +07:00
1 parent 3df9f6ec5d
commit e60ab63154
129 files changed
+8520 -6684

No files matched your search

+124 -25
View File
@@ -3,9 +3,12 @@ import io
import os
import re
import traceback
import pickle
from datetime import date
from pathlib import Path
import torch
from torchvision import transforms
import requests
from fastapi import FastAPI, HTTPException
from fastapi.middleware.cors import CORSMiddleware
@@ -94,6 +97,43 @@ app.add_middleware(
allow_headers=["*"],
)
# DINOv2 Image preprocessing
DINOV2_TRANSFORMS = transforms.Compose([
transforms.Resize((224, 224)),
transforms.ToTensor(),
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
])
# Load DINOv2 Vector Similarity Search Model on startup
print("Loading DINOv2 for similarity search...")
dinov2_model = None
dinov2_index = None
models_dir = resolve_classifier_models_dir()
dinov2_index_path = models_dir / "dinov2_index.pkl" if models_dir else None
if dinov2_index_path and dinov2_index_path.is_file():
try:
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"Using DINOv2 on device: {device}")
# Load the index
print(f"Loading DINOv2 index from {dinov2_index_path}...")
with open(dinov2_index_path, "rb") as f:
dinov2_index = pickle.load(f)
print(f"DINOv2 index loaded with {len(dinov2_index['metadata'])} reference images.")
# Load DINOv2 model
dinov2_model = torch.hub.load("facebookresearch/dinov2", "dinov2_vits14").to(device)
dinov2_model.eval()
print("DINOv2 model loaded successfully.")
except Exception as e:
print(f"Error loading DINOv2 model or index: {e}")
dinov2_model = None
dinov2_index = None
else:
print(f"DINOv2 index not found at {dinov2_index_path}. DINOv2 search disabled.")
# Load models on startup
print("Loading YOLO model...")
yolo_model = None
@@ -431,33 +471,92 @@ async def classify_ocr(payload: ScanRequest):
raw_image = Image.open(io.BytesIO(img_data))
image = ImageOps.exif_transpose(raw_image).convert("RGB")
# 1. Run YOLO Classification
# 1. Run DINOv2 Similarity Search or YOLO Classification
classification_result = {}
top1_name = None
if yolo_model:
results = yolo_model(image)
probs = results[0].probs
top1_idx = probs.top1
top1_conf = float(probs.top1conf)
top1_name = results[0].names[top1_idx]
all_probs = []
for idx, val in enumerate(probs.data):
all_probs.append({
"name": results[0].names[idx],
"confidence": float(val)
})
all_probs.sort(key=lambda x: x["confidence"], reverse=True)
classification_result = {
"top1_name": top1_name,
"top1_confidence": top1_conf,
"all_probabilities": all_probs
}
else:
classification_result = {
"error": "YOLO model not loaded"
}
# Try DINOv2 first if index exists
if dinov2_model and dinov2_index:
try:
device = "cuda" if torch.cuda.is_available() else "cpu"
# Preprocess image
preprocessed = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
# Extract query embedding
with torch.no_grad():
query_emb = dinov2_model(preprocessed)
query_emb = query_emb / query_emb.norm(dim=-1, keepdim=True)
query_emb = query_emb.squeeze(0).cpu().numpy()
ref_embeddings = dinov2_index["embeddings"] # (N, 384)
ref_metadata = dinov2_index["metadata"] # List of dicts
# Compute Cosine Similarity
similarities = np.dot(ref_embeddings, query_emb)
# Aggregate similarities per class (max similarity of any reference image in the class)
class_sims = {}
for idx, meta in enumerate(ref_metadata):
c_name = meta["class_name"]
sim = float(similarities[idx])
if c_name not in class_sims or sim > class_sims[c_name]:
class_sims[c_name] = sim
# Sort classes by similarity
sorted_classes = sorted(class_sims.items(), key=lambda x: x[1], reverse=True)
all_probs = []
for c_name, sim in sorted_classes:
all_probs.append({
"name": c_name,
"confidence": sim
})
if all_probs:
top1_name = all_probs[0]["name"]
top1_conf = all_probs[0]["confidence"]
classification_result = {
"top1_name": top1_name,
"top1_confidence": top1_conf,
"all_probabilities": all_probs,
"method": "dinov2_similarity"
}
print(f"[DINOv2] Best match: {top1_name} ({top1_conf:.4f})")
except Exception as dinov2_err:
print(f"[DINOv2 Error] Similarity search failed, falling back to YOLO: {dinov2_err}")
traceback.print_exc()
top1_name = None
# Fallback to YOLO if DINOv2 was not run or failed
if not top1_name:
if yolo_model:
results = yolo_model(image)
probs = results[0].probs
top1_idx = probs.top1
top1_conf = float(probs.top1conf)
top1_name = results[0].names[top1_idx]
all_probs = []
for idx, val in enumerate(probs.data):
all_probs.append({
"name": results[0].names[idx],
"confidence": float(val)
})
all_probs.sort(key=lambda x: x["confidence"], reverse=True)
classification_result = {
"top1_name": top1_name,
"top1_confidence": top1_conf,
"all_probabilities": all_probs,
"method": "yolo_classifier"
}
print(f"[YOLO Fallback] Best match: {top1_name} ({top1_conf:.4f})")
else:
classification_result = {
"error": "Both DINOv2 index and YOLO model are unavailable"
}
# 2. Run PaddleOCR
ocr_result = {}