Adopt agents-settings kit, ship Product/SKU scan models, harden auth, verify OCR accuracy
Backend (app-pfm-ocr-v2/backend): - Product/SKU scan feature complete: trained DINOv2 index (118 reference photos, 16 SKU classes) and YOLO classifier (83.3% top-1 val accuracy), fixed scripts/install-pipeline.sh (was missing ultralytics/torch), fully browser-verified end-to-end on /scan-pfm. Mobile m-scan-pfm page cancelled (Flutter app handles mobile; web UI is desktop-only for pipeline testing). - Fixed a real data-loss bug: Save Ground Truth (scan-pfm and the DO-flow's manual-label) was silently writing into the pfm-web-app container's ephemeral filesystem instead of the host, because /sources wasn't bind-mounted in docker-compose.yml. Added the mount, recovered an orphaned entry. - accounts.password is now bcrypt-hashed (bcryptjs, idempotent migration in db/init.ts) instead of plaintext; login route compares hashes. - /api/v1/documents/* (list, PUT, upload) now enforces real 401 auth, matching what the Flutter client already sends. The "classic" routes deliberately stay open — they're dev-only web UI with no login flow and won't exist in production. - OCR accuracy investigated end-to-end: real baseline is 95.10% overall (target met; accuracy_report.md was stale at 75.04%, now flagged). Fixed one genuine parser.ts bug (SO/DO field duplication in the global fallback regex); remaining gaps are OCR/layout-model limitations, not parser bugs. - Adopted a standalone copy of the fhanyuh/agents-settings e/n workflow scoped to backend/ (AGENTS.md Part A/B split, SKILLS.md, plans/, docs/), independent of the root copy which now covers Flutter only. - next-implementation.md deleted; content folded into backend/plans/next-enhancements.md for traceability. Root: - Adopted fhanyuh/agents-settings kit (AGENTS.md, SKILLS.md, plans/, docs/feature-list.md), scoped to the Flutter app only. - Pending documents queue now persists to Hive (lib/core/storage) instead of memory-only, surviving an app kill mid-upload. Removed backend_backup/ (stale Express/Prisma prototype, superseded by pfm-web-app) and the completed plans/next-enhancement-plan.md checklist. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
3df9f6ec5d
commit
e60ab63154
129 files changed
+8520
-6684
No files matched your search
@@ -3,9 +3,12 @@ import io
|
||||
import os
|
||||
import re
|
||||
import traceback
|
||||
import pickle
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
from torchvision import transforms
|
||||
import requests
|
||||
from fastapi import FastAPI, HTTPException
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
@@ -94,6 +97,43 @@ app.add_middleware(
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
# DINOv2 Image preprocessing
|
||||
DINOV2_TRANSFORMS = transforms.Compose([
|
||||
transforms.Resize((224, 224)),
|
||||
transforms.ToTensor(),
|
||||
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
|
||||
])
|
||||
|
||||
# Load DINOv2 Vector Similarity Search Model on startup
|
||||
print("Loading DINOv2 for similarity search...")
|
||||
dinov2_model = None
|
||||
dinov2_index = None
|
||||
|
||||
models_dir = resolve_classifier_models_dir()
|
||||
dinov2_index_path = models_dir / "dinov2_index.pkl" if models_dir else None
|
||||
|
||||
if dinov2_index_path and dinov2_index_path.is_file():
|
||||
try:
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using DINOv2 on device: {device}")
|
||||
|
||||
# Load the index
|
||||
print(f"Loading DINOv2 index from {dinov2_index_path}...")
|
||||
with open(dinov2_index_path, "rb") as f:
|
||||
dinov2_index = pickle.load(f)
|
||||
print(f"DINOv2 index loaded with {len(dinov2_index['metadata'])} reference images.")
|
||||
|
||||
# Load DINOv2 model
|
||||
dinov2_model = torch.hub.load("facebookresearch/dinov2", "dinov2_vits14").to(device)
|
||||
dinov2_model.eval()
|
||||
print("DINOv2 model loaded successfully.")
|
||||
except Exception as e:
|
||||
print(f"Error loading DINOv2 model or index: {e}")
|
||||
dinov2_model = None
|
||||
dinov2_index = None
|
||||
else:
|
||||
print(f"DINOv2 index not found at {dinov2_index_path}. DINOv2 search disabled.")
|
||||
|
||||
# Load models on startup
|
||||
print("Loading YOLO model...")
|
||||
yolo_model = None
|
||||
@@ -431,33 +471,92 @@ async def classify_ocr(payload: ScanRequest):
|
||||
raw_image = Image.open(io.BytesIO(img_data))
|
||||
image = ImageOps.exif_transpose(raw_image).convert("RGB")
|
||||
|
||||
# 1. Run YOLO Classification
|
||||
# 1. Run DINOv2 Similarity Search or YOLO Classification
|
||||
classification_result = {}
|
||||
top1_name = None
|
||||
if yolo_model:
|
||||
results = yolo_model(image)
|
||||
probs = results[0].probs
|
||||
top1_idx = probs.top1
|
||||
top1_conf = float(probs.top1conf)
|
||||
top1_name = results[0].names[top1_idx]
|
||||
|
||||
all_probs = []
|
||||
for idx, val in enumerate(probs.data):
|
||||
all_probs.append({
|
||||
"name": results[0].names[idx],
|
||||
"confidence": float(val)
|
||||
})
|
||||
all_probs.sort(key=lambda x: x["confidence"], reverse=True)
|
||||
|
||||
classification_result = {
|
||||
"top1_name": top1_name,
|
||||
"top1_confidence": top1_conf,
|
||||
"all_probabilities": all_probs
|
||||
}
|
||||
else:
|
||||
classification_result = {
|
||||
"error": "YOLO model not loaded"
|
||||
}
|
||||
|
||||
# Try DINOv2 first if index exists
|
||||
if dinov2_model and dinov2_index:
|
||||
try:
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
|
||||
# Preprocess image
|
||||
preprocessed = DINOV2_TRANSFORMS(image).unsqueeze(0).to(device)
|
||||
|
||||
# Extract query embedding
|
||||
with torch.no_grad():
|
||||
query_emb = dinov2_model(preprocessed)
|
||||
query_emb = query_emb / query_emb.norm(dim=-1, keepdim=True)
|
||||
query_emb = query_emb.squeeze(0).cpu().numpy()
|
||||
|
||||
ref_embeddings = dinov2_index["embeddings"] # (N, 384)
|
||||
ref_metadata = dinov2_index["metadata"] # List of dicts
|
||||
|
||||
# Compute Cosine Similarity
|
||||
similarities = np.dot(ref_embeddings, query_emb)
|
||||
|
||||
# Aggregate similarities per class (max similarity of any reference image in the class)
|
||||
class_sims = {}
|
||||
for idx, meta in enumerate(ref_metadata):
|
||||
c_name = meta["class_name"]
|
||||
sim = float(similarities[idx])
|
||||
if c_name not in class_sims or sim > class_sims[c_name]:
|
||||
class_sims[c_name] = sim
|
||||
|
||||
# Sort classes by similarity
|
||||
sorted_classes = sorted(class_sims.items(), key=lambda x: x[1], reverse=True)
|
||||
|
||||
all_probs = []
|
||||
for c_name, sim in sorted_classes:
|
||||
all_probs.append({
|
||||
"name": c_name,
|
||||
"confidence": sim
|
||||
})
|
||||
|
||||
if all_probs:
|
||||
top1_name = all_probs[0]["name"]
|
||||
top1_conf = all_probs[0]["confidence"]
|
||||
|
||||
classification_result = {
|
||||
"top1_name": top1_name,
|
||||
"top1_confidence": top1_conf,
|
||||
"all_probabilities": all_probs,
|
||||
"method": "dinov2_similarity"
|
||||
}
|
||||
print(f"[DINOv2] Best match: {top1_name} ({top1_conf:.4f})")
|
||||
except Exception as dinov2_err:
|
||||
print(f"[DINOv2 Error] Similarity search failed, falling back to YOLO: {dinov2_err}")
|
||||
traceback.print_exc()
|
||||
top1_name = None
|
||||
|
||||
# Fallback to YOLO if DINOv2 was not run or failed
|
||||
if not top1_name:
|
||||
if yolo_model:
|
||||
results = yolo_model(image)
|
||||
probs = results[0].probs
|
||||
top1_idx = probs.top1
|
||||
top1_conf = float(probs.top1conf)
|
||||
top1_name = results[0].names[top1_idx]
|
||||
|
||||
all_probs = []
|
||||
for idx, val in enumerate(probs.data):
|
||||
all_probs.append({
|
||||
"name": results[0].names[idx],
|
||||
"confidence": float(val)
|
||||
})
|
||||
all_probs.sort(key=lambda x: x["confidence"], reverse=True)
|
||||
|
||||
classification_result = {
|
||||
"top1_name": top1_name,
|
||||
"top1_confidence": top1_conf,
|
||||
"all_probabilities": all_probs,
|
||||
"method": "yolo_classifier"
|
||||
}
|
||||
print(f"[YOLO Fallback] Best match: {top1_name} ({top1_conf:.4f})")
|
||||
else:
|
||||
classification_result = {
|
||||
"error": "Both DINOv2 index and YOLO model are unavailable"
|
||||
}
|
||||
|
||||
# 2. Run PaddleOCR
|
||||
ocr_result = {}
|
||||
|
||||
Reference in new issue
Block a user