diff --git a/.gitignore b/.gitignore index 6e8cda1..b2642ea 100644 --- a/.gitignore +++ b/.gitignore @@ -51,3 +51,8 @@ app.*.map.json # Graphify code knowledge graph (regenerated locally via git hooks) graphify-out/ + +# Client business documents kept local-only (user decision 2026-07-16) +proposals/ +screenshots/v2/OCR-Architecture-Diagram.pptx +screenshots/v2/OCR-Project-Presentation.pptx diff --git a/backend/config/classify_ocr_server.py b/backend/config/classify_ocr_server.py index ed51599..01fbd10 100644 --- a/backend/config/classify_ocr_server.py +++ b/backend/config/classify_ocr_server.py @@ -164,19 +164,6 @@ except Exception as e: class ScanRequest(BaseModel): image_base64: str -class ProbeRequest(BaseModel): - # Temporary debug endpoint input: container-local image path + crop box. - path: str - x0: int - y0: int - x1: int - y1: int - # Each recipe is a comma-separated op chain applied left to right, e.g. - # "s2,blur5" = upscale 2x then Gaussian-blur k=5. Ops: sN (scale xN, - # floats ok), blurN, closeN (morph close), clahe, gray, inv, thrN - # (adaptive threshold, block N). - recipes: list = ["none", "s2", "blur5", "s2,blur5"] - def clean_ocr_text(text: str) -> str: return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip() @@ -262,142 +249,6 @@ def extract_product_name(text_lines, classified_name=None): return "Unknown Product" -@app.post("/probe-ocr") -async def probe_ocr(payload: ProbeRequest): - # Temporary debug endpoint: OCR a crop of a container-local image at - # several upscale factors (optionally CLAHE-enhanced) using the - # already-loaded GPU OCR model, and run the expiry cascade on each - # variant's lines. Lets us test enhancement recipes without loading a - # second model instance (GPU is full) - remove once tuning is done. - from PIL import ImageOps - import cv2 - img = ImageOps.exif_transpose(Image.open(payload.path)).convert("RGB") - # Clamp to image bounds - PIL pads out-of-bounds crops onto a giant canvas. - crop = img.crop(( - max(0, payload.x0), max(0, payload.y0), - min(img.width, payload.x1), min(img.height, payload.y1), - )) - out = {"image_size": img.size, "variants": {}} - - def apply_ops(arr, recipe): - for op in recipe.split(","): - op = op.strip().lower() - if not op or op == "none": - continue - if op.startswith("s"): - f = float(op[1:]) - arr = cv2.resize(arr, None, fx=f, fy=f, interpolation=cv2.INTER_LANCZOS4) - elif op.startswith("blur"): - k = int(op[4:]) | 1 - arr = cv2.GaussianBlur(arr, (k, k), 0) - elif op.startswith("close"): - k = int(op[5:]) - kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (k, k)) - arr = cv2.morphologyEx(arr, cv2.MORPH_CLOSE, kernel) - elif op == "clahe": - g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr - cl = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8, 8)).apply(g) - arr = cv2.cvtColor(cl, cv2.COLOR_GRAY2RGB) - elif op == "gray": - g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr - arr = cv2.cvtColor(g, cv2.COLOR_GRAY2RGB) - elif op == "inv": - arr = 255 - arr - elif op.startswith("thr"): - b = int(op[3:]) | 1 - g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr - t = cv2.adaptiveThreshold(g, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, - cv2.THRESH_BINARY, b, 10) - arr = cv2.cvtColor(t, cv2.COLOR_GRAY2RGB) - return arr - - def vl_read(arr, use_layout=True): - # Route the crop through the vLLM-backed VL pipeline instead of the - # local PP-OCR model. Returns text lines from its markdown output. - buffered = io.BytesIO() - Image.fromarray(arr).save(buffered, format="JPEG") - resp = requests.post( - os.environ.get("VL_PIPELINE_URL", "http://localhost:8090/layout-parsing"), - json={ - "file": base64.b64encode(buffered.getvalue()).decode("utf-8"), - "matchHistoryJob": False, - "useLayoutDetection": use_layout, - "fileType": 1, - "useDocUnwarping": False, - "useDocOrientationClassify": True, - }, - timeout=120, - ) - resp.raise_for_status() - data = resp.json() - results = data.get("result", {}).get("layoutParsingResults", []) - md = (results[0].get("markdown") or {}).get("text", "") if results else "" - return [ln.strip() for ln in md.splitlines() if ln.strip()] - - def vlm_read(arr): - # Ask the PaddleOCR-VL model on the vLLM genai server (:8118) to OCR - # the crop directly, skipping the pipeline's layout detection (which - # tags dot-matrix prints as pictures and refuses to read them). - buffered = io.BytesIO() - Image.fromarray(arr).save(buffered, format="JPEG") - b64 = base64.b64encode(buffered.getvalue()).decode("utf-8") - resp = requests.post( - os.environ.get("VLM_CHAT_URL", "http://paddleocr-vllm-server:8118/v1/chat/completions"), - json={ - "model": os.environ.get("VLM_MODEL", "PaddleOCR-VL-1.6-0.9B"), - "messages": [{ - "role": "user", - "content": [ - {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{b64}"}}, - {"type": "text", "text": "OCR:"}, - ], - }], - "temperature": 0.0, - "max_tokens": 256, - }, - timeout=120, - ) - resp.raise_for_status() - content = resp.json()["choices"][0]["message"]["content"] or "" - return [ln.strip() for ln in content.splitlines() if ln.strip()] - - def v5s_ocr(): - # Lazily load the heavier PP-OCRv5 server det/rec pair (the main - # pipeline runs PP-OCRv6_medium). Cached on the app object so - # repeated probes don't reload weights. - if not hasattr(app.state, "v5s_ocr"): - app.state.v5s_ocr = PaddleOCR( - text_detection_model_name="PP-OCRv5_server_det", - text_recognition_model_name="PP-OCRv5_server_rec", - use_textline_orientation=True, - ) - return app.state.v5s_ocr - - base = np.array(crop) - readers = ("vl", "vlnl", "vlm", "v5s") - for recipe in payload.recipes: - try: - ops = [op.strip().lower() for op in recipe.split(",")] - reader = next((o for o in ops if o in readers), None) - arr = apply_ops(base.copy(), ",".join(o for o in ops if o not in readers)) - if reader == "vl": - lines = vl_read(arr, use_layout=True) - elif reader == "vlnl": - lines = vl_read(arr, use_layout=False) - elif reader == "vlm": - lines = vlm_read(arr) - elif reader == "v5s": - res = list(v5s_ocr().predict(arr)) - lines = res[0].get("rec_texts", []) if res else [] - else: - res = list(ocr.predict(arr)) - lines = res[0].get("rec_texts", []) if res else [] - d, _i, src = extract_expired_date(lines) - out["variants"][recipe] = {"size": [arr.shape[1], arr.shape[0]], "lines": lines, "date": d, "source": src} - except Exception as e: - out["variants"][recipe] = {"error": str(e)} - return out - @app.post("/classify-ocr") async def classify_ocr(payload: ScanRequest): try: diff --git a/lib/config/app_config.dart b/lib/config/app_config.dart index 02e7757..674a207 100644 --- a/lib/config/app_config.dart +++ b/lib/config/app_config.dart @@ -1,9 +1,9 @@ -import 'dart:io'; +import 'dart:io'; import 'package:flutter/material.dart'; import 'package:google_fonts/google_fonts.dart'; class AppConfig { - // ─── API Configuration ──────────────────────────────────────────────────── + // ─── API Configuration ──────────────────────────────────────────────────── // // Dual-Mode Configuration, tried in this order at every app startup: // 1. Local/LAN backend (fast, low-latency - the common case when the phone @@ -19,7 +19,7 @@ class AppConfig { // - Physical Device on same Wi-Fi: use the backend machine's current LAN IP // - Android Emulator: use 'http://10.0.2.2:8000/api/v1' // - iOS Simulator: use 'http://localhost:8000/api/v1' - static const String _lanBaseUrl = 'http://192.168.100.20:8000/api/v1'; + static const String _lanBaseUrl = 'http://192.168.80.84:8000/api/v1'; // Ngrok public tunnel URL (update ini setiap ngrok di-restart) static const String _ngrokBaseUrl = 'https://unlocated-waylon-potently.ngrok-free.dev/api/v1'; @@ -107,7 +107,7 @@ class AppConfig { static const Color surfaceColor = Colors.white; static const Color errorColor = Color(0xFFD32F2F); static const Color successColor = Color(0xFF388E3C); - // DO mode color — orange cue used throughout for the DO Scan mode: + // DO mode color — orange cue used throughout for the DO Scan mode: // document_card.dart top-label, documents_tab_switcher.dart active DO tab, // camera_drawer_mode_toggle.dart DO segment decoration. Formalized as a // shared constant so all three call sites pull from one place (was an