Save local changes before splitting branches
This commit is contained in:
1 parent
e6daa9b053
commit
15566a6951
3 files changed
+9
-153
No files matched your search
@@ -51,3 +51,8 @@ app.*.map.json
|
|||||||
|
|
||||||
# Graphify code knowledge graph (regenerated locally via git hooks)
|
# Graphify code knowledge graph (regenerated locally via git hooks)
|
||||||
graphify-out/
|
graphify-out/
|
||||||
|
|
||||||
|
# Client business documents kept local-only (user decision 2026-07-16)
|
||||||
|
proposals/
|
||||||
|
screenshots/v2/OCR-Architecture-Diagram.pptx
|
||||||
|
screenshots/v2/OCR-Project-Presentation.pptx
|
||||||
@@ -164,19 +164,6 @@ except Exception as e:
|
|||||||
class ScanRequest(BaseModel):
|
class ScanRequest(BaseModel):
|
||||||
image_base64: str
|
image_base64: str
|
||||||
|
|
||||||
class ProbeRequest(BaseModel):
|
|
||||||
# Temporary debug endpoint input: container-local image path + crop box.
|
|
||||||
path: str
|
|
||||||
x0: int
|
|
||||||
y0: int
|
|
||||||
x1: int
|
|
||||||
y1: int
|
|
||||||
# Each recipe is a comma-separated op chain applied left to right, e.g.
|
|
||||||
# "s2,blur5" = upscale 2x then Gaussian-blur k=5. Ops: sN (scale xN,
|
|
||||||
# floats ok), blurN, closeN (morph close), clahe, gray, inv, thrN
|
|
||||||
# (adaptive threshold, block N).
|
|
||||||
recipes: list = ["none", "s2", "blur5", "s2,blur5"]
|
|
||||||
|
|
||||||
def clean_ocr_text(text: str) -> str:
|
def clean_ocr_text(text: str) -> str:
|
||||||
return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip()
|
return re.sub(r'^[^\w\s./-]+|[^\w\s./-]+$', '', text).strip()
|
||||||
|
|
||||||
@@ -262,142 +249,6 @@ def extract_product_name(text_lines, classified_name=None):
|
|||||||
|
|
||||||
return "Unknown Product"
|
return "Unknown Product"
|
||||||
|
|
||||||
@app.post("/probe-ocr")
|
|
||||||
async def probe_ocr(payload: ProbeRequest):
|
|
||||||
# Temporary debug endpoint: OCR a crop of a container-local image at
|
|
||||||
# several upscale factors (optionally CLAHE-enhanced) using the
|
|
||||||
# already-loaded GPU OCR model, and run the expiry cascade on each
|
|
||||||
# variant's lines. Lets us test enhancement recipes without loading a
|
|
||||||
# second model instance (GPU is full) - remove once tuning is done.
|
|
||||||
from PIL import ImageOps
|
|
||||||
import cv2
|
|
||||||
img = ImageOps.exif_transpose(Image.open(payload.path)).convert("RGB")
|
|
||||||
# Clamp to image bounds - PIL pads out-of-bounds crops onto a giant canvas.
|
|
||||||
crop = img.crop((
|
|
||||||
max(0, payload.x0), max(0, payload.y0),
|
|
||||||
min(img.width, payload.x1), min(img.height, payload.y1),
|
|
||||||
))
|
|
||||||
out = {"image_size": img.size, "variants": {}}
|
|
||||||
|
|
||||||
def apply_ops(arr, recipe):
|
|
||||||
for op in recipe.split(","):
|
|
||||||
op = op.strip().lower()
|
|
||||||
if not op or op == "none":
|
|
||||||
continue
|
|
||||||
if op.startswith("s"):
|
|
||||||
f = float(op[1:])
|
|
||||||
arr = cv2.resize(arr, None, fx=f, fy=f, interpolation=cv2.INTER_LANCZOS4)
|
|
||||||
elif op.startswith("blur"):
|
|
||||||
k = int(op[4:]) | 1
|
|
||||||
arr = cv2.GaussianBlur(arr, (k, k), 0)
|
|
||||||
elif op.startswith("close"):
|
|
||||||
k = int(op[5:])
|
|
||||||
kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (k, k))
|
|
||||||
arr = cv2.morphologyEx(arr, cv2.MORPH_CLOSE, kernel)
|
|
||||||
elif op == "clahe":
|
|
||||||
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
|
|
||||||
cl = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8, 8)).apply(g)
|
|
||||||
arr = cv2.cvtColor(cl, cv2.COLOR_GRAY2RGB)
|
|
||||||
elif op == "gray":
|
|
||||||
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
|
|
||||||
arr = cv2.cvtColor(g, cv2.COLOR_GRAY2RGB)
|
|
||||||
elif op == "inv":
|
|
||||||
arr = 255 - arr
|
|
||||||
elif op.startswith("thr"):
|
|
||||||
b = int(op[3:]) | 1
|
|
||||||
g = cv2.cvtColor(arr, cv2.COLOR_RGB2GRAY) if arr.ndim == 3 else arr
|
|
||||||
t = cv2.adaptiveThreshold(g, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C,
|
|
||||||
cv2.THRESH_BINARY, b, 10)
|
|
||||||
arr = cv2.cvtColor(t, cv2.COLOR_GRAY2RGB)
|
|
||||||
return arr
|
|
||||||
|
|
||||||
def vl_read(arr, use_layout=True):
|
|
||||||
# Route the crop through the vLLM-backed VL pipeline instead of the
|
|
||||||
# local PP-OCR model. Returns text lines from its markdown output.
|
|
||||||
buffered = io.BytesIO()
|
|
||||||
Image.fromarray(arr).save(buffered, format="JPEG")
|
|
||||||
resp = requests.post(
|
|
||||||
os.environ.get("VL_PIPELINE_URL", "http://localhost:8090/layout-parsing"),
|
|
||||||
json={
|
|
||||||
"file": base64.b64encode(buffered.getvalue()).decode("utf-8"),
|
|
||||||
"matchHistoryJob": False,
|
|
||||||
"useLayoutDetection": use_layout,
|
|
||||||
"fileType": 1,
|
|
||||||
"useDocUnwarping": False,
|
|
||||||
"useDocOrientationClassify": True,
|
|
||||||
},
|
|
||||||
timeout=120,
|
|
||||||
)
|
|
||||||
resp.raise_for_status()
|
|
||||||
data = resp.json()
|
|
||||||
results = data.get("result", {}).get("layoutParsingResults", [])
|
|
||||||
md = (results[0].get("markdown") or {}).get("text", "") if results else ""
|
|
||||||
return [ln.strip() for ln in md.splitlines() if ln.strip()]
|
|
||||||
|
|
||||||
def vlm_read(arr):
|
|
||||||
# Ask the PaddleOCR-VL model on the vLLM genai server (:8118) to OCR
|
|
||||||
# the crop directly, skipping the pipeline's layout detection (which
|
|
||||||
# tags dot-matrix prints as pictures and refuses to read them).
|
|
||||||
buffered = io.BytesIO()
|
|
||||||
Image.fromarray(arr).save(buffered, format="JPEG")
|
|
||||||
b64 = base64.b64encode(buffered.getvalue()).decode("utf-8")
|
|
||||||
resp = requests.post(
|
|
||||||
os.environ.get("VLM_CHAT_URL", "http://paddleocr-vllm-server:8118/v1/chat/completions"),
|
|
||||||
json={
|
|
||||||
"model": os.environ.get("VLM_MODEL", "PaddleOCR-VL-1.6-0.9B"),
|
|
||||||
"messages": [{
|
|
||||||
"role": "user",
|
|
||||||
"content": [
|
|
||||||
{"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{b64}"}},
|
|
||||||
{"type": "text", "text": "OCR:"},
|
|
||||||
],
|
|
||||||
}],
|
|
||||||
"temperature": 0.0,
|
|
||||||
"max_tokens": 256,
|
|
||||||
},
|
|
||||||
timeout=120,
|
|
||||||
)
|
|
||||||
resp.raise_for_status()
|
|
||||||
content = resp.json()["choices"][0]["message"]["content"] or ""
|
|
||||||
return [ln.strip() for ln in content.splitlines() if ln.strip()]
|
|
||||||
|
|
||||||
def v5s_ocr():
|
|
||||||
# Lazily load the heavier PP-OCRv5 server det/rec pair (the main
|
|
||||||
# pipeline runs PP-OCRv6_medium). Cached on the app object so
|
|
||||||
# repeated probes don't reload weights.
|
|
||||||
if not hasattr(app.state, "v5s_ocr"):
|
|
||||||
app.state.v5s_ocr = PaddleOCR(
|
|
||||||
text_detection_model_name="PP-OCRv5_server_det",
|
|
||||||
text_recognition_model_name="PP-OCRv5_server_rec",
|
|
||||||
use_textline_orientation=True,
|
|
||||||
)
|
|
||||||
return app.state.v5s_ocr
|
|
||||||
|
|
||||||
base = np.array(crop)
|
|
||||||
readers = ("vl", "vlnl", "vlm", "v5s")
|
|
||||||
for recipe in payload.recipes:
|
|
||||||
try:
|
|
||||||
ops = [op.strip().lower() for op in recipe.split(",")]
|
|
||||||
reader = next((o for o in ops if o in readers), None)
|
|
||||||
arr = apply_ops(base.copy(), ",".join(o for o in ops if o not in readers))
|
|
||||||
if reader == "vl":
|
|
||||||
lines = vl_read(arr, use_layout=True)
|
|
||||||
elif reader == "vlnl":
|
|
||||||
lines = vl_read(arr, use_layout=False)
|
|
||||||
elif reader == "vlm":
|
|
||||||
lines = vlm_read(arr)
|
|
||||||
elif reader == "v5s":
|
|
||||||
res = list(v5s_ocr().predict(arr))
|
|
||||||
lines = res[0].get("rec_texts", []) if res else []
|
|
||||||
else:
|
|
||||||
res = list(ocr.predict(arr))
|
|
||||||
lines = res[0].get("rec_texts", []) if res else []
|
|
||||||
d, _i, src = extract_expired_date(lines)
|
|
||||||
out["variants"][recipe] = {"size": [arr.shape[1], arr.shape[0]], "lines": lines, "date": d, "source": src}
|
|
||||||
except Exception as e:
|
|
||||||
out["variants"][recipe] = {"error": str(e)}
|
|
||||||
return out
|
|
||||||
|
|
||||||
@app.post("/classify-ocr")
|
@app.post("/classify-ocr")
|
||||||
async def classify_ocr(payload: ScanRequest):
|
async def classify_ocr(payload: ScanRequest):
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -1,9 +1,9 @@
|
|||||||
import 'dart:io';
|
import 'dart:io';
|
||||||
import 'package:flutter/material.dart';
|
import 'package:flutter/material.dart';
|
||||||
import 'package:google_fonts/google_fonts.dart';
|
import 'package:google_fonts/google_fonts.dart';
|
||||||
|
|
||||||
class AppConfig {
|
class AppConfig {
|
||||||
// ─── API Configuration ────────────────────────────────────────────────────
|
// ─── API Configuration ────────────────────────────────────────────────────
|
||||||
//
|
//
|
||||||
// Dual-Mode Configuration, tried in this order at every app startup:
|
// Dual-Mode Configuration, tried in this order at every app startup:
|
||||||
// 1. Local/LAN backend (fast, low-latency - the common case when the phone
|
// 1. Local/LAN backend (fast, low-latency - the common case when the phone
|
||||||
@@ -19,7 +19,7 @@ class AppConfig {
|
|||||||
// - Physical Device on same Wi-Fi: use the backend machine's current LAN IP
|
// - Physical Device on same Wi-Fi: use the backend machine's current LAN IP
|
||||||
// - Android Emulator: use 'http://10.0.2.2:8000/api/v1'
|
// - Android Emulator: use 'http://10.0.2.2:8000/api/v1'
|
||||||
// - iOS Simulator: use 'http://localhost:8000/api/v1'
|
// - iOS Simulator: use 'http://localhost:8000/api/v1'
|
||||||
static const String _lanBaseUrl = 'http://192.168.100.20:8000/api/v1';
|
static const String _lanBaseUrl = 'http://192.168.80.84:8000/api/v1';
|
||||||
|
|
||||||
// Ngrok public tunnel URL (update ini setiap ngrok di-restart)
|
// Ngrok public tunnel URL (update ini setiap ngrok di-restart)
|
||||||
static const String _ngrokBaseUrl = 'https://unlocated-waylon-potently.ngrok-free.dev/api/v1';
|
static const String _ngrokBaseUrl = 'https://unlocated-waylon-potently.ngrok-free.dev/api/v1';
|
||||||
@@ -107,7 +107,7 @@ class AppConfig {
|
|||||||
static const Color surfaceColor = Colors.white;
|
static const Color surfaceColor = Colors.white;
|
||||||
static const Color errorColor = Color(0xFFD32F2F);
|
static const Color errorColor = Color(0xFFD32F2F);
|
||||||
static const Color successColor = Color(0xFF388E3C);
|
static const Color successColor = Color(0xFF388E3C);
|
||||||
// DO mode color — orange cue used throughout for the DO Scan mode:
|
// DO mode color — orange cue used throughout for the DO Scan mode:
|
||||||
// document_card.dart top-label, documents_tab_switcher.dart active DO tab,
|
// document_card.dart top-label, documents_tab_switcher.dart active DO tab,
|
||||||
// camera_drawer_mode_toggle.dart DO segment decoration. Formalized as a
|
// camera_drawer_mode_toggle.dart DO segment decoration. Formalized as a
|
||||||
// shared constant so all three call sites pull from one place (was an
|
// shared constant so all three call sites pull from one place (was an
|
||||||
|
|||||||
Reference in new issue
Block a user